diff --git a/scripts/b6_run.py b/scripts/b6_run.py index 3acbc95..aa39589 100644 --- a/scripts/b6_run.py +++ b/scripts/b6_run.py @@ -83,12 +83,16 @@ DEFAULT_PER_RUNG_H100_HR_RATE = 2.0 # Shadeform spot reference SEED_RETRY_OFFSET = 1000 # second-attempt seed = primary + 1000 -# Standard rungs for B6 — locked to the v0.10 ladder. Each rung -# carries its dim + n_layers for audit reproducibility. +# Standard rungs for B6 — a SINGLE S3 rung, matching the validated +# ρ=0.614 transfer run (experiments/2026-06-transfer-credibility/run_b6_s3.sh). +# Only the S3 cell feeds the Spearman score vector, so evaluating the same +# checkpoint at S1/S2/S3 (which differ only by an advisory scale_label — +# run_ladder_eval re-runs the same checkpoint+patch per rung) is 3× redundant +# compute. `n_examples_per_task=100` caps per-task eval so the run stays inside +# the validator time budget — without it, mode-0 evaluates ALL examples (e.g. +# bigbench_qa_wikidata ≈ 20k rows), which blows the cap. _B6_STANDARD_RUNGS: tuple[LadderRungSpec, ...] = ( - LadderRungSpec(scale_label="S1", dim=256, n_layers=4), - LadderRungSpec(scale_label="S2", dim=512, n_layers=12), - LadderRungSpec(scale_label="S3", dim=768, n_layers=12), + LadderRungSpec(scale_label="S3", dim=768, n_layers=12, n_examples_per_task=100), ) @@ -189,8 +193,8 @@ def estimate_cost( """Estimate USD cost of a recipe's wall-clock at the spot rate. The wall_clock_s is from the run_ladder_eval combined report; it's - the max across rungs (per merge_rung_reports), so a 3-rung recipe at - S3=300s would produce wall_clock_s ≈ 300s. Multiply by spot rate. + the max across rungs (per merge_rung_reports), so the single S3 rung at + ≈300s produces wall_clock_s ≈ 300s. Multiply by spot rate. """ h100_hr = wall_clock_s / 3600.0 return h100_hr * per_rung_h100_hr_rate diff --git a/tests/test_b6_run.py b/tests/test_b6_run.py index 1b1921d..bf826e7 100644 --- a/tests/test_b6_run.py +++ b/tests/test_b6_run.py @@ -42,11 +42,12 @@ def _set_test_mode(monkeypatch): def _b6_recipe_with_single_rung_only() -> tuple: - """Override the standard rungs to a single S3 rung for tests against - the synthetic CLI (which hard-codes its cell key). + """Pin the rungs to a bare single S3 rung (no per-task cap) for tests + against the synthetic CLI (which hard-codes its cell key). - NOTE: run_one_recipe + run_b6 use _B6_STANDARD_RUNGS (3-rung) by - contract. For tests we monkey-patch the constant. + The canonical `_B6_STANDARD_RUNGS` is also a single S3 rung but carries + `n_examples_per_task=100`; tests use synthetic fixtures where the cap is + irrelevant, so this keeps them decoupled from that value. """ from validator.ladder import LadderRungSpec return (LadderRungSpec(scale_label="S3", dim=768, n_layers=12),)