|
16 | 16 | import pytest |
17 | 17 |
|
18 | 18 | from hpcagent_bench import experiments |
| 19 | +from hpcagent_bench.stats import population |
19 | 20 |
|
20 | 21 |
|
21 | 22 | def test_a_blank_and_filled_arm_reads_as_one_identity() -> None: |
@@ -178,37 +179,39 @@ def clean_frame() -> pd.DataFrame: |
178 | 179 | ) |
179 | 180 |
|
180 | 181 |
|
181 | | -def test_an_arm_superseded_by_a_clean_rerun_is_dropped_with_a_warning() -> None: |
182 | | - """Spec X9: the clean arm re-ran the condition from scratch because the earlier wave was wrong, |
183 | | - and the two carry one identity -- pooled, the defect the re-run exists to escape is averaged |
184 | | - back in. What survives is reported under the CONDITION's name: the suffix named the wave.""" |
185 | | - with pytest.warns(UserWarning, match="dropped 2 row"): |
186 | | - kept = experiments.drop_superseded_arm_rows(clean_frame()) |
187 | | - assert set(kept.arm) == {"cpf-llr-focus40-qwen38-c-cpf", "cpf-llr-focus40-qwen38-c-cpfsrc"} |
188 | | - assert len(kept[kept.arm == "cpf-llr-focus40-qwen38-c-cpf"]) == 2 |
| 182 | +def test_a_campaign_with_no_clean_arm_is_left_alone() -> None: |
| 183 | + frame = clean_frame() |
| 184 | + frame = frame[~frame.arm.str.endswith("-clean")] |
| 185 | + assert experiments.fold_clean_arms(frame).equals(frame) |
189 | 186 |
|
190 | 187 |
|
191 | | -def test_a_clean_rerun_supersedes_only_its_own_identity_group() -> None: |
192 | | - """The suffix says which TASKS are live for one condition, not that every other arm of the |
193 | | - campaign was re-run; a cpfsrc arm with no clean wave keeps every row.""" |
194 | | - with pytest.warns(UserWarning, match="spec X9"): |
195 | | - kept = experiments.drop_superseded_arm_rows(clean_frame()) |
196 | | - assert (kept.packet == "cpfsrc").sum() == 2 |
| 188 | +def test_a_clean_rerun_folds_into_the_arm_it_re_ran_and_keeps_every_row() -> None: |
| 189 | + """Spec X9 (2026-09-18 user rule): the suffix names a wave, not a condition, so the clean arm is |
| 190 | + reported under the arm it re-ran and both waves' rows stay for the latest run to choose from.""" |
| 191 | + kept = experiments.fold_clean_arms(clean_frame()) |
| 192 | + assert len(kept) == len(clean_frame()) |
| 193 | + assert (kept.arm == "cpf-llr-focus40-qwen38-c-cpf").sum() == 4 |
| 194 | + assert (kept.arm == "cpf-llr-focus40-qwen38-c-cpfsrc").sum() == 2 |
197 | 195 |
|
198 | 196 |
|
199 | | -def test_a_campaign_with_no_clean_arm_is_left_alone() -> None: |
200 | | - """Every wave so far ran without the suffix, and X9 must be invisible to them.""" |
201 | | - frame = clean_frame() |
202 | | - frame = frame[~frame.arm.str.endswith("-clean")] |
203 | | - assert len(experiments.drop_superseded_arm_rows(frame)) == len(frame) |
| 197 | +def test_a_one_kernel_owed_rerun_keeps_the_arms_other_kernels_and_wins_its_own() -> None: |
| 198 | + """The bug: an owed rerun is named -clean and covers a few kernels, and X9 used to drop every row |
| 199 | + of the wave it topped up -- 40 kernels became the rerun's one.""" |
| 200 | + common = {"record": "task", "run_root": "r", "language": "c", "packet": "", "harness": "claude"} |
| 201 | + rows = [ |
| 202 | + {**common, "arm": "x-qwen38-c", "job": "1", "run_id": f"a{i}", "benchmark": f"k{i}", "ts_ms": 1} |
| 203 | + for i in range(3) |
| 204 | + ] + [{**common, "arm": "x-qwen38-c-clean", "job": "2", "run_id": "b0", "benchmark": "k0", "ts_ms": 2}] |
| 205 | + latest = population.latest_runs(experiments.fold_clean_arms(pd.DataFrame(rows))) |
| 206 | + assert sorted(latest.benchmark) == ["k0", "k1", "k2"] |
| 207 | + assert latest.set_index("benchmark").loc["k0", "job"] == "2" |
| 208 | + assert set(latest.arm) == {"x-qwen38-c"} |
204 | 209 |
|
205 | 210 |
|
206 | | -def test_a_clean_arm_with_no_task_row_supersedes_nothing() -> None: |
207 | | - """A judge row alone does not say a re-run happened: the task row is what records that an agent |
208 | | - was launched under the clean arm, the same evidence X6-X8 read.""" |
209 | | - frame = clean_frame() |
210 | | - frame = frame[(frame.record != "task") | (~frame.arm.str.endswith("-clean"))] |
211 | | - assert len(experiments.drop_superseded_arm_rows(frame)) == len(frame) |
| 211 | +def test_a_blank_arm_stays_blank_through_the_fold() -> None: |
| 212 | + frame = pd.DataFrame({"arm": [math.nan, "x-c-clean"], "record": ["call", "task"]}) |
| 213 | + kept = experiments.fold_clean_arms(frame) |
| 214 | + assert experiments.is_blank(kept.arm.iloc[0]) and kept.arm.iloc[1] == "x-c" |
212 | 215 |
|
213 | 216 |
|
214 | 217 | @pytest.mark.parametrize( |
@@ -287,56 +290,6 @@ def test_nan_is_blank_but_zero_is_not() -> None: |
287 | 290 | assert not experiments.is_blank("0") |
288 | 291 |
|
289 | 292 |
|
290 | | -def test_a_clean_rerun_of_one_model_does_not_supersede_another_model(tmp_path: pathlib.Path) -> None: |
291 | | - """Spec X9 groups by identity, and an extracted table records language, packet and harness but |
292 | | - not the model. Grouping on those alone put every model's C control in one identity, so six |
293 | | - finished GPT-OSS-120B re-runs deleted Qwen3.8-27B's and Kimi-K2.7-Code's arms as well.""" |
294 | | - rows = [ |
295 | | - { |
296 | | - "arm": "cpf-llr-focus40-oss120b-c-clean", |
297 | | - "record": "task", |
298 | | - "language": "c", |
299 | | - "packet": "", |
300 | | - "harness": "claude", |
301 | | - }, |
302 | | - {"arm": "cpf-llr-focus40-oss120b-c", "record": "task", "language": "c", "packet": "", "harness": "claude"}, |
303 | | - {"arm": "cpf-llr-focus40-qwen38-c", "record": "task", "language": "c", "packet": "", "harness": "claude"}, |
304 | | - {"arm": "cpf-llr-focus40-kimi27sglang-c", "record": "task", "language": "c", "packet": "", "harness": "claude"}, |
305 | | - ] |
306 | | - with pytest.warns(UserWarning, match="superseded by a clean re-run"): |
307 | | - kept = experiments.drop_superseded_arm_rows(pd.DataFrame(rows)) |
308 | | - assert sorted(kept.arm) == [ |
309 | | - "cpf-llr-focus40-kimi27sglang-c", |
310 | | - "cpf-llr-focus40-oss120b-c", |
311 | | - "cpf-llr-focus40-qwen38-c", |
312 | | - ] |
313 | | - |
314 | | - |
315 | | -def test_a_clean_rerun_supersedes_the_arm_of_its_own_name(tmp_path: pathlib.Path) -> None: |
316 | | - """The suffix names no condition, so the clean wave replaces the wave it re-ran and nothing else.""" |
317 | | - rows = [ |
318 | | - {"arm": "x-qwen38-c-skills-clean", "record": "task", "language": "c", "packet": "lang-skills", "harness": "h"}, |
319 | | - {"arm": "x-qwen38-c-skills", "record": "task", "language": "c", "packet": "lang-skills", "harness": "h"}, |
320 | | - {"arm": "x-qwen38-c", "record": "task", "language": "c", "packet": "", "harness": "h"}, |
321 | | - ] |
322 | | - with pytest.warns(UserWarning, match="superseded by a clean re-run"): |
323 | | - kept = experiments.drop_superseded_arm_rows(pd.DataFrame(rows)) |
324 | | - assert sorted(kept.arm) == ["x-qwen38-c", "x-qwen38-c-skills"] |
325 | | - |
326 | | - |
327 | | -def test_a_surviving_clean_arm_is_reported_under_the_condition_it_re_ran() -> None: |
328 | | - """The suffix names a wave, not a condition. Left on the arm, it renames the condition for every |
329 | | - consumer downstream: a pair list, an --arms regex and a figure's arm pattern all ask by name.""" |
330 | | - rows = [ |
331 | | - {"arm": "x-qwen38-c-clean", "record": "task", "language": "c", "packet": "", "harness": "h"}, |
332 | | - {"arm": "x-qwen38-c", "record": "task", "language": "c", "packet": "", "harness": "h"}, |
333 | | - {"arm": "x-oss120b-c", "record": "task", "language": "c", "packet": "", "harness": "h"}, |
334 | | - ] |
335 | | - with pytest.warns(UserWarning, match="superseded by a clean re-run"): |
336 | | - kept = experiments.drop_superseded_arm_rows(pd.DataFrame(rows)) |
337 | | - assert sorted(kept.arm) == ["x-oss120b-c", "x-qwen38-c"] |
338 | | - |
339 | | - |
340 | 293 | def test_a_column_no_row_in_the_table_ever_recorded_still_fills_from_the_arm_name() -> None: |
341 | 294 | """The bug this guards: a column NOTHING recorded reads back from CSV as all-NaN float64, and |
342 | 295 | writing an arm's recovered language into that raised ``Invalid value 'c' for dtype 'float64'`` |
|
0 commit comments