From e7c0a3c16e1e7825e4ba589792ba04f77ebc495e Mon Sep 17 00:00:00 2001 From: John Menke Date: Sat, 26 Sep 2026 20:48:16 -0400 Subject: [PATCH 1/3] Fail closed when a benchmark baseline case is missing from the current scores. Co-authored-by: Cursor --- src/docgen/cli.py | 3 +++ src/docgen/scene_benchmark.py | 20 +++++++++++++++++++- tests/test_scene_benchmark.py | 31 +++++++++++++++++++++++++++++++ 3 files changed, 53 insertions(+), 1 deletion(-) diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 9501b4c..4a4426d 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -1465,6 +1465,7 @@ def benchmark( holds, title skip, page transitions). """ from docgen.scene_benchmark import ( + baseline_scoped_to_case, compare_to_baseline, default_baseline_path, format_table, @@ -1495,6 +1496,8 @@ def benchmark( written = write_baseline(scores, base_path) click.echo(f"wrote baseline {written}") baseline = load_baseline(base_path) + if case_id: + baseline = baseline_scoped_to_case(baseline, case_id) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) if output_path: diff --git a/src/docgen/scene_benchmark.py b/src/docgen/scene_benchmark.py index 7315778..7204d7e 100644 --- a/src/docgen/scene_benchmark.py +++ b/src/docgen/scene_benchmark.py @@ -391,13 +391,29 @@ def write_baseline(scores: list[CaseScore], path: Path | None = None) -> Path: return p +def baseline_scoped_to_case(baseline: dict[str, Any], case_id: str) -> dict[str, Any]: + """Limit a baseline to one case so a filtered run is not a deleted-corpus failure.""" + stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} + scoped = dict(baseline) + scoped["cases"] = {key: value for key, value in stored.items() if key == case_id} + return scoped + + def compare_to_baseline( scores: list[CaseScore], baseline: dict[str, Any], ) -> list[str]: - """Return regression notes. Empty means the run meets or beats the baseline.""" + """Return regression notes. Empty means the run meets or beats the baseline. + + A baseline case id absent from ``scores`` is a failure. Deleting a corpus + case must not leave that committed id unchecked. + """ notes: list[str] = [] stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} + scored_ids = {score.case_id for score in scores} + for case_id in stored: + if case_id not in scored_ids: + notes.append(f"{case_id}: missing from current scores") for score in scores: prev = stored.get(score.case_id) if not isinstance(prev, dict): @@ -482,6 +498,8 @@ def build_benchmark_report( scores = run_benchmark(case_id=case_id) path = baseline_path or default_baseline_path() baseline = load_baseline(path) + if case_id: + baseline = baseline_scoped_to_case(baseline, case_id) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} diff --git a/tests/test_scene_benchmark.py b/tests/test_scene_benchmark.py index 6377b02..7f09548 100644 --- a/tests/test_scene_benchmark.py +++ b/tests/test_scene_benchmark.py @@ -11,6 +11,7 @@ from docgen.cli import main from docgen.scene_benchmark import ( BenchmarkCase, + CaseScore, compare_to_baseline, default_baseline_path, format_table, @@ -133,6 +134,36 @@ def test_full_corpus_meets_committed_baseline() -> None: assert "quality average" in table +def _tiny_score(case_id: str) -> CaseScore: + return CaseScore( + case_id=case_id, + title=case_id, + role="quality", + wait_skips=0, + overshoots=0, + hold_idle_violations=0, + cadence_violations=0, + sim_drift=0, + mid_hold_pulses=1, + box_reveals=1, + last_motion_frac=1.0, + audio_end=1.0, + defect_points=0, + quality_points=10, + score=100, + ) + + +def test_compare_flags_baseline_id_missing_from_current_scores() -> None: + """A baseline id dropped from the score list must fail (leftover #17).""" + scores = [_tiny_score("alpha"), _tiny_score("beta")] + baseline = {"version": 1, "cases": {score.case_id: score.snapshot() for score in scores}} + assert compare_to_baseline(scores, baseline) == [] + reduced = [score for score in scores if score.case_id != "beta"] + notes = compare_to_baseline(reduced, baseline) + assert notes == ["beta: missing from current scores"] + + def test_compare_flags_skip_regression() -> None: scores = run_benchmark() dump = load_baseline() From 3460370be4cb1283e2dc68f1df9374a3f293aa29 Mon Sep 17 00:00:00 2001 From: John Menke Date: Sat, 26 Sep 2026 21:26:19 -0400 Subject: [PATCH 2/3] Keep baseline-id checks without raising cyclomatic complexity. --- src/docgen/cli.py | 4 +--- src/docgen/scene_benchmark.py | 24 +++++++++++++++--------- 2 files changed, 16 insertions(+), 12 deletions(-) diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 4a4426d..1f96752 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -1495,9 +1495,7 @@ def benchmark( raise click.ClickException("--update-baseline requires the full corpus (omit --case)") written = write_baseline(scores, base_path) click.echo(f"wrote baseline {written}") - baseline = load_baseline(base_path) - if case_id: - baseline = baseline_scoped_to_case(baseline, case_id) + baseline = baseline_scoped_to_case(load_baseline(base_path), case_id) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) if output_path: diff --git a/src/docgen/scene_benchmark.py b/src/docgen/scene_benchmark.py index 7204d7e..7c545c9 100644 --- a/src/docgen/scene_benchmark.py +++ b/src/docgen/scene_benchmark.py @@ -391,14 +391,26 @@ def write_baseline(scores: list[CaseScore], path: Path | None = None) -> Path: return p -def baseline_scoped_to_case(baseline: dict[str, Any], case_id: str) -> dict[str, Any]: +def baseline_scoped_to_case(baseline: dict[str, Any], case_id: str | None) -> dict[str, Any]: """Limit a baseline to one case so a filtered run is not a deleted-corpus failure.""" + if not case_id: + return baseline stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} scoped = dict(baseline) scoped["cases"] = {key: value for key, value in stored.items() if key == case_id} return scoped +def missing_baseline_case_notes(scores: list[CaseScore], stored: dict[str, Any]) -> list[str]: + """Baseline ids that this run did not score. Deleting a case must fail.""" + scored_ids = {score.case_id for score in scores} + return [ + f"{case_id}: missing from current scores" + for case_id in stored + if case_id not in scored_ids + ] + + def compare_to_baseline( scores: list[CaseScore], baseline: dict[str, Any], @@ -408,12 +420,8 @@ def compare_to_baseline( A baseline case id absent from ``scores`` is a failure. Deleting a corpus case must not leave that committed id unchecked. """ - notes: list[str] = [] stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} - scored_ids = {score.case_id for score in scores} - for case_id in stored: - if case_id not in scored_ids: - notes.append(f"{case_id}: missing from current scores") + notes: list[str] = missing_baseline_case_notes(scores, stored) for score in scores: prev = stored.get(score.case_id) if not isinstance(prev, dict): @@ -497,9 +505,7 @@ def build_benchmark_report( """JSON payload for the CLI, wizard Vue view, and desktop GUI.""" scores = run_benchmark(case_id=case_id) path = baseline_path or default_baseline_path() - baseline = load_baseline(path) - if case_id: - baseline = baseline_scoped_to_case(baseline, case_id) + baseline = baseline_scoped_to_case(load_baseline(path), case_id) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} From 3d159859ce1b748a7c2be7f3c8e6b62fa78d52a2 Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 19:16:15 -0400 Subject: [PATCH 3/3] Keep the benchmark case filter out of cli.py. The hotspot gate fails any edit to that file, so a filtered run marks its scores and compare_to_baseline scopes the missing-id check. Co-authored-by: Cursor --- src/docgen/cli.py | 3 +-- src/docgen/scene_benchmark.py | 38 ++++++++++++++++++++++++++++++++--- tests/test_scene_benchmark.py | 3 +++ 3 files changed, 39 insertions(+), 5 deletions(-) diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 1f96752..9501b4c 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -1465,7 +1465,6 @@ def benchmark( holds, title skip, page transitions). """ from docgen.scene_benchmark import ( - baseline_scoped_to_case, compare_to_baseline, default_baseline_path, format_table, @@ -1495,7 +1494,7 @@ def benchmark( raise click.ClickException("--update-baseline requires the full corpus (omit --case)") written = write_baseline(scores, base_path) click.echo(f"wrote baseline {written}") - baseline = baseline_scoped_to_case(load_baseline(base_path), case_id) + baseline = load_baseline(base_path) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) if output_path: diff --git a/src/docgen/scene_benchmark.py b/src/docgen/scene_benchmark.py index 7c545c9..cad08ab 100644 --- a/src/docgen/scene_benchmark.py +++ b/src/docgen/scene_benchmark.py @@ -367,7 +367,38 @@ def run_benchmark( if not cases: known = ", ".join(c.id for c in standard_cases()) raise ValueError(f"unknown benchmark case {case_id!r}; known: {known}") - return [score_case(c) for c in cases] + return _mark_case_filter([score_case(c) for c in cases], case_id) + + +def _mark_case_filter(scores: list[CaseScore], case_id: str | None) -> list[CaseScore]: + """Remember an explicit ``--case`` filter so other baseline ids are not deletions.""" + if not case_id: + return scores + for score in scores: + score.filtered_case_id = case_id + return scores + + +def _filtered_case_id(scores: list[CaseScore]) -> str | None: + if not scores: + return None + marked = [getattr(score, "filtered_case_id", None) for score in scores] + if any(item is None for item in marked): + return None + if len(set(marked)) != 1: + return None + return marked[0] + + +def _stored_cases_for_scores( + baseline: dict[str, Any], + scores: list[CaseScore], +) -> dict[str, Any]: + stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} + case_id = _filtered_case_id(scores) + if not case_id: + return stored + return {key: value for key, value in stored.items() if key == case_id} def load_baseline(path: Path | None = None) -> dict[str, Any]: @@ -418,9 +449,10 @@ def compare_to_baseline( """Return regression notes. Empty means the run meets or beats the baseline. A baseline case id absent from ``scores`` is a failure. Deleting a corpus - case must not leave that committed id unchecked. + case must not leave that committed id unchecked. Scores from + ``run_benchmark(case_id=...)`` only check that one id. """ - stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} + stored = _stored_cases_for_scores(baseline, scores) notes: list[str] = missing_baseline_case_notes(scores, stored) for score in scores: prev = stored.get(score.case_id) diff --git a/tests/test_scene_benchmark.py b/tests/test_scene_benchmark.py index 7f09548..65c271c 100644 --- a/tests/test_scene_benchmark.py +++ b/tests/test_scene_benchmark.py @@ -162,6 +162,9 @@ def test_compare_flags_baseline_id_missing_from_current_scores() -> None: reduced = [score for score in scores if score.case_id != "beta"] notes = compare_to_baseline(reduced, baseline) assert notes == ["beta: missing from current scores"] + kept = next(score for score in scores if score.case_id == "alpha") + kept.filtered_case_id = "alpha" + assert compare_to_baseline([kept], baseline) == [] def test_compare_flags_skip_regression() -> None: