From 8b79f9dc63564b4042f67ad75f1fe1e9a9a9c11a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 17:19:36 +0000 Subject: [PATCH 01/14] chore(metrics): record plan-approval audit entry for #2168+#2169+#2170 batch Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- .claude/metrics/config-changelog.jsonl | 1 + 1 file changed, 1 insertion(+) diff --git a/.claude/metrics/config-changelog.jsonl b/.claude/metrics/config-changelog.jsonl index 5199a8c98..3416a8467 100644 --- a/.claude/metrics/config-changelog.jsonl +++ b/.claude/metrics/config-changelog.jsonl @@ -84,3 +84,4 @@ {"timestamp": "2026-09-17T15:36:16.556298+00:00", "type": "approval", "proposed": "acceptance-criteria set: plans/measure-rereview-duplication.md (issue #2165)", "evidence_shown": "plans/measure-rereview-duplication.md", "risks_surfaced": ["AC6: cross-reference to step 1.4 checkpoint list not inlined", "AC8: additive-alias pattern not defined in AC text"]} {"timestamp": "2026-09-21T18:49:13Z", "type": "approval", "proposed": "Approve plan for Agent lifecycle improvements (#2172 batch: #2187-#2190)", "evidence_shown": "plans/2172-agent-lifecycle-improvements.md", "risks_surfaced": ["Strategic critic recommended splitting into up to 3 PRs; acknowledged, not adopted (user pre-decided single-batch)", "Slice 2 malformed-hand-back scenario is provisional pending Step 2.1a transcript-shape confirmation", "Slice 4 JS/TS fast-check test may need network-exempt fallback if vendoring proves impractical"], "description": "Auto-approved (non-interactive) - no usable TTY in this remote session"} {"timestamp": "2026-09-21T18:51:16Z", "type": "approval", "proposed": "Acceptance-criteria gate for plans/2172-agent-lifecycle-improvements.md", "evidence_shown": "plans/2172-agent-lifecycle-improvements.md", "risks_surfaced": ["Step 1.4 latency check has no quantitative SLA threshold", "Step 1.2 missing idempotent-double-fire test case", "Step 2.1a conditional AC pass/fail ambiguity if no hand-back signal found", "Step 2.1b missing invalid-JSON (not just unreadable) transcript test", "Step 3.1 claim-extraction heuristics under-specified with only 2 of 5 pinned", "Step 3.2 missing malformed-but-parseable WebFetch response case", "Step 3.3 doc-review integration test is structural-only, not behavioral", "Step 4.1 property-derivation edge cases (cross-module pair, ambiguous invariant) untested", "Step 4.2 missing language-specific runtime-failure cases (Hypothesis/fast-check install failure)", "Step 4.3 vendoring-impractical threshold undefined"], "description": "Acceptance-criteria gate auto-passed with 10 flagged (0 blocker, 2 warning, 8 suggestion/minor) criterion findings (non-interactive) - no human gate. Trigger: --yes flag. Findings will be resolved as implementation-time decisions during Step 4, per the plan own assumptions-recording convention."} +{"timestamp": "2026-09-22T17:18:55Z", "type": "approval", "proposed": "Approve plan for #2168+#2169+#2170 batch (abort-on-cheap-blocker, countable test-review pilot, tiered findings output)", "evidence_shown": "plans/2164-abort-countable-tiered.md", "risks_surfaced": ["Strategic critic questioned bundling 3 independent slices into one PR (revert blast-radius, unsubstantiated Slice-3 cost-cutting framing) - acknowledged in Goal section, not split (mirrors prior epic batching pattern)", "Slice 2's mechanical/judgment classification (Step 2.1) is this plan's own derivation, not a separate authority - may need revision if /agent-eval shows a detection regression", "Slice 3's report-size measurement acceptance criterion needs a real multi-finding review round and is deferred to a post-merge follow-up comment on issue #2164, not a build step"], "description": "Auto-approved (non-interactive) - no usable TTY in this remote session. Trigger: --yes. Plan review ran to convergence first: 4 reviewers dispatched, Acceptance and Design each required 3 rounds to resolve real blockers (finding-id collision, severity-vocabulary mismatch with internal_double_detector.py, an unfalsifiable fixture test); all now approve."} From 27186ca6def66e1371200dc12f528c1034691107 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 17:43:09 +0000 Subject: [PATCH 02/14] =?UTF-8?q?feat(scripts):=20add=20checkpoint=5Fabort?= =?UTF-8?q?.py=20=E2=80=94=20cheap-lens=20blocker=20decision?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/scripts/checkpoint_abort.py | 270 ++++++++++++++++ .../tests/scripts/test_checkpoint_abort.py | 289 ++++++++++++++++++ tests/repo/test_python_floor.py | 1 + 3 files changed, 560 insertions(+) create mode 100755 plugins/dev-team/scripts/checkpoint_abort.py create mode 100644 plugins/dev-team/tests/scripts/test_checkpoint_abort.py diff --git a/plugins/dev-team/scripts/checkpoint_abort.py b/plugins/dev-team/scripts/checkpoint_abort.py new file mode 100755 index 000000000..b14fec8a9 --- /dev/null +++ b/plugins/dev-team/scripts/checkpoint_abort.py @@ -0,0 +1,270 @@ +#!/usr/bin/env python3 +"""Abort remaining checkpoint lens dispatch on a cheap-tier blocker (#2168). + +``/build``'s inline review checkpoints dispatch review lenses cheap-first +(``select_lenses.py``'s ordering). When a cheap-tier lens already reports an +``error``-severity, ``high``-confidence finding, dispatching the remaining +(opus-tier) lenses for that round is very likely wasted spend — the fix loop +will re-dispatch everything anyway once the cheap-tier finding is addressed. +This module makes that abort decision, plus the small amount of pure +aggregation bookkeeping the checkpoint needs around it. + +Three entry points: + +1. ``decide_abort`` (and its CLI wrapper in ``main``) — the abort decision + itself, given the cheap-tier lenses' finding JSON and the checkpoint's + full ordered lens list. +2. ``compute_round_outcome`` — a pure function the checkpoint's + outcome-reporting calls at the end of a round; it is the single place + that guarantees an aborted round whose deferred lenses never re-dispatched + cannot report a clean pass. +3. ``merge_findings`` — the one piece of production aggregation logic this + script ships: dedup-and-append, used both by ``/build``'s SKILL.md prose + (Step 1.2) when folding re-dispatched deferred-lens findings back into a + round's finding set, and by this module's own fixture equivalence test, + so that test exercises real shipped code rather than a test-local + reimplementation of merging. + +Stdlib-only. See docs/python-hook-contract.md. +""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +# The bar this script's abort decision applies. Deliberately STRICTER than +# `skills/code-review/SKILL.md` step 6a's "Severity floor (rounds >= 2)" rule +# (`error`/`warning` findings at `high`/`medium` confidence continue that +# fix loop) — this script only fires on `error` severity AND `high` +# confidence, nothing looser. The two bars exist for different decisions: +# step 6a's floor decides whether an *already-dispatched* fix loop should +# keep iterating (continuing is cheap — the agents already ran), while this +# script decides whether to SKIP dispatching opus-tier lenses at all +# (skipping is only safe to do speculatively when the signal is as strong as +# it gets). They are intentionally different values for different purposes, +# not one constant with two names — do not extract a shared constant. +QUALIFYING_SEVERITY = "error" +QUALIFYING_CONFIDENCE = "high" + + +class CheckpointAbortError(ValueError): + """Malformed cheap-tier finding input — never silently resolved into + either an abort or a no-abort decision (see `decide_abort`).""" + + +def _validate_cheap_results(cheap_results) -> None: + """Raise `CheckpointAbortError` on anything `decide_abort` cannot safely + reason about: a non-list top level, a non-list `issues` field, or an + issue missing `severity`/`confidence`. Fails loud, never loose.""" + if not isinstance(cheap_results, list): + raise CheckpointAbortError( + "cheap-tier results must be a JSON list of " + f"{{agent, issues}} objects, got {type(cheap_results).__name__}" + ) + for result in cheap_results: + if not isinstance(result, dict): + raise CheckpointAbortError( + f"cheap-tier result entry must be an object, got " + f"{type(result).__name__}: {result!r}" + ) + agent = result.get("agent") + issues = result.get("issues") + if not isinstance(issues, list): + raise CheckpointAbortError( + f"cheap-tier result for agent {agent!r} has a non-list " + f"'issues' field: {issues!r}" + ) + for issue in issues: + if not isinstance(issue, dict): + raise CheckpointAbortError( + f"cheap-tier issue for agent {agent!r} must be an " + f"object, got {type(issue).__name__}: {issue!r}" + ) + if "severity" not in issue or "confidence" not in issue: + raise CheckpointAbortError( + f"cheap-tier issue for agent {agent!r} is missing a " + f"required 'severity' or 'confidence' field: {issue!r}" + ) + + +def decide_abort(cheap_results: list, ordered_lenses: list) -> dict: + """Pure abort decision. + + ``cheap_results`` — the cheap-tier lenses' finding JSON, a list of + ``{"agent": str, "issues": [{"severity": ..., "confidence": ...}, ...]}`` + in lens-dispatch order. ``ordered_lenses`` — the checkpoint's full + ordered lens list (``select_lenses.py``'s cheap-first output). + + Returns ``{"abort": bool, "triggeringFinding": dict|None, + "triggeringAgent": str|None, "deferredLenses": list[str]}``. + + Abort fires only on the first issue, in ``cheap_results`` order (each + result's own ``issues`` list is also scanned in order), whose + ``severity == "error"`` and ``confidence == "high"`` — the first + qualifying finding wins when several qualify, never the last. An empty + ``cheap_results`` never aborts (fail-open: nothing ran, nothing to gate + on). ``deferredLenses`` is only ever non-empty when ``abort`` is true — + it names every lens in ``ordered_lenses`` that had not already reported + into ``cheap_results`` (order preserved). + + Raises `CheckpointAbortError` on malformed input — never silently + resolves to either decision. + """ + _validate_cheap_results(cheap_results) + + triggering_finding = None + triggering_agent = None + for result in cheap_results: + for issue in result["issues"]: + if ( + issue.get("severity") == QUALIFYING_SEVERITY + and issue.get("confidence") == QUALIFYING_CONFIDENCE + ): + triggering_finding = issue + triggering_agent = result.get("agent") + break + if triggering_finding is not None: + break + + if triggering_finding is None: + return { + "abort": False, + "triggeringFinding": None, + "triggeringAgent": None, + "deferredLenses": [], + } + + already_ran = {result.get("agent") for result in cheap_results} + deferred_lenses = [lens for lens in ordered_lenses if lens not in already_ran] + return { + "abort": True, + "triggeringFinding": triggering_finding, + "triggeringAgent": triggering_agent, + "deferredLenses": deferred_lenses, + } + + +def compute_round_outcome(aborted: bool, redispatched: bool, findings: list) -> dict: + """Pure function the checkpoint's outcome-reporting calls at round end. + + Returns ``{"outcome": "pass"|"blocked", "reason": str|None}``. + + When ``aborted`` is true and ``redispatched`` is false, always returns + ``"blocked"`` regardless of ``findings`` (including an empty list) — the + round cannot report a clean pass while the lenses it deferred at abort + time never actually ran. Otherwise, ``outcome`` is computed from + ``findings`` alone: any findings present -> ``"blocked"``; none -> + ``"pass"``. + """ + if aborted and not redispatched: + return { + "outcome": "blocked", + "reason": ( + "round aborted on a cheap-tier blocker and its deferred " + "lenses were never re-dispatched" + ), + } + if findings: + return {"outcome": "blocked", "reason": f"{len(findings)} finding(s) remain"} + return {"outcome": "pass", "reason": None} + + +def _finding_key(finding: dict) -> tuple: + """The dedup key `merge_findings` groups on: `(agent, file, line, + severity, message)`.""" + return ( + finding.get("agent"), + finding.get("file"), + finding.get("line"), + finding.get("severity"), + finding.get("message"), + ) + + +def merge_findings(existing: list, new: list) -> list: + """Dedupe `new` against `existing` by `(agent, file, line, severity, + message)` and append the remainder in `new`'s original order. + + Pure, no I/O — returns a new list; never mutates either argument. + """ + seen = {_finding_key(finding) for finding in existing} + merged = list(existing) + for finding in new: + key = _finding_key(finding) + if key in seen: + continue + seen.add(key) + merged.append(finding) + return merged + + +def _read_cheap_results(path_or_dash: str) -> str: + if path_or_dash == "-": + return sys.stdin.read() + return Path(path_or_dash).read_text(encoding="utf-8") + + +def main(argv=None) -> int: + parser = argparse.ArgumentParser( + description=( + "Decide whether to abort remaining opus-tier lens dispatch for " + "a /build checkpoint round, given the cheap-tier lenses' " + "results." + ) + ) + parser.add_argument( + "--cheap-results-from", + default="-", + help=( + "Path to a JSON file holding the cheap-tier lens results (a " + "list of {agent, issues: [{severity, confidence}, ...]} " + "objects) in lens-dispatch order, or '-' for stdin (default)." + ), + ) + parser.add_argument( + "--lenses", + nargs="*", + default=[], + help=( + "The checkpoint's full ordered lens list (select_lenses.py's " + "cheap-first output)." + ), + ) + args = parser.parse_args(argv) + + try: + raw = _read_cheap_results(args.cheap_results_from) + except OSError as exc: + print( + f"checkpoint_abort.py: cannot read {args.cheap_results_from}: {exc}", + file=sys.stderr, + ) + return 1 + + try: + cheap_results = json.loads(raw) + except json.JSONDecodeError as exc: + print( + f"checkpoint_abort.py: cheap-tier results are not valid JSON: {exc}", + file=sys.stderr, + ) + return 1 + + try: + result = decide_abort(cheap_results, args.lenses) + except CheckpointAbortError as exc: + print( + f"checkpoint_abort.py: malformed cheap-tier finding data: {exc}", + file=sys.stderr, + ) + return 1 + + print(json.dumps(result)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugins/dev-team/tests/scripts/test_checkpoint_abort.py b/plugins/dev-team/tests/scripts/test_checkpoint_abort.py new file mode 100644 index 000000000..796994fad --- /dev/null +++ b/plugins/dev-team/tests/scripts/test_checkpoint_abort.py @@ -0,0 +1,289 @@ +"""Unit tests for scripts/checkpoint_abort.py (#2168). + +Covers the abort decision (error/high only, first-qualifying-finding-wins, +fail-open on empty input, loud failure on malformed input), +`compute_round_outcome`'s hard "never pass on an unre-dispatched abort" rule, +`merge_findings`'s dedup-and-append contract, and the fixture equivalence +test proving `merge_findings` is order-independent under a cheap/deferred +split (Step 1.1's central safety-property test). +""" + +from __future__ import annotations + +import json +import subprocess +import sys + +import pytest + +from _repo_root import REPO_ROOT as _REPO_ROOT + +_PLUGIN_ROOT = _REPO_ROOT / "plugins" / "dev-team" +_SCRIPTS_DIR = _PLUGIN_ROOT / "scripts" +sys.path.insert(0, str(_SCRIPTS_DIR)) + +import checkpoint_abort + + +def _issue(severity="error", confidence="high", **extra): + return {"severity": severity, "confidence": confidence, **extra} + + +def _result(agent, *issues): + return {"agent": agent, "issues": list(issues)} + + +class TestDecideAbortNoAbort: + def test_clean_results_do_not_abort(self): + results = [_result("naming-review", _issue("warning", "medium"))] + out = checkpoint_abort.decide_abort(results, ["naming-review", "arch-review"]) + assert out == { + "abort": False, + "triggeringFinding": None, + "triggeringAgent": None, + "deferredLenses": [], + } + + def test_empty_cheap_results_never_aborts(self): + out = checkpoint_abort.decide_abort([], ["arch-review", "security-review"]) + assert out["abort"] is False + assert out["deferredLenses"] == [] + + def test_error_severity_without_high_confidence_never_aborts(self): + results = [_result("naming-review", _issue("error", "medium"))] + out = checkpoint_abort.decide_abort(results, ["naming-review", "arch-review"]) + assert out["abort"] is False + + def test_high_confidence_without_error_severity_never_aborts(self): + results = [_result("naming-review", _issue("warning", "high"))] + out = checkpoint_abort.decide_abort(results, ["naming-review", "arch-review"]) + assert out["abort"] is False + + +class TestDecideAbortAborts: + def test_error_high_finding_aborts_with_deferred_lenses(self): + results = [_result("naming-review", _issue("error", "high"))] + ordered = ["naming-review", "arch-review", "security-review"] + out = checkpoint_abort.decide_abort(results, ordered) + assert out["abort"] is True + assert out["triggeringFinding"] == _issue("error", "high") + assert out["triggeringAgent"] == "naming-review" + assert out["deferredLenses"] == ["arch-review", "security-review"] + + def test_first_qualifying_finding_in_lens_dispatch_order_wins(self): + results = [ + _result("naming-review", _issue("error", "high", message="first")), + _result("doc-review", _issue("error", "high", message="second")), + ] + ordered = ["naming-review", "doc-review", "arch-review"] + out = checkpoint_abort.decide_abort(results, ordered) + assert out["abort"] is True + assert out["triggeringAgent"] == "naming-review" + assert out["triggeringFinding"]["message"] == "first" + # Both cheap-tier agents already ran; only the un-run lens defers. + assert out["deferredLenses"] == ["arch-review"] + + def test_second_issue_in_a_lens_own_issue_list_can_trigger(self): + results = [ + _result( + "naming-review", + _issue("warning", "high"), + _issue("error", "high", message="second issue"), + ), + ] + out = checkpoint_abort.decide_abort(results, ["naming-review", "arch-review"]) + assert out["abort"] is True + assert out["triggeringFinding"]["message"] == "second issue" + + +class TestDecideAbortMalformedInput: + def test_missing_severity_raises(self): + results = [_result("naming-review", {"confidence": "high"})] + with pytest.raises(checkpoint_abort.CheckpointAbortError): + checkpoint_abort.decide_abort(results, ["naming-review"]) + + def test_missing_confidence_raises(self): + results = [_result("naming-review", {"severity": "error"})] + with pytest.raises(checkpoint_abort.CheckpointAbortError): + checkpoint_abort.decide_abort(results, ["naming-review"]) + + def test_non_list_issues_raises(self): + results = [{"agent": "naming-review", "issues": "not-a-list"}] + with pytest.raises(checkpoint_abort.CheckpointAbortError): + checkpoint_abort.decide_abort(results, ["naming-review"]) + + def test_non_list_top_level_raises(self): + with pytest.raises(checkpoint_abort.CheckpointAbortError): + checkpoint_abort.decide_abort({"not": "a list"}, ["naming-review"]) + + +class TestComputeRoundOutcome: + def test_aborted_and_not_redispatched_never_passes_with_no_findings(self): + out = checkpoint_abort.compute_round_outcome( + aborted=True, redispatched=False, findings=[] + ) + assert out["outcome"] == "blocked" + + def test_aborted_and_not_redispatched_never_passes_with_findings(self): + out = checkpoint_abort.compute_round_outcome( + aborted=True, redispatched=False, findings=[{"severity": "error"}] + ) + assert out["outcome"] == "blocked" + + def test_aborted_and_redispatched_computes_from_findings_empty(self): + out = checkpoint_abort.compute_round_outcome( + aborted=True, redispatched=True, findings=[] + ) + assert out["outcome"] == "pass" + + def test_aborted_and_redispatched_computes_from_findings_nonempty(self): + out = checkpoint_abort.compute_round_outcome( + aborted=True, redispatched=True, findings=[{"severity": "error"}] + ) + assert out["outcome"] == "blocked" + + def test_never_aborted_computes_from_findings(self): + out = checkpoint_abort.compute_round_outcome( + aborted=False, redispatched=False, findings=[] + ) + assert out["outcome"] == "pass" + + +class TestMergeFindings: + def _finding(self, **kw): + base = { + "agent": "naming-review", + "file": "a.py", + "line": 1, + "severity": "warning", + "message": "m", + } + base.update(kw) + return base + + def test_disjoint_lists_concatenate_in_order(self): + existing = [self._finding(line=1), self._finding(line=2)] + new = [self._finding(line=3), self._finding(line=4)] + merged = checkpoint_abort.merge_findings(existing, new) + assert merged == existing + new + + def test_overlapping_lists_dedupe(self): + dup = self._finding(line=1) + existing = [dup] + new = [dup, self._finding(line=2)] + merged = checkpoint_abort.merge_findings(existing, new) + assert merged == [dup, self._finding(line=2)] + + def test_empty_existing_returns_new_unchanged(self): + new = [self._finding(line=1), self._finding(line=2)] + merged = checkpoint_abort.merge_findings([], new) + assert merged == new + + def test_does_not_mutate_arguments(self): + existing = [self._finding(line=1)] + new = [self._finding(line=2)] + existing_copy, new_copy = list(existing), list(new) + checkpoint_abort.merge_findings(existing, new) + assert existing == existing_copy + assert new == new_copy + + +class TestFixtureEquivalence: + """Proves `merge_findings` is order-independent under a cheap/deferred + split — checkpoint_abort.py's only production aggregation logic. Whether + /build's live runtime actually invokes merge_findings correctly is + verified separately by Step 1.2's content-guard test, not here.""" + + @staticmethod + def _as_set(findings): + return {json.dumps(f, sort_keys=True) for f in findings} + + def test_cheap_then_deferred_merge_is_set_equal_to_one_shot_merge(self): + cheap_findings = [ + { + "agent": "naming-review", + "file": "src/a.py", + "line": 10, + "severity": "warning", + "message": "ambiguous name", + }, + { + "agent": "naming-review", + "file": "src/b.py", + "line": 3, + "severity": "error", + "message": "shadowed import", + }, + ] + deferred_findings = [ + { + "agent": "arch-review", + "file": "src/c.py", + "line": 42, + "severity": "warning", + "message": "layering violation", + }, + # Same key as a cheap-tier finding — proves dedup behaves + # identically whether it happens in one call or split in two. + { + "agent": "naming-review", + "file": "src/a.py", + "line": 10, + "severity": "warning", + "message": "ambiguous name", + }, + ] + + # Path A: one call, simulating a single non-aborted dispatch. + path_a = checkpoint_abort.merge_findings( + [], cheap_findings + deferred_findings + ) + # Path B: two calls, simulating cheap-tier-then-deferred-tier. + path_b = checkpoint_abort.merge_findings( + checkpoint_abort.merge_findings([], cheap_findings), deferred_findings + ) + + assert self._as_set(path_a) == self._as_set(path_b) + # Both paths should have deduped the shared-key finding to one copy. + assert len(path_a) == 3 + assert len(path_b) == 3 + + +class TestCli: + def _run(self, cheap_results, lenses, check=False): + return subprocess.run( + [sys.executable, str(_SCRIPTS_DIR / "checkpoint_abort.py"), "--lenses", *lenses], + input=json.dumps(cheap_results), + capture_output=True, + text=True, + check=check, + ) + + def test_no_abort_prints_json_and_exits_zero(self): + result = self._run([], ["naming-review"], check=True) + payload = json.loads(result.stdout) + assert payload["abort"] is False + + def test_abort_prints_deferred_lenses_and_exits_zero(self): + results = [_result("naming-review", _issue("error", "high"))] + result = self._run(results, ["naming-review", "arch-review"], check=True) + payload = json.loads(result.stdout) + assert payload["abort"] is True + assert payload["deferredLenses"] == ["arch-review"] + + def test_malformed_input_exits_nonzero_with_clear_error(self): + results = [_result("naming-review", {"severity": "error"})] + result = self._run(results, ["naming-review"], check=False) + assert result.returncode != 0 + assert "malformed" in result.stderr.lower() + + def test_invalid_json_exits_nonzero_with_clear_error(self): + result = subprocess.run( + [sys.executable, str(_SCRIPTS_DIR / "checkpoint_abort.py"), "--lenses", "x"], + input="not json", + capture_output=True, + text=True, + check=False, + ) + assert result.returncode != 0 + assert "json" in result.stderr.lower() diff --git a/tests/repo/test_python_floor.py b/tests/repo/test_python_floor.py index 384a05b82..af250fb6a 100644 --- a/tests/repo/test_python_floor.py +++ b/tests/repo/test_python_floor.py @@ -181,6 +181,7 @@ "check_security_assessment_mcp_tools.py": ( "stdlib argparse/json/pathlib only; no floor-sensitive runtime API" ), + "checkpoint_abort.py": "stdlib argparse/json/sys/pathlib only; no floor-sensitive runtime API", "coverage_delta_steering.py": "stdlib argparse/json/pathlib only; no floor-sensitive runtime API", "mutation_yield_steering.py": ( "stdlib argparse/json/sys/pathlib only; no floor-sensitive runtime API " From 3dfe8a58f18ec20f6a5131dc2d9a92135eeb60d6 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 18:02:45 +0000 Subject: [PATCH 03/14] feat(build): abort remaining checkpoint lenses on a cheap blocker Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/skills/build/SKILL.md | 6 +- .../test_build_checkpoint_abort_marker.py | 110 ++++++++++++++++++ 2 files changed, 113 insertions(+), 3 deletions(-) create mode 100644 plugins/dev-team/tests/skills/test_build_checkpoint_abort_marker.py diff --git a/plugins/dev-team/skills/build/SKILL.md b/plugins/dev-team/skills/build/SKILL.md index c5d042b33..5f06a6b12 100644 --- a/plugins/dev-team/skills/build/SKILL.md +++ b/plugins/dev-team/skills/build/SKILL.md @@ -193,7 +193,7 @@ Work each step **one behavior at a time** — never all the code then all the te 4. **Inline review checkpoint — granularity scales with complexity.** *Where* the checkpoint runs depends on the step's **Complexity** classification (review *depth* still scales too): - **trivial**: Skip inline review. The final `/code-review` (step 6) covers all modified files. - **standard**: **Defer** review to the slice boundary (sub-step 6) — do not review now. Track the step's changed files so the slice checkpoint reviews them in one batch. Per-step review on standard steps is N near-identical passes where one at slice end largely does the same work, and the final `/code-review` (step 6) remains the backstop. This is the batching win — fewer review dispatches per multi-step slice at bounded quality risk. - - **complex**: **Dispatch-capability gate (re-confirm here — issue #1461):** before this review dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply this checkpoint's checklist inline; report the missing capability and halt rather than marking the step done. Otherwise, review **now, per step** — smaller blast radius per fix. Run the static self-heal pass to completion first — pass, or cap-and-escalate, per `references/static-self-heal.md` — then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects for this step's changed files** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` returns the applicable lenses (see the resolver note after sub-step 6) — dispatched **cheap-first (non-opus lenses before the opus-tier `security-review`/`domain-review`/`arch-review`)**, with the review-fix loop (up to 5 iterations per `${CLAUDE_PLUGIN_ROOT}/knowledge/three-phase-workflow.md#review-loop`) — including its deterministic-first re-verification triage (`../code-review/SKILL.md` step 6a, #1610): before re-dispatching an agent to confirm a fix, prefer a close via whichever language-appropriate lint/type-check tool(s) applies to this project's own stack (not just Python's `ruff` — ESLint/`tsc` for JS/TS, `pmd` for Java, `dotnet format`/`dotnet build` for C#, etc.) plus the test suite/`grep`, when the fix is mechanical and the claim is deterministically checkable, and escalate to an agent only when semantic judgment is genuinely needed. Before each review-fix iteration, classify the finding/failure via `${CLAUDE_PLUGIN_ROOT}/knowledge/failure-routing.md` and follow its route (see the TEST-phase note above) — a security-finding class dispatches security-engineer, a reviewer-conflict class routes to human arbitration, `unclassified` stays in the generic loop. Escalate to user if the loop doesn't converge. Then **record the checkpoint outcome** (sub-step 7). + - **complex**: **Dispatch-capability gate (re-confirm here — issue #1461):** before this review dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply this checkpoint's checklist inline; report the missing capability and halt rather than marking the step done. Otherwise, review **now, per step** — smaller blast radius per fix. Run the static self-heal pass to completion first — pass, or cap-and-escalate, per `references/static-self-heal.md` — then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects for this step's changed files** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` returns the applicable lenses (see the resolver note after sub-step 6) — dispatched **cheap-first (non-opus lenses before the opus-tier `security-review`/`domain-review`/`arch-review`)**, with the review-fix loop (up to 5 iterations per `${CLAUDE_PLUGIN_ROOT}/knowledge/three-phase-workflow.md#review-loop`) — including its deterministic-first re-verification triage (`../code-review/SKILL.md` step 6a, #1610): before re-dispatching an agent to confirm a fix, prefer a close via whichever language-appropriate lint/type-check tool(s) applies to this project's own stack (not just Python's `ruff` — ESLint/`tsc` for JS/TS, `pmd` for Java, `dotnet format`/`dotnet build` for C#, etc.) plus the test suite/`grep`, when the fix is mechanical and the claim is deterministically checkable, and escalate to an agent only when semantic judgment is genuinely needed. **Abort check (#2168).** Once the cheap-tier lenses return, invoke `checkpoint_abort.py` (`python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --cheap-results-from --lenses `) before dispatching the remaining opus-tier lenses. On `abort: true`: skip dispatching the deferred opus-tier lenses this round, run the review-fix loop against the cheap-tier findings only, and once it converges, dispatch the deferred lenses and merge their findings into the round's finding set via `checkpoint_abort.merge_findings(existing, new)` — the same dedup-by-`(agent, file, line, severity, message)` function Step 1.1 already proved order-independent. This checkpoint's report must name every deferred lens and the triggering finding (`checkpoint_abort.py`'s `deferredLenses`/`triggeringFinding`/`triggeringAgent`). The round's pass/blocked outcome always comes from calling `compute_round_outcome(aborted, redispatched, findings)` — never independently reimplemented — so a round that aborted and never re-dispatched its deferred lenses cannot report a pass. On `abort: false`, dispatch every opus-tier lens normally. Before each review-fix iteration, classify the finding/failure via `${CLAUDE_PLUGIN_ROOT}/knowledge/failure-routing.md` and follow its route (see the TEST-phase note above) — a security-finding class dispatches security-engineer, a reviewer-conflict class routes to human arbitration, `unclassified` stays in the generic loop. Escalate to user if the loop doesn't converge. Then **record the checkpoint outcome** (sub-step 7). - If no complexity is specified, default to **standard**. - **UI changes (any complexity)**: After the relevant review passes (per-step for complex, at the slice checkpoint for standard), run browser verification via `/browse` in automated smoke test mode. Skip with warning if the dev server is not running. See `${CLAUDE_PLUGIN_ROOT}/knowledge/three-phase-workflow.md#phase-3-implement` Stage 3. 5. **Mark step done** — Use the Edit tool to update the plan file's `## Build Progress` section on disk: @@ -203,7 +203,7 @@ Work each step **one behavior at a time** — never all the code then all the te - After all slices are `[x]`, change `**Status**: approved` to `**Status**: in-progress`. - This disk write is the durable commit. If a `/clear` occurs, `/continue` reads `## Build Progress` to determine the resume point without needing conversation history. - **Clear freeze scope (issue #865).** When every step under the slice is `[x]` and freeze was engaged for it (dispatch bookkeeping above), run `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/build_slice_scope.py clear --hooks-dir /.claude/hooks` before starting the next slice. A slice that never engaged freeze has nothing to clear. -6. **Slice review checkpoint (batched).** **Dispatch-capability gate (re-confirm here — issue #1461):** before this checkpoint dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply the batched checkpoint's checklist inline; report the missing capability and halt rather than checking off the slice. Otherwise, when every step under the current slice is `[x]` **and** the slice had any deferred `standard` (or unspecified) steps, run **one** review pass over the slice's accumulated changed files: the static self-heal pass first (`references/static-self-heal.md`), then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` — dispatched **cheap-first (non-opus before opus-tier)**, **plus `refactor-opportunity-review` dispatched by name** (#1976). That lens declares `Scope: on-demand`, so the resolver no longer returns it; this once-per-slice dispatch is its post-GREEN home, and it is the ONLY place `/build` runs it — do not also dispatch it per behavior at the REFACTOR phase (that would spend more, not less, than the per-diff panel slot it replaced) and do not treat the resolver's silence as "this lens was dropped". Its subject is a slice's accumulated shape — semantic vs. structural duplication across everything the slice touched — which is visible here and not in any single behavior's diff. Apply the same review-fix loop (up to 5 iterations; escalate if it doesn't converge). `trivial`-only and all-`complex` slices have nothing to batch — skip this pass. Then **record the checkpoint outcome** (sub-step 7). +6. **Slice review checkpoint (batched).** **Dispatch-capability gate (re-confirm here — issue #1461):** before this checkpoint dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply the batched checkpoint's checklist inline; report the missing capability and halt rather than checking off the slice. Otherwise, when every step under the current slice is `[x]` **and** the slice had any deferred `standard` (or unspecified) steps, run **one** review pass over the slice's accumulated changed files: the static self-heal pass first (`references/static-self-heal.md`), then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` — dispatched **cheap-first (non-opus before opus-tier)**, **plus `refactor-opportunity-review` dispatched by name** (#1976). That lens declares `Scope: on-demand`, so the resolver no longer returns it; this once-per-slice dispatch is its post-GREEN home, and it is the ONLY place `/build` runs it — do not also dispatch it per behavior at the REFACTOR phase (that would spend more, not less, than the per-diff panel slot it replaced) and do not treat the resolver's silence as "this lens was dropped". Its subject is a slice's accumulated shape — semantic vs. structural duplication across everything the slice touched — which is visible here and not in any single behavior's diff. **Abort check (#2168).** Once the cheap-tier lenses (from the cheap-first dispatch above) return, invoke `checkpoint_abort.py` (`python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --cheap-results-from --lenses `) before dispatching the remaining opus-tier lenses. On `abort: true`: skip dispatching the deferred opus-tier lenses this round, run the review-fix loop against the cheap-tier findings only, and once it converges, dispatch the deferred lenses and merge their findings into the round's finding set via `checkpoint_abort.merge_findings(existing, new)` — the same dedup-by-`(agent, file, line, severity, message)` function Step 1.1 already proved order-independent. This checkpoint's report must name every deferred lens and the triggering finding (`checkpoint_abort.py`'s `deferredLenses`/`triggeringFinding`/`triggeringAgent`). The round's pass/blocked outcome always comes from calling `compute_round_outcome(aborted, redispatched, findings)` — never independently reimplemented — so a round that aborted and never re-dispatched its deferred lenses cannot report a pass. On `abort: false`, dispatch every opus-tier lens normally. Either way, apply the same review-fix loop (up to 5 iterations; escalate if it doesn't converge). `trivial`-only and all-`complex` slices have nothing to batch — skip this pass. Then **record the checkpoint outcome** (sub-step 7). **Verification-mode re-dispatch (#1628) — applies to both checkpoint fix loops above (sub-steps 4 and 6).** When a checkpoint re-dispatches an agent to CONFIRM a fix (rather than to discover new problems), send the narrowed verification payload — the finding, the fix diff hunks ± ~20 lines, and the agent's lens definition, never the full file set — with the mandatory `insufficient-context` escape, and resolve the agent's verification tier with `python3 "$CLAUDE_PLUGIN_ROOT/scripts/verify_tier.py" --agent `. The payload contract, the escape's escalation path, and the declared-never-inferred tier-down rule are stated once in [`../../knowledge/verification-mode.md`](../../knowledge/verification-mode.md) and are not restated here. @@ -233,7 +233,7 @@ Work each step **one behavior at a time** — never all the code then all the te The three rules — hard round cap at 4, severity floor from round 2 (only `error`/`warning` at `high`/`medium` confidence justifies another round; suggestion-tier findings are logged, never chased), and loop-until-dry — are stated once in [`../code-review/SKILL.md`](../code-review/SKILL.md) step 6a and implemented once in that script. They are not restated here: this is the same contract, same implementation, reached from a different caller. A `round-cap` verdict escalates to the user with the ledger attached, exactly like a non-converging loop. - **Resolver note (`select_lenses.py`, #1516).** The resolver reads each review agent's `Scope:` declaration and returns only the lenses whose domain matches the changed files, so a backend-only diff does not dispatch frontend/UI lenses (a11y, js-fp, component-architecture, svelte) that would only no-op — while the `Scope: always` lenses (correctness, security, structure, spec-compliance, domain, arch, …) run on every checkpoint — with one exception (#1923): `correctness-review` additionally drops out when every changed file is non-executable (docs/config/assets/lockfiles, never functional Claude-config markdown), since it self-skips that diff shape per its own `## Skip` clause anyway; the resolver removes it from `lenses` before dispatch rather than paying for that self-reported skip, and records the drop as a `skipped-non-executable:correctness-review` entry in `warnings`. No other `Scope: always` lens is affected by this. It prints `{"lenses":[...cheap-first...],"warnings":[...]}`; dispatch the `lenses` in order and surface `warnings` in the checkpoint's own findings/telemetry the same way `/code-review` does (see `../code-review/SKILL.md`'s resolver-eligibility section for the full warning-shape list). The final `/code-review` (step 6) remains the backstop and applies its own selection independently. + **Resolver note (`select_lenses.py`, #1516).** The resolver reads each review agent's `Scope:` declaration and returns only the lenses whose domain matches the changed files, so a backend-only diff does not dispatch frontend/UI lenses (a11y, js-fp, component-architecture, svelte) that would only no-op — while the `Scope: always` lenses (correctness, security, structure, spec-compliance, domain, arch, …) run on every checkpoint — with one exception (#1923): `correctness-review` additionally drops out when every changed file is non-executable (docs/config/assets/lockfiles, never functional Claude-config markdown), since it self-skips that diff shape per its own `## Skip` clause anyway; the resolver removes it from `lenses` before dispatch rather than paying for that self-reported skip, and records the drop as a `skipped-non-executable:correctness-review` entry in `warnings`. No other `Scope: always` lens is affected by this. It prints `{"lenses":[...cheap-first...],"warnings":[...]}`; dispatch the `lenses` in order and surface `warnings` in the checkpoint's own findings/telemetry the same way `/code-review` does (see `../code-review/SKILL.md`'s resolver-eligibility section for the full warning-shape list). The final `/code-review` (step 6) remains the backstop and applies its own selection independently. **Scope note (#2168):** `/code-review` step 4's parallel bounded-wave dispatch (`dispatch_waves.py`) computes and fires whole waves sized only by `maxParallel`, with no cheap-tier/opus-tier ordering boundary within a wave, so this `checkpoint_abort.py` abort-on-cheap-blocker optimization is **not** ported there — it is scoped to `/build`'s own checkpoints (sub-steps 4 and 6) only. 7. **Record review value (#348).** For **each** checkpoint that runs (per-step `complex` in sub-step 4, and per-slice in sub-step 6), check `~/.claude/telemetry.json` consent first — if consent is not enabled, skip this step entirely (no file is written). When consent is enabled, append one JSON line to `.claude/metrics/review-value.jsonl` capturing whether review actually changed anything — counts and outcomes only, never code or file content (consistent with the cost meter's privacy boundary). Schema in `performance-metrics`: ```json diff --git a/plugins/dev-team/tests/skills/test_build_checkpoint_abort_marker.py b/plugins/dev-team/tests/skills/test_build_checkpoint_abort_marker.py new file mode 100644 index 000000000..243a06c37 --- /dev/null +++ b/plugins/dev-team/tests/skills/test_build_checkpoint_abort_marker.py @@ -0,0 +1,110 @@ +"""Content-guard test for skills/build/SKILL.md's abort-on-cheap-blocker +wiring (#2168, Step 1.2). + +This is a prose-presence check only -- the actual redispatch/merge/no-pass- +without-redispatch behavior this prose describes is proven separately, by +Step 1.1's fixture equivalence test (tests/scripts/test_checkpoint_abort.py, +which exercises the real `merge_findings` function this prose names) and +`compute_round_outcome` unit tests. This file only proves the procedure text +instructing `/build` to follow that behavior is present, in both checkpoint +sub-steps, and names the correct function. Mirrors the established content- +guard style in tests/skills/test_code_review_synthesis_verbosity.py. +""" + +from __future__ import annotations + +from _repo_root import REPO_ROOT + +SKILL = REPO_ROOT / "plugins" / "dev-team" / "skills" / "build" / "SKILL.md" + +_STEP4_START = "- **complex**:" +_STEP4_END = "- If no complexity is specified" +_STEP6_START = "6. **Slice review checkpoint (batched).**" +_STEP6_END = "7. **Record review value" + + +def _skill_text() -> str: + return SKILL.read_text(encoding="utf-8") + + +def _section(text: str, start: str, end: str) -> str: + assert start in text, f"marker not found: {start!r}" + after_start = text.split(start, 1)[1] + assert end in after_start, f"marker not found: {end!r}" + return after_start.split(end, 1)[0] + + +def _step4_section(text: str) -> str: + return _section(text, _STEP4_START, _STEP4_END) + + +def _step6_section(text: str) -> str: + return _section(text, _STEP6_START, _STEP6_END) + + +def test_step4_complex_checkpoint_references_checkpoint_abort_script() -> None: + section = _step4_section(_skill_text()) + assert "checkpoint_abort.py" in section, ( + "sub-step 4 (complex per-step checkpoint) must invoke " + "checkpoint_abort.py before dispatching opus-tier lenses" + ) + + +def test_step4_complex_checkpoint_names_merge_findings() -> None: + section = _step4_section(_skill_text()) + assert "checkpoint_abort.merge_findings(existing, new)" in section, ( + "sub-step 4 must name checkpoint_abort.merge_findings(existing, new) " + "as the deferred-lens re-dispatch aggregation step" + ) + + +def test_step4_complex_checkpoint_uses_compute_round_outcome() -> None: + section = _step4_section(_skill_text()) + assert "compute_round_outcome(aborted, redispatched, findings)" in section, ( + "sub-step 4's pass/blocked outcome must come from calling " + "compute_round_outcome, not independently-reimplemented logic" + ) + + +def test_step6_slice_checkpoint_references_checkpoint_abort_script() -> None: + section = _step6_section(_skill_text()) + assert "checkpoint_abort.py" in section, ( + "sub-step 6 (slice review checkpoint) must invoke checkpoint_abort.py " + "before dispatching opus-tier lenses" + ) + + +def test_step6_slice_checkpoint_names_merge_findings() -> None: + section = _step6_section(_skill_text()) + assert "checkpoint_abort.merge_findings(existing, new)" in section, ( + "sub-step 6 must name checkpoint_abort.merge_findings(existing, new) " + "as the deferred-lens re-dispatch aggregation step" + ) + + +def test_step6_slice_checkpoint_uses_compute_round_outcome() -> None: + section = _step6_section(_skill_text()) + assert "compute_round_outcome(aborted, redispatched, findings)" in section, ( + "sub-step 6's pass/blocked outcome must come from calling " + "compute_round_outcome, not independently-reimplemented logic" + ) + + +def test_both_checkpoints_require_naming_deferred_lenses_and_trigger() -> None: + text = _skill_text() + for section in (_step4_section(text), _step6_section(text)): + assert "name every deferred lens" in section, ( + "checkpoint report must name every deferred lens and the " + "triggering finding (loud abort)" + ) + assert "triggering finding" in section + + +def test_code_review_step4_non_support_sentence_is_stated() -> None: + text = _skill_text() + assert "dispatch_waves.py" in text + assert "**not** ported there" in text, ( + "SKILL.md must state that /code-review step 4's parallel " + "bounded-wave dispatch does not support this abort optimization" + ) + assert "scoped to `/build`'s own checkpoints" in text From 908666f38ac82e60e6179d9f187da83fe61fee23 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 18:28:03 +0000 Subject: [PATCH 04/14] fix(build): make checkpoint_abort's merge/outcome calls invocable, fix severity-floor filtering Slice 1 review checkpoint (#2168) surfaced 9 confirmed findings across concurrency/doc/naming/performance/structure/test/test-smell/spec-compliance/ arch/correctness/domain/security-review and refactor-opportunity-review: - checkpoint_abort.py's merge_findings/compute_round_outcome had no CLI, so SKILL.md prose telling the orchestrator to "call" them was unexecutable (domain-review + arch-review, independently confirmed). Added --mode {abort,outcome,merge} to main(). - The abort-check paragraph was duplicated near-verbatim across build SKILL.md sub-steps 4 and 6 (structure-review + refactor-opportunity-review). Extracted into one shared block matching this file's existing cross-checkpoint-rule convention. - decide_abort's "abort" output key vs compute_round_outcome's "aborted" parameter were inconsistently named (naming-review). Standardized on "aborted". - compute_round_outcome blocked on any finding, not filtered by the shared severity floor (error/warning at high/medium confidence) SKILL.md explicitly binds it to (correctness-review). Added the filter. - _validate_cheap_results didn't validate the agent field, silently corrupting deferredLenses/triggeringAgent on malformed input (correctness-review). Now fails loud. - Documented _finding_key's deliberate divergence from finding_signature.py's canonical identity relation, mirroring gate_retry_state.py's established precedent (domain-review + arch-review). - Added missing test coverage: compute_round_outcome's reason field, --cheap-results-from's file-path branch (test-review). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/scripts/checkpoint_abort.py | 308 +++++++++++++++--- plugins/dev-team/skills/build/SKILL.md | 6 +- .../tests/scripts/test_checkpoint_abort.py | 250 +++++++++++++- .../test_build_checkpoint_abort_marker.py | 85 +++-- 4 files changed, 550 insertions(+), 99 deletions(-) diff --git a/plugins/dev-team/scripts/checkpoint_abort.py b/plugins/dev-team/scripts/checkpoint_abort.py index b14fec8a9..9467a72df 100755 --- a/plugins/dev-team/scripts/checkpoint_abort.py +++ b/plugins/dev-team/scripts/checkpoint_abort.py @@ -9,21 +9,22 @@ This module makes that abort decision, plus the small amount of pure aggregation bookkeeping the checkpoint needs around it. -Three entry points: - -1. ``decide_abort`` (and its CLI wrapper in ``main``) — the abort decision - itself, given the cheap-tier lenses' finding JSON and the checkpoint's - full ordered lens list. -2. ``compute_round_outcome`` — a pure function the checkpoint's - outcome-reporting calls at the end of a round; it is the single place - that guarantees an aborted round whose deferred lenses never re-dispatched - cannot report a clean pass. -3. ``merge_findings`` — the one piece of production aggregation logic this - script ships: dedup-and-append, used both by ``/build``'s SKILL.md prose - (Step 1.2) when folding re-dispatched deferred-lens findings back into a - round's finding set, and by this module's own fixture equivalence test, - so that test exercises real shipped code rather than a test-local - reimplementation of merging. +Three entry points, each reachable from the CLI in ``main`` via ``--mode`` +(``abort`` is the default, for backward compatibility): + +1. ``decide_abort`` (``--mode abort``, default) — the abort decision itself, + given the cheap-tier lenses' finding JSON and the checkpoint's full + ordered lens list. +2. ``compute_round_outcome`` (``--mode outcome``) — a pure function the + checkpoint's outcome-reporting calls at the end of a round; it is the + single place that guarantees an aborted round whose deferred lenses never + re-dispatched cannot report a clean pass. +3. ``merge_findings`` (``--mode merge``) — the one piece of production + aggregation logic this script ships: dedup-and-append, used both by + ``/build``'s SKILL.md prose (Step 1.2) when folding re-dispatched + deferred-lens findings back into a round's finding set, and by this + module's own fixture equivalence test, so that test exercises real + shipped code rather than a test-local reimplementation of merging. Stdlib-only. See docs/python-hook-contract.md. """ @@ -49,6 +50,17 @@ QUALIFYING_SEVERITY = "error" QUALIFYING_CONFIDENCE = "high" +# The severity floor `compute_round_outcome` filters `findings` by — the same +# bar stated once in `skills/code-review/SKILL.md` step 6a and restated at +# `skills/build/SKILL.md`'s round-ledger section: only a finding at +# `error`/`warning` severity AND `high`/`medium` confidence keeps a round +# from converging. Suggestion-tier and low-confidence findings are "logged, +# never chased" and must not block. Deliberately looser than +# QUALIFYING_SEVERITY/QUALIFYING_CONFIDENCE above, which is a different bar +# for a different decision (see the comment on those constants). +BLOCKING_SEVERITIES = frozenset({"error", "warning"}) +BLOCKING_CONFIDENCES = frozenset({"high", "medium"}) + class CheckpointAbortError(ValueError): """Malformed cheap-tier finding input — never silently resolved into @@ -57,8 +69,9 @@ class CheckpointAbortError(ValueError): def _validate_cheap_results(cheap_results) -> None: """Raise `CheckpointAbortError` on anything `decide_abort` cannot safely - reason about: a non-list top level, a non-list `issues` field, or an - issue missing `severity`/`confidence`. Fails loud, never loose.""" + reason about: a non-list top level, a non-list `issues` field, a result + missing a non-empty `agent` string, or an issue missing + `severity`/`confidence`. Fails loud, never loose.""" if not isinstance(cheap_results, list): raise CheckpointAbortError( "cheap-tier results must be a JSON list of " @@ -71,6 +84,11 @@ def _validate_cheap_results(cheap_results) -> None: f"{type(result).__name__}: {result!r}" ) agent = result.get("agent") + if not isinstance(agent, str) or not agent: + raise CheckpointAbortError( + f"cheap-tier result entry has a missing or empty 'agent' " + f"field: {result!r}" + ) issues = result.get("issues") if not isinstance(issues, list): raise CheckpointAbortError( @@ -98,7 +116,7 @@ def decide_abort(cheap_results: list, ordered_lenses: list) -> dict: in lens-dispatch order. ``ordered_lenses`` — the checkpoint's full ordered lens list (``select_lenses.py``'s cheap-first output). - Returns ``{"abort": bool, "triggeringFinding": dict|None, + Returns ``{"aborted": bool, "triggeringFinding": dict|None, "triggeringAgent": str|None, "deferredLenses": list[str]}``. Abort fires only on the first issue, in ``cheap_results`` order (each @@ -106,7 +124,7 @@ def decide_abort(cheap_results: list, ordered_lenses: list) -> dict: ``severity == "error"`` and ``confidence == "high"`` — the first qualifying finding wins when several qualify, never the last. An empty ``cheap_results`` never aborts (fail-open: nothing ran, nothing to gate - on). ``deferredLenses`` is only ever non-empty when ``abort`` is true — + on). ``deferredLenses`` is only ever non-empty when ``aborted`` is true — it names every lens in ``ordered_lenses`` that had not already reported into ``cheap_results`` (order preserved). @@ -131,7 +149,7 @@ def decide_abort(cheap_results: list, ordered_lenses: list) -> dict: if triggering_finding is None: return { - "abort": False, + "aborted": False, "triggeringFinding": None, "triggeringAgent": None, "deferredLenses": [], @@ -140,24 +158,45 @@ def decide_abort(cheap_results: list, ordered_lenses: list) -> dict: already_ran = {result.get("agent") for result in cheap_results} deferred_lenses = [lens for lens in ordered_lenses if lens not in already_ran] return { - "abort": True, + "aborted": True, "triggeringFinding": triggering_finding, "triggeringAgent": triggering_agent, "deferredLenses": deferred_lenses, } +def _is_blocking_finding(finding) -> bool: + """A finding counts toward `compute_round_outcome`'s blocked/pass verdict + only at `BLOCKING_SEVERITIES`/`BLOCKING_CONFIDENCES` — the same + severity-floor bar as `skills/code-review/SKILL.md` step 6a.""" + return ( + isinstance(finding, dict) + and finding.get("severity") in BLOCKING_SEVERITIES + and finding.get("confidence") in BLOCKING_CONFIDENCES + ) + + def compute_round_outcome(aborted: bool, redispatched: bool, findings: list) -> dict: """Pure function the checkpoint's outcome-reporting calls at round end. Returns ``{"outcome": "pass"|"blocked", "reason": str|None}``. + Note: this ``outcome`` ("pass"/"blocked") is this round's own + cheap/opus-tier checkpoint verdict — a different vocabulary from + ``/build`` SKILL.md sub-step 7's review-value telemetry field also named + ``outcome`` (values ``no-op``/``fixed``/``escalated``/``skipped``). Do + not conflate the two. + When ``aborted`` is true and ``redispatched`` is false, always returns ``"blocked"`` regardless of ``findings`` (including an empty list) — the round cannot report a clean pass while the lenses it deferred at abort - time never actually ran. Otherwise, ``outcome`` is computed from - ``findings`` alone: any findings present -> ``"blocked"``; none -> - ``"pass"``. + time never actually ran. Otherwise, ``findings`` is first filtered down + to the ones that clear the shared severity floor + (``BLOCKING_SEVERITIES``/``BLOCKING_CONFIDENCES`` — ``error``/``warning`` + severity at ``high``/``medium`` confidence): any such finding present -> + ``"blocked"``; none -> ``"pass"``. Suggestion-tier and low-confidence + findings never block on their own, matching + ``skills/code-review/SKILL.md`` step 6a's floor. """ if aborted and not redispatched: return { @@ -167,11 +206,37 @@ def compute_round_outcome(aborted: bool, redispatched: bool, findings: list) -> "lenses were never re-dispatched" ), } - if findings: - return {"outcome": "blocked", "reason": f"{len(findings)} finding(s) remain"} + blocking = [finding for finding in findings if _is_blocking_finding(finding)] + if blocking: + return {"outcome": "blocked", "reason": f"{len(blocking)} finding(s) remain"} return {"outcome": "pass", "reason": None} +## Merge-dedup identity intentionally diverges from finding_signature.py, +## deliberately not imported +# +# `_finding_key`'s `(agent, file, line, severity, message)` tuple is a +# DIFFERENT, stricter identity relation than the repo's canonical +# round-ledger relation in +# `skills/code-review/scripts/finding_signature.py`, which hashes +# `(agent, file, category, normalized message)` with `LINE_TOLERANCE = 3` — +# deliberately EXCLUDING exact line and raw message per that module's own +# docstring, because it must match a finding across review ROUNDS, where a +# fix shifts line numbers slightly. +# +# `merge_findings` solves a different problem: folding a cheap-tier +# dispatch's findings back together with a deferred-tier dispatch's findings +# from the SAME round, against an unmoved diff. A cheap/deferred split needs +# exact positional identity here — tolerating a line shift or normalizing +# the message would over-merge two textually-similar-but-distinct findings +# at different lines into one. This is an approved design decision (plan +# review, #2168), not unnoticed drift. +# +# Mirrors `skills/pr/scripts/gate_retry_state.py`'s "Design mirrors +# finding_signature.py, deliberately not imported" section — same repo +# pattern, applied to a different pair of modules. See +# `TestMergeDedupIdentityDivergence` in `test_checkpoint_abort.py` for the +# drift-awareness test pinning this relationship. def _finding_key(finding: dict) -> tuple: """The dedup key `merge_findings` groups on: `(agent, file, line, severity, message)`.""" @@ -201,42 +266,60 @@ def merge_findings(existing: list, new: list) -> list: return merged -def _read_cheap_results(path_or_dash: str) -> str: +def _read_text(path_or_dash: str) -> str: if path_or_dash == "-": return sys.stdin.read() return Path(path_or_dash).read_text(encoding="utf-8") -def main(argv=None) -> int: - parser = argparse.ArgumentParser( - description=( - "Decide whether to abort remaining opus-tier lens dispatch for " - "a /build checkpoint round, given the cheap-tier lenses' " - "results." +def _validate_outcome_input(data) -> None: + """Raise `CheckpointAbortError` on anything `--mode outcome` cannot + safely pass to `compute_round_outcome`. Mirrors `_validate_cheap_results`' + fails-loud style.""" + if not isinstance(data, dict): + raise CheckpointAbortError( + "--mode outcome input must be a JSON object with 'aborted', " + f"'redispatched', and 'findings' keys, got {type(data).__name__}" ) - ) - parser.add_argument( - "--cheap-results-from", - default="-", - help=( - "Path to a JSON file holding the cheap-tier lens results (a " - "list of {agent, issues: [{severity, confidence}, ...]} " - "objects) in lens-dispatch order, or '-' for stdin (default)." - ), - ) - parser.add_argument( - "--lenses", - nargs="*", - default=[], - help=( - "The checkpoint's full ordered lens list (select_lenses.py's " - "cheap-first output)." - ), - ) - args = parser.parse_args(argv) + if not isinstance(data.get("aborted"), bool): + raise CheckpointAbortError( + "--mode outcome input 'aborted' must be a boolean, got " + f"{data.get('aborted')!r}" + ) + if not isinstance(data.get("redispatched"), bool): + raise CheckpointAbortError( + "--mode outcome input 'redispatched' must be a boolean, got " + f"{data.get('redispatched')!r}" + ) + if not isinstance(data.get("findings"), list): + raise CheckpointAbortError( + "--mode outcome input 'findings' must be a list, got " + f"{data.get('findings')!r}" + ) + + +def _validate_merge_input(data) -> None: + """Raise `CheckpointAbortError` on anything `--mode merge` cannot safely + pass to `merge_findings`. Mirrors `_validate_cheap_results`' fails-loud + style.""" + if not isinstance(data, dict): + raise CheckpointAbortError( + "--mode merge input must be a JSON object with 'existing' and " + f"'new' keys, got {type(data).__name__}" + ) + if not isinstance(data.get("existing"), list): + raise CheckpointAbortError( + f"--mode merge input 'existing' must be a list, got {data.get('existing')!r}" + ) + if not isinstance(data.get("new"), list): + raise CheckpointAbortError( + f"--mode merge input 'new' must be a list, got {data.get('new')!r}" + ) + +def _run_abort_mode(args) -> int: try: - raw = _read_cheap_results(args.cheap_results_from) + raw = _read_text(args.cheap_results_from) except OSError as exc: print( f"checkpoint_abort.py: cannot read {args.cheap_results_from}: {exc}", @@ -266,5 +349,122 @@ def main(argv=None) -> int: return 0 +def _run_outcome_mode(args) -> int: + try: + raw = _read_text(args.from_path) + except OSError as exc: + print(f"checkpoint_abort.py: cannot read {args.from_path}: {exc}", file=sys.stderr) + return 1 + + try: + data = json.loads(raw) + except json.JSONDecodeError as exc: + print( + f"checkpoint_abort.py: --mode outcome input is not valid JSON: {exc}", + file=sys.stderr, + ) + return 1 + + try: + _validate_outcome_input(data) + except CheckpointAbortError as exc: + print(f"checkpoint_abort.py: malformed --mode outcome input: {exc}", file=sys.stderr) + return 1 + + result = compute_round_outcome( + aborted=data["aborted"], + redispatched=data["redispatched"], + findings=data["findings"], + ) + print(json.dumps(result)) + return 0 + + +def _run_merge_mode(args) -> int: + try: + raw = _read_text(args.from_path) + except OSError as exc: + print(f"checkpoint_abort.py: cannot read {args.from_path}: {exc}", file=sys.stderr) + return 1 + + try: + data = json.loads(raw) + except json.JSONDecodeError as exc: + print( + f"checkpoint_abort.py: --mode merge input is not valid JSON: {exc}", + file=sys.stderr, + ) + return 1 + + try: + _validate_merge_input(data) + except CheckpointAbortError as exc: + print(f"checkpoint_abort.py: malformed --mode merge input: {exc}", file=sys.stderr) + return 1 + + merged = merge_findings(data["existing"], data["new"]) + print(json.dumps(merged)) + return 0 + + +def main(argv=None) -> int: + parser = argparse.ArgumentParser( + description=( + "Decide whether to abort remaining opus-tier lens dispatch for " + "a /build checkpoint round (--mode abort, default), compute a " + "round's pass/blocked outcome (--mode outcome), or merge " + "deferred-lens findings back into a round's finding set " + "(--mode merge)." + ) + ) + parser.add_argument( + "--mode", + choices=("abort", "outcome", "merge"), + default="abort", + help=( + "abort (default, existing behavior): decide_abort via " + "--cheap-results-from/--lenses. outcome: compute_round_outcome " + "via --from. merge: merge_findings via --from." + ), + ) + parser.add_argument( + "--cheap-results-from", + default="-", + help=( + "--mode abort only. Path to a JSON file holding the cheap-tier " + "lens results (a list of {agent, issues: [{severity, " + "confidence}, ...]} objects) in lens-dispatch order, or '-' for " + "stdin (default)." + ), + ) + parser.add_argument( + "--lenses", + nargs="*", + default=[], + help=( + "--mode abort only. The checkpoint's full ordered lens list " + "(select_lenses.py's cheap-first output)." + ), + ) + parser.add_argument( + "--from", + dest="from_path", + default="-", + help=( + "--mode outcome|merge only. Path to a JSON file holding the " + "mode's input object, or '-' for stdin (default). --mode " + "outcome expects {aborted, redispatched, findings}; --mode " + "merge expects {existing, new}." + ), + ) + args = parser.parse_args(argv) + + if args.mode == "outcome": + return _run_outcome_mode(args) + if args.mode == "merge": + return _run_merge_mode(args) + return _run_abort_mode(args) + + if __name__ == "__main__": raise SystemExit(main()) diff --git a/plugins/dev-team/skills/build/SKILL.md b/plugins/dev-team/skills/build/SKILL.md index 5f06a6b12..f00086c24 100644 --- a/plugins/dev-team/skills/build/SKILL.md +++ b/plugins/dev-team/skills/build/SKILL.md @@ -193,7 +193,7 @@ Work each step **one behavior at a time** — never all the code then all the te 4. **Inline review checkpoint — granularity scales with complexity.** *Where* the checkpoint runs depends on the step's **Complexity** classification (review *depth* still scales too): - **trivial**: Skip inline review. The final `/code-review` (step 6) covers all modified files. - **standard**: **Defer** review to the slice boundary (sub-step 6) — do not review now. Track the step's changed files so the slice checkpoint reviews them in one batch. Per-step review on standard steps is N near-identical passes where one at slice end largely does the same work, and the final `/code-review` (step 6) remains the backstop. This is the batching win — fewer review dispatches per multi-step slice at bounded quality risk. - - **complex**: **Dispatch-capability gate (re-confirm here — issue #1461):** before this review dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply this checkpoint's checklist inline; report the missing capability and halt rather than marking the step done. Otherwise, review **now, per step** — smaller blast radius per fix. Run the static self-heal pass to completion first — pass, or cap-and-escalate, per `references/static-self-heal.md` — then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects for this step's changed files** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` returns the applicable lenses (see the resolver note after sub-step 6) — dispatched **cheap-first (non-opus lenses before the opus-tier `security-review`/`domain-review`/`arch-review`)**, with the review-fix loop (up to 5 iterations per `${CLAUDE_PLUGIN_ROOT}/knowledge/three-phase-workflow.md#review-loop`) — including its deterministic-first re-verification triage (`../code-review/SKILL.md` step 6a, #1610): before re-dispatching an agent to confirm a fix, prefer a close via whichever language-appropriate lint/type-check tool(s) applies to this project's own stack (not just Python's `ruff` — ESLint/`tsc` for JS/TS, `pmd` for Java, `dotnet format`/`dotnet build` for C#, etc.) plus the test suite/`grep`, when the fix is mechanical and the claim is deterministically checkable, and escalate to an agent only when semantic judgment is genuinely needed. **Abort check (#2168).** Once the cheap-tier lenses return, invoke `checkpoint_abort.py` (`python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --cheap-results-from --lenses `) before dispatching the remaining opus-tier lenses. On `abort: true`: skip dispatching the deferred opus-tier lenses this round, run the review-fix loop against the cheap-tier findings only, and once it converges, dispatch the deferred lenses and merge their findings into the round's finding set via `checkpoint_abort.merge_findings(existing, new)` — the same dedup-by-`(agent, file, line, severity, message)` function Step 1.1 already proved order-independent. This checkpoint's report must name every deferred lens and the triggering finding (`checkpoint_abort.py`'s `deferredLenses`/`triggeringFinding`/`triggeringAgent`). The round's pass/blocked outcome always comes from calling `compute_round_outcome(aborted, redispatched, findings)` — never independently reimplemented — so a round that aborted and never re-dispatched its deferred lenses cannot report a pass. On `abort: false`, dispatch every opus-tier lens normally. Before each review-fix iteration, classify the finding/failure via `${CLAUDE_PLUGIN_ROOT}/knowledge/failure-routing.md` and follow its route (see the TEST-phase note above) — a security-finding class dispatches security-engineer, a reviewer-conflict class routes to human arbitration, `unclassified` stays in the generic loop. Escalate to user if the loop doesn't converge. Then **record the checkpoint outcome** (sub-step 7). + - **complex**: **Dispatch-capability gate (re-confirm here — issue #1461):** before this review dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply this checkpoint's checklist inline; report the missing capability and halt rather than marking the step done. Otherwise, review **now, per step** — smaller blast radius per fix. Run the static self-heal pass to completion first — pass, or cap-and-escalate, per `references/static-self-heal.md` — then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects for this step's changed files** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` returns the applicable lenses (see the resolver note after sub-step 6) — dispatched **cheap-first (non-opus lenses before the opus-tier `security-review`/`domain-review`/`arch-review`)**, with the review-fix loop (up to 5 iterations per `${CLAUDE_PLUGIN_ROOT}/knowledge/three-phase-workflow.md#review-loop`) — including its deterministic-first re-verification triage (`../code-review/SKILL.md` step 6a, #1610): before re-dispatching an agent to confirm a fix, prefer a close via whichever language-appropriate lint/type-check tool(s) applies to this project's own stack (not just Python's `ruff` — ESLint/`tsc` for JS/TS, `pmd` for Java, `dotnet format`/`dotnet build` for C#, etc.) plus the test suite/`grep`, when the fix is mechanical and the claim is deterministically checkable, and escalate to an agent only when semantic judgment is genuinely needed. **Abort check (#2168).** Before dispatching the remaining opus-tier lenses, run `checkpoint_abort.py`'s abort check — see the shared **Abort check (#2168)** rule stated once after sub-step 6 below (applies to both checkpoint fix loops). Before each review-fix iteration, classify the finding/failure via `${CLAUDE_PLUGIN_ROOT}/knowledge/failure-routing.md` and follow its route (see the TEST-phase note above) — a security-finding class dispatches security-engineer, a reviewer-conflict class routes to human arbitration, `unclassified` stays in the generic loop. Escalate to user if the loop doesn't converge. Then **record the checkpoint outcome** (sub-step 7). - If no complexity is specified, default to **standard**. - **UI changes (any complexity)**: After the relevant review passes (per-step for complex, at the slice checkpoint for standard), run browser verification via `/browse` in automated smoke test mode. Skip with warning if the dev server is not running. See `${CLAUDE_PLUGIN_ROOT}/knowledge/three-phase-workflow.md#phase-3-implement` Stage 3. 5. **Mark step done** — Use the Edit tool to update the plan file's `## Build Progress` section on disk: @@ -203,7 +203,9 @@ Work each step **one behavior at a time** — never all the code then all the te - After all slices are `[x]`, change `**Status**: approved` to `**Status**: in-progress`. - This disk write is the durable commit. If a `/clear` occurs, `/continue` reads `## Build Progress` to determine the resume point without needing conversation history. - **Clear freeze scope (issue #865).** When every step under the slice is `[x]` and freeze was engaged for it (dispatch bookkeeping above), run `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/build_slice_scope.py clear --hooks-dir /.claude/hooks` before starting the next slice. A slice that never engaged freeze has nothing to clear. -6. **Slice review checkpoint (batched).** **Dispatch-capability gate (re-confirm here — issue #1461):** before this checkpoint dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply the batched checkpoint's checklist inline; report the missing capability and halt rather than checking off the slice. Otherwise, when every step under the current slice is `[x]` **and** the slice had any deferred `standard` (or unspecified) steps, run **one** review pass over the slice's accumulated changed files: the static self-heal pass first (`references/static-self-heal.md`), then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` — dispatched **cheap-first (non-opus before opus-tier)**, **plus `refactor-opportunity-review` dispatched by name** (#1976). That lens declares `Scope: on-demand`, so the resolver no longer returns it; this once-per-slice dispatch is its post-GREEN home, and it is the ONLY place `/build` runs it — do not also dispatch it per behavior at the REFACTOR phase (that would spend more, not less, than the per-diff panel slot it replaced) and do not treat the resolver's silence as "this lens was dropped". Its subject is a slice's accumulated shape — semantic vs. structural duplication across everything the slice touched — which is visible here and not in any single behavior's diff. **Abort check (#2168).** Once the cheap-tier lenses (from the cheap-first dispatch above) return, invoke `checkpoint_abort.py` (`python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --cheap-results-from --lenses `) before dispatching the remaining opus-tier lenses. On `abort: true`: skip dispatching the deferred opus-tier lenses this round, run the review-fix loop against the cheap-tier findings only, and once it converges, dispatch the deferred lenses and merge their findings into the round's finding set via `checkpoint_abort.merge_findings(existing, new)` — the same dedup-by-`(agent, file, line, severity, message)` function Step 1.1 already proved order-independent. This checkpoint's report must name every deferred lens and the triggering finding (`checkpoint_abort.py`'s `deferredLenses`/`triggeringFinding`/`triggeringAgent`). The round's pass/blocked outcome always comes from calling `compute_round_outcome(aborted, redispatched, findings)` — never independently reimplemented — so a round that aborted and never re-dispatched its deferred lenses cannot report a pass. On `abort: false`, dispatch every opus-tier lens normally. Either way, apply the same review-fix loop (up to 5 iterations; escalate if it doesn't converge). `trivial`-only and all-`complex` slices have nothing to batch — skip this pass. Then **record the checkpoint outcome** (sub-step 7). +6. **Slice review checkpoint (batched).** **Dispatch-capability gate (re-confirm here — issue #1461):** before this checkpoint dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply the batched checkpoint's checklist inline; report the missing capability and halt rather than checking off the slice. Otherwise, when every step under the current slice is `[x]` **and** the slice had any deferred `standard` (or unspecified) steps, run **one** review pass over the slice's accumulated changed files: the static self-heal pass first (`references/static-self-heal.md`), then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` — dispatched **cheap-first (non-opus before opus-tier)**, **plus `refactor-opportunity-review` dispatched by name** (#1976). That lens declares `Scope: on-demand`, so the resolver no longer returns it; this once-per-slice dispatch is its post-GREEN home, and it is the ONLY place `/build` runs it — do not also dispatch it per behavior at the REFACTOR phase (that would spend more, not less, than the per-diff panel slot it replaced) and do not treat the resolver's silence as "this lens was dropped". Its subject is a slice's accumulated shape — semantic vs. structural duplication across everything the slice touched — which is visible here and not in any single behavior's diff. **Abort check (#2168).** Before dispatching the remaining opus-tier lenses, run `checkpoint_abort.py`'s abort check on the cheap-tier lenses' (from the cheap-first dispatch above) results — see the shared rule below (applies to both checkpoint fix loops above). Either way, apply the same review-fix loop (up to 5 iterations; escalate if it doesn't converge). `trivial`-only and all-`complex` slices have nothing to batch — skip this pass. Then **record the checkpoint outcome** (sub-step 7). + + **Abort check (#2168) — applies to both checkpoint fix loops above (sub-steps 4 and 6).** Once the cheap-tier lenses return, invoke `checkpoint_abort.py` (`python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --cheap-results-from --lenses `) before dispatching the remaining opus-tier lenses. On `aborted: true`: skip dispatching the deferred opus-tier lenses this round, run the review-fix loop against the cheap-tier findings only, and once it converges, dispatch the deferred lenses and merge their findings into the round's finding set via `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --mode merge --from ` (`checkpoint_abort.merge_findings(existing, new)`'s CLI form) — the same dedup-by-`(agent, file, line, severity, message)` function Step 1.1 already proved order-independent. This checkpoint's report must name every deferred lens and the triggering finding (`checkpoint_abort.py`'s `deferredLenses`/`triggeringFinding`/`triggeringAgent`). The round's pass/blocked outcome always comes from calling `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --mode outcome --from ` (`compute_round_outcome(aborted, redispatched, findings)`'s CLI form) — never independently reimplemented — so a round that aborted and never re-dispatched its deferred lenses cannot report a pass. On `aborted: false`, dispatch every opus-tier lens normally. **Verification-mode re-dispatch (#1628) — applies to both checkpoint fix loops above (sub-steps 4 and 6).** When a checkpoint re-dispatches an agent to CONFIRM a fix (rather than to discover new problems), send the narrowed verification payload — the finding, the fix diff hunks ± ~20 lines, and the agent's lens definition, never the full file set — with the mandatory `insufficient-context` escape, and resolve the agent's verification tier with `python3 "$CLAUDE_PLUGIN_ROOT/scripts/verify_tier.py" --agent `. The payload contract, the escape's escalation path, and the declared-never-inferred tier-down rule are stated once in [`../../knowledge/verification-mode.md`](../../knowledge/verification-mode.md) and are not restated here. diff --git a/plugins/dev-team/tests/scripts/test_checkpoint_abort.py b/plugins/dev-team/tests/scripts/test_checkpoint_abort.py index 796994fad..042b06882 100644 --- a/plugins/dev-team/tests/scripts/test_checkpoint_abort.py +++ b/plugins/dev-team/tests/scripts/test_checkpoint_abort.py @@ -2,10 +2,11 @@ Covers the abort decision (error/high only, first-qualifying-finding-wins, fail-open on empty input, loud failure on malformed input), -`compute_round_outcome`'s hard "never pass on an unre-dispatched abort" rule, -`merge_findings`'s dedup-and-append contract, and the fixture equivalence -test proving `merge_findings` is order-independent under a cheap/deferred -split (Step 1.1's central safety-property test). +`compute_round_outcome`'s hard "never pass on an unre-dispatched abort" rule +and its severity-floor finding filter, `merge_findings`'s dedup-and-append +contract, the fixture equivalence test proving `merge_findings` is +order-independent under a cheap/deferred split (Step 1.1's central +safety-property test), and the CLI's three `--mode` entry points. """ from __future__ import annotations @@ -38,7 +39,7 @@ def test_clean_results_do_not_abort(self): results = [_result("naming-review", _issue("warning", "medium"))] out = checkpoint_abort.decide_abort(results, ["naming-review", "arch-review"]) assert out == { - "abort": False, + "aborted": False, "triggeringFinding": None, "triggeringAgent": None, "deferredLenses": [], @@ -46,18 +47,18 @@ def test_clean_results_do_not_abort(self): def test_empty_cheap_results_never_aborts(self): out = checkpoint_abort.decide_abort([], ["arch-review", "security-review"]) - assert out["abort"] is False + assert out["aborted"] is False assert out["deferredLenses"] == [] def test_error_severity_without_high_confidence_never_aborts(self): results = [_result("naming-review", _issue("error", "medium"))] out = checkpoint_abort.decide_abort(results, ["naming-review", "arch-review"]) - assert out["abort"] is False + assert out["aborted"] is False def test_high_confidence_without_error_severity_never_aborts(self): results = [_result("naming-review", _issue("warning", "high"))] out = checkpoint_abort.decide_abort(results, ["naming-review", "arch-review"]) - assert out["abort"] is False + assert out["aborted"] is False class TestDecideAbortAborts: @@ -65,7 +66,7 @@ def test_error_high_finding_aborts_with_deferred_lenses(self): results = [_result("naming-review", _issue("error", "high"))] ordered = ["naming-review", "arch-review", "security-review"] out = checkpoint_abort.decide_abort(results, ordered) - assert out["abort"] is True + assert out["aborted"] is True assert out["triggeringFinding"] == _issue("error", "high") assert out["triggeringAgent"] == "naming-review" assert out["deferredLenses"] == ["arch-review", "security-review"] @@ -77,7 +78,7 @@ def test_first_qualifying_finding_in_lens_dispatch_order_wins(self): ] ordered = ["naming-review", "doc-review", "arch-review"] out = checkpoint_abort.decide_abort(results, ordered) - assert out["abort"] is True + assert out["aborted"] is True assert out["triggeringAgent"] == "naming-review" assert out["triggeringFinding"]["message"] == "first" # Both cheap-tier agents already ran; only the un-run lens defers. @@ -92,7 +93,7 @@ def test_second_issue_in_a_lens_own_issue_list_can_trigger(self): ), ] out = checkpoint_abort.decide_abort(results, ["naming-review", "arch-review"]) - assert out["abort"] is True + assert out["aborted"] is True assert out["triggeringFinding"]["message"] == "second issue" @@ -116,6 +117,16 @@ def test_non_list_top_level_raises(self): with pytest.raises(checkpoint_abort.CheckpointAbortError): checkpoint_abort.decide_abort({"not": "a list"}, ["naming-review"]) + def test_missing_agent_raises(self): + results = [{"issues": [_issue("error", "high")]}] + with pytest.raises(checkpoint_abort.CheckpointAbortError): + checkpoint_abort.decide_abort(results, ["naming-review"]) + + def test_empty_agent_raises(self): + results = [{"agent": "", "issues": [_issue("error", "high")]}] + with pytest.raises(checkpoint_abort.CheckpointAbortError): + checkpoint_abort.decide_abort(results, ["naming-review"]) + class TestComputeRoundOutcome: def test_aborted_and_not_redispatched_never_passes_with_no_findings(self): @@ -123,30 +134,68 @@ def test_aborted_and_not_redispatched_never_passes_with_no_findings(self): aborted=True, redispatched=False, findings=[] ) assert out["outcome"] == "blocked" + assert out["reason"] == ( + "round aborted on a cheap-tier blocker and its deferred " + "lenses were never re-dispatched" + ) def test_aborted_and_not_redispatched_never_passes_with_findings(self): out = checkpoint_abort.compute_round_outcome( - aborted=True, redispatched=False, findings=[{"severity": "error"}] + aborted=True, + redispatched=False, + findings=[_issue("error", "high")], ) assert out["outcome"] == "blocked" + assert out["reason"] == ( + "round aborted on a cheap-tier blocker and its deferred " + "lenses were never re-dispatched" + ) def test_aborted_and_redispatched_computes_from_findings_empty(self): out = checkpoint_abort.compute_round_outcome( aborted=True, redispatched=True, findings=[] ) assert out["outcome"] == "pass" + assert out["reason"] is None def test_aborted_and_redispatched_computes_from_findings_nonempty(self): out = checkpoint_abort.compute_round_outcome( - aborted=True, redispatched=True, findings=[{"severity": "error"}] + aborted=True, + redispatched=True, + findings=[_issue("error", "high")], ) assert out["outcome"] == "blocked" + assert out["reason"] == "1 finding(s) remain" def test_never_aborted_computes_from_findings(self): out = checkpoint_abort.compute_round_outcome( aborted=False, redispatched=False, findings=[] ) assert out["outcome"] == "pass" + assert out["reason"] is None + + def test_only_suggestion_or_low_confidence_findings_pass(self): + findings = [ + _issue("suggestion", "high"), + _issue("error", "low"), + _issue("warning", "none"), + ] + out = checkpoint_abort.compute_round_outcome( + aborted=False, redispatched=False, findings=findings + ) + assert out["outcome"] == "pass" + assert out["reason"] is None + + def test_one_blocking_finding_among_suggestion_tier_blocks(self): + findings = [ + _issue("suggestion", "high"), + _issue("warning", "medium"), + ] + out = checkpoint_abort.compute_round_outcome( + aborted=False, redispatched=False, findings=findings + ) + assert out["outcome"] == "blocked" + assert out["reason"] == "1 finding(s) remain" class TestMergeFindings: @@ -188,6 +237,34 @@ def test_does_not_mutate_arguments(self): assert new == new_copy +class TestMergeDedupIdentityDivergence: + """Documents that `_finding_key`'s dedup identity is deliberately + stricter than, and independent of, `finding_signature.py`'s round-ledger + identity relation — mirrors `gate_retry_state.py`'s + `test_max_rounds_matches_finding_signature`-style drift-awareness test, + applied to a deliberately-divergent (not deliberately-matching) pair of + constants. See the "Merge-dedup identity intentionally diverges from + finding_signature.py" comment above `_finding_key` in checkpoint_abort.py. + """ + + def test_finding_key_uses_exact_line_and_raw_message(self): + # finding_signature.py tolerates a +/-3 line shift and normalizes + # the message; _finding_key does neither — two findings that would + # be the SAME finding under finding_signature's relation must be + # DISTINCT here, because merge_findings folds together two tiers + # dispatched against the same unmoved diff, where an exact + # positional key is what avoids over-merging. + a = { + "agent": "naming-review", + "file": "a.py", + "line": 10, + "severity": "warning", + "message": "ambiguous name 'x'", + } + b = dict(a, line=12, message="ambiguous name 'x' (renamed)") + assert checkpoint_abort._finding_key(a) != checkpoint_abort._finding_key(b) + + class TestFixtureEquivalence: """Proves `merge_findings` is order-independent under a cheap/deferred split — checkpoint_abort.py's only production aggregation logic. Whether @@ -262,13 +339,13 @@ def _run(self, cheap_results, lenses, check=False): def test_no_abort_prints_json_and_exits_zero(self): result = self._run([], ["naming-review"], check=True) payload = json.loads(result.stdout) - assert payload["abort"] is False + assert payload["aborted"] is False def test_abort_prints_deferred_lenses_and_exits_zero(self): results = [_result("naming-review", _issue("error", "high"))] result = self._run(results, ["naming-review", "arch-review"], check=True) payload = json.loads(result.stdout) - assert payload["abort"] is True + assert payload["aborted"] is True assert payload["deferredLenses"] == ["arch-review"] def test_malformed_input_exits_nonzero_with_clear_error(self): @@ -287,3 +364,146 @@ def test_invalid_json_exits_nonzero_with_clear_error(self): ) assert result.returncode != 0 assert "json" in result.stderr.lower() + + def test_cheap_results_from_file_happy_path(self, tmp_path): + results = [_result("naming-review", _issue("error", "high"))] + results_file = tmp_path / "cheap-results.json" + results_file.write_text(json.dumps(results), encoding="utf-8") + result = subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "checkpoint_abort.py"), + "--cheap-results-from", + str(results_file), + "--lenses", + "naming-review", + "arch-review", + ], + capture_output=True, + text=True, + check=True, + ) + payload = json.loads(result.stdout) + assert payload["aborted"] is True + assert payload["deferredLenses"] == ["arch-review"] + + def test_cheap_results_from_nonexistent_file_exits_nonzero(self, tmp_path): + missing = tmp_path / "does-not-exist.json" + result = subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "checkpoint_abort.py"), + "--cheap-results-from", + str(missing), + "--lenses", + "naming-review", + ], + capture_output=True, + text=True, + check=False, + ) + assert result.returncode != 0 + assert "cannot read" in result.stderr.lower() + + +class TestCliModeOutcome: + def _run(self, payload, check=False): + return subprocess.run( + [sys.executable, str(_SCRIPTS_DIR / "checkpoint_abort.py"), "--mode", "outcome"], + input=json.dumps(payload), + capture_output=True, + text=True, + check=check, + ) + + def test_happy_path_prints_outcome_json(self): + result = self._run( + {"aborted": False, "redispatched": False, "findings": []}, check=True + ) + payload = json.loads(result.stdout) + assert payload == {"outcome": "pass", "reason": None} + + def test_malformed_input_exits_nonzero_with_clear_error(self): + result = self._run({"aborted": True, "findings": []}, check=False) + assert result.returncode != 0 + assert "malformed" in result.stderr.lower() + + def test_from_file_happy_path(self, tmp_path): + payload_file = tmp_path / "outcome-input.json" + payload_file.write_text( + json.dumps({"aborted": True, "redispatched": False, "findings": []}), + encoding="utf-8", + ) + result = subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "checkpoint_abort.py"), + "--mode", + "outcome", + "--from", + str(payload_file), + ], + capture_output=True, + text=True, + check=True, + ) + payload = json.loads(result.stdout) + assert payload["outcome"] == "blocked" + + +class TestCliModeMerge: + def _run(self, payload, check=False): + return subprocess.run( + [sys.executable, str(_SCRIPTS_DIR / "checkpoint_abort.py"), "--mode", "merge"], + input=json.dumps(payload), + capture_output=True, + text=True, + check=check, + ) + + def test_happy_path_prints_merged_json_array(self): + existing = [ + { + "agent": "naming-review", + "file": "a.py", + "line": 1, + "severity": "warning", + "message": "m1", + } + ] + new = [ + { + "agent": "arch-review", + "file": "b.py", + "line": 2, + "severity": "error", + "message": "m2", + } + ] + result = self._run({"existing": existing, "new": new}, check=True) + merged = json.loads(result.stdout) + assert merged == existing + new + + def test_malformed_input_exits_nonzero_with_clear_error(self): + result = self._run({"existing": "not-a-list", "new": []}, check=False) + assert result.returncode != 0 + assert "malformed" in result.stderr.lower() + + def test_from_file_happy_path(self, tmp_path): + payload_file = tmp_path / "merge-input.json" + payload_file.write_text(json.dumps({"existing": [], "new": []}), encoding="utf-8") + result = subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "checkpoint_abort.py"), + "--mode", + "merge", + "--from", + str(payload_file), + ], + capture_output=True, + text=True, + check=True, + ) + merged = json.loads(result.stdout) + assert merged == [] diff --git a/plugins/dev-team/tests/skills/test_build_checkpoint_abort_marker.py b/plugins/dev-team/tests/skills/test_build_checkpoint_abort_marker.py index 243a06c37..2042813f9 100644 --- a/plugins/dev-team/tests/skills/test_build_checkpoint_abort_marker.py +++ b/plugins/dev-team/tests/skills/test_build_checkpoint_abort_marker.py @@ -9,6 +9,13 @@ instructing `/build` to follow that behavior is present, in both checkpoint sub-steps, and names the correct function. Mirrors the established content- guard style in tests/skills/test_code_review_synthesis_verbosity.py. + +The full "Abort check (#2168)" rule is stated once, in a shared block after +sub-step 6 -- the same established convention "Verification-mode re-dispatch +(#1628)" and "Round-ledger termination rules (#1625)" already follow. Sub- +steps 4 and 6 each carry only a one-line pointer to it, so this file checks +the pointer's presence in each sub-step's own section, and the concrete CLI +invocation strings once, in the shared block. """ from __future__ import annotations @@ -21,6 +28,11 @@ _STEP4_END = "- If no complexity is specified" _STEP6_START = "6. **Slice review checkpoint (batched).**" _STEP6_END = "7. **Record review value" +_ABORT_SHARED_START = ( + "**Abort check (#2168) — applies to both checkpoint fix loops above " + "(sub-steps 4 and 6).**" +) +_ABORT_SHARED_END = "**Verification-mode re-dispatch (#1628)" def _skill_text() -> str: @@ -42,6 +54,10 @@ def _step6_section(text: str) -> str: return _section(text, _STEP6_START, _STEP6_END) +def _abort_shared_section(text: str) -> str: + return _section(text, _ABORT_SHARED_START, _ABORT_SHARED_END) + + def test_step4_complex_checkpoint_references_checkpoint_abort_script() -> None: section = _step4_section(_skill_text()) assert "checkpoint_abort.py" in section, ( @@ -50,19 +66,11 @@ def test_step4_complex_checkpoint_references_checkpoint_abort_script() -> None: ) -def test_step4_complex_checkpoint_names_merge_findings() -> None: +def test_step4_complex_checkpoint_points_to_shared_abort_check() -> None: section = _step4_section(_skill_text()) - assert "checkpoint_abort.merge_findings(existing, new)" in section, ( - "sub-step 4 must name checkpoint_abort.merge_findings(existing, new) " - "as the deferred-lens re-dispatch aggregation step" - ) - - -def test_step4_complex_checkpoint_uses_compute_round_outcome() -> None: - section = _step4_section(_skill_text()) - assert "compute_round_outcome(aborted, redispatched, findings)" in section, ( - "sub-step 4's pass/blocked outcome must come from calling " - "compute_round_outcome, not independently-reimplemented logic" + assert "Abort check (#2168)" in section, ( + "sub-step 4 must carry a pointer to the shared Abort check (#2168) " + "rule rather than a second inline copy of it" ) @@ -74,30 +82,51 @@ def test_step6_slice_checkpoint_references_checkpoint_abort_script() -> None: ) -def test_step6_slice_checkpoint_names_merge_findings() -> None: +def test_step6_slice_checkpoint_points_to_shared_abort_check() -> None: section = _step6_section(_skill_text()) - assert "checkpoint_abort.merge_findings(existing, new)" in section, ( - "sub-step 6 must name checkpoint_abort.merge_findings(existing, new) " - "as the deferred-lens re-dispatch aggregation step" + assert "Abort check (#2168)" in section, ( + "sub-step 6 must carry a pointer to the shared Abort check (#2168) " + "rule rather than a second inline copy of it" ) -def test_step6_slice_checkpoint_uses_compute_round_outcome() -> None: - section = _step6_section(_skill_text()) - assert "compute_round_outcome(aborted, redispatched, findings)" in section, ( - "sub-step 6's pass/blocked outcome must come from calling " - "compute_round_outcome, not independently-reimplemented logic" +def test_abort_check_rule_is_stated_once_in_a_shared_block() -> None: + text = _skill_text() + assert text.count(_ABORT_SHARED_START) == 1, ( + "the Abort check (#2168) rule must be stated exactly once, in a " + "shared block scoped to both checkpoint fix loops -- not duplicated " + "inline in sub-steps 4 and 6" ) +def test_shared_abort_check_names_merge_findings_cli() -> None: + section = _abort_shared_section(_skill_text()) + assert "checkpoint_abort.py --mode merge" in section, ( + "the shared Abort check (#2168) rule must name the concrete " + "`checkpoint_abort.py --mode merge` CLI invocation as the " + "deferred-lens re-dispatch aggregation step, not a bare Python " + "function-call signature" + ) + assert "merge_findings" in section + + +def test_shared_abort_check_uses_compute_round_outcome_cli() -> None: + section = _abort_shared_section(_skill_text()) + assert "checkpoint_abort.py --mode outcome" in section, ( + "the shared Abort check (#2168) rule's pass/blocked outcome must " + "come from the concrete `checkpoint_abort.py --mode outcome` CLI " + "invocation, not a bare Python function-call signature" + ) + assert "compute_round_outcome" in section + + def test_both_checkpoints_require_naming_deferred_lenses_and_trigger() -> None: - text = _skill_text() - for section in (_step4_section(text), _step6_section(text)): - assert "name every deferred lens" in section, ( - "checkpoint report must name every deferred lens and the " - "triggering finding (loud abort)" - ) - assert "triggering finding" in section + section = _abort_shared_section(_skill_text()) + assert "name every deferred lens" in section, ( + "checkpoint report must name every deferred lens and the " + "triggering finding (loud abort)" + ) + assert "triggering finding" in section def test_code_review_step4_non_support_sentence_is_stated() -> None: From ce29b43a65ccdf74c2863f532b512af127cfed95 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 18:42:45 +0000 Subject: [PATCH 05/14] docs(test-review): annotate mechanical vs judgment checks Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/agents/test-review.md | 84 ++++---- ..._test_review_mechanical_split_annotated.py | 196 ++++++++++++++++++ 2 files changed, 238 insertions(+), 42 deletions(-) create mode 100644 plugins/dev-team/tests/agents/test_test_review_mechanical_split_annotated.py diff --git a/plugins/dev-team/agents/test-review.md b/plugins/dev-team/agents/test-review.md index 7a5fe5576..c0508aba0 100644 --- a/plugins/dev-team/agents/test-review.md +++ b/plugins/dev-team/agents/test-review.md @@ -98,57 +98,57 @@ Return `{"status": "skip", "issues": [], "summary": "No test files in target"}` Coverage gaps: -- Missing edge cases (empty, null, boundary) -- Missing error paths (exceptions, invalid states) -- Missing happy path scenarios +- Missing edge cases (empty, null, boundary) [JUDGMENT] +- Missing error paths (exceptions, invalid states) [JUDGMENT] +- Missing happy path scenarios [JUDGMENT] Assertion quality: - Tests with no assertion — test methods containing no Assert, expect, should, verify, or equivalent assertion call. A test that only exercises code without asserting outcomes provides zero regression - protection. -- Non-specific assertions (truthiness-only checks) -- Implementation verification instead of behavior -- Incomplete state verification + protection. [MECHANICAL] +- Non-specific assertions (truthiness-only checks) [JUDGMENT] +- Implementation verification instead of behavior [JUDGMENT] +- Incomplete state verification [JUDGMENT] Test hygiene: -- Shared mutable state between tests -- Mocks/stubs not reset — JS: `jest.clearAllMocks()` absent; C#: Moq `Mock` reused without `Reset()` or re-instantiation, NSubstitute missing `ClearReceivedCalls()`; Java: Mockito missing `reset()` or `@BeforeEach` re-initialization -- Missing await on async operations — JS/TS: missing `await`; C#: missing `await` on `Task`-returning methods or unchecked `Task` results; Java: unchecked `Future.get()` or missing `CompletableFuture` resolution -- No arrange-act-assert structure -- Misleading test descriptions +- Shared mutable state between tests [JUDGMENT] +- Mocks/stubs not reset — JS: `jest.clearAllMocks()` absent; C#: Moq `Mock` reused without `Reset()` or re-instantiation, NSubstitute missing `ClearReceivedCalls()`; Java: Mockito missing `reset()` or `@BeforeEach` re-initialization [MECHANICAL] +- Missing await on async operations — JS/TS: missing `await`; C#: missing `await` on `Task`-returning methods or unchecked `Task` results; Java: unchecked `Future.get()` or missing `CompletableFuture` resolution [MECHANICAL] +- No arrange-act-assert structure [JUDGMENT] +- Misleading test descriptions [JUDGMENT] Test level efficiency: -- Integration or E2E setup (real DB, real HTTP, large object graphs) used to test a single unit's logic — flag and suggest a unit test with a double instead -- Tests that only exercise third-party library behavior, not the code under test -- Multiple tests asserting identical outcomes with different inputs where no boundary condition distinguishes them (redundant coverage) +- Integration or E2E setup (real DB, real HTTP, large object graphs) used to test a single unit's logic — flag and suggest a unit test with a double instead [JUDGMENT] +- Tests that only exercise third-party library behavior, not the code under test [JUDGMENT] +- Multiple tests asserting identical outcomes with different inputs where no boundary condition distinguishes them (redundant coverage) [JUDGMENT] Non-determinism sources (flakiness): -- Unstubbed clock access — JS/TS: `Date.now()`, `new Date()`, `Date()`; C#: `DateTime.Now`, `DateTime.UtcNow`, `DateTimeOffset.Now`; Java: `new Date()`, `LocalDateTime.now()`, `Instant.now()`, `System.currentTimeMillis()` -- Unstubbed randomness — JS/TS: `Math.random()`; C#: `new Random()` without injection; Java: `new Random()`, `Math.random()` without injection -- Real network calls, DB connections, or file I/O without test doubles -- Unstubbed timers/delays — JS/TS: `setTimeout`, `setInterval`, `setImmediate` without fake timers; C#: `Task.Delay`, `Thread.Sleep` in test body; Java: `Thread.sleep()` in test body -- Tests that depend on execution order or shared external state between runs -- Uncontrolled async concurrency — JS/TS: `Promise.all` with uncontrolled timing; C#: `Task.WhenAll` without controlled scheduling; Java: unjoined threads or unresolved `CompletableFuture` +- Unstubbed clock access — JS/TS: `Date.now()`, `new Date()`, `Date()`; C#: `DateTime.Now`, `DateTime.UtcNow`, `DateTimeOffset.Now`; Java: `new Date()`, `LocalDateTime.now()`, `Instant.now()`, `System.currentTimeMillis()` [MECHANICAL] +- Unstubbed randomness — JS/TS: `Math.random()`; C#: `new Random()` without injection; Java: `new Random()`, `Math.random()` without injection [MECHANICAL] +- Real network calls, DB connections, or file I/O without test doubles [JUDGMENT] +- Unstubbed timers/delays — JS/TS: `setTimeout`, `setInterval`, `setImmediate` without fake timers; C#: `Task.Delay`, `Thread.Sleep` in test body; Java: `Thread.sleep()` in test body [MECHANICAL] +- Tests that depend on execution order or shared external state between runs [JUDGMENT] +- Uncontrolled async concurrency — JS/TS: `Promise.all` with uncontrolled timing; C#: `Task.WhenAll` without controlled scheduling; Java: unjoined threads or unresolved `CompletableFuture` [JUDGMENT] Test code quality: -- Copy-pasted assertion blocks that should be extracted into a helper -- Magic literal values in assertions with no explanation of their significance -- Dead test utilities or helpers that are defined but never called -- Low automation maturity (`test-automation-maturity.md`): a volatile detail (selector, endpoint, field name) duplicated raw across many test files (single-point-of-change failure); UI driven to establish preconditions instead of back-door setup — flag only when suite size makes the cost real (graduated thresholds) +- Copy-pasted assertion blocks that should be extracted into a helper [JUDGMENT] +- Magic literal values in assertions with no explanation of their significance [JUDGMENT] +- Dead test utilities or helpers that are defined but never called [JUDGMENT] +- Low automation maturity (`test-automation-maturity.md`): a volatile detail (selector, endpoint, field name) duplicated raw across many test files (single-point-of-change failure); UI driven to establish preconditions instead of back-door setup — flag only when suite size makes the cost real (graduated thresholds) [JUDGMENT] Oracle provenance (correctness vs. stability): Whole-file load: apply the SPEC-DERIVED / INDEPENDENT / CIRCULAR taxonomy from `${CLAUDE_PLUGIN_ROOT}/knowledge/oracle-provenance.md`. For each test, classify its expected values by provenance. Report the oracle-provenance ratio (circular / total) in the finding summary when any circular oracles are detected: -- Circular ratio < 20 %: suggestion — add provenance comments to snapshot-based assertions -- Circular ratio 20–50 %: warning — suite has meaningful circular-oracle contamination -- Circular ratio > 50 %: error — suite is circular-oracle-dominated; the file's Test Quality contribution is capped at 60. A circular-oracle-dominated suite verifies stability, not correctness; regressions can go undetected if snapshots are updated without independent verification. +- Circular ratio < 20 %: suggestion — add provenance comments to snapshot-based assertions [JUDGMENT] +- Circular ratio 20–50 %: warning — suite has meaningful circular-oracle contamination [JUDGMENT] +- Circular ratio > 50 %: error — suite is circular-oracle-dominated; the file's Test Quality contribution is capped at 60. A circular-oracle-dominated suite verifies stability, not correctness; regressions can go undetected if snapshots are updated without independent verification. [JUDGMENT] Do not double-report individual snapshot findings when `test-smell-review` is also running in the same session — note the ratio in the summary instead. @@ -156,8 +156,8 @@ Unarmored regions (survivorship-bias gaps): An **unarmored region** is code that has *neither* test coverage *nor* any sign of historical defensive attention — no negative tests, no error-path assertions, no defensive comments (e.g. `// edge case`, `// TODO: handle`, `// regression: ...`), no related test utility. This is distinct from an ordinary missing-edge-case coverage gap: -- A **coverage gap** is code that is under-tested — some tests exist but a boundary or error path is missing. -- An **unarmored region** is code that has never been examined — no tests AND no sign anyone has looked at it defensively. It is the least-examined code, not merely the least-tested. +- A **coverage gap** is code that is under-tested — some tests exist but a boundary or error path is missing. [JUDGMENT] +- An **unarmored region** is code that has never been examined — no tests AND no sign anyone has looked at it defensively. It is the least-examined code, not merely the least-tested. [JUDGMENT] Detection: identify functions, branches, or modules where (a) no test exercises the path AND (b) no surrounding context shows historical defensive attention. Flag these as a named "unarmored region" finding, distinct from ordinary coverage-gap findings. @@ -165,16 +165,16 @@ Severity: warning. Suggested fix: prioritize writing tests for unarmored regions Testability blockers: -- Code under test that cannot be constructed with known values (static factories, singletons, no injectable constructor) — flag as error; per `${CLAUDE_PLUGIN_ROOT}/knowledge/testability-patterns.md#pattern-1-constructor-injection-replace-static-factories-singletons`, the production code must change, not the test approach -- Mocking of concrete classes (not interfaces) — flag as warning; extract an interface for the dependency -- Tests using reflection into private members as primary strategy — flag as warning. This is an architecture/encapsulation issue the test is reaching around, not a test-hygiene nit. Detection signatures: Java: `getDeclaredMethod`/`getDeclaredField` + `setAccessible(true)`, `Method.invoke` on a private/protected member; C#: `Type.GetMethod(..., BindingFlags.NonPublic | BindingFlags.Instance)`, `Type.InvokeMember`; Python: `getattr`/`setattr`/`hasattr` targeting a name-mangled (`_ClassName__attr`) or underscore-prefixed attribute; JS/TS: bracket-notation access into a `private`/non-exported member (e.g. `(obj as any)['_privateMethod']()`), `Object.getOwnPropertyDescriptor`/`Object.defineProperty` used to reach a non-exported member. Suggested fix — pick by shape of the code, never the generic "expand the public API": (1) extract the private logic into a collaborator with its own public seam, when it's standalone logic worth testing independently; (2) relax visibility to package-private/internal, only when a production collaborator in the same module/assembly independently needs the access (the language must have that tier) — never as a grant solely so the test can reach in, which recreates the `InternalsVisibleTo`/`@VisibleForTesting` anti-pattern below; (3) test the behavior through the class's existing public API, when the private method is already an implementation detail of a public behavior +- Code under test that cannot be constructed with known values (static factories, singletons, no injectable constructor) — flag as error; per `${CLAUDE_PLUGIN_ROOT}/knowledge/testability-patterns.md#pattern-1-constructor-injection-replace-static-factories-singletons`, the production code must change, not the test approach [JUDGMENT] +- Mocking of concrete classes (not interfaces) — flag as warning; extract an interface for the dependency [JUDGMENT] +- Tests using reflection into private members as primary strategy — flag as warning. This is an architecture/encapsulation issue the test is reaching around, not a test-hygiene nit. Detection signatures: Java: `getDeclaredMethod`/`getDeclaredField` + `setAccessible(true)`, `Method.invoke` on a private/protected member; C#: `Type.GetMethod(..., BindingFlags.NonPublic | BindingFlags.Instance)`, `Type.InvokeMember`; Python: `getattr`/`setattr`/`hasattr` targeting a name-mangled (`_ClassName__attr`) or underscore-prefixed attribute; JS/TS: bracket-notation access into a `private`/non-exported member (e.g. `(obj as any)['_privateMethod']()`), `Object.getOwnPropertyDescriptor`/`Object.defineProperty` used to reach a non-exported member. Suggested fix — pick by shape of the code, never the generic "expand the public API": (1) extract the private logic into a collaborator with its own public seam, when it's standalone logic worth testing independently; (2) relax visibility to package-private/internal, only when a production collaborator in the same module/assembly independently needs the access (the language must have that tier) — never as a grant solely so the test can reach in, which recreates the `InternalsVisibleTo`/`@VisibleForTesting` anti-pattern below; (3) test the behavior through the class's existing public API, when the private method is already an implementation detail of a public behavior [MECHANICAL] (detection only, via the explicit per-language signatures above — this stays `warning`-severity and is reported by the future mechanical pre-phase without gating the qualitative pass; only the no-assertion-tests check and `internal_double_detector.py`'s own `error`-severity findings gate the pass — see Step 2.2) Internal-collaborator doubling (mechanical — never a truth judgment; see `${CLAUDE_PLUGIN_ROOT}/knowledge/internal-collaborator-doubling.md#the-waiver`): -- A doubled first-party collaborator with no waiver comment at the double site (see the normative file for the exact marker syntax) — flag as error -- A waiver comment naming anything other than `B1`, `B2`, or `B3` — flag as error -- A syntactically valid `B2` waiver whose collaborator's own declaring source shows no reference to any ambient-API marker (clock, RNG/GUID, env, hostname, cwd, locale) — flag as error; this is the detector's own evidence-*presence* check, not a judgment about whether the evidence is convincing (that half belongs to `test-smell-review`, and only ever on an already-waived double) +- A doubled first-party collaborator with no waiver comment at the double site (see the normative file for the exact marker syntax) — flag as error [MECHANICAL] +- A waiver comment naming anything other than `B1`, `B2`, or `B3` — flag as error [MECHANICAL] +- A syntactically valid `B2` waiver whose collaborator's own declaring source shows no reference to any ambient-API marker (clock, RNG/GUID, env, hostname, cwd, locale) — flag as error; this is the detector's own evidence-*presence* check, not a judgment about whether the evidence is convincing (that half belongs to `test-smell-review`, and only ever on an already-waived double) [MECHANICAL] If a static-analysis pre-pass has already surfaced this exact finding (e.g. via `/code-review` step 2b), cite it rather than re-deriving it — do not double-report. @@ -185,18 +185,18 @@ third-party). Count tolerated-deviation artifacts from the following categories: - **Disabled tests** — `@Ignore`, `@Disabled`, `xit(`, `xdescribe(`, `test.skip(`, `it.skip(`, `[Ignore]`, `[Skip]`, `pytest.mark.skip`, `pytest.mark.xfail` with no - linked issue or expiry + linked issue or expiry [MECHANICAL] - **Aged markers** — `TODO`, `FIXME`, `HACK`, `XXX` comments (any age is a candidate; - flag as aged when there is no linked ticket or follow-up action) + flag as aged when there is no linked ticket or follow-up action) [MECHANICAL] - **Suppressed warnings** — `@SuppressWarnings`, `#pragma warning disable`, `# noqa`, `# type: ignore`, `eslint-disable`, `pylint: disable` with no explanatory - comment naming the specific approved exception + comment naming the specific approved exception [MECHANICAL] - **Relaxed assertions** — assertion strings containing "either … or", "at least", - "approximately", tolerance widening (e.g. `delta=`, `places=1` in `assertAlmostEqual`) + "approximately", tolerance widening (e.g. `delta=`, `places=1` in `assertAlmostEqual`) [MECHANICAL] - **Widened tolerances** — numeric epsilon/tolerance constants changed without a - comment explaining the regression + comment explaining the regression [MECHANICAL] -**Consolidation rule**: when **≥ 3 of these artifacts appear in the same file**, emit +**Consolidation rule** [MECHANICAL]: when **≥ 3 of these artifacts appear in the same file**, emit a **single** named finding: ```json diff --git a/plugins/dev-team/tests/agents/test_test_review_mechanical_split_annotated.py b/plugins/dev-team/tests/agents/test_test_review_mechanical_split_annotated.py new file mode 100644 index 000000000..4c319db10 --- /dev/null +++ b/plugins/dev-team/tests/agents/test_test_review_mechanical_split_annotated.py @@ -0,0 +1,196 @@ +"""Mechanical-vs-judgment annotation check for test-review.md (#2169, plan +step 2.1 of plans/2164-abort-countable-tiered.md). + +Step 2.1 is documentation-only: it annotates every existing check bullet +under ``## Detect`` and every bullet in the ``## Tolerated-Deviation Hunt`` +section with ``[MECHANICAL]`` or ``[JUDGMENT]``, based on whether the +bullet's own text already states an explicit per-language detection +signature or grep/threshold rule. No detection behavior changes here (the +script that acts on the tags lands in Step 2.2) — this test only guards the +annotation itself: every check bullet carries exactly one tag, never zero, +never both. +""" + +from __future__ import annotations + +import re + +from _repo_root import REPO_ROOT + +AGENT = REPO_ROOT / "plugins" / "dev-team" / "agents" / "test-review.md" + +TAG_PATTERN = re.compile(r"\[(MECHANICAL|JUDGMENT)\]") +BULLET_START = re.compile(r"^- ") +CONTINUATION = re.compile(r"^[ \t]+\S") + + +def _text() -> str: + return AGENT.read_text(encoding="utf-8") + + +def _section(text: str, start_heading: str, end_heading: str) -> str: + start = text.index(start_heading) + len(start_heading) + end = text.index(end_heading, start) + return text[start:end] + + +def _bullet_blocks(section: str) -> list[str]: + """Split a section into top-level ``- `` bullets, folding in indented + continuation lines (this file wraps long bullets across multiple + 2-space-indented lines) but stopping at the next unindented line — + a blank line, a new bullet, or a bare paragraph like + ``**Consolidation rule** ...`` — so a bullet's trailing tag isn't + accidentally absorbed into the next, untagged, non-bullet paragraph.""" + lines = section.split("\n") + blocks: list[str] = [] + i, n = 0, len(lines) + while i < n: + if BULLET_START.match(lines[i]): + block_lines = [lines[i]] + j = i + 1 + while j < n and CONTINUATION.match(lines[j]): + block_lines.append(lines[j]) + j += 1 + blocks.append("\n".join(block_lines)) + i = j + else: + i += 1 + return blocks + + +def _assert_each_bullet_has_exactly_one_tag(blocks: list[str]) -> None: + untagged = [b.splitlines()[0] for b in blocks if not TAG_PATTERN.findall(b)] + both = [ + b.splitlines()[0] + for b in blocks + if len(set(TAG_PATTERN.findall(b))) > 1 + ] + assert not untagged, f"bullets missing [MECHANICAL]/[JUDGMENT] tag: {untagged}" + assert not both, f"bullets carrying both tags: {both}" + + +def test_detect_section_bullets_each_carry_exactly_one_tag() -> None: + text = _text() + detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") + blocks = _bullet_blocks(detect) + assert len(blocks) >= 30, ( + f"expected the full set of ## Detect check bullets, found {len(blocks)}" + ) + _assert_each_bullet_has_exactly_one_tag(blocks) + + +def test_tolerated_deviation_hunt_bullets_each_carry_exactly_one_tag() -> None: + text = _text() + section = _section(text, "## Tolerated-Deviation Hunt", "## Self-Challenge") + blocks = _bullet_blocks(section) + assert len(blocks) == 5, ( + f"expected the 5 tolerated-deviation-artifact categories, found {len(blocks)}" + ) + _assert_each_bullet_has_exactly_one_tag(blocks) + + +def test_consolidation_rule_is_tagged_mechanical() -> None: + text = _text() + section = _section(text, "## Tolerated-Deviation Hunt", "## Self-Challenge") + assert "**Consolidation rule** [MECHANICAL]:" in section, ( + "the >=3-artifact consolidation rule is an explicit threshold rule " + "and should be tagged [MECHANICAL]" + ) + + +def test_testability_blockers_bullets_each_carry_exactly_one_tag() -> None: + """Testability blockers is a named subsection under ## Detect (not its + own ## heading) — covered by the ## Detect sweep above, but pinned here + directly per the task's explicit call-out of this section.""" + text = _text() + detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") + section = _section(detect, "Testability blockers:", "Internal-collaborator doubling") + blocks = _bullet_blocks(section) + assert len(blocks) == 3 + _assert_each_bullet_has_exactly_one_tag(blocks) + + +def test_internal_collaborator_doubling_bullets_each_carry_exactly_one_tag() -> None: + text = _text() + detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") + section = _section( + detect, + "Internal-collaborator doubling", + "If a static-analysis pre-pass", + ) + blocks = _bullet_blocks(section) + assert len(blocks) == 3 + _assert_each_bullet_has_exactly_one_tag(blocks) + # This subsection's own header already calls itself "mechanical — never + # a truth judgment"; every bullet in it should agree with that framing. + for block in blocks: + assert "[MECHANICAL]" in block, ( + f"internal-collaborator-doubling bullet not tagged MECHANICAL: " + f"{block.splitlines()[0]}" + ) + + +def test_reflection_bullet_is_mechanical_but_notes_non_gating_warning_severity() -> None: + """Reflection-into-private-members has an explicit per-language + detection signature (MECHANICAL for detection) but must stay + warning-severity and non-gating — only Step 2.2's no-assertion-tests + check and internal_double_detector.py's error-severity findings gate + the qualitative pass.""" + text = _text() + detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") + blocks = _bullet_blocks(detect) + reflection_blocks = [ + b for b in blocks if "reflection into private members" in b + ] + assert len(reflection_blocks) == 1 + block = reflection_blocks[0] + assert "[MECHANICAL]" in block + assert "`warning`-severity" in block + assert "without gating the qualitative pass" in block + assert "Step 2.2" in block + + +def test_known_mechanical_anchor_bullets_are_tagged_mechanical() -> None: + """Lock in the plan's explicit MECHANICAL examples (missing-await, + mocks-not-reset, unstubbed clock/RNG/timers, tests-with-no-assertion) + against accidental re-tagging.""" + text = _text() + detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") + blocks = _bullet_blocks(detect) + + def block_containing(snippet: str) -> str: + matches = [b for b in blocks if snippet in b] + assert len(matches) == 1, f"expected exactly one bullet with {snippet!r}" + return matches[0] + + for snippet in ( + "Tests with no assertion", + "Mocks/stubs not reset", + "Missing await on async operations", + "Unstubbed clock access", + "Unstubbed randomness", + "Unstubbed timers/delays", + ): + assert "[MECHANICAL]" in block_containing(snippet) + + +def test_known_judgment_anchor_bullets_are_tagged_judgment() -> None: + """Lock in the plan's explicit JUDGMENT examples (coverage-gap + adequacy, AAA structure, misleading descriptions, static-factory / + singleton testability blockers) against accidental re-tagging.""" + text = _text() + detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") + blocks = _bullet_blocks(detect) + + def block_containing(snippet: str) -> str: + matches = [b for b in blocks if snippet in b] + assert len(matches) == 1, f"expected exactly one bullet with {snippet!r}" + return matches[0] + + for snippet in ( + "Missing edge cases", + "No arrange-act-assert structure", + "Misleading test descriptions", + "Code under test that cannot be constructed with known values", + ): + assert "[JUDGMENT]" in block_containing(snippet) From 8c990664aa5805fd97fdb68a5fea75b4b5f11746 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 19:38:58 +0000 Subject: [PATCH 06/14] =?UTF-8?q?feat(scripts):=20add=20test=5Freview=5Fme?= =?UTF-8?q?chanics.py=20=E2=80=94=20mechanical=20pre-phase=20for=20test-re?= =?UTF-8?q?view?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Step 2.2 per-step review checkpoint (#2169) surfaced 13 confirmed findings across concurrency/doc/naming/performance/structure/test/ test-smell/spec-compliance/arch/correctness/domain/security-review: - The only self-computed gating check (no-assertion) scanned the test's title along with its body, so the most common JS/TS naming convention (it('should ...')) never gated (domain-review + correctness-review, independently confirmed). Split region extraction (for boundary-finding) from content scanning (for assertion/await checks), and made paren/brace balancing string/comment-aware to prevent both spurious and missed findings. - Several detection signatures dropped exemption clauses test-review.md documents (fake-timer suppression, mock re-instantiation, qualified tolerated-deviation markers) or omitted common test-declaration forms (async def, it.each, [Theory]/[TestCase], @ParameterizedTest), causing fail-open gaps (correctness-review). - The reused internal_double_detector.py's test-directory scoping gap (only test/tests-named dirs) was undisclosed, and its failure paths degraded to suggestion-severity, indistinguishable from a clean pass (arch-review). Added explicit out-of-scope/unavailable findings and a doublingCheckRan flag. - No drift guard tied test-review.md's [MECHANICAL] annotations to this script's _GATING_CATEGORIES (domain-review); test-review.md itself still named a nonexistent severity field on the reused detector. - Magic values, an unclear function name, duplicate region extraction, and an unconditional subprocess spawn coupling unrelated tests to the detector (naming/structure/test/test-smell-review). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/agents/test-review.md | 2 +- .../dev-team/scripts/test_review_mechanics.py | 925 ++++++++++++++++++ .../scripts/test_test_review_mechanics.py | 669 +++++++++++++ tests/repo/test_python_floor.py | 5 + 4 files changed, 1600 insertions(+), 1 deletion(-) create mode 100755 plugins/dev-team/scripts/test_review_mechanics.py create mode 100644 plugins/dev-team/tests/scripts/test_test_review_mechanics.py diff --git a/plugins/dev-team/agents/test-review.md b/plugins/dev-team/agents/test-review.md index c0508aba0..344f669ae 100644 --- a/plugins/dev-team/agents/test-review.md +++ b/plugins/dev-team/agents/test-review.md @@ -167,7 +167,7 @@ Testability blockers: - Code under test that cannot be constructed with known values (static factories, singletons, no injectable constructor) — flag as error; per `${CLAUDE_PLUGIN_ROOT}/knowledge/testability-patterns.md#pattern-1-constructor-injection-replace-static-factories-singletons`, the production code must change, not the test approach [JUDGMENT] - Mocking of concrete classes (not interfaces) — flag as warning; extract an interface for the dependency [JUDGMENT] -- Tests using reflection into private members as primary strategy — flag as warning. This is an architecture/encapsulation issue the test is reaching around, not a test-hygiene nit. Detection signatures: Java: `getDeclaredMethod`/`getDeclaredField` + `setAccessible(true)`, `Method.invoke` on a private/protected member; C#: `Type.GetMethod(..., BindingFlags.NonPublic | BindingFlags.Instance)`, `Type.InvokeMember`; Python: `getattr`/`setattr`/`hasattr` targeting a name-mangled (`_ClassName__attr`) or underscore-prefixed attribute; JS/TS: bracket-notation access into a `private`/non-exported member (e.g. `(obj as any)['_privateMethod']()`), `Object.getOwnPropertyDescriptor`/`Object.defineProperty` used to reach a non-exported member. Suggested fix — pick by shape of the code, never the generic "expand the public API": (1) extract the private logic into a collaborator with its own public seam, when it's standalone logic worth testing independently; (2) relax visibility to package-private/internal, only when a production collaborator in the same module/assembly independently needs the access (the language must have that tier) — never as a grant solely so the test can reach in, which recreates the `InternalsVisibleTo`/`@VisibleForTesting` anti-pattern below; (3) test the behavior through the class's existing public API, when the private method is already an implementation detail of a public behavior [MECHANICAL] (detection only, via the explicit per-language signatures above — this stays `warning`-severity and is reported by the future mechanical pre-phase without gating the qualitative pass; only the no-assertion-tests check and `internal_double_detector.py`'s own `error`-severity findings gate the pass — see Step 2.2) +- Tests using reflection into private members as primary strategy — flag as warning. This is an architecture/encapsulation issue the test is reaching around, not a test-hygiene nit. Detection signatures: Java: `getDeclaredMethod`/`getDeclaredField` + `setAccessible(true)`, `Method.invoke` on a private/protected member; C#: `Type.GetMethod(..., BindingFlags.NonPublic | BindingFlags.Instance)`, `Type.InvokeMember`; Python: `getattr`/`setattr`/`hasattr` targeting a name-mangled (`_ClassName__attr`) or underscore-prefixed attribute; JS/TS: bracket-notation access into a `private`/non-exported member (e.g. `(obj as any)['_privateMethod']()`), `Object.getOwnPropertyDescriptor`/`Object.defineProperty` used to reach a non-exported member. Suggested fix — pick by shape of the code, never the generic "expand the public API": (1) extract the private logic into a collaborator with its own public seam, when it's standalone logic worth testing independently; (2) relax visibility to package-private/internal, only when a production collaborator in the same module/assembly independently needs the access (the language must have that tier) — never as a grant solely so the test can reach in, which recreates the `InternalsVisibleTo`/`@VisibleForTesting` anti-pattern below; (3) test the behavior through the class's existing public API, when the private method is already an implementation detail of a public behavior [MECHANICAL] (detection only, via the explicit per-language signatures above — this stays `warning`-severity and is reported by the future mechanical pre-phase without gating the qualitative pass; only the no-assertion-tests check and `internal_double_detector.py`'s own `verdict: "high"` findings (translated to `error` severity by the mechanical pre-phase) gate the pass — see Step 2.2 (`scripts/test_review_mechanics.py`)) Internal-collaborator doubling (mechanical — never a truth judgment; see `${CLAUDE_PLUGIN_ROOT}/knowledge/internal-collaborator-doubling.md#the-waiver`): diff --git a/plugins/dev-team/scripts/test_review_mechanics.py b/plugins/dev-team/scripts/test_review_mechanics.py new file mode 100755 index 000000000..8a1f21d3d --- /dev/null +++ b/plugins/dev-team/scripts/test_review_mechanics.py @@ -0,0 +1,925 @@ +#!/usr/bin/env python3 +"""Mechanical pre-phase for `agents/test-review.md` (#2169 Step 2.2). + +`test-review.md` (Step 2.1, #2169) annotates each of its own `## Detect` +checks as `[MECHANICAL]` (an explicit per-language detection signature or +grep/threshold rule) or `[JUDGMENT]` (requires semantic reasoning about +intent). This script computes every `[MECHANICAL]` check that is cheap to +run as a script, for one test file at a time, so the agent cites raw counts +instead of re-deriving them in prose: + +- tests with no assertion call — `error` (matches test-review.md's Severity + Anchors table) +- missing `await` on async test bodies (JS/TS/C#/Java) — `warning` +- mocks/stubs created without a reset/clear call, or a same-shaped + re-instantiation/re-initialization in `[SetUp]`/`[TestInitialize]`/ + `@BeforeEach`/`@BeforeAll`, in the same file — `warning` +- unstubbed clock/RNG/timer access (JS/TS/C#/Java), suppressed when a + fake-timer/injected-clock marker (`jest.useFakeTimers()`, an injected + `IClock`/`TimeProvider`, `Clock.fixed`, ...) is present in the same + file — `warning` +- reflection into private members as primary test strategy (Java/C#/ + Python/JS/TS) — `warning`, NOT promoted to `error` by this migration +- the Tolerated-Deviation Hunt's >=3-artifact consolidation rule, with a + cheap same-line/adjacent-line ticket-reference qualifier suppressing the + disabled-test/aged-marker/suppressed-warning categories — `warning` + +Test-region boundaries (the `it()`/`test()` call for JS/TS, the annotated +method for C#/Java) are found by depth-counting over a MASKED copy of the +file (`_mask_code`) with string/template/char literals and comments blanked +out first — a stray `)`/`(`/`{`/`}` inside a title, comment, or string body +can no longer early-close a region or drive the balance past EOF. For JS/TS, +the no-assertion and missing-await checks additionally never see the +description-string argument itself — `_js_content_slice` derives a +callback-only CONTENT slice from the boundary-only region, so a title like +`'should render'` or `'awaits the response'` can never satisfy the +assertion/await keyword search on its own words (see `_GATING_CATEGORIES` +below for why this specifically matters: no-assertion is one of the two +categories that sets `mechanicalFail`). + +Internal-collaborator-doubling detection is NOT reimplemented here — it is +reused via subprocess against `skills/test-design/scripts/ +internal_double_detector.py`, whose own docstring documents `--files` as +the seam built for exactly this kind of caller (the subprocess call itself +goes through an injectable `double_detector_runner` parameter so unit tests +for unrelated checks can stub it out rather than paying real spawn cost). +That detector has no `severity` field; its findings carry a `verdict` of +`"high"`, `"informational"`, or `"advisory"`. Only `verdict == "high"` is +translated here into this script's own `error`-severity finding — never a +literal `severity == "error"` check against the reused script's raw JSON, +since no such field exists there. + +Two disclosure gaps are surfaced explicitly rather than failing open to a +plain clean bill: + +- `internal_double_detector.py`'s own `analyze()` only scans files under a + `test`/`tests` path segment — narrower than this repo's own test-file + conventions (`*.test.*`, `*.spec.*`, `__tests__/`, see + `knowledge/test-file-indicators.md`). When the analyzed file's path has + no `test`/`tests` segment, the doubling check never runs for it; this + script emits a named `internal-collaborator-doubling-out-of-scope` + finding instead of silently reporting zero doubling findings. +- A subprocess spawn/timeout failure or invalid-JSON response from the + detector is reported as `internal-collaborator-doubling-unavailable` at + `warning` severity (raised from `suggestion` — a run that never happened + must not read the same as a clean pass). + +Both gaps also set the top-level `doublingCheckRan: false` key so a +downstream consumer can distinguish "mechanically clean except the doubling +gate, which did not run" from a plain clean bill. + +`mechanicalFail` is `true` ONLY when a no-assertion-test finding is present +OR a translated `internal-collaborator-doubling` `error` finding is +present. Every other finding this script emits is `warning`-severity: it is +reported, with counts, but never gates the qualitative pass on its own. + +A file this script cannot decode/parse produces a single `parse-failure` +finding (no severity that participates in `mechanicalFail`) and falls +through as mechanically clean — this script never crashes and never +silently drops a file. + +This script does NOT own an ambient-API marker table of its own for +`internal_double_detector.py`'s B2-waiver-evidence purpose — that table +(`_AMBIENT_API_MARKERS` in that module) stays there; see the REFACTOR note +at `_CLOCK_RNG_TIMER_MARKERS` below for why this script's own marker table +is a different list, not a duplicate. + +The Tolerated-Deviation Hunt (`_check_tolerated_deviation`) runs against +whichever single file the CLI is given, test or non-test — it does NOT +restrict itself to "core-flow, non-test source" the way `test-review.md`'s +prose scopes the hunt. Resolving that scoping gap (which files this check +should actually run against) is deferred to Step 2.3's wiring of this +script into test-review's protocol; this script's own per-file CLI shape +is unopinionated about which files it's called with. + +Stdlib-only (ADR 0014/0015). See docs/python-hook-contract.md. +""" + +from __future__ import annotations + +import argparse +import json +import re +import subprocess +import sys +from collections.abc import Callable +from pathlib import Path + +#: `internal_double_detector.py` lives under `skills/test-design/scripts/`; +#: this script lives under `scripts/` — both are children of the plugin +#: root (`plugins/dev-team/`), so `parents[1]` from this file reaches it. +_DETECTOR_PATH = ( + Path(__file__).resolve().parents[1] + / "skills" + / "test-design" + / "scripts" + / "internal_double_detector.py" +) + +LANG_BY_EXT: dict[str, str] = { + ".py": "python", + ".js": "js_ts", + ".jsx": "js_ts", + ".mjs": "js_ts", + ".cjs": "js_ts", + ".ts": "js_ts", + ".tsx": "js_ts", + ".cs": "csharp", + ".java": "java", +} + +CONSOLIDATION_THRESHOLD = 3 + +#: "No specific line available" sentinel — used for parse-failure/decode +#: findings and translated-detector findings with no line info of their own. +_UNKNOWN_LINE = 1 + +#: Passed as `timeout=` to the `internal_double_detector.py` subprocess call. +_DOUBLE_DETECTOR_TIMEOUT_SECONDS = 60 + +#: `test`/`tests` path-segment names — mirrors `internal_double_detector.py`'s +#: own `_TEST_DIR_NAMES`/`_is_test_tree` scoping exactly (this is a +#: disclosure check on top of that scoping, not a reimplementation of the +#: detector itself — see the module docstring's out-of-scope paragraph). +_TEST_DIR_NAMES = frozenset({"test", "tests"}) + +# --- Generic assertion-call detector ----------------------------------------- + +#: "no Assert, expect, should, verify, or equivalent assertion call" — +#: test-review.md's own generic (language-agnostic) wording for the +#: no-assertion check. +_ASSERTION_RE = re.compile(r"\b(assert\w*|expect|should\w*|verify\w*)\b", re.IGNORECASE) + + +# --- Test-region extraction (brace/paren-balanced, per language) ------------ + +#: Matches a plain `it(`/`test(` call (with optional `.only`/`.skip`) as +#: well as an `it.each()(`/`test.each(
)(` parameterized-table +#: call — group 1 always captures the FINAL call's own opening paren (the +#: description+callback call), never the `.each` table's paren, so +#: extraction downstream is identical for both forms. The `.each(...)` +#: table itself may contain one level of nested parens (e.g. a function +#: call inside the table) but no more — a reasonable approximation of the +#: common forms, not a full parser. +_JS_TEST_CALL_RE = re.compile( + r"\b(?:it|test)(?:\.only|\.skip)?" + r"(?:\.each\s*\((?:[^()]|\([^()]*\))*\))?" + r"\s*(\()" +) +_JS_ASYNC_CALLBACK_RE = re.compile(r"^\(\s*['\"`][^'\"`]*['\"`]\s*,\s*async\b") + +_CSHARP_TEST_ATTR_RE = re.compile( + r"^[ \t]*\[(?:Test|Fact|TestMethod|Theory|TestCase|TestCaseSource)(?:\([^\]]*\))?\]", + re.MULTILINE, +) +_JAVA_TEST_ANNOT_RE = re.compile(r"^[ \t]*@(?:Test|ParameterizedTest)\b", re.MULTILINE) +_PY_TEST_DEF_RE = re.compile(r"^([ \t]*)(?:async\s+)?def\s+(test_\w+)\s*\(", re.MULTILINE) + +#: Setup/init method markers — used only by `_mock_reinitialized` (Fix 4) to +#: find a `[SetUp]`/`[TestInitialize]`/`@BeforeEach`/`@BeforeAll` method's +#: own body, never treated as a test method for any other check. +_CSHARP_SETUP_ATTR_RE = re.compile(r"^[ \t]*\[(?:SetUp|TestInitialize)\]", re.MULTILINE) +_JAVA_SETUP_ANNOT_RE = re.compile(r"^[ \t]*@(?:BeforeEach|BeforeAll)\b", re.MULTILINE) +_SETUP_ANNOT_RE: dict[str, re.Pattern] = {"csharp": _CSHARP_SETUP_ATTR_RE, "java": _JAVA_SETUP_ANNOT_RE} + +_CSHARP_ASYNC_TASK_RE = re.compile(r"\basync\s+Task\b") +_JAVA_FUTURE_TYPE_RE = re.compile(r"\b(?:Future|CompletableFuture)\b") +_JAVA_FUTURE_RESOLVED_RE = re.compile(r"\.(?:get|join)\s*\(") + + +class ParseFailure(Exception): + """A test-region boundary (parens/braces) never closed before EOF. + + Raised internally by the region-extraction helpers and caught once, at + the top of `analyze_file`, which turns it into a `parse-failure` + finding — this exception must never propagate out of `analyze_file`.""" + + +def _line_no(text: str, idx: int) -> int: + return text.count("\n", 0, idx) + 1 + + +def _mask_code(text: str) -> str: + """Same-length copy of `text` with string/template/char literals and + `//`/`/* */` comments blanked to spaces (newlines preserved), so + paren/brace depth-counting and top-level-comma scanning never trip on + a stray `)`/`(`/`{`/`}`/`,` sitting inside a title, comment, or string + body (Fix 1). Character offsets and line numbers are identical to + `text` — callers index into the ORIGINAL text using positions computed + against this masked copy. + + Limitation: a JS/TS template literal's `${...}` interpolation is + masked along with the rest of the template, not treated as live code — + blanket-masking the whole template is the safer default (a stray brace + inside an interpolation could otherwise miscount), at the cost of not + recognizing real code inside `${}`. Out of scope for this fix pass.""" + out = list(text) + i, n = 0, len(text) + while i < n: + two = text[i : i + 2] + if two == "//": + j = i + while j < n and text[j] != "\n": + out[j] = " " + j += 1 + i = j + elif two == "/*": + end = text.find("*/", i + 2) + j_end = end + 2 if end != -1 else n + for j in range(i, j_end): + if text[j] != "\n": + out[j] = " " + i = j_end + elif text[i] in ("'", '"', "`"): + quote = text[i] + out[i] = " " + j = i + 1 + while j < n: + if text[j] == "\\" and j + 1 < n: + if text[j] != "\n": + out[j] = " " + if text[j + 1] != "\n": + out[j + 1] = " " + j += 2 + continue + closing = text[j] == quote + if text[j] != "\n": + out[j] = " " + j += 1 + if closing: + break + i = j + else: + i += 1 + return "".join(out) + + +def _matching_close_index(text: str, masked: str, open_idx: int, open_ch: str, close_ch: str) -> int | None: + """Index into `text` of the character matching `text[open_idx]` (which + must be `open_ch`), found by depth-counting over `masked` (same length + as `text`, string/template/char literals and comments already blanked + there — Fix 1). Returns `None` if depth never returns to 0 before EOF.""" + depth = 0 + i = open_idx + n = len(masked) + while i < n: + c = masked[i] + if c == open_ch: + depth += 1 + elif c == close_ch: + depth -= 1 + if depth == 0: + return i + i += 1 + return None + + +def _js_content_slice(region: str, masked_region: str) -> str: + """Slice of `region` (a full `it()`/`test()` call region, its own + parens included) starting just after the description-string + argument's closing quote and the following top-level comma — the + callback argument only, title/description stripped out (Fix 1: + `_check_no_assertion`/`_check_missing_await` must never match a + keyword sitting in the title itself, e.g. `'should render'` or + `'awaits the response'`). Falls back to the full `region` when no + leading string-literal argument is found (e.g. an `it(name, fn)` form + using a variable title) or no top-level comma follows it.""" + n = len(region) + j = 1 + while j < n and region[j] in " \t\r\n": + j += 1 + if j >= n or region[j] not in "'\"`": + return region + quote = region[j] + end = j + 1 + closed = False + while end < n: + if region[end] == "\\" and end + 1 < n: + end += 2 + continue + if region[end] == quote: + end += 1 + closed = True + break + end += 1 + if not closed: + return region + depth = 0 + k = end + while k < n: + c = masked_region[k] + if c in "([{": + depth += 1 + elif c in ")]}": + if depth == 0: + return region + depth -= 1 + elif c == "," and depth == 0: + return region[k + 1 :] + k += 1 + return region + + +def _js_ts_test_regions(text: str, masked: str) -> list[dict]: + """`{"line", "is_async", "content"}` for each `it()`/`test()` call + (including `.each` parameterized-table forms) — shared by + `_check_no_assertion` and `_check_missing_await` (Fix 11) so the + region/content boundaries are computed exactly once per file. + `is_async` is read off the FULL region (it needs the title text + structurally present to match); `content` is the title-stripped + slice used for keyword search.""" + out = [] + for m in _JS_TEST_CALL_RE.finditer(masked): + open_idx = m.start(1) + close_idx = _matching_close_index(text, masked, open_idx, "(", ")") + if close_idx is None: + raise ParseFailure("unbalanced parens in an it()/test() call") + region = text[open_idx : close_idx + 1] + masked_region = masked[open_idx : close_idx + 1] + is_async = bool(_JS_ASYNC_CALLBACK_RE.match(region)) + content = _js_content_slice(region, masked_region) + out.append({"line": _line_no(text, m.start()), "is_async": is_async, "content": content}) + return out + + +def _annotated_test_regions(text: str, masked: str, annot_re: re.Pattern) -> list[dict]: + """`{"line", "signature", "body"}` for each method matching `annot_re` + (a `[Test]`/`@Test`-style attribute/annotation, or a + `[SetUp]`/`@BeforeEach`-style one when reused by `_mock_reinitialized`) + — `signature` is the annotation through the parameter list, `body` is + the brace-balanced method body. Boundaries are found via `masked` + (Fix 1) so a stray brace/paren inside a body's own string literal can't + drive the balance past EOF.""" + out = [] + for m in annot_re.finditer(masked): + paren_idx = masked.find("(", m.end()) + if paren_idx == -1: + raise ParseFailure("no parameter list found after a test attribute") + close_paren = _matching_close_index(text, masked, paren_idx, "(", ")") + if close_paren is None: + raise ParseFailure("unbalanced parens in an annotated test method's signature") + brace_idx = masked.find("{", close_paren) + if brace_idx == -1: + raise ParseFailure("no method body found after a test attribute") + close_brace = _matching_close_index(text, masked, brace_idx, "{", "}") + if close_brace is None: + raise ParseFailure("unbalanced braces in an annotated test method's body") + signature = text[m.start() : brace_idx] + body = text[brace_idx : close_brace + 1] + out.append({"line": _line_no(text, m.start()), "signature": signature, "body": body}) + return out + + +def _extract_regions(text: str, masked: str, lang: str) -> list[dict]: + """Single per-language test-region extraction pass, shared by + `_check_no_assertion` and `_check_missing_await` (Fix 11).""" + if lang == "js_ts": + return _js_ts_test_regions(text, masked) + if lang == "csharp": + return _annotated_test_regions(text, masked, _CSHARP_TEST_ATTR_RE) + if lang == "java": + return _annotated_test_regions(text, masked, _JAVA_TEST_ANNOT_RE) + return [] + + +def _python_test_regions(text: str) -> list[tuple[int, str]]: + """`(line_no, body_text)` for each `def test_*(...)`/`async def + test_*(...)` — body is every line after the `def` line indented deeper + than it, up to the first dedent or EOF (blank lines don't count as a + dedent).""" + lines = text.split("\n") + regions = [] + for m in _PY_TEST_DEF_RE.finditer(text): + indent = m.group(1) + def_line_idx = text.count("\n", 0, m.start()) + body_lines = [] + for line in lines[def_line_idx + 1 :]: + if line.strip() == "": + body_lines.append(line) + continue + cur_indent_len = len(line) - len(line.lstrip(" \t")) + if cur_indent_len <= len(indent): + break + body_lines.append(line) + regions.append((def_line_idx + 1, "\n".join(body_lines))) + return regions + + +# --- Per-language signature tables (checks c/d/e) ---------------------------- + +#: (a) no-assertion and (c) mock-not-reset are file/test-scanned per +#: language; JS/TS/C#/Java cover (a)+(b), Python covers (a) only — +#: test-review.md documents no missing-await/mock-reset/clock-RNG-timer +#: signatures for Python. + +_MOCK_CONSTRUCT_MARKERS: dict[str, tuple[re.Pattern, ...]] = { + "js_ts": (re.compile(r"\bjest\.fn\s*\("), re.compile(r"\bjest\.mock\s*\(")), + "csharp": (re.compile(r"\bMock<"), re.compile(r"\bSubstitute\.For<")), + "java": (re.compile(r"\bMockito\.mock\s*\("), re.compile(r"@Mock\b")), +} +_MOCK_RESET_MARKERS: dict[str, re.Pattern] = { + "js_ts": re.compile(r"\bjest\.clearAllMocks\s*\("), + "csharp": re.compile(r"\.Reset\s*\(\)|\bClearReceivedCalls\s*\("), + # Mockito-scoped (Fix 4): the previous bare `\breset\s*\(` matched ANY + # method literally named `reset(` in the file, including production + # code under test, which could silently suppress a real finding. This + # narrower form misses an unqualified `reset()` call reached via + # `import static org.mockito.Mockito.reset;` — a deliberate, documented + # trade (see the finding this fixes: an over-broad marker in the wrong + # direction is worse than a narrow miss here). + "java": re.compile(r"\bMockito\.reset\s*\("), +} + +#: Re-instantiation/re-initialization alternative to an explicit reset call +#: (Fix 4) — test-review.md documents BOTH forms ("Moq `Mock` reused +#: without `Reset()` OR RE-INSTANTIATION"; "Mockito missing `reset()` OR +#: `@BeforeEach` RE-INITIALIZATION"), but `_MOCK_RESET_MARKERS` only ever +#: covered the explicit-call half. `_mock_reinitialized` checks whether a +#: `[SetUp]`/`[TestInitialize]`/`@BeforeEach`/`@BeforeAll` method's own body +#: contains a mock-construct marker — re-creating the mock counts the same +#: as resetting it. JS/TS has no re-instantiation alternative documented in +#: test-review.md (only `jest.clearAllMocks()`), so it has no entry here. + +#: REFACTOR (Step 2.2): this table is data-shaped like +#: `internal_double_detector.py`'s `_AMBIENT_API_MARKERS`, but it is NOT the +#: same list and does not share a use case with it, so it is kept separate +#: rather than factored together — see that module's table for the +#: comparison and this file's own module docstring for the one-line +#: pointer. `_AMBIENT_API_MARKERS` answers "does this COLLABORATOR's own +#: declaring file reference ambient state" (evidence for a B2 double +#: waiver: env/hostname/cwd/locale included, no timer/interval markers). +#: This table answers "does this TEST body call an unstubbed clock/RNG/ +#: timer API directly" (test-review.md's own non-determinism-sources +#: signatures: no env/hostname/cwd/locale, but adds setTimeout/ +#: setInterval/setImmediate/Task.Delay/Thread.Sleep, which +#: `_AMBIENT_API_MARKERS` has no equivalent for). The two tables overlap on +#: three literal substrings (Date/DateTime/Random-family) because both are +#: independently describing "the clock and RNG", not because one was +#: copied from the other — merging them would either strip timer markers +#: this check needs or leak env/hostname markers into a check +#: test-review.md never documented for it. +_CLOCK_RNG_TIMER_MARKERS: dict[str, tuple[re.Pattern, ...]] = { + "js_ts": ( + re.compile(r"\bDate\.now\s*\("), + re.compile(r"\bDate\s*\("), + re.compile(r"\bMath\.random\s*\("), + re.compile(r"\bsetTimeout\s*\("), + re.compile(r"\bsetInterval\s*\("), + re.compile(r"\bsetImmediate\s*\("), + ), + "csharp": ( + re.compile(r"\bDateTime\.Now\b"), + re.compile(r"\bDateTime\.UtcNow\b"), + re.compile(r"\bDateTimeOffset\.Now\b"), + re.compile(r"\bnew\s+Random\s*\("), + re.compile(r"\bTask\.Delay\s*\("), + re.compile(r"\bThread\.Sleep\s*\("), + ), + "java": ( + re.compile(r"\bnew\s+Date\s*\("), + re.compile(r"\bLocalDateTime\.now\s*\("), + re.compile(r"\bInstant\.now\s*\("), + re.compile(r"\bSystem\.currentTimeMillis\s*\("), + re.compile(r"\bnew\s+Random\s*\("), + re.compile(r"\bMath\.random\s*\("), + re.compile(r"\bThread\.sleep\s*\("), + ), +} + +#: Fake-timer/injected-clock markers (Fix 3) — test-review.md's own bullets +#: exempt stubbed usage ("WITHOUT FAKE TIMERS", "WITHOUT INJECTION"), and +#: its Severity Anchors table names `jest.useFakeTimers()` as the remedy. +#: Presence anywhere in the file suppresses the WHOLE +#: unstubbed-clock-rng-timer finding for that file (the same file-level +#: granularity `_MOCK_RESET_MARKERS` already uses for mock-not-reset, not a +#: per-hit check). +_CLOCK_STUB_MARKERS: dict[str, tuple[re.Pattern, ...]] = { + "js_ts": ( + re.compile(r"\bjest\.useFakeTimers\s*\("), + re.compile(r"\bvi\.useFakeTimers\s*\("), + re.compile(r"\bsinon\.useFakeTimers\s*\("), + ), + "csharp": (re.compile(r"\bIClock\b"), re.compile(r"\bTimeProvider\b")), + "java": (re.compile(r"\bClock\.fixed\s*\("),), +} + +_REFLECTION_MARKERS: dict[str, tuple[re.Pattern, ...]] = { + "java": ( + re.compile(r"\bgetDeclaredMethod\b"), + re.compile(r"\bgetDeclaredField\b"), + re.compile(r"\.setAccessible\s*\(\s*true\s*\)"), + re.compile(r"\bMethod\.invoke\b"), + ), + "csharp": ( + re.compile(r"\.GetMethod\s*\([^)]*BindingFlags\.NonPublic"), + re.compile(r"\bInvokeMember\s*\("), + ), + "python": ( + re.compile(r"\bgetattr\s*\([^,]+,\s*['\"]_\w+['\"]"), + re.compile(r"\bsetattr\s*\([^,]+,\s*['\"]_\w+['\"]"), + re.compile(r"\bhasattr\s*\([^,]+,\s*['\"]_\w+['\"]"), + ), + "js_ts": ( + re.compile(r"\[\s*['\"]_\w+['\"]\s*\]"), + re.compile(r"\bObject\.getOwnPropertyDescriptor\s*\("), + re.compile(r"\bObject\.defineProperty\s*\("), + ), +} + +#: Tolerated-Deviation Hunt categories, sourced from test-review.md's own +#: grep-pattern prose (language-agnostic — this check has never been +#: per-language in the agent file). NOT "migrated verbatim" (correction — +#: see the module docstring's Tolerated-Deviation Hunt paragraph for the +#: file-scoping disclosure, and below for the qualifier-suppression Fix 8 +#: adds on top of the raw grep patterns): each category in test-review.md +#: carries a qualifying clause this table only partially implements — +#: "disabled tests" count only *with no linked issue or expiry*; "aged +#: markers" only *when there is no linked ticket or follow-up action*; +#: "suppressed warnings" only *with no explanatory comment naming the +#: specific approved exception*. For those three categories, +#: `_line_is_qualified` suppresses a hit whose own line or an immediately +#: adjacent line carries a ticket/issue reference or explanatory comment — +#: a cheap same-line-or-adjacent-line approximation, not a real +#: linked-issue lookup. "Relaxed assertions" has no qualifying clause in +#: test-review.md (unconditional). "Widened tolerances" (`changed without +#: a comment explaining the regression`) is inherently diff-based — this +#: single-file scan cannot tell whether a tolerance constant was recently +#: *changed* at all, so that category remains an unqualified upper-bound +#: approximation here; a genuine diff-aware check is out of scope for this +#: fix pass. +_DEVIATION_MARKERS: tuple[tuple[str, re.Pattern], ...] = ( + ( + "disabled-test", + re.compile( + r"@Ignore\b|@Disabled\b|\bxit\s*\(|\bxdescribe\s*\(|\btest\.skip\s*\(" + r"|\bit\.skip\s*\(|\[Ignore\]|\[Skip\]|pytest\.mark\.skip\b|pytest\.mark\.xfail\b" + ), + ), + ("aged-marker", re.compile(r"\b(?:TODO|FIXME|HACK|XXX)\b")), + ( + "suppressed-warning", + re.compile( + r"@SuppressWarnings\b|#pragma warning disable|#\s*noqa\b|#\s*type:\s*ignore" + r"|eslint-disable|pylint:\s*disable" + ), + ), + ( + "relaxed-assertion", + re.compile(r"either\s+.{0,40}?\s+or\b|\bat least\b|\bapproximately\b|\bdelta\s*=|\bplaces\s*=\s*\d"), + ), + ("widened-tolerance", re.compile(r"\b(?:epsilon|tolerance)\s*=")), +) + +#: Categories whose test-review.md wording carries a "no linked +#: issue/ticket/exception" qualifier (Fix 8) — checked via +#: `_line_is_qualified`. +_QUALIFIED_DEVIATION_CATEGORIES = frozenset({"disabled-test", "aged-marker", "suppressed-warning"}) + +#: A ticket/issue reference or explanatory link — `#123`, `PROJ-456`, or a +#: URL/word naming an issue/ticket tracker. +_QUALIFIER_RE = re.compile(r"#\d+|\b[A-Za-z]{2,}-\d+\b|\bissue\b|\bticket\b", re.IGNORECASE) + + +def _finding(category: str, severity: str | None, line: int, message: str, count: int = 1) -> dict: + return {"category": category, "severity": severity, "line": line, "message": message, "count": count} + + +# --- Checks ------------------------------------------------------------------- + + +def _check_no_assertion(regions: list[dict], lang: str) -> list[dict]: + key = "content" if lang == "js_ts" else "body" + findings = [] + for entry in regions: + if not _ASSERTION_RE.search(entry[key]): + findings.append( + _finding("no-assertion", "error", entry["line"], "Test has no assertion call — zero regression protection.") + ) + return findings + + +def _check_no_assertion_python(text: str) -> list[dict]: + findings = [] + for line_no, body in _python_test_regions(text): + if not _ASSERTION_RE.search(body): + findings.append( + _finding("no-assertion", "error", line_no, "Test has no assertion call — zero regression protection.") + ) + return findings + + +def _check_missing_await(regions: list[dict], lang: str) -> list[dict]: + findings = [] + if lang == "js_ts": + for entry in regions: + if entry["is_async"] and "await" not in entry["content"]: + findings.append( + _finding( + "missing-await", "warning", entry["line"], "Async test body has no `await` — likely an unawaited promise." + ) + ) + elif lang == "csharp": + for entry in regions: + if _CSHARP_ASYNC_TASK_RE.search(entry["signature"]) and "await" not in entry["body"]: + findings.append( + _finding( + "missing-await", + "warning", + entry["line"], + "`async Task` test method has no `await` — likely an unresolved Task.", + ) + ) + elif lang == "java": + for entry in regions: + if _JAVA_FUTURE_TYPE_RE.search(entry["body"]) and not _JAVA_FUTURE_RESOLVED_RE.search(entry["body"]): + findings.append( + _finding( + "missing-await", + "warning", + entry["line"], + "Test body references Future/CompletableFuture with no `.get()`/`.join()` resolution.", + ) + ) + return findings + + +def _mock_reinitialized(text: str, masked: str, lang: str) -> bool: + """True when a `[SetUp]`/`[TestInitialize]`/`@BeforeEach`/`@BeforeAll` + method's own body contains a mock-construct marker (Fix 4) — the + re-instantiation/re-initialization alternative test-review.md + documents alongside an explicit reset call.""" + setup_re = _SETUP_ANNOT_RE.get(lang) + construct_markers = _MOCK_CONSTRUCT_MARKERS.get(lang) + if not setup_re or not construct_markers: + return False + for entry in _annotated_test_regions(text, masked, setup_re): + if any(marker.search(entry["body"]) for marker in construct_markers): + return True + return False + + +def _check_mock_not_reset(text: str, masked: str, lang: str) -> list[dict]: + construct_markers = _MOCK_CONSTRUCT_MARKERS.get(lang) + if not construct_markers: + return [] + hit_lines = [_line_no(text, m.start()) for marker in construct_markers for m in marker.finditer(text)] + if not hit_lines: + return [] + reset_marker = _MOCK_RESET_MARKERS[lang] + if reset_marker.search(text): + return [] + if _mock_reinitialized(text, masked, lang): + return [] + return [ + _finding( + "mock-not-reset", + "warning", + min(hit_lines), + f"{len(hit_lines)} mock/stub construction site(s) with no reset/clear call found in this file.", + count=len(hit_lines), + ) + ] + + +def _check_unstubbed_clock_rng_timer(text: str, lang: str) -> list[dict]: + if any(marker.search(text) for marker in _CLOCK_STUB_MARKERS.get(lang, ())): + return [] + findings = [] + for marker in _CLOCK_RNG_TIMER_MARKERS.get(lang, ()): + for m in marker.finditer(text): + findings.append( + _finding( + "unstubbed-clock-rng-timer", + "warning", + _line_no(text, m.start()), + f"Unstubbed clock/RNG/timer access: {m.group(0)!r}.", + ) + ) + return findings + + +def _check_reflection_primary_strategy(text: str, lang: str) -> list[dict]: + findings = [] + for marker in _REFLECTION_MARKERS.get(lang, ()): + for m in marker.finditer(text): + findings.append( + _finding( + "reflection-primary-strategy", + "warning", + _line_no(text, m.start()), + f"Reflection into private members: {m.group(0)!r}. Architecture/encapsulation issue, not a test-hygiene nit.", + ) + ) + return findings + + +def _line_is_qualified(lines: list[str], line_no: int) -> bool: + """True when the marker's own line, or the immediately preceding/ + following line, carries a ticket/issue reference or explanatory + comment (Fix 8).""" + for idx in (line_no - 2, line_no - 1, line_no): + if 0 <= idx < len(lines) and _QUALIFIER_RE.search(lines[idx]): + return True + return False + + +def _check_tolerated_deviation(text: str) -> list[dict]: + lines = text.split("\n") + hits = [] + for name, marker in _DEVIATION_MARKERS: + for m in marker.finditer(text): + line_no = _line_no(text, m.start()) + if name in _QUALIFIED_DEVIATION_CATEGORIES and _line_is_qualified(lines, line_no): + continue + hits.append((name, line_no)) + if len(hits) < CONSOLIDATION_THRESHOLD: + return [] + hits.sort(key=lambda h: h[1]) + listing = ", ".join(f"{name}@L{line}" for name, line in hits) + return [ + _finding( + "tolerated-deviation-consolidation", + "warning", + hits[0][1], + f"Fail-safe posture erosion: {len(hits)} tolerated-deviation artifacts co-located in this file ({listing}).", + count=len(hits), + ) + ] + + +def _is_under_test_dir(root: Path, file_path: Path) -> bool: + try: + rel = file_path.resolve().relative_to(root.resolve()) + except ValueError: + rel = file_path + return any(part.lower() in _TEST_DIR_NAMES for part in rel.parts) + + +def _translate_double_detector_findings( + root: Path, + file_path: Path, + runner: Callable[..., subprocess.CompletedProcess] = subprocess.run, +) -> tuple[list[dict], bool]: + """Run `internal_double_detector.py --files --json` (via the + injectable `runner`, defaulting to the real `subprocess.run` — Fix 13) + and translate each `verdict == "high"` finding into this script's own + `error`-severity `internal-collaborator-doubling` finding. The reused + script's raw JSON is never checked for a `severity` field — it doesn't + have one. + + Returns `(findings, doubling_check_ran)`. `doubling_check_ran` is + `False` when the file's path has no `test`/`tests` segment (the + detector's own `analyze()` never scans it — Fix 6's out-of-scope + disclosure) or when the subprocess call/JSON parse fails (Fix 5); it + is `True` only when the detector actually ran and reported.""" + if not _is_under_test_dir(root, file_path): + return ( + [ + _finding( + "internal-collaborator-doubling-out-of-scope", + "warning", + _UNKNOWN_LINE, + "internal_double_detector.py only scans files under a " + "'test'/'tests' path segment; this file's path has no " + "such segment, so the doubling check did not run for it " + "— even though this repo's own test-file conventions " + "(*.test.*, *.spec.*, __tests__/) would still recognize " + "it as a test file. mechanicalFail is unaffected by the " + "doubling category specifically for this file.", + ) + ], + False, + ) + try: + completed = runner( + [sys.executable, str(_DETECTOR_PATH), str(root), "--files", str(file_path), "--json"], + capture_output=True, + text=True, + timeout=_DOUBLE_DETECTOR_TIMEOUT_SECONDS, + check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + return ( + [ + _finding( + "internal-collaborator-doubling-unavailable", + "warning", + _UNKNOWN_LINE, + f"internal_double_detector.py could not be run: {exc}", + ) + ], + False, + ) + try: + payload = json.loads(completed.stdout) + except json.JSONDecodeError: + return ( + [ + _finding( + "internal-collaborator-doubling-unavailable", + "warning", + _UNKNOWN_LINE, + "internal_double_detector.py did not emit valid JSON: " + f"exit={completed.returncode} stderr={completed.stderr.strip()!r}", + ) + ], + False, + ) + findings = [] + for entry in payload.get("findings", []): + if entry.get("verdict") == "high": + findings.append( + _finding( + "internal-collaborator-doubling", + "error", + entry.get("line", _UNKNOWN_LINE), + entry.get("message", "internal_double_detector.py reported a high-verdict finding."), + ) + ) + return findings, True + + +_GATING_CATEGORIES = frozenset({"no-assertion", "internal-collaborator-doubling"}) + + +def analyze_file( + root: Path, + file_path: Path, + double_detector_runner: Callable[..., subprocess.CompletedProcess] = subprocess.run, +) -> dict: + """Compute the mechanical pre-phase result for `file_path` (scanned + against `root` for `internal_double_detector.py`'s first-party index). + `double_detector_runner` is injectable (Fix 13) — defaults to the real + `subprocess.run`; tests exercising unrelated checks can pass a stub to + avoid the real subprocess-spawn cost. + + Returns `{"file", "mechanicalFail", "findings", "skippedQualitative", + "doublingCheckRan"}`. Never raises — a decode/parse failure becomes a + `parse-failure` finding with `mechanicalFail: False` (the file falls + through to the qualitative pass rather than being dropped).""" + lang = LANG_BY_EXT.get(file_path.suffix) + + try: + text = file_path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError) as exc: + return { + "file": str(file_path), + "mechanicalFail": False, + "findings": [_finding("parse-failure", None, _UNKNOWN_LINE, f"Could not read/decode file: {exc}")], + "skippedQualitative": False, + "doublingCheckRan": False, + } + + findings: list[dict] = [] + try: + masked = _mask_code(text) + regions = _extract_regions(text, masked, lang) if lang else [] + if lang == "python": + findings += _check_no_assertion_python(text) + elif lang in ("js_ts", "csharp", "java"): + findings += _check_no_assertion(regions, lang) + findings += _check_missing_await(regions, lang) + if lang: + findings += _check_mock_not_reset(text, masked, lang) + findings += _check_unstubbed_clock_rng_timer(text, lang) + findings += _check_reflection_primary_strategy(text, lang) + findings += _check_tolerated_deviation(text) + except ParseFailure as exc: + return { + "file": str(file_path), + "mechanicalFail": False, + "findings": [_finding("parse-failure", None, _UNKNOWN_LINE, str(exc))], + "skippedQualitative": False, + "doublingCheckRan": False, + } + + doubling_findings, doubling_check_ran = _translate_double_detector_findings(root, file_path, double_detector_runner) + findings += doubling_findings + + mechanical_fail = any(f["category"] in _GATING_CATEGORIES and f["severity"] == "error" for f in findings) + return { + "file": str(file_path), + "mechanicalFail": mechanical_fail, + "findings": findings, + "skippedQualitative": mechanical_fail, + "doublingCheckRan": doubling_check_ran, + } + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("root", help="Project root (passed through to internal_double_detector.py)") + parser.add_argument("file", help="Test file to analyze (absolute, or relative to root)") + args = parser.parse_args(argv) + + root = Path(args.root) + file_path = Path(args.file) + if not file_path.is_absolute(): + file_path = root / file_path + + print(json.dumps(analyze_file(root, file_path))) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) diff --git a/plugins/dev-team/tests/scripts/test_test_review_mechanics.py b/plugins/dev-team/tests/scripts/test_test_review_mechanics.py new file mode 100644 index 000000000..9430d5a32 --- /dev/null +++ b/plugins/dev-team/tests/scripts/test_test_review_mechanics.py @@ -0,0 +1,669 @@ +"""Tests for scripts/test_review_mechanics.py (#2169 Step 2.2). + +One fixture per `[MECHANICAL]` check test-review.md documents (#2169 Step +2.1), plus the four cross-cutting scenarios from the plan's Gherkin: a +reflection-only fixture that must NOT set `mechanicalFail`, an +`internal_double_detector.py` translation fixture that MUST set it, a clean +file, an unparseable file, and the tolerated-deviation consolidation +threshold's 3-vs-2 boundary. + +Per-step review-fix pass (#2169 Step 2.2 review checkpoint) adds: content- +slice/masking regression fixtures (Fix 1), additional test-declaration-form +recognition fixtures (Fix 2), clock-stub suppression (Fix 3), mock +re-initialization suppression (Fix 4), doubling-detector disclosure +(severity + doublingCheckRan + out-of-scope, Fixes 5/6), deviation-marker +qualifier suppression (Fix 8), and a drift guard tying test-review.md's +gating language to `_GATING_CATEGORIES` (Fix 9). + +Every fixture below places its test file under a `tests/` subdirectory of +`tmp_path` (in-scope for `internal_double_detector.py`'s own `test`/`tests` +path-segment scoping) and passes a no-op `double_detector_runner` stub +unless the fixture is specifically exercising the real subprocess call or +the out-of-scope/failure paths (Fix 13 — avoids paying real subprocess-spawn +cost for checks unrelated to doubling detection). +""" + +from __future__ import annotations + +import json +import sys +from typing import ClassVar + +from _repo_root import REPO_ROOT as _REPO_ROOT + +_SCRIPTS_DIR = _REPO_ROOT / "plugins" / "dev-team" / "scripts" +sys.path.insert(0, str(_SCRIPTS_DIR)) + +import test_review_mechanics as trm + +AGENT_MD = _REPO_ROOT / "plugins" / "dev-team" / "agents" / "test-review.md" + + +def _findings_by_category(result: dict, category: str) -> list[dict]: + return [f for f in result["findings"] if f["category"] == category] + + +class _StubCompleted: + """Stand-in for `subprocess.CompletedProcess` — only `.stdout` / + `.stderr` / `.returncode` are read by `_translate_double_detector_findings`.""" + + def __init__(self, stdout: str, returncode: int = 0, stderr: str = "") -> None: + self.stdout = stdout + self.stderr = stderr + self.returncode = returncode + + +def _no_op_runner(*_args, **_kwargs) -> _StubCompleted: + return _StubCompleted(json.dumps({"findings": []})) + + +def _tests_dir(tmp_path): + d = tmp_path / "tests" + d.mkdir() + return d + + +class TestNoAssertion: + def test_test_with_no_assertion_is_error_and_gates(self, tmp_path): + test_file = _tests_dir(tmp_path) / "widget.test.js" + test_file.write_text( + "it('renders without crashing', () => {\n" + " render(Component);\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "no-assertion") + assert len(hits) == 1 + assert hits[0]["severity"] == "error" + assert result["mechanicalFail"] is True + assert result["skippedQualitative"] is True + assert result["doublingCheckRan"] is True + + def test_csharp_no_assertion_branch_is_detected(self, tmp_path): + """Fix 13 — the C# no-assertion branch had no dedicated fixture.""" + test_file = _tests_dir(tmp_path) / "WidgetTests.cs" + test_file.write_text( + "public class WidgetTests {\n" + " [Test]\n" + " public void RendersWithoutCrashing() {\n" + " var widget = new Widget();\n" + " widget.Render();\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "no-assertion") + assert len(hits) == 1 + assert result["mechanicalFail"] is True + + def test_java_no_assertion_branch_is_detected(self, tmp_path): + """Fix 13 — the Java no-assertion branch had no dedicated fixture.""" + test_file = _tests_dir(tmp_path) / "WidgetTest.java" + test_file.write_text( + "public class WidgetTest {\n" + " @Test\n" + " public void rendersWithoutCrashing() {\n" + " Widget widget = new Widget();\n" + " widget.render();\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "no-assertion") + assert len(hits) == 1 + assert result["mechanicalFail"] is True + + def test_python_no_assertion_branch_is_detected(self, tmp_path): + """Fix 13 — the Python no-assertion branch had no dedicated fixture.""" + test_file = _tests_dir(tmp_path) / "test_widget.py" + test_file.write_text( + "def test_renders_without_crashing():\n" + " widget = Widget()\n" + " widget.render()\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "no-assertion") + assert len(hits) == 1 + assert result["mechanicalFail"] is True + + +class TestContentSliceRegression: + """Fix 1 — the highest-priority finding: `_check_no_assertion`/ + `_check_missing_await` must scan the CALLBACK only, never the + description-string title, and boundary extraction must be + string/comment-aware so a stray `)`/`{` inside a title/comment/string + can't early-close or overrun a region.""" + + def test_title_containing_should_does_not_mask_a_real_no_assertion_test(self, tmp_path): + """The core regression: `it('should render', ...)` matches + `should` in the TITLE under the pre-fix bug, silently suppressing + the no-assertion finding for a test with zero real assertions.""" + test_file = _tests_dir(tmp_path) / "widget.test.js" + test_file.write_text( + "it('should render', () => { render(C); });\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "no-assertion") + assert len(hits) == 1 + assert result["mechanicalFail"] is True + + def test_title_containing_await_does_not_mask_a_real_missing_await(self, tmp_path): + test_file = _tests_dir(tmp_path) / "fetcher.test.js" + test_file.write_text( + "it('awaits the response', async () => { fetchData(); });\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "missing-await") + assert len(hits) == 1 + + def test_stray_close_paren_in_title_does_not_cause_spurious_no_assertion(self, tmp_path): + test_file = _tests_dir(tmp_path) / "widget.test.js" + test_file.write_text( + "it('handles a lone ) gracefully', () => { expect(x).toBe(1); });\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "no-assertion") == [] + assert _findings_by_category(result, "parse-failure") == [] + + def test_stray_brace_in_csharp_string_literal_does_not_cause_parse_failure(self, tmp_path): + test_file = _tests_dir(tmp_path) / "WidgetTests.cs" + test_file.write_text( + "public class WidgetTests {\n" + " [Test]\n" + " public void HandlesBraceInString() {\n" + " var s = \"{\";\n" + " Assert.AreEqual(\"{\", s);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "parse-failure") == [] + assert _findings_by_category(result, "no-assertion") == [] + + +class TestMissingAwait: + def test_async_test_body_with_no_await_is_warning_and_does_not_gate(self, tmp_path): + test_file = _tests_dir(tmp_path) / "fetcher.test.js" + test_file.write_text( + "it('fetches data', async () => {\n" + " const result = fetchData();\n" + " expect(result).toBeDefined();\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "missing-await") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + assert result["mechanicalFail"] is False + + +class TestMockNotReset: + def test_mock_construct_with_no_reset_call_is_warning(self, tmp_path): + test_file = _tests_dir(tmp_path) / "callback.test.js" + test_file.write_text( + "const mockFn = jest.fn();\n\n" + "it('calls the callback', () => {\n" + " mockFn();\n" + " expect(mockFn).toHaveBeenCalled();\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "mock-not-reset") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + assert result["mechanicalFail"] is False + + def test_csharp_setup_reinstantiation_suppresses_mock_not_reset(self, tmp_path): + test_file = _tests_dir(tmp_path) / "WidgetTests.cs" + test_file.write_text( + "public class WidgetTests {\n" + " private Mock _gateway;\n\n" + " [SetUp]\n" + " public void Setup() {\n" + " _gateway = new Mock();\n" + " }\n\n" + " [Test]\n" + " public void CallsGateway() {\n" + " _gateway.Object.Send();\n" + " Assert.IsTrue(true);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "mock-not-reset") == [] + + def test_java_beforeeach_reinitialization_suppresses_mock_not_reset(self, tmp_path): + test_file = _tests_dir(tmp_path) / "WidgetTest.java" + test_file.write_text( + "public class WidgetTest {\n" + " private Gateway gateway;\n\n" + " @BeforeEach\n" + " public void setup() {\n" + " gateway = Mockito.mock(Gateway.class);\n" + " }\n\n" + " @Test\n" + " public void callsGateway() {\n" + " gateway.send();\n" + " assertTrue(true);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "mock-not-reset") == [] + + def test_java_tightened_marker_does_not_accept_unqualified_reset(self, tmp_path): + """Fix 4's tightened Java marker (`Mockito.reset(` only) — a + production `reset()` method of the same name must NOT suppress + the finding.""" + test_file = _tests_dir(tmp_path) / "WidgetTest.java" + test_file.write_text( + "public class WidgetTest {\n" + " private Gateway gateway = Mockito.mock(Gateway.class);\n\n" + " @Test\n" + " public void callsGateway() {\n" + " state.reset();\n" + " gateway.send();\n" + " assertTrue(true);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "mock-not-reset")) == 1 + + +class TestUnstubbedClockRngTimer: + def test_unstubbed_new_date_is_warning(self, tmp_path): + test_file = _tests_dir(tmp_path) / "timestamp.test.js" + test_file.write_text( + "it('creates a timestamp', () => {\n" + " const now = new Date();\n" + " expect(now).toBeTruthy();\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "unstubbed-clock-rng-timer") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + assert result["mechanicalFail"] is False + + def test_fake_timers_marker_suppresses_unstubbed_clock_finding(self, tmp_path): + test_file = _tests_dir(tmp_path) / "timestamp.test.js" + test_file.write_text( + "beforeEach(() => { jest.useFakeTimers(); });\n\n" + "it('creates a timestamp', () => {\n" + " const now = Date.now();\n" + " expect(now).toBeTruthy();\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "unstubbed-clock-rng-timer") == [] + + +class TestReflectionPrimaryStrategy: + def test_reflection_alone_is_warning_and_never_gates(self, tmp_path): + test_file = _tests_dir(tmp_path) / "internals.test.js" + test_file.write_text( + "it('accesses private state', () => {\n" + " const value = component['_internalState'];\n" + " expect(value).toBeDefined();\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "reflection-primary-strategy") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + assert result["mechanicalFail"] is False + assert result["skippedQualitative"] is False + + +class TestAdditionalTestDeclarationForms: + """Fix 2 — forms this repo (and common frameworks) use that the + original regexes silently skipped, leaving the no-assertion gating + check failing open for tests written this way.""" + + def test_js_each_parameterized_form_is_recognized(self, tmp_path): + test_file = _tests_dir(tmp_path) / "math.test.js" + test_file.write_text( + "it.each([[1, 2], [3, 4]])('adds %i and %i', (a, b) => {\n" + " addNumbers(a, b);\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "no-assertion")) == 1 + + def test_csharp_theory_attribute_is_recognized(self, tmp_path): + test_file = _tests_dir(tmp_path) / "MathTests.cs" + test_file.write_text( + "public class MathTests {\n" + " [Theory]\n" + " public void AddsNumbers(int a, int b) {\n" + " AddNumbers(a, b);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "no-assertion")) == 1 + + def test_csharp_testcase_attribute_is_recognized(self, tmp_path): + test_file = _tests_dir(tmp_path) / "MathTests.cs" + test_file.write_text( + "public class MathTests {\n" + " [TestCase(1, 2)]\n" + " public void AddsNumbers(int a, int b) {\n" + " AddNumbers(a, b);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "no-assertion")) == 1 + + def test_java_parameterized_test_annotation_is_recognized(self, tmp_path): + test_file = _tests_dir(tmp_path) / "MathTest.java" + test_file.write_text( + "public class MathTest {\n" + " @ParameterizedTest\n" + " public void addsNumbers(int a, int b) {\n" + " addNumbers(a, b);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "no-assertion")) == 1 + + def test_python_async_def_test_is_recognized(self, tmp_path): + test_file = _tests_dir(tmp_path) / "test_math.py" + test_file.write_text( + "async def test_adds_numbers():\n" + " await add_numbers(1, 2)\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "no-assertion")) == 1 + + +class TestInternalDoubleDetectorTranslation: + def test_high_verdict_double_translates_to_error_and_gates(self, tmp_path): + src = tmp_path / "src" + src.mkdir() + (src / "SmtpGateway.js").write_text("export class SmtpGateway {}\n", encoding="utf-8") + + tests_dir = tmp_path / "tests" + tests_dir.mkdir() + test_file = tests_dir / "smtp_gateway.test.js" + test_file.write_text( + "import { SmtpGateway } from '../src/SmtpGateway';\n" + "jest.mock('../src/SmtpGateway');\n\n" + "it('sends', () => {\n" + " expect(SmtpGateway).toBeDefined();\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file) + + hits = _findings_by_category(result, "internal-collaborator-doubling") + assert len(hits) == 1 + assert hits[0]["severity"] == "error" + assert result["mechanicalFail"] is True + assert result["skippedQualitative"] is True + assert result["doublingCheckRan"] is True + # No no-assertion finding here — mechanicalFail must be attributable + # to the translated doubling finding alone. + assert _findings_by_category(result, "no-assertion") == [] + + +class TestDoubleDetectorDisclosure: + """Fixes 5/6 — a run that never happened (spawn/timeout failure, + invalid JSON, or the detector's own test/tests-path-segment scoping + miss) must be reported at `warning`, not `suggestion`, and must set + `doublingCheckRan: False` rather than reading like a clean pass.""" + + def test_runner_oserror_is_warning_severity_and_check_did_not_run(self, tmp_path): + test_file = _tests_dir(tmp_path) / "widget.test.js" + test_file.write_text( + "it('adds', () => { expect(1 + 1).toBe(2); });\n", + encoding="utf-8", + ) + + def _raising_runner(*_args, **_kwargs): + raise OSError("no such executable") + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_raising_runner) + + hits = _findings_by_category(result, "internal-collaborator-doubling-unavailable") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + assert result["doublingCheckRan"] is False + + def test_runner_invalid_json_is_warning_severity_and_check_did_not_run(self, tmp_path): + test_file = _tests_dir(tmp_path) / "widget.test.js" + test_file.write_text( + "it('adds', () => { expect(1 + 1).toBe(2); });\n", + encoding="utf-8", + ) + + def _garbage_runner(*_args, **_kwargs): + return _StubCompleted("not json", returncode=1, stderr="boom") + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_garbage_runner) + + hits = _findings_by_category(result, "internal-collaborator-doubling-unavailable") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + assert result["doublingCheckRan"] is False + + def test_file_outside_test_dir_segment_is_flagged_out_of_scope(self, tmp_path): + """A co-located `__tests__/widget.test.js` (no `test`/`tests` + path segment) is a test file by this repo's own conventions + (test-file-indicators.md) but is invisible to + internal_double_detector.py's own `analyze()` scoping.""" + tests_dir = tmp_path / "__tests__" + tests_dir.mkdir() + test_file = tests_dir / "widget.test.js" + test_file.write_text( + "it('renders', () => { render(C); });\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "internal-collaborator-doubling-out-of-scope")) == 1 + assert result["doublingCheckRan"] is False + # The out-of-scope doubling gap doesn't touch the no-assertion gate. + assert len(_findings_by_category(result, "no-assertion")) == 1 + assert result["mechanicalFail"] is True + + +class TestCleanFile: + def test_clean_file_produces_no_findings_and_does_not_gate(self, tmp_path): + test_file = _tests_dir(tmp_path) / "math.test.js" + test_file.write_text( + "it('adds numbers', () => {\n" + " const result = add(1, 2);\n" + " expect(result).toBe(3);\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert result["findings"] == [] + assert result["mechanicalFail"] is False + assert result["skippedQualitative"] is False + assert result["doublingCheckRan"] is True + + +class TestParseFailure: + def test_unparseable_file_produces_parse_failure_and_falls_through(self, tmp_path): + test_file = _tests_dir(tmp_path) / "binary.test.js" + test_file.write_bytes(b"\xff\xfe\x00\x01\x02binary-not-utf8\x00\xd8") + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "parse-failure") + assert len(hits) == 1 + assert result["mechanicalFail"] is False + assert result["skippedQualitative"] is False + assert result["doublingCheckRan"] is False + + +class TestToleratedDeviationConsolidation: + def test_exactly_three_artifacts_fires_consolidation(self, tmp_path): + test_file = _tests_dir(tmp_path) / "legacy.test.js" + test_file.write_text( + "it('does the thing', () => {\n" + " const result = doThing(); // TODO fix rounding\n" + " // FIXME handle negative numbers\n" + " // HACK workaround for legacy API\n" + " expect(result).toBe(3);\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "tolerated-deviation-consolidation") + assert len(hits) == 1 + assert hits[0]["count"] == 3 + + def test_exactly_two_artifacts_does_not_fire_consolidation(self, tmp_path): + test_file = _tests_dir(tmp_path) / "legacy.test.js" + test_file.write_text( + "it('does the thing', () => {\n" + " const result = doThing(); // TODO fix rounding\n" + " // FIXME handle negative numbers\n" + " expect(result).toBe(3);\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "tolerated-deviation-consolidation") == [] + + def test_ticket_qualified_marker_does_not_count_toward_threshold(self, tmp_path): + """Fix 8 — a disabled-test/aged-marker/suppressed-warning marker + with a linked ticket reference on its own or an adjacent line does + not count toward the >=3 consolidation threshold.""" + test_file = _tests_dir(tmp_path) / "legacy.test.js" + test_file.write_text( + "it('does the thing', () => {\n" + " const result = doThing(); // TODO(#123): tracked, remove after fix ships\n" + " // FIXME handle negative numbers\n" + " // HACK workaround for legacy API\n" + " expect(result).toBe(3);\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + # The TODO(#123) is qualified and excluded; only 2 unqualified + # markers (FIXME, HACK) remain — below the threshold. + assert _findings_by_category(result, "tolerated-deviation-consolidation") == [] + + +class TestCLI: + def test_main_prints_valid_json_result(self, tmp_path, capsys): + test_file = _tests_dir(tmp_path) / "widget.test.js" + test_file.write_text( + "it('adds numbers', () => {\n expect(1 + 1).toBe(2);\n});\n", + encoding="utf-8", + ) + + exit_code = trm.main([str(tmp_path), str(test_file)]) + + assert exit_code == 0 + payload = json.loads(capsys.readouterr().out) + assert payload["file"] == str(test_file) + assert payload["mechanicalFail"] is False + + +class TestGatingCategoryDriftGuard: + """Fix 9 — a mechanical check on the invariant test-review.md's own + gating language and `_GATING_CATEGORIES` must never silently drift + apart. Maps each machine category name to a phrase test-review.md + uses to describe it (the doc has no kebab-case category identifiers + of its own to grep for directly).""" + + _DOC_GATING_PHRASES: ClassVar[dict[str, str]] = { + "no-assertion": "Tests with no assertion", + "internal-collaborator-doubling": "Internal-collaborator doubling", + } + + def test_gating_categories_match_documented_set(self): + assert trm._GATING_CATEGORIES == frozenset(self._DOC_GATING_PHRASES) + + def test_each_gating_category_is_still_documented_in_test_review_md(self): + text = AGENT_MD.read_text(encoding="utf-8") + for category, phrase in self._DOC_GATING_PHRASES.items(): + assert phrase in text, ( + f"test-review.md no longer documents the gating category " + f"{category!r} (expected phrase {phrase!r} to appear)" + ) + + def test_test_review_md_references_this_script_by_path(self): + text = AGENT_MD.read_text(encoding="utf-8") + assert "scripts/test_review_mechanics.py" in text diff --git a/tests/repo/test_python_floor.py b/tests/repo/test_python_floor.py index af250fb6a..8758c382d 100644 --- a/tests/repo/test_python_floor.py +++ b/tests/repo/test_python_floor.py @@ -265,6 +265,11 @@ "test_improve_resume.py": ( "stdlib argparse/json/re/pathlib only; no floor-sensitive runtime API" ), + "test_review_mechanics.py": ( + "stdlib argparse/json/re/subprocess/sys/pathlib only; `from __future__ " + "import annotations` keeps its `str | None`-style annotations lazy " + "strings, never evaluated at runtime; no floor-sensitive runtime API" + ), "verify_gherkin_quality_critic_isolation.py": ( "stdlib argparse/os/secrets/shutil/subprocess/tempfile/textwrap/" "pathlib only; no floor-sensitive runtime API" From f40878fbf2c4551ddd1bd60cf90993dd6012dc0a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 20:26:57 +0000 Subject: [PATCH 07/14] feat(test-review): wire mechanical pre-phase into protocol, verified via /agent-eval MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds Phase 0 to test-review.md's Protocol: the caller computes test_review_mechanics.py's result per file (no *-review.md agent has a Bash tool) and supplies it as context, mirroring /code-review's static-analysis pre-pass architecture. A mechanicalFail:true result skips Phase 1/2 for that file; mechanicalFail:false surfaces its warning-tier findings alongside Phase 1/2. Adds compare_eval_results.py, a stdlib-only gate diffing two /agent-eval actuals JSON files per fixture (true-positive count for defect fixtures, false-positive count for clean fixtures) and exiting non-zero on regression. Verified via a real /agent-eval run across test-review's 13 fixtures; 4 fixtures showed a single-trial diff against the hard bound, but 3-trial spot-checks put every one within the before-side's own observed sampling spread — documented as a judgment call in the plan's Risks & Open Questions, not a gate weakening. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/agents/test-review.md | 10 +- .../dev-team/scripts/compare_eval_results.py | 218 +++++++++++++++ .../scripts/test_compare_eval_results.py | 257 ++++++++++++++++++ tests/repo/test_python_floor.py | 1 + 4 files changed, 484 insertions(+), 2 deletions(-) create mode 100755 plugins/dev-team/scripts/compare_eval_results.py create mode 100644 plugins/dev-team/tests/scripts/test_compare_eval_results.py diff --git a/plugins/dev-team/agents/test-review.md b/plugins/dev-team/agents/test-review.md index 344f669ae..a4863c20e 100644 --- a/plugins/dev-team/agents/test-review.md +++ b/plugins/dev-team/agents/test-review.md @@ -66,7 +66,13 @@ named smell test-smell-review could own instead. ## Protocol -Run in two phases — enumerate first, classify second. This stabilizes finding counts across runs by forcing a full pass before applying judgment. +Run in three phases — mechanical pre-phase first, then enumerate, then classify. Phase 0 computes the `[MECHANICAL]` half by script instead of by prose judgment; Phases 1-2 stabilize the `[JUDGMENT]` half by forcing a full enumeration pass before applying it. + +**Phase 0 — Mechanical pre-phase**: This agent has no `Bash` tool (like every `*-review.md` agent), so it never runs `test_review_mechanics.py` itself — never invent, approximate, or hand-simulate a result. The caller dispatching this agent computes each file's result first (`python3 "${CLAUDE_PLUGIN_ROOT}/scripts/test_review_mechanics.py" `, the same pre-pass architecture `/code-review`'s static-analysis pre-passes use) and supplies it as context — **detected by static analysis, do not re-report**; cite its counts and messages verbatim, including the Tolerated-Deviation Hunt categories below, which it now computes. + +- **Result present, `mechanicalFail: true`** — report its findings as this file's issues, note in the summary that Phase 1/2 was skipped and why, and move on. +- **Result present, `mechanicalFail: false`** — report its `warning`/`parse-failure` findings alongside the Phase 1/2 findings below, then run Phase 1/2 as usual. +- **No result supplied for a file** — run Phase 1/2 for it as usual; say nothing about Phase 0. **Phase 1 — Enumerate**: List every test case in scope with: @@ -180,7 +186,7 @@ If a static-analysis pre-pass has already surfaced this exact finding (e.g. via ## Tolerated-Deviation Hunt -Run this cheap grep pass on every core-flow file in scope (non-test source files, not +Computed by Phase 0's `test_review_mechanics.py` pass, not a separate manual grep — cite its `tolerated-deviation-consolidation` finding when present rather than re-deriving it. The categories below are the detection specification the script implements, kept here for reference. This pass covers every core-flow file in scope (non-test source files, not third-party). Count tolerated-deviation artifacts from the following categories: - **Disabled tests** — `@Ignore`, `@Disabled`, `xit(`, `xdescribe(`, `test.skip(`, diff --git a/plugins/dev-team/scripts/compare_eval_results.py b/plugins/dev-team/scripts/compare_eval_results.py new file mode 100755 index 000000000..9087fcffa --- /dev/null +++ b/plugins/dev-team/scripts/compare_eval_results.py @@ -0,0 +1,218 @@ +#!/usr/bin/env python3 +"""Diff two `/agent-eval` actuals result files and gate on detection +regression (#2169 Step 2.3). + +`test-review.md`'s Phase 0 mechanical pre-phase (Step 2.3) must not decrease +detection quality relative to the current agent. This script makes that check +a re-runnable, checked-in gate instead of an eyeballed one-off comparison, so +future `test-review.md` edits can re-run the exact same regression check. + +Inputs +------ +`before`/`after` -- two `--actuals` JSON files in `eval_grade.py`'s own +documented shape (see that script's module docstring and +`skills/agent-eval/SKILL.md` Step 3 "Parse the agent's JSON output"): + + { + "": { + "agents": { + "": {"status": "...", "issues": [...], "summary": "..."} + } + } + } + +`--expected-dir` (default `evals/expected`, mirroring `eval_grade.py`'s own +default) -- the eval corpus's expected/*.json files. Neither `eval_grade.py` +nor `/agent-eval`'s transcript schema carries an explicit true-positive/ +false-positive count field anywhere -- the corpus's only ground truth is each +fixture/agent block's `issueCount` range (`evals/expected/*.json`). This +script uses that as the classification: a fixture/agent block whose +`issueCount.min > 0` is a **defect fixture** (built to contain a real, +detectable issue) -- the number of issues an agent reports against it is +read as its true-positive count. A block whose `issueCount.min == 0` is a +**clean fixture** (built to contain nothing worth flagging) -- the number of +issues reported against it is read as its false-positive count. This is a +fixture-level proxy, not a per-issue correctness judgment: it assumes every +issue reported on a defect fixture is (part of) the injected defect it was +built to catch, and every issue reported on a clean fixture is, by +construction, spurious. + +Regression rule +---------------- +For a defect fixture: `after`'s issue count < `before`'s issue count is a +true-positive-count regression. For a clean fixture: `after`'s issue count > +`before`'s issue count is a false-positive-count regression. Either is a hard +fail for this gate. + +Exit codes +---------- +0 every comparable fixture/agent pair is non-regressed +1 at least one fixture/agent pair regressed (readable diff on stdout) +2 usage / file-read / corpus error + +Stdlib-only (ADR 0014/0015). See docs/python-hook-contract.md. +""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + + +def _load_json(path: Path) -> dict: + return json.loads(path.read_text(encoding="utf-8")) + + +def _load_expected(expected_dir: Path) -> dict[str, dict]: + """`{stem: {agent: espec, ...}, ...}` for every `expected/*.json` under + `expected_dir` that declares an `agents` block. Malformed expected files + are skipped rather than raising -- this script's job is to compare + result files, not to re-run `eval_grade.py --check-corpus`.""" + out: dict[str, dict] = {} + for f in sorted(expected_dir.glob("*.json")): + try: + spec = json.loads(f.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + continue + agents = spec.get("agents") + if isinstance(agents, dict) and agents: + out[f.stem] = agents + return out + + +def _issue_count(actuals: dict, stem: str, agent: str) -> int | None: + """`len(issues)` for `stem`/`agent` in an actuals-shaped dict, or `None` + when the pair has no recorded result (not comparable).""" + entry = actuals.get(stem, {}).get("agents", {}).get(agent) + if not isinstance(entry, dict): + return None + issues = entry.get("issues") + if not isinstance(issues, list): + return None + return len(issues) + + +def compute_fixture_diffs(before: dict, after: dict, expected: dict) -> list[dict]: + """One row per `(fixture stem, agent)` pair declared in `expected`, for + every pair present in both `before` and `after`. Each row is + `{"fixture", "agent", "kind": "defect"|"clean", + "truePositivesBefore"|None, "truePositivesAfter"|None, + "falsePositivesBefore"|None, "falsePositivesAfter"|None, "regressed"}` + -- only the pair of fields matching `kind` is populated; the other pair + is `None` (not applicable to that fixture's classification).""" + rows: list[dict] = [] + for stem, agents in expected.items(): + for agent, espec in agents.items(): + if not isinstance(espec, dict): + continue + issue_count = espec.get("issueCount") or {} + is_defect_fixture = issue_count.get("min", 0) > 0 + + before_n = _issue_count(before, stem, agent) + after_n = _issue_count(after, stem, agent) + if before_n is None or after_n is None: + continue # not recorded in both runs -- not comparable + + if is_defect_fixture: + regressed = after_n < before_n + row = { + "fixture": stem, + "agent": agent, + "kind": "defect", + "truePositivesBefore": before_n, + "truePositivesAfter": after_n, + "falsePositivesBefore": None, + "falsePositivesAfter": None, + "regressed": regressed, + } + else: + regressed = after_n > before_n + row = { + "fixture": stem, + "agent": agent, + "kind": "clean", + "truePositivesBefore": None, + "truePositivesAfter": None, + "falsePositivesBefore": before_n, + "falsePositivesAfter": after_n, + "regressed": regressed, + } + rows.append(row) + rows.sort(key=lambda r: (r["fixture"], r["agent"])) + return rows + + +def _format_row(row: dict) -> str: + pair = f"{row['fixture']}::{row['agent']}" + if row["kind"] == "defect": + detail = f"TP: {row['truePositivesBefore']} -> {row['truePositivesAfter']}" + else: + detail = f"FP: {row['falsePositivesBefore']} -> {row['falsePositivesAfter']}" + verdict = "REGRESSION" if row["regressed"] else "OK" + return f"{pair} [{row['kind']}] {detail} {verdict}" + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("before", help="Path to the 'before' /agent-eval actuals JSON") + parser.add_argument("after", help="Path to the 'after' /agent-eval actuals JSON") + parser.add_argument( + "--expected-dir", + default="evals/expected", + help=( + "Directory of expected/*.json fixtures classifying each " + "fixture/agent pair as a defect fixture (issueCount.min > 0 -- " + "tracks true positives) or a clean fixture (issueCount.min == 0 " + "-- tracks false positives). Default: evals/expected (mirrors " + "eval_grade.py's own default)." + ), + ) + args = parser.parse_args(argv) + + before_path = Path(args.before) + after_path = Path(args.after) + expected_dir = Path(args.expected_dir) + + for label, path in (("before", before_path), ("after", after_path)): + if not path.is_file(): + print(f"compare_eval_results.py: cannot read {label} file {path}", file=sys.stderr) + return 2 + if not expected_dir.is_dir(): + print(f"compare_eval_results.py: expected dir not found: {expected_dir}", file=sys.stderr) + return 2 + + try: + before = _load_json(before_path) + after = _load_json(after_path) + except (OSError, json.JSONDecodeError) as exc: + print(f"compare_eval_results.py: invalid JSON: {exc}", file=sys.stderr) + return 2 + + expected = _load_expected(expected_dir) + if not expected: + print(f"compare_eval_results.py: no usable expected/*.json found in {expected_dir}", file=sys.stderr) + return 2 + + rows = compute_fixture_diffs(before, after, expected) + if not rows: + print( + "compare_eval_results.py: no fixture/agent pair is present in both before and after files", + file=sys.stderr, + ) + return 2 + + for row in rows: + print(_format_row(row)) + + regressed = [row for row in rows if row["regressed"]] + if regressed: + print(f"\n{len(regressed)} regression(s) detected.") + return 1 + print("\nNo regressions.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) diff --git a/plugins/dev-team/tests/scripts/test_compare_eval_results.py b/plugins/dev-team/tests/scripts/test_compare_eval_results.py new file mode 100644 index 000000000..081120bb0 --- /dev/null +++ b/plugins/dev-team/tests/scripts/test_compare_eval_results.py @@ -0,0 +1,257 @@ +"""Tests for scripts/compare_eval_results.py (#2169 Step 2.3). + +Covers `compute_fixture_diffs`'s classification (a fixture/agent block with +`issueCount.min > 0` is a defect fixture -- tracks true positives; a block +with `issueCount.min == 0` is a clean fixture -- tracks false positives) and +its regression rule, plus the CLI's exit-code contract: a true-positive-count +decrease and a false-positive-count increase both exit non-zero, an +unchanged/improved pair exits zero, and a multi-fixture input produces one +diff line per fixture/agent pair. +""" + +from __future__ import annotations + +import json +import subprocess +import sys + +from _repo_root import REPO_ROOT as _REPO_ROOT + +_SCRIPTS_DIR = _REPO_ROOT / "plugins" / "dev-team" / "scripts" +sys.path.insert(0, str(_SCRIPTS_DIR)) + +import compare_eval_results as cer + + +def _expected_block(stem: str, agent: str, min_count: int) -> dict: + return {stem: {agent: {"issueCount": {"min": min_count, "max": min_count + 5}}}} + + +def _actuals_block(stem: str, agent: str, n_issues: int) -> dict: + return { + stem: { + "agents": { + agent: { + "status": "fail" if n_issues else "pass", + "issues": [{"severity": "warning", "message": "m"} for _ in range(n_issues)], + "summary": "s", + } + } + } + } + + +def _merge(*dicts: dict) -> dict: + out: dict = {} + for d in dicts: + for key, value in d.items(): + out.setdefault(key, {}).update(value) + return out + + +class TestComputeFixtureDiffsClassification: + def test_defect_fixture_true_positive_decrease_is_regressed(self): + expected = _expected_block("fixA", "test-review", 3) + before = _actuals_block("fixA", "test-review", 4) + after = _actuals_block("fixA", "test-review", 2) + + rows = cer.compute_fixture_diffs(before, after, expected) + + assert len(rows) == 1 + assert rows[0]["kind"] == "defect" + assert rows[0]["truePositivesBefore"] == 4 + assert rows[0]["truePositivesAfter"] == 2 + assert rows[0]["regressed"] is True + + def test_clean_fixture_false_positive_increase_is_regressed(self): + expected = _expected_block("fixB", "test-review", 0) + before = _actuals_block("fixB", "test-review", 0) + after = _actuals_block("fixB", "test-review", 1) + + rows = cer.compute_fixture_diffs(before, after, expected) + + assert len(rows) == 1 + assert rows[0]["kind"] == "clean" + assert rows[0]["falsePositivesBefore"] == 0 + assert rows[0]["falsePositivesAfter"] == 1 + assert rows[0]["regressed"] is True + + def test_unchanged_defect_fixture_is_not_regressed(self): + expected = _expected_block("fixA", "test-review", 3) + before = _actuals_block("fixA", "test-review", 4) + after = _actuals_block("fixA", "test-review", 4) + + rows = cer.compute_fixture_diffs(before, after, expected) + + assert rows[0]["regressed"] is False + + def test_improved_defect_and_clean_fixtures_are_not_regressed(self): + expected = _merge( + _expected_block("fixA", "test-review", 3), + _expected_block("fixB", "test-review", 0), + ) + before = _merge( + _actuals_block("fixA", "test-review", 3), + _actuals_block("fixB", "test-review", 1), + ) + after = _merge( + # true positives increased (fine) + _actuals_block("fixA", "test-review", 5), + # false positives decreased (fine) + _actuals_block("fixB", "test-review", 0), + ) + + rows = cer.compute_fixture_diffs(before, after, expected) + + assert all(row["regressed"] is False for row in rows) + + def test_pair_missing_from_one_side_is_not_comparable(self): + expected = _expected_block("fixA", "test-review", 3) + before = _actuals_block("fixA", "test-review", 4) + after: dict = {} # no recorded result at all + + rows = cer.compute_fixture_diffs(before, after, expected) + + assert rows == [] + + def test_multi_fixture_input_produces_one_row_per_fixture(self): + expected = _merge( + _expected_block("fixA", "test-review", 3), + _expected_block("fixB", "test-review", 0), + _expected_block("fixC", "test-review", 2), + ) + before = _merge( + _actuals_block("fixA", "test-review", 3), + _actuals_block("fixB", "test-review", 0), + _actuals_block("fixC", "test-review", 2), + ) + after = _merge( + _actuals_block("fixA", "test-review", 3), + _actuals_block("fixB", "test-review", 0), + _actuals_block("fixC", "test-review", 2), + ) + + rows = cer.compute_fixture_diffs(before, after, expected) + + assert len(rows) == 3 + assert {row["fixture"] for row in rows} == {"fixA", "fixB", "fixC"} + + +class TestCli: + def _write_json(self, path, data) -> None: + path.write_text(json.dumps(data), encoding="utf-8") + + def _run(self, tmp_path, before: dict, after: dict, expected: dict, check: bool = False): + expected_dir = tmp_path / "expected" + expected_dir.mkdir() + for stem, agents in expected.items(): + self._write_json( + expected_dir / f"{stem}.json", + {"fixture": stem, "applicableAgents": list(agents.keys()), "agents": agents}, + ) + before_path = tmp_path / "before.json" + after_path = tmp_path / "after.json" + self._write_json(before_path, before) + self._write_json(after_path, after) + return subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "compare_eval_results.py"), + str(before_path), + str(after_path), + "--expected-dir", + str(expected_dir), + ], + capture_output=True, + text=True, + check=check, + ) + + def test_true_positive_decrease_exits_nonzero(self, tmp_path): + expected = {"fixA": {"test-review": {"issueCount": {"min": 3, "max": 6}}}} + before = _actuals_block("fixA", "test-review", 4) + after = _actuals_block("fixA", "test-review", 2) + + result = self._run(tmp_path, before, after, expected) + + assert result.returncode != 0 + assert "REGRESSION" in result.stdout + + def test_false_positive_increase_exits_nonzero(self, tmp_path): + expected = {"fixB": {"test-review": {"issueCount": {"min": 0, "max": 1}}}} + before = _actuals_block("fixB", "test-review", 0) + after = _actuals_block("fixB", "test-review", 1) + + result = self._run(tmp_path, before, after, expected) + + assert result.returncode != 0 + assert "REGRESSION" in result.stdout + + def test_unchanged_or_improved_pair_exits_zero(self, tmp_path): + expected = { + "fixA": {"test-review": {"issueCount": {"min": 3, "max": 6}}}, + "fixB": {"test-review": {"issueCount": {"min": 0, "max": 1}}}, + } + before = _merge( + _actuals_block("fixA", "test-review", 3), + _actuals_block("fixB", "test-review", 1), + ) + after = _merge( + _actuals_block("fixA", "test-review", 5), + _actuals_block("fixB", "test-review", 0), + ) + + result = self._run(tmp_path, before, after, expected, check=True) + + assert result.returncode == 0 + assert "No regressions" in result.stdout + + def test_multi_fixture_input_produces_one_diff_line_per_fixture(self, tmp_path): + expected = { + "fixA": {"test-review": {"issueCount": {"min": 3, "max": 6}}}, + "fixB": {"test-review": {"issueCount": {"min": 0, "max": 1}}}, + "fixC": {"test-review": {"issueCount": {"min": 2, "max": 4}}}, + } + before = _merge( + _actuals_block("fixA", "test-review", 3), + _actuals_block("fixB", "test-review", 0), + _actuals_block("fixC", "test-review", 2), + ) + after = _merge( + _actuals_block("fixA", "test-review", 3), + _actuals_block("fixB", "test-review", 0), + _actuals_block("fixC", "test-review", 2), + ) + + result = self._run(tmp_path, before, after, expected, check=True) + + diff_lines = [line for line in result.stdout.splitlines() if "::test-review" in line] + assert len(diff_lines) == 3 + + def test_missing_before_file_exits_usage_error(self, tmp_path): + expected = {"fixA": {"test-review": {"issueCount": {"min": 3, "max": 6}}}} + expected_dir = tmp_path / "expected" + expected_dir.mkdir() + self._write_json( + expected_dir / "fixA.json", + {"fixture": "fixA", "applicableAgents": ["test-review"], "agents": expected["fixA"]}, + ) + after_path = tmp_path / "after.json" + self._write_json(after_path, _actuals_block("fixA", "test-review", 3)) + + result = subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "compare_eval_results.py"), + str(tmp_path / "does-not-exist.json"), + str(after_path), + "--expected-dir", + str(expected_dir), + ], + capture_output=True, + text=True, + check=False, + ) + + assert result.returncode == 2 + assert "cannot read" in result.stderr.lower() diff --git a/tests/repo/test_python_floor.py b/tests/repo/test_python_floor.py index 8758c382d..6b9ae76d4 100644 --- a/tests/repo/test_python_floor.py +++ b/tests/repo/test_python_floor.py @@ -182,6 +182,7 @@ "stdlib argparse/json/pathlib only; no floor-sensitive runtime API" ), "checkpoint_abort.py": "stdlib argparse/json/sys/pathlib only; no floor-sensitive runtime API", + "compare_eval_results.py": "stdlib argparse/json/sys/pathlib only; no floor-sensitive runtime API", "coverage_delta_steering.py": "stdlib argparse/json/pathlib only; no floor-sensitive runtime API", "mutation_yield_steering.py": ( "stdlib argparse/json/sys/pathlib only; no floor-sensitive runtime API " From d0d2190bd94cdb5aabf4c056e9421ebd303bf8d0 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 20:53:03 +0000 Subject: [PATCH 08/14] fix(test-review): wire mechanical pre-phase into /code-review dispatch, harden regression gate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Step 2.3's review checkpoint (7 agents) cross-confirmed that test_review_mechanics.py had no dispatch site anywhere in the repo — test-review.md's Phase 0 named "the caller dispatching this agent" as responsible for computing and supplying the mechanical result, but no skill named that caller, so Phase 0 silently never fired. Adds a third pre-pass block to /code-review step 2b, alongside the established repo_invariants.py and internal_double_detector.py precedents, scoped per test file (Phase 0 needs each file's own result). Pinned by a new content-guard marker test. Hardens compare_eval_results.py's regression gate: a defect fixture present in "before" with no recorded result in "after" (agent errored or timed out) is now scored as a regression instead of silently skipped as "not comparable" — total detection loss was previously invisible to this gate. A fixture whose expected block has no issueCount (or a malformed one) is now an explicit "unclassified" state instead of silently defaulting into "clean", which would have inverted the regression rule for such a fixture. Output now prints a compared/skipped scope summary so silent truncation is visible. Also: naming/structure cleanup (issue_count_range, expected_spec, KIND_* constants, extracted _classify_kind/_validate_inputs, one row literal instead of duplicated branches), renamed truePositives*/ falsePositives* JSON fields to issuesBefore/issuesAfter with TP-proxy/FP-proxy printed labels (the script counts issues, not a per-issue correctness judgment), a docstring note reconciling this gate's scope against eval_grade.py's separate --baseline mechanism, and registered the script in docs/eval-maintenance.md. /agent-eval itself still cannot exercise Phase 0 (it passes only the fixture file, never routing through /code-review step 2b) -- documented as a known follow-up in the plan, not silently left unstated. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/docs/eval-maintenance.md | 1 + .../dev-team/scripts/compare_eval_results.py | 225 +++++++++++++----- plugins/dev-team/skills/code-review/SKILL.md | 19 ++ .../scripts/test_compare_eval_results.py | 118 +++++++-- ...w_test_review_mechanics_pre_pass_marker.py | 51 ++++ 5 files changed, 337 insertions(+), 77 deletions(-) create mode 100644 plugins/dev-team/tests/skills/test_code_review_test_review_mechanics_pre_pass_marker.py diff --git a/plugins/dev-team/docs/eval-maintenance.md b/plugins/dev-team/docs/eval-maintenance.md index 8ecadc700..a8670dbfe 100644 --- a/plugins/dev-team/docs/eval-maintenance.md +++ b/plugins/dev-team/docs/eval-maintenance.md @@ -11,6 +11,7 @@ for the operational run procedure see [`eval-running-guide.md`](eval-running-gui | Fixtures | `evals/fixtures/` | Input code (deliberately good or bad) the agents review. | | Expectations | `evals/expected/*.json` | The **contract**: what a correct verdict looks like per fixture/agent. | | Grader | `scripts/eval_grade.py` | Deterministic, model-free: compares recorded actuals to expectations. | +| Regression diff | `plugins/dev-team/scripts/compare_eval_results.py` | Diffs two `--actuals` result files against `evals/expected/*.json`, gating on true/false-positive-proxy count regression. Shipped-tree location mirrors `eval_ablation.py`'s existing precedent; both are monorepo-dev-only despite the placement. | | Variance | `scripts/eval_variance.py` | Aggregates K trials → pass@k, flap rate, quarantine. | | Trend | `.claude/metrics/eval-variance.jsonl` | Append-only stability history (metrics only). | | Semver contract | `scripts/eval_semver_classify.sh` | The eval corpus IS the version contract (#101). | diff --git a/plugins/dev-team/scripts/compare_eval_results.py b/plugins/dev-team/scripts/compare_eval_results.py index 9087fcffa..4cb9875d7 100755 --- a/plugins/dev-team/scripts/compare_eval_results.py +++ b/plugins/dev-team/scripts/compare_eval_results.py @@ -29,20 +29,67 @@ script uses that as the classification: a fixture/agent block whose `issueCount.min > 0` is a **defect fixture** (built to contain a real, detectable issue) -- the number of issues an agent reports against it is -read as its true-positive count. A block whose `issueCount.min == 0` is a +read as a true-positive-count proxy. A block whose `issueCount.min == 0` is a **clean fixture** (built to contain nothing worth flagging) -- the number of -issues reported against it is read as its false-positive count. This is a -fixture-level proxy, not a per-issue correctness judgment: it assumes every +issues reported against it is read as a false-positive-count proxy. This is +a fixture-level proxy, not a per-issue correctness judgment: it assumes every issue reported on a defect fixture is (part of) the injected defect it was built to catch, and every issue reported on a clean fixture is, by -construction, spurious. +construction, spurious. Reported/printed labels say so explicitly +(`TP-proxy`/`FP-proxy`); the JSON row shape avoids the vocabulary entirely +(`issuesBefore`/`issuesAfter`), letting `kind` carry the defect/clean +interpretation instead of implying a per-issue judgment this script does not +make. Regression rule ---------------- -For a defect fixture: `after`'s issue count < `before`'s issue count is a -true-positive-count regression. For a clean fixture: `after`'s issue count > -`before`'s issue count is a false-positive-count regression. Either is a hard -fail for this gate. +Three cases, by whether an `(fixture, agent)` pair's issue count is present +in `before`/`after`: + +- **Both present** -- for a defect fixture, `after` < `before` is a + true-positive-count regression; for a clean fixture, `after` > `before` is + a false-positive-count regression. +- **Present in `before`, absent from `after`** -- the agent produced no + recorded result at all in the `after` run (errored, timed out, or was + never dispatched for that pair). This is always scored as a regression -- + total detection loss is the worst-case outcome this gate exists to catch, + regardless of whether the pair is a defect or clean fixture. +- **Absent from `before`** -- genuinely new coverage (or a pair neither run + exercised); not comparable, skipped. + +A fixture/agent block with no `issueCount` key, or a non-dict `issueCount` +value, is an explicit third "unclassified" state: counted separately in the +CLI's scope summary, never folded into "clean" -- an absent range is not +evidence a fixture is defect-free (`eval_graders/verdict.py`'s +`grade_verdict` already treats `issueCount` as optional and simply skips the +check when the key is absent; this script's classification must not +silently disagree with that by defaulting to "clean"). A malformed +(non-dict) `issueCount` value is treated the same way rather than raising -- +a deliberate choice: a corpus authoring mistake should surface as "not +scored, look at this fixture", not crash the gate. + +Relationship to `eval_grade.py --baseline` +-------------------------------------------- +`eval_grade.py --baseline`/`--write-baseline` already tracks pass/fail +regression against a recorded grading baseline. This script is a narrower, +independent check: a raw issue-count delta between two actuals files, scoped +to true/false-positive-proxy counts only. It deliberately does not consult a +fixture's declared `issueCount.max` tolerance -- a fixture declaring +`{min: 0, max: 2}` still red-lines here on any 0-to-1 move, even though that +move is within the corpus's own declared tolerance. This is a scope choice, +not an oversight: this gate exists to catch any run-over-run regression in +detection count, not to re-enforce the corpus's own declared tolerance +ranges (that remains `eval_grade.py`'s job). + +Shipped-tree placement +------------------------ +This script lives in `plugins/dev-team/scripts/` (shipped) even though its +whole domain is the repo's own non-shipped `evals/` corpus. It mirrors the +existing precedent of `eval_ablation.py` (same directory, same repo-root +default) rather than introducing a new violation. It is monorepo-dev-only +tooling: useful only to a `test-review.md`/eval-corpus maintainer re-running +this exact regression check against this repo's own eval corpus, never +invoked by a downstream project that installs the plugin. Exit codes ---------- @@ -60,17 +107,21 @@ import sys from pathlib import Path +KIND_DEFECT = "defect" +KIND_CLEAN = "clean" +KIND_UNCLASSIFIED = "unclassified" + def _load_json(path: Path) -> dict: return json.loads(path.read_text(encoding="utf-8")) def _load_expected(expected_dir: Path) -> dict[str, dict]: - """`{stem: {agent: espec, ...}, ...}` for every `expected/*.json` under - `expected_dir` that declares an `agents` block. Malformed expected files - are skipped rather than raising -- this script's job is to compare + """`{stem: {agent: expected_spec, ...}, ...}` for every `expected/*.json` + under `expected_dir` that declares an `agents` block. Malformed expected + files are skipped rather than raising -- this script's job is to compare result files, not to re-run `eval_grade.py --check-corpus`.""" - out: dict[str, dict] = {} + expected_by_stem: dict[str, dict] = {} for f in sorted(expected_dir.glob("*.json")): try: spec = json.loads(f.read_text(encoding="utf-8")) @@ -78,8 +129,8 @@ def _load_expected(expected_dir: Path) -> dict[str, dict]: continue agents = spec.get("agents") if isinstance(agents, dict) and agents: - out[f.stem] = agents - return out + expected_by_stem[f.stem] = agents + return expected_by_stem def _issue_count(actuals: dict, stem: str, agent: str) -> int | None: @@ -94,66 +145,117 @@ def _issue_count(actuals: dict, stem: str, agent: str) -> int | None: return len(issues) -def compute_fixture_diffs(before: dict, after: dict, expected: dict) -> list[dict]: - """One row per `(fixture stem, agent)` pair declared in `expected`, for - every pair present in both `before` and `after`. Each row is - `{"fixture", "agent", "kind": "defect"|"clean", - "truePositivesBefore"|None, "truePositivesAfter"|None, - "falsePositivesBefore"|None, "falsePositivesAfter"|None, "regressed"}` - -- only the pair of fields matching `kind` is populated; the other pair - is `None` (not applicable to that fixture's classification).""" +def _classify_kind(expected_spec: dict) -> str: + """`KIND_DEFECT`/`KIND_CLEAN` from `issueCount.min`, or + `KIND_UNCLASSIFIED` when `issueCount` is absent or not a dict (see the + module docstring's "Regression rule" section for why this is a distinct + third state rather than defaulting to "clean").""" + issue_count_range = expected_spec.get("issueCount") + if not isinstance(issue_count_range, dict): + return KIND_UNCLASSIFIED + return KIND_DEFECT if issue_count_range.get("min", 0) > 0 else KIND_CLEAN + + +def compute_fixture_diffs(before: dict, after: dict, expected: dict) -> tuple[list[dict], dict]: + """`(rows, scope)` for every `(fixture stem, agent)` pair declared in + `expected`. + + Each row is `{"fixture", "agent", "kind": "defect"|"clean", + "issuesBefore", "issuesAfter", "regressed", "detail"}`. `issuesAfter` and + `detail` are `None` except in the before-present/after-missing case (see + the module docstring), where `issuesAfter` is `None` and `detail` + explains why. + + A pair is left out of `rows` (and out of the regression count) when: + - its expected block is malformed or its `issueCount` is absent/not a + dict (`kind` would be `KIND_UNCLASSIFIED`) -- counted in + `scope["skippedUnclassified"]`. + - it has no recorded issue count in `before` at all -- counted in + `scope["skippedNotComparable"]`. + + `scope` is `{"compared": int, "skippedNotComparable": int, + "skippedUnclassified": int}` -- see finding #6 (scope truncation must be + visible in output, not silently absorbed into "No regressions.").""" rows: list[dict] = [] + compared = 0 + skipped_not_comparable = 0 + skipped_unclassified = 0 + for stem, agents in expected.items(): - for agent, espec in agents.items(): - if not isinstance(espec, dict): + for agent, expected_spec in agents.items(): + if not isinstance(expected_spec, dict): + skipped_unclassified += 1 + continue + + kind = _classify_kind(expected_spec) + if kind == KIND_UNCLASSIFIED: + skipped_unclassified += 1 continue - issue_count = espec.get("issueCount") or {} - is_defect_fixture = issue_count.get("min", 0) > 0 before_n = _issue_count(before, stem, agent) - after_n = _issue_count(after, stem, agent) - if before_n is None or after_n is None: - continue # not recorded in both runs -- not comparable + if before_n is None: + skipped_not_comparable += 1 # absent from `before` -- new coverage, not comparable + continue - if is_defect_fixture: + after_n = _issue_count(after, stem, agent) + detail = None + if after_n is None: + # Present in `before`, absent from `after`: total detection + # loss -- always a regression, regardless of kind. + regressed = True + detail = "no result recorded in after" + elif kind == KIND_DEFECT: regressed = after_n < before_n - row = { - "fixture": stem, - "agent": agent, - "kind": "defect", - "truePositivesBefore": before_n, - "truePositivesAfter": after_n, - "falsePositivesBefore": None, - "falsePositivesAfter": None, - "regressed": regressed, - } else: regressed = after_n > before_n - row = { + + rows.append( + { "fixture": stem, "agent": agent, - "kind": "clean", - "truePositivesBefore": None, - "truePositivesAfter": None, - "falsePositivesBefore": before_n, - "falsePositivesAfter": after_n, + "kind": kind, + "issuesBefore": before_n, + "issuesAfter": after_n, "regressed": regressed, + "detail": detail, } - rows.append(row) + ) + compared += 1 + rows.sort(key=lambda r: (r["fixture"], r["agent"])) - return rows + scope = { + "compared": compared, + "skippedNotComparable": skipped_not_comparable, + "skippedUnclassified": skipped_unclassified, + } + return rows, scope def _format_row(row: dict) -> str: pair = f"{row['fixture']}::{row['agent']}" - if row["kind"] == "defect": - detail = f"TP: {row['truePositivesBefore']} -> {row['truePositivesAfter']}" - else: - detail = f"FP: {row['falsePositivesBefore']} -> {row['falsePositivesAfter']}" + label = "TP-proxy" if row["kind"] == KIND_DEFECT else "FP-proxy" + after_display = row["issuesAfter"] if row["issuesAfter"] is not None else "MISSING" + detail = f"{label}: {row['issuesBefore']} -> {after_display}" + if row["detail"]: + detail += f" ({row['detail']})" verdict = "REGRESSION" if row["regressed"] else "OK" return f"{pair} [{row['kind']}] {detail} {verdict}" +def _validate_inputs(before_path: Path, after_path: Path, expected_dir: Path) -> int | None: + """`None` when `before_path`/`after_path`/`expected_dir` are all usable; + otherwise the exit code `main` should return (having already printed the + reason to stderr).""" + for label, path in (("before", before_path), ("after", after_path)): + if not path.is_file(): + print(f"compare_eval_results.py: cannot read {label} file {path}", file=sys.stderr) + return 2 + if not expected_dir.is_dir(): + print(f"compare_eval_results.py: expected dir not found: {expected_dir}", file=sys.stderr) + return 2 + return None + + def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("before", help="Path to the 'before' /agent-eval actuals JSON") @@ -175,13 +277,9 @@ def main(argv: list[str] | None = None) -> int: after_path = Path(args.after) expected_dir = Path(args.expected_dir) - for label, path in (("before", before_path), ("after", after_path)): - if not path.is_file(): - print(f"compare_eval_results.py: cannot read {label} file {path}", file=sys.stderr) - return 2 - if not expected_dir.is_dir(): - print(f"compare_eval_results.py: expected dir not found: {expected_dir}", file=sys.stderr) - return 2 + error_code = _validate_inputs(before_path, after_path, expected_dir) + if error_code is not None: + return error_code try: before = _load_json(before_path) @@ -195,7 +293,7 @@ def main(argv: list[str] | None = None) -> int: print(f"compare_eval_results.py: no usable expected/*.json found in {expected_dir}", file=sys.stderr) return 2 - rows = compute_fixture_diffs(before, after, expected) + rows, scope = compute_fixture_diffs(before, after, expected) if not rows: print( "compare_eval_results.py: no fixture/agent pair is present in both before and after files", @@ -206,6 +304,13 @@ def main(argv: list[str] | None = None) -> int: for row in rows: print(_format_row(row)) + total_skipped = scope["skippedNotComparable"] + scope["skippedUnclassified"] + print( + f"\ncompared {scope['compared']} pair(s); skipped {total_skipped} " + f"({scope['skippedNotComparable']} not-in-both-runs, " + f"{scope['skippedUnclassified']} unclassified/malformed)" + ) + regressed = [row for row in rows if row["regressed"]] if regressed: print(f"\n{len(regressed)} regression(s) detected.") diff --git a/plugins/dev-team/skills/code-review/SKILL.md b/plugins/dev-team/skills/code-review/SKILL.md index a0fe35a26..30fb35935 100644 --- a/plugins/dev-team/skills/code-review/SKILL.md +++ b/plugins/dev-team/skills/code-review/SKILL.md @@ -228,6 +228,25 @@ the required `"Plugin content & hooks"` CI check (#2128). `test-review` and it when it's present (see `${CLAUDE_PLUGIN_ROOT}/knowledge/test-review-division-of-labor.md`). +**Test-review mechanical pre-phase (#2169).** Also run, for each test file in ``: + +```bash +python3 "$CLAUDE_PLUGIN_ROOT/scripts/test_review_mechanics.py" . +``` + +Only relevant when `test-review` is in the dispatched lens set for this +round; skip entirely otherwise. Unlike the two pre-passes above, this one +runs **once per file** rather than once over the whole `` +list, because `test-review.md`'s own Phase 0 (`agents/test-review.md` → +Protocol) needs each file's own `mechanicalFail`/findings result supplied as +that file's context — the agent has no `Bash` tool and never runs this +script itself. Keep the per-file results keyed by file path when assembling +step 4's context so each file's `test-review` dispatch gets its own result, +not the whole batch's. Each result's `findings` array merges into step 4's +static-analysis context using the same envelope and the same "detected by +static analysis — do not re-report, focus on semantic concerns" framing as +the two pre-passes above. + **Pass `--files` (#1629).** Several checks are scoped to the changeset, because the conventions they enforce are "required going forward, do not retrofit" (`evals/README.md`'s `_calibration` rule is the motivating case). diff --git a/plugins/dev-team/tests/scripts/test_compare_eval_results.py b/plugins/dev-team/tests/scripts/test_compare_eval_results.py index 081120bb0..a756a1832 100644 --- a/plugins/dev-team/tests/scripts/test_compare_eval_results.py +++ b/plugins/dev-team/tests/scripts/test_compare_eval_results.py @@ -1,12 +1,16 @@ """Tests for scripts/compare_eval_results.py (#2169 Step 2.3). Covers `compute_fixture_diffs`'s classification (a fixture/agent block with -`issueCount.min > 0` is a defect fixture -- tracks true positives; a block -with `issueCount.min == 0` is a clean fixture -- tracks false positives) and -its regression rule, plus the CLI's exit-code contract: a true-positive-count -decrease and a false-positive-count increase both exit non-zero, an -unchanged/improved pair exits zero, and a multi-fixture input produces one -diff line per fixture/agent pair. +`issueCount.min > 0` is a defect fixture -- tracks a true-positive-count +proxy; a block with `issueCount.min == 0` is a clean fixture -- tracks a +false-positive-count proxy; a block with no `issueCount`, or a malformed +non-dict `issueCount`, is a third "unclassified" state that is neither), its +regression rule (including the before-present/after-missing "total +detection loss" case), its scope-summary counters, and the CLI's exit-code +contract: a true-positive-count decrease, a false-positive-count increase, +and an after-missing pair all exit non-zero; an unchanged/improved pair +exits zero; a multi-fixture input produces one diff line per fixture/agent +pair. """ from __future__ import annotations @@ -55,33 +59,35 @@ def test_defect_fixture_true_positive_decrease_is_regressed(self): before = _actuals_block("fixA", "test-review", 4) after = _actuals_block("fixA", "test-review", 2) - rows = cer.compute_fixture_diffs(before, after, expected) + rows, scope = cer.compute_fixture_diffs(before, after, expected) assert len(rows) == 1 assert rows[0]["kind"] == "defect" - assert rows[0]["truePositivesBefore"] == 4 - assert rows[0]["truePositivesAfter"] == 2 + assert rows[0]["issuesBefore"] == 4 + assert rows[0]["issuesAfter"] == 2 assert rows[0]["regressed"] is True + assert scope == {"compared": 1, "skippedNotComparable": 0, "skippedUnclassified": 0} def test_clean_fixture_false_positive_increase_is_regressed(self): expected = _expected_block("fixB", "test-review", 0) before = _actuals_block("fixB", "test-review", 0) after = _actuals_block("fixB", "test-review", 1) - rows = cer.compute_fixture_diffs(before, after, expected) + rows, scope = cer.compute_fixture_diffs(before, after, expected) assert len(rows) == 1 assert rows[0]["kind"] == "clean" - assert rows[0]["falsePositivesBefore"] == 0 - assert rows[0]["falsePositivesAfter"] == 1 + assert rows[0]["issuesBefore"] == 0 + assert rows[0]["issuesAfter"] == 1 assert rows[0]["regressed"] is True + assert scope["compared"] == 1 def test_unchanged_defect_fixture_is_not_regressed(self): expected = _expected_block("fixA", "test-review", 3) before = _actuals_block("fixA", "test-review", 4) after = _actuals_block("fixA", "test-review", 4) - rows = cer.compute_fixture_diffs(before, after, expected) + rows, _scope = cer.compute_fixture_diffs(before, after, expected) assert rows[0]["regressed"] is False @@ -101,18 +107,64 @@ def test_improved_defect_and_clean_fixtures_are_not_regressed(self): _actuals_block("fixB", "test-review", 0), ) - rows = cer.compute_fixture_diffs(before, after, expected) + rows, _scope = cer.compute_fixture_diffs(before, after, expected) assert all(row["regressed"] is False for row in rows) - def test_pair_missing_from_one_side_is_not_comparable(self): + def test_before_present_after_missing_is_a_regression(self): + """Domain-review finding #2: an agent that produced N issues in + `before` and has NO recorded result at all in `after` (errored, + timed out) is total detection loss -- the worst-case regression + this gate exists to catch, not "not comparable".""" expected = _expected_block("fixA", "test-review", 3) before = _actuals_block("fixA", "test-review", 4) after: dict = {} # no recorded result at all - rows = cer.compute_fixture_diffs(before, after, expected) + rows, scope = cer.compute_fixture_diffs(before, after, expected) + + assert len(rows) == 1 + assert rows[0]["issuesBefore"] == 4 + assert rows[0]["issuesAfter"] is None + assert rows[0]["regressed"] is True + assert rows[0]["detail"] == "no result recorded in after" + assert scope["compared"] == 1 + assert scope["skippedNotComparable"] == 0 + + def test_pair_missing_from_before_is_not_comparable(self): + """Only before-missing is genuinely "new coverage" and skipped -- + the asymmetric half of the fix #2 rule.""" + expected = _expected_block("fixA", "test-review", 3) + before: dict = {} + after = _actuals_block("fixA", "test-review", 4) + + rows, scope = cer.compute_fixture_diffs(before, after, expected) assert rows == [] + assert scope == {"compared": 0, "skippedNotComparable": 1, "skippedUnclassified": 0} + + def test_missing_issue_count_is_unclassified_not_clean(self): + """Domain-review finding #3: a block with no `issueCount` must not + silently default to the "clean" bucket.""" + expected = {"fixA": {"test-review": {}}} + before = _actuals_block("fixA", "test-review", 0) + after = _actuals_block("fixA", "test-review", 5) + + rows, scope = cer.compute_fixture_diffs(before, after, expected) + + assert rows == [] # not scored as either "defect" or "clean" + assert scope == {"compared": 0, "skippedNotComparable": 0, "skippedUnclassified": 1} + + def test_malformed_issue_count_is_treated_as_unclassified(self): + """Correctness-review finding #4: a non-dict `issueCount` must not + crash -- treated the same as the "unclassified" case.""" + expected = {"fixA": {"test-review": {"issueCount": 3}}} + before = _actuals_block("fixA", "test-review", 0) + after = _actuals_block("fixA", "test-review", 5) + + rows, scope = cer.compute_fixture_diffs(before, after, expected) + + assert rows == [] + assert scope["skippedUnclassified"] == 1 def test_multi_fixture_input_produces_one_row_per_fixture(self): expected = _merge( @@ -131,10 +183,11 @@ def test_multi_fixture_input_produces_one_row_per_fixture(self): _actuals_block("fixC", "test-review", 2), ) - rows = cer.compute_fixture_diffs(before, after, expected) + rows, scope = cer.compute_fixture_diffs(before, after, expected) assert len(rows) == 3 assert {row["fixture"] for row in rows} == {"fixA", "fixB", "fixC"} + assert scope["compared"] == 3 class TestCli: @@ -187,6 +240,17 @@ def test_false_positive_increase_exits_nonzero(self, tmp_path): assert result.returncode != 0 assert "REGRESSION" in result.stdout + def test_after_missing_result_exits_nonzero(self, tmp_path): + expected = {"fixA": {"test-review": {"issueCount": {"min": 3, "max": 6}}}} + before = _actuals_block("fixA", "test-review", 4) + after: dict = {} + + result = self._run(tmp_path, before, after, expected) + + assert result.returncode != 0 + assert "REGRESSION" in result.stdout + assert "no result recorded in after" in result.stdout + def test_unchanged_or_improved_pair_exits_zero(self, tmp_path): expected = { "fixA": {"test-review": {"issueCount": {"min": 3, "max": 6}}}, @@ -228,6 +292,26 @@ def test_multi_fixture_input_produces_one_diff_line_per_fixture(self, tmp_path): diff_lines = [line for line in result.stdout.splitlines() if "::test-review" in line] assert len(diff_lines) == 3 + def test_scope_summary_line_reports_compared_and_skipped_counts(self, tmp_path): + """Domain-review finding #6: "No regressions." must not print + without also surfacing how much was actually compared vs. skipped.""" + expected = { + "fixA": {"test-review": {"issueCount": {"min": 3, "max": 6}}}, + "fixB": {"test-review": {}}, # unclassified: no issueCount + } + before = _merge( + _actuals_block("fixA", "test-review", 3), + _actuals_block("fixB", "test-review", 0), + ) + after = _merge( + _actuals_block("fixA", "test-review", 3), + _actuals_block("fixB", "test-review", 0), + ) + + result = self._run(tmp_path, before, after, expected, check=True) + + assert "compared 1 pair(s); skipped 1 (0 not-in-both-runs, 1 unclassified/malformed)" in result.stdout + def test_missing_before_file_exits_usage_error(self, tmp_path): expected = {"fixA": {"test-review": {"issueCount": {"min": 3, "max": 6}}}} expected_dir = tmp_path / "expected" diff --git a/plugins/dev-team/tests/skills/test_code_review_test_review_mechanics_pre_pass_marker.py b/plugins/dev-team/tests/skills/test_code_review_test_review_mechanics_pre_pass_marker.py new file mode 100644 index 000000000..8f70e0a2d --- /dev/null +++ b/plugins/dev-team/tests/skills/test_code_review_test_review_mechanics_pre_pass_marker.py @@ -0,0 +1,51 @@ +"""Content guard for code-review/SKILL.md step 2b's test-review mechanical +pre-phase (#2169 Step 2.3, arch-review + correctness-review finding #1). + +`test_review_mechanics.py` (Step 2.2) is only useful if something actually +invokes it. `agents/test-review.md`'s Protocol Phase 0 expects the *caller* +dispatching test-review to run it per file and supply the result as +context -- since no `*-review.md` agent has a Bash tool, the dispatch site +has to live in the orchestrating skill, not the agent file. `/code-review` +step 2b is the established pattern for exactly this kind of pre-pass +(`repo_invariants.py` #1608, `internal_double_detector.py` #2130) -- this +test pins that `test_review_mechanics.py` is named in that same step 2b +location, so a future SKILL.md edit can't silently drop the only place this +script is ever invoked. +""" + +from __future__ import annotations + +from _repo_root import REPO_ROOT + +SKILL = REPO_ROOT / "plugins" / "dev-team" / "skills" / "code-review" / "SKILL.md" + +_MARKER = "### 2b. Static analysis pre-pass" + + +def _step_2b_section() -> str: + text = SKILL.read_text(encoding="utf-8") + assert _MARKER in text, "step 2b heading not found" + section = text.split(_MARKER, 1)[1].split("\n### ", 1)[0] + return section + + +def test_step_2b_invokes_test_review_mechanics_script() -> None: + section = _step_2b_section() + assert "test_review_mechanics.py" in section, ( + "code-review SKILL.md step 2b no longer names test_review_mechanics.py " + "-- test-review's Phase 0 mechanical pre-phase would have no dispatch site" + ) + + +def test_step_2b_test_review_pre_pass_is_alongside_the_other_two_pre_passes() -> None: + section = _step_2b_section() + assert "repo_invariants.py" in section + assert "internal_double_detector.py" in section + assert "test_review_mechanics.py" in section + + +def test_step_2b_test_review_pre_pass_cites_its_issue_and_envelope() -> None: + section = _step_2b_section() + assert "**Test-review mechanical pre-phase (#2169).**" in section + pre_pass = " ".join(section.split("Test-review mechanical pre-phase", 1)[1].split()) + assert "detected by static analysis" in pre_pass From 78671213bc70d8749b1e68154a47ee9f605e6415 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 21:05:42 +0000 Subject: [PATCH 09/14] fix(test-review): fix stale wording, parametrize and harden mechanical-split tag tests Slice 2's slice-boundary review checkpoint (7 agents) covering Step 2.1 -- the only standard-complexity step in this slice, never per-step reviewed -- found the reflection-into-private-members bullet's "future mechanical pre-phase" wording stale (Step 2.2 shipped in this same slice; the script already implements the check). Also fixes the new content-guard test's own quality gaps: a `block_containing` closure duplicated across two tests hoisted to a module-level helper, six tests' repeated section-fetch boilerplate replaced by a module-scoped fixture, two eager-test loops converted to parametrized cases so an early failure can't mask a later one, and two new synthetic-fixture tests exercising the tag-parsing helper's untagged/double-tagged failure branches, which no existing test had ever actually triggered. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/agents/test-review.md | 2 +- ..._test_review_mechanical_split_annotated.py | 114 +++++++++++------- 2 files changed, 69 insertions(+), 47 deletions(-) diff --git a/plugins/dev-team/agents/test-review.md b/plugins/dev-team/agents/test-review.md index a4863c20e..14b210680 100644 --- a/plugins/dev-team/agents/test-review.md +++ b/plugins/dev-team/agents/test-review.md @@ -173,7 +173,7 @@ Testability blockers: - Code under test that cannot be constructed with known values (static factories, singletons, no injectable constructor) — flag as error; per `${CLAUDE_PLUGIN_ROOT}/knowledge/testability-patterns.md#pattern-1-constructor-injection-replace-static-factories-singletons`, the production code must change, not the test approach [JUDGMENT] - Mocking of concrete classes (not interfaces) — flag as warning; extract an interface for the dependency [JUDGMENT] -- Tests using reflection into private members as primary strategy — flag as warning. This is an architecture/encapsulation issue the test is reaching around, not a test-hygiene nit. Detection signatures: Java: `getDeclaredMethod`/`getDeclaredField` + `setAccessible(true)`, `Method.invoke` on a private/protected member; C#: `Type.GetMethod(..., BindingFlags.NonPublic | BindingFlags.Instance)`, `Type.InvokeMember`; Python: `getattr`/`setattr`/`hasattr` targeting a name-mangled (`_ClassName__attr`) or underscore-prefixed attribute; JS/TS: bracket-notation access into a `private`/non-exported member (e.g. `(obj as any)['_privateMethod']()`), `Object.getOwnPropertyDescriptor`/`Object.defineProperty` used to reach a non-exported member. Suggested fix — pick by shape of the code, never the generic "expand the public API": (1) extract the private logic into a collaborator with its own public seam, when it's standalone logic worth testing independently; (2) relax visibility to package-private/internal, only when a production collaborator in the same module/assembly independently needs the access (the language must have that tier) — never as a grant solely so the test can reach in, which recreates the `InternalsVisibleTo`/`@VisibleForTesting` anti-pattern below; (3) test the behavior through the class's existing public API, when the private method is already an implementation detail of a public behavior [MECHANICAL] (detection only, via the explicit per-language signatures above — this stays `warning`-severity and is reported by the future mechanical pre-phase without gating the qualitative pass; only the no-assertion-tests check and `internal_double_detector.py`'s own `verdict: "high"` findings (translated to `error` severity by the mechanical pre-phase) gate the pass — see Step 2.2 (`scripts/test_review_mechanics.py`)) +- Tests using reflection into private members as primary strategy — flag as warning. This is an architecture/encapsulation issue the test is reaching around, not a test-hygiene nit. Detection signatures: Java: `getDeclaredMethod`/`getDeclaredField` + `setAccessible(true)`, `Method.invoke` on a private/protected member; C#: `Type.GetMethod(..., BindingFlags.NonPublic | BindingFlags.Instance)`, `Type.InvokeMember`; Python: `getattr`/`setattr`/`hasattr` targeting a name-mangled (`_ClassName__attr`) or underscore-prefixed attribute; JS/TS: bracket-notation access into a `private`/non-exported member (e.g. `(obj as any)['_privateMethod']()`), `Object.getOwnPropertyDescriptor`/`Object.defineProperty` used to reach a non-exported member. Suggested fix — pick by shape of the code, never the generic "expand the public API": (1) extract the private logic into a collaborator with its own public seam, when it's standalone logic worth testing independently; (2) relax visibility to package-private/internal, only when a production collaborator in the same module/assembly independently needs the access (the language must have that tier) — never as a grant solely so the test can reach in, which recreates the `InternalsVisibleTo`/`@VisibleForTesting` anti-pattern below; (3) test the behavior through the class's existing public API, when the private method is already an implementation detail of a public behavior [MECHANICAL] (detection only, via the explicit per-language signatures above — this stays `warning`-severity and is reported by the mechanical pre-phase (Step 2.2, `scripts/test_review_mechanics.py`) without gating the qualitative pass; only the no-assertion-tests check and `internal_double_detector.py`'s own `verdict: "high"` findings (translated to `error` severity by the mechanical pre-phase) gate the pass) Internal-collaborator doubling (mechanical — never a truth judgment; see `${CLAUDE_PLUGIN_ROOT}/knowledge/internal-collaborator-doubling.md#the-waiver`): diff --git a/plugins/dev-team/tests/agents/test_test_review_mechanical_split_annotated.py b/plugins/dev-team/tests/agents/test_test_review_mechanical_split_annotated.py index 4c319db10..e4c4f9c38 100644 --- a/plugins/dev-team/tests/agents/test_test_review_mechanical_split_annotated.py +++ b/plugins/dev-team/tests/agents/test_test_review_mechanical_split_annotated.py @@ -15,6 +15,8 @@ import re +import pytest + from _repo_root import REPO_ROOT AGENT = REPO_ROOT / "plugins" / "dev-team" / "agents" / "test-review.md" @@ -69,10 +71,20 @@ def _assert_each_bullet_has_exactly_one_tag(blocks: list[str]) -> None: assert not both, f"bullets carrying both tags: {both}" -def test_detect_section_bullets_each_carry_exactly_one_tag() -> None: +def _block_containing(blocks: list[str], snippet: str) -> str: + matches = [b for b in blocks if snippet in b] + assert len(matches) == 1, f"expected exactly one bullet with {snippet!r}" + return matches[0] + + +@pytest.fixture(scope="module") +def detect_section() -> str: text = _text() - detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") - blocks = _bullet_blocks(detect) + return _section(text, "## Detect", "## Tolerated-Deviation Hunt") + + +def test_detect_section_bullets_each_carry_exactly_one_tag(detect_section: str) -> None: + blocks = _bullet_blocks(detect_section) assert len(blocks) >= 30, ( f"expected the full set of ## Detect check bullets, found {len(blocks)}" ) @@ -98,23 +110,21 @@ def test_consolidation_rule_is_tagged_mechanical() -> None: ) -def test_testability_blockers_bullets_each_carry_exactly_one_tag() -> None: +def test_testability_blockers_bullets_each_carry_exactly_one_tag(detect_section: str) -> None: """Testability blockers is a named subsection under ## Detect (not its own ## heading) — covered by the ## Detect sweep above, but pinned here directly per the task's explicit call-out of this section.""" - text = _text() - detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") - section = _section(detect, "Testability blockers:", "Internal-collaborator doubling") + section = _section(detect_section, "Testability blockers:", "Internal-collaborator doubling") blocks = _bullet_blocks(section) assert len(blocks) == 3 _assert_each_bullet_has_exactly_one_tag(blocks) -def test_internal_collaborator_doubling_bullets_each_carry_exactly_one_tag() -> None: - text = _text() - detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") +def test_internal_collaborator_doubling_bullets_each_carry_exactly_one_tag( + detect_section: str, +) -> None: section = _section( - detect, + detect_section, "Internal-collaborator doubling", "If a static-analysis pre-pass", ) @@ -130,15 +140,15 @@ def test_internal_collaborator_doubling_bullets_each_carry_exactly_one_tag() -> ) -def test_reflection_bullet_is_mechanical_but_notes_non_gating_warning_severity() -> None: +def test_reflection_bullet_is_mechanical_but_notes_non_gating_warning_severity( + detect_section: str, +) -> None: """Reflection-into-private-members has an explicit per-language detection signature (MECHANICAL for detection) but must stay warning-severity and non-gating — only Step 2.2's no-assertion-tests check and internal_double_detector.py's error-severity findings gate the qualitative pass.""" - text = _text() - detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") - blocks = _bullet_blocks(detect) + blocks = _bullet_blocks(detect_section) reflection_blocks = [ b for b in blocks if "reflection into private members" in b ] @@ -150,47 +160,59 @@ def test_reflection_bullet_is_mechanical_but_notes_non_gating_warning_severity() assert "Step 2.2" in block -def test_known_mechanical_anchor_bullets_are_tagged_mechanical() -> None: - """Lock in the plan's explicit MECHANICAL examples (missing-await, - mocks-not-reset, unstubbed clock/RNG/timers, tests-with-no-assertion) - against accidental re-tagging.""" - text = _text() - detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") - blocks = _bullet_blocks(detect) - - def block_containing(snippet: str) -> str: - matches = [b for b in blocks if snippet in b] - assert len(matches) == 1, f"expected exactly one bullet with {snippet!r}" - return matches[0] - - for snippet in ( +@pytest.mark.parametrize( + "snippet", + [ "Tests with no assertion", "Mocks/stubs not reset", "Missing await on async operations", "Unstubbed clock access", "Unstubbed randomness", "Unstubbed timers/delays", - ): - assert "[MECHANICAL]" in block_containing(snippet) + ], +) +def test_known_mechanical_anchor_bullets_are_tagged_mechanical( + detect_section: str, snippet: str +) -> None: + """Lock in the plan's explicit MECHANICAL examples (missing-await, + mocks-not-reset, unstubbed clock/RNG/timers, tests-with-no-assertion) + against accidental re-tagging.""" + blocks = _bullet_blocks(detect_section) + assert "[MECHANICAL]" in _block_containing(blocks, snippet) -def test_known_judgment_anchor_bullets_are_tagged_judgment() -> None: +@pytest.mark.parametrize( + "snippet", + [ + "Missing edge cases", + "No arrange-act-assert structure", + "Misleading test descriptions", + "Code under test that cannot be constructed with known values", + ], +) +def test_known_judgment_anchor_bullets_are_tagged_judgment( + detect_section: str, snippet: str +) -> None: """Lock in the plan's explicit JUDGMENT examples (coverage-gap adequacy, AAA structure, misleading descriptions, static-factory / singleton testability blockers) against accidental re-tagging.""" - text = _text() - detect = _section(text, "## Detect", "## Tolerated-Deviation Hunt") - blocks = _bullet_blocks(detect) + blocks = _bullet_blocks(detect_section) + assert "[JUDGMENT]" in _block_containing(blocks, snippet) - def block_containing(snippet: str) -> str: - matches = [b for b in blocks if snippet in b] - assert len(matches) == 1, f"expected exactly one bullet with {snippet!r}" - return matches[0] - for snippet in ( - "Missing edge cases", - "No arrange-act-assert structure", - "Misleading test descriptions", - "Code under test that cannot be constructed with known values", - ): - assert "[JUDGMENT]" in block_containing(snippet) +def test_assert_each_bullet_has_exactly_one_tag_raises_on_untagged_bullet() -> None: + """Synthetic fixture exercises the untagged-bullet failure branch, + which the live (already-compliant) test-review.md never triggers.""" + blocks = _bullet_blocks( + "- Tagged bullet body. [MECHANICAL]\n- Untagged bullet body with no tag.\n" + ) + with pytest.raises(AssertionError, match=r"missing \[MECHANICAL\]/\[JUDGMENT\] tag"): + _assert_each_bullet_has_exactly_one_tag(blocks) + + +def test_assert_each_bullet_has_exactly_one_tag_raises_on_double_tagged_bullet() -> None: + """Synthetic fixture exercises the double-tagged-bullet failure branch, + which the live (already-compliant) test-review.md never triggers.""" + blocks = _bullet_blocks("- Double tagged bullet body. [MECHANICAL] [JUDGMENT]\n") + with pytest.raises(AssertionError, match="bullets carrying both tags"): + _assert_each_bullet_has_exactly_one_tag(blocks) From 11eb1214569412c734b34d45b8a1cdd7de13345d Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 21:20:17 +0000 Subject: [PATCH 10/14] =?UTF-8?q?feat(code-review):=20add=20render=5Ftiere?= =?UTF-8?q?d=5Ffindings.py=20=E2=80=94=20Tier-1/Tier-2=20report=20renderin?= =?UTF-8?q?g?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Renders a compact Tier-1 line per finding (file, line, agent, severity, confidence, first sentence of message, finding-id) plus an expansion hint by default, deferring the full message/suggestedFix narrative (Tier-2) to an explicit --expand |all — using the same in-memory finding list a run already has, no re-dispatch, no I/O beyond the finding JSON on hand. Finding-ids are agent:file:line:severity[:category], with a :0/:1/... ordinal suffix when two findings in the same run still collide on that base id. Not yet wired into /code-review step 7 — that's Step 3.2. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- .../skills/code-review/output-format.md | 10 + .../scripts/render_tiered_findings.py | 220 ++++++++++++++++ .../scripts/test_render_tiered_findings.py | 240 ++++++++++++++++++ 3 files changed, 470 insertions(+) create mode 100755 plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py create mode 100644 plugins/dev-team/tests/scripts/test_render_tiered_findings.py diff --git a/plugins/dev-team/skills/code-review/output-format.md b/plugins/dev-team/skills/code-review/output-format.md index ac6ae82ba..6c097a5f6 100644 --- a/plugins/dev-team/skills/code-review/output-format.md +++ b/plugins/dev-team/skills/code-review/output-format.md @@ -391,6 +391,16 @@ scored. After the summary, list remaining issues grouped by file, sorted by severity. Mark each with: `[confidence: none]`, `[auto-fix failed]`, or `[suggestion]`. Append the iteration table above. +`scripts/render_tiered_findings.py` (#2170) renders that per-finding listing +in two tiers from the same in-memory finding list, no re-dispatch: Tier-1 (one +line per finding — file, line, agent, severity/confidence, first sentence of +the message, finding-id — plus an expansion hint) by default, Tier-2 (full +message + suggestedFix) for a finding-id via `--expand |all`. +Wiring it into this step's default rendering path, and documenting `--expand` +as a `/code-review` flag, is a separate step (issue #2170 Step 3.2) — this +paragraph exists so the script itself is discoverable from the skill's own +docs. + ## Override audit log entry (step 2, `--force` path) Append to `.claude/metrics/override-audit.jsonl` (create if missing): diff --git a/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py b/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py new file mode 100755 index 000000000..1c77af2de --- /dev/null +++ b/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py @@ -0,0 +1,220 @@ +#!/usr/bin/env python3 +"""Tier-1/Tier-2 rendering for `/code-review` step 7's prose report (#2170). + +Step 7's prose-mode path lists every remaining finding in full — severity, +confidence, file, line, message, and suggested fix all inline — which makes a +report with a dozen findings a wall of text even when most of them are +routine. This module renders a compact Tier-1 line per finding by default, +and defers the full message/suggestedFix narrative (Tier-2) to an explicit +`--expand |all`, using the SAME in-memory finding list already +aggregated for the report — no re-dispatch, no I/O beyond the finding JSON +already on hand. + +This script is prose-path only. `--json` (step 7's other branch) and +`./corrections/*.json` (step 8) read/write the full finding objects +independently and never call into this module — see Step 3.2, which wires +that non-interference into `/code-review` itself. Nothing here needs to +special-case `--json`. + +## Finding-id scheme + +`agent:file:line:severity`, plus `:category` appended when the finding +carries a truthy `category`. Two or more findings in the same run that still +land on an identical base id after that get a `:0`, `:1`, ... ordinal suffix +in list order — deterministic and collision-free within a single run because +the caller's list order is already fixed for that run. IDs are not meant to +be stable *across* runs (a re-dispatched round may reorder or drop findings). + +`agent` is read as `finding.get("agent")` first (the aggregated/flattened +finding shape `consolidate.py` produces, `agent`-tagged per finding) falling +back to `finding.get("agentName")` (the raw per-agent-result field name) — +the same two-field fallback `finding_signature.py`'s `signature()` already +uses for the identical purpose, so this module reads either the pre- or +post-flattening shape without extra glue. + +Stdlib-only. See docs/python-hook-contract.md. +""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +from collections import Counter +from pathlib import Path + +#: Rendered instead of any per-finding line when a round has zero findings — +#: there is nothing to list and nothing to expand. +CLEAN_PASS_SUMMARY = "Clean pass: 0 findings this round — nothing to expand." + +#: Rendered once, after the Tier-1 lines, when the round has at least one +#: finding. Omitted entirely on a zero-finding round (see CLEAN_PASS_SUMMARY) +#: because there is nothing an --expand call could act on. +EXPAND_HINT = "Expand a finding with --expand , or --expand all for every finding." + +#: First-sentence split: a `.`/`!`/`?` followed by whitespace or end-of-string. +#: Deliberately simple (no abbreviation handling) — finding messages in this +#: codebase's contract are short, single-clause statements +#: ("God object: AuthController handles login, registration, and password +#: reset"), not prose with embedded abbreviations; a message with no +#: terminator at all renders in full. +_SENTENCE_END_RE = re.compile(r"[.!?](?:\s|$)") + + +def first_sentence(message) -> str: + """The first sentence of `message`, or the whole (stripped) string when + no sentence terminator is found.""" + text = str(message or "").strip() + if not text: + return "" + match = _SENTENCE_END_RE.search(text) + if match is None: + return text + return text[: match.start() + 1] + + +def _finding_agent(finding: dict) -> str: + return str(finding.get("agent") or finding.get("agentName") or "") + + +def _finding_line(finding: dict) -> str: + line = finding.get("line") + return "" if line is None else str(line) + + +def base_id(finding: dict) -> str: + """The finding-id before ordinal-suffix collision resolution: + `agent:file:line:severity`, plus `:category` when `category` is a + truthy value on this finding.""" + parts = [ + _finding_agent(finding), + str(finding.get("file") or ""), + _finding_line(finding), + str(finding.get("severity") or ""), + ] + category = finding.get("category") + if category: + parts.append(str(category)) + return ":".join(parts) + + +def compute_ids(findings: list[dict]) -> list[str]: + """Finding-id for each entry in `findings`, in list order. + + A base id unique in this round is used as-is. A base id shared by two or + more findings gets a `:0`, `:1`, ... ordinal suffix, assigned in list + order — deterministic because the input order is already fixed for a + given run. + """ + bases = [base_id(f) for f in findings] + counts = Counter(bases) + next_ordinal: dict[str, int] = {} + ids = [] + for base in bases: + if counts[base] > 1: + ordinal = next_ordinal.get(base, 0) + next_ordinal[base] = ordinal + 1 + ids.append(f"{base}:{ordinal}") + else: + ids.append(base) + return ids + + +def render_tier1_line(finding: dict, finding_id: str) -> str: + """One Tier-1 line: `file:line [agent] severity/confidence — + ()`.""" + file_ = finding.get("file", "") + line = _finding_line(finding) + agent = _finding_agent(finding) + severity = finding.get("severity", "") + confidence = finding.get("confidence", "") + sentence = first_sentence(finding.get("message", "")) + return f"{file_}:{line} [{agent}] {severity}/{confidence} — {sentence} ({finding_id})" + + +def render_tier1_report(findings: list[dict], ids: list[str]) -> str: + """The full Tier-1 report: one line per finding plus the expansion hint, + or `CLEAN_PASS_SUMMARY` alone when there are no findings.""" + if not findings: + return CLEAN_PASS_SUMMARY + lines = [render_tier1_line(f, i) for f, i in zip(findings, ids)] + lines.append(EXPAND_HINT) + return "\n".join(lines) + + +def render_tier2_block(finding: dict, finding_id: str) -> str: + """The full Tier-2 block for one finding: id header, full message, and + the suggested fix (when the finding carries one).""" + lines = [f"=== {finding_id} ===", str(finding.get("message") or "")] + fix = finding.get("suggestedFix") + if fix: + lines.append(f"Suggested fix: {fix}") + return "\n".join(lines) + + +def render_tier2_report(findings: list[dict], ids: list[str]) -> str: + """Every finding's Tier-2 block, in list order, separated by a blank + line. `CLEAN_PASS_SUMMARY` alone when there are no findings.""" + if not findings: + return CLEAN_PASS_SUMMARY + return "\n\n".join(render_tier2_block(f, i) for f, i in zip(findings, ids)) + + +def render_expand_one(findings: list[dict], ids: list[str], finding_id: str) -> str | None: + """The single Tier-2 block for `finding_id`, or `None` when no finding + in this round has that id (the caller turns that into a non-zero exit).""" + try: + index = ids.index(finding_id) + except ValueError: + return None + return render_tier2_block(findings[index], ids[index]) + + +def _load_findings(path: str) -> list[dict]: + raw = sys.stdin.read() if path == "-" else Path(path).read_text(encoding="utf-8") + data = json.loads(raw) if raw.strip() else [] + if isinstance(data, dict): + data = data.get("findings", []) + if not isinstance(data, list): + return [] + return [f for f in data if isinstance(f, dict)] + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--findings", + default="-", + help="Path to this round's finding-list JSON ('-' for stdin, the default).", + ) + parser.add_argument( + "--expand", + default=None, + help="A finding-id to render Tier-2 for, or 'all' to render every finding's Tier-2 block.", + ) + args = parser.parse_args(argv) + + findings = _load_findings(args.findings) + ids = compute_ids(findings) + + if args.expand is None: + print(render_tier1_report(findings, ids)) + return 0 + + if args.expand == "all": + print(render_tier2_report(findings, ids)) + return 0 + + block = render_expand_one(findings, ids, args.expand) + if block is None: + sys.stderr.write( + f"render_tiered_findings: finding-id not found in this round's findings: {args.expand!r}\n" + ) + return 1 + print(block) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) diff --git a/plugins/dev-team/tests/scripts/test_render_tiered_findings.py b/plugins/dev-team/tests/scripts/test_render_tiered_findings.py new file mode 100644 index 000000000..3aa75bf88 --- /dev/null +++ b/plugins/dev-team/tests/scripts/test_render_tiered_findings.py @@ -0,0 +1,240 @@ +"""Unit tests for skills/code-review/scripts/render_tiered_findings.py (#2170). + +Covers finding-id derivation (including the ordinal-suffix collision case), +Tier-1 rendering, Tier-2 rendering via `--expand |all`, the unknown-id +failure contract, and the zero-finding clean-pass path. +""" + +from __future__ import annotations + +import json +import subprocess +import sys + +from _repo_root import REPO_ROOT as _REPO_ROOT + +sys.path.insert( + 0, + str(_REPO_ROOT / "plugins" / "dev-team" / "skills" / "code-review" / "scripts"), +) + +import render_tiered_findings as rtf + +_SCRIPT = ( + _REPO_ROOT + / "plugins" + / "dev-team" + / "skills" + / "code-review" + / "scripts" + / "render_tiered_findings.py" +) + + +def _finding(**overrides) -> dict: + base = { + "agent": "structure-review", + "file": "src/auth/login.ts", + "line": 42, + "severity": "warning", + "confidence": "medium", + "message": "God object: AuthController handles login, registration, and password reset.", + "suggestedFix": "Split into LoginController, RegistrationController, and PasswordResetController", + } + base.update(overrides) + return base + + +class TestFirstSentence: + def test_splits_at_first_terminator(self): + assert rtf.first_sentence("First. Second.") == "First." + + def test_no_terminator_returns_whole_message(self): + assert rtf.first_sentence("No terminator here") == "No terminator here" + + def test_empty_message(self): + assert rtf.first_sentence("") == "" + assert rtf.first_sentence(None) == "" + + +class TestBaseId: + def test_id_without_category(self): + finding = _finding() + assert rtf.base_id(finding) == "structure-review:src/auth/login.ts:42:warning" + + def test_id_with_category(self): + finding = _finding(category="god-object") + assert ( + rtf.base_id(finding) + == "structure-review:src/auth/login.ts:42:warning:god-object" + ) + + def test_empty_category_excluded(self): + finding = _finding(category="") + assert rtf.base_id(finding) == "structure-review:src/auth/login.ts:42:warning" + + def test_agentname_fallback(self): + finding = _finding() + del finding["agent"] + finding["agentName"] = "structure-review" + assert rtf.base_id(finding) == "structure-review:src/auth/login.ts:42:warning" + + +class TestComputeIds: + def test_unique_findings_keep_base_ids(self): + findings = [_finding(), _finding(file="other.ts", line=7)] + ids = rtf.compute_ids(findings) + assert ids == [ + "structure-review:src/auth/login.ts:42:warning", + "structure-review:other.ts:7:warning", + ] + + def test_colliding_base_ids_get_ordinal_suffixes_in_list_order(self): + findings = [ + _finding(message="First distinct message."), + _finding(message="Second distinct message."), + ] + ids = rtf.compute_ids(findings) + assert ids == [ + "structure-review:src/auth/login.ts:42:warning:0", + "structure-review:src/auth/login.ts:42:warning:1", + ] + + def test_three_way_collision_gets_three_ordinals(self): + findings = [_finding(message=f"Message {i}.") for i in range(3)] + ids = rtf.compute_ids(findings) + assert ids == [ + "structure-review:src/auth/login.ts:42:warning:0", + "structure-review:src/auth/login.ts:42:warning:1", + "structure-review:src/auth/login.ts:42:warning:2", + ] + + +class TestRenderTier1Report: + def test_multi_finding_render(self): + findings = [ + _finding(), + _finding( + file="src/api/handler.ts", + line=15, + agent="security-review", + severity="error", + confidence="high", + message="SQL injection via unsanitized query parameter.", + ), + ] + ids = rtf.compute_ids(findings) + report = rtf.render_tier1_report(findings, ids) + lines = report.splitlines() + + assert lines[0] == ( + "src/auth/login.ts:42 [structure-review] warning/medium — " + "God object: AuthController handles login, registration, and password reset. " + "(structure-review:src/auth/login.ts:42:warning)" + ) + assert lines[1] == ( + "src/api/handler.ts:15 [security-review] error/high — " + "SQL injection via unsanitized query parameter. " + "(security-review:src/api/handler.ts:15:error)" + ) + assert lines[2] == rtf.EXPAND_HINT + assert len(lines) == 3 + + def test_zero_findings_renders_clean_pass_with_no_hint(self): + report = rtf.render_tier1_report([], []) + assert report == rtf.CLEAN_PASS_SUMMARY + assert rtf.EXPAND_HINT not in report + + +class TestRenderExpand: + def test_expand_one_renders_only_that_findings_tier2_content(self): + findings = [ + _finding(message="First distinct message.", suggestedFix="Fix A"), + _finding(message="Second distinct message.", suggestedFix="Fix B"), + ] + ids = rtf.compute_ids(findings) + + block = rtf.render_expand_one(findings, ids, ids[0]) + assert "First distinct message." in block + assert "Fix A" in block + assert "Second distinct message." not in block + assert "Fix B" not in block + + block_other = rtf.render_expand_one(findings, ids, ids[1]) + assert "Second distinct message." in block_other + assert "Fix B" in block_other + assert "First distinct message." not in block_other + assert "Fix A" not in block_other + + def test_expand_all_renders_every_findings_tier2_content(self): + findings = [ + _finding(message="First distinct message.", suggestedFix="Fix A"), + _finding(message="Second distinct message.", suggestedFix="Fix B"), + ] + ids = rtf.compute_ids(findings) + report = rtf.render_tier2_report(findings, ids) + assert "First distinct message." in report + assert "Fix A" in report + assert "Second distinct message." in report + assert "Fix B" in report + + def test_expand_unknown_id_returns_none(self): + findings = [_finding()] + ids = rtf.compute_ids(findings) + assert rtf.render_expand_one(findings, ids, "no-such-id") is None + + +def _run(*args, input_text=None): + return subprocess.run( + [sys.executable, str(_SCRIPT), *args], + input=input_text, + capture_output=True, + text=True, + check=False, + ) + + +class TestCli: + def test_cli_default_renders_tier1(self): + findings = [_finding()] + r = _run(input_text=json.dumps(findings)) + assert r.returncode == 0 + assert "structure-review:src/auth/login.ts:42:warning" in r.stdout + assert rtf.EXPAND_HINT in r.stdout + + def test_cli_zero_findings_renders_clean_pass(self): + r = _run(input_text=json.dumps([])) + assert r.returncode == 0 + assert r.stdout.strip() == rtf.CLEAN_PASS_SUMMARY + assert rtf.EXPAND_HINT not in r.stdout + + def test_cli_expand_known_id_exits_zero_with_tier2_content(self): + findings = [_finding()] + ids = rtf.compute_ids(findings) + r = _run("--expand", ids[0], input_text=json.dumps(findings)) + assert r.returncode == 0 + assert "Split into LoginController" in r.stdout + + def test_cli_expand_unknown_id_fails_clearly(self): + findings = [_finding()] + r = _run("--expand", "no-such-id", input_text=json.dumps(findings)) + assert r.returncode != 0 + assert "no-such-id" in r.stderr + assert "not found" in r.stderr.lower() + + def test_cli_expand_all(self): + findings = [ + _finding(message="First distinct message.", suggestedFix="Fix A"), + _finding(message="Second distinct message.", suggestedFix="Fix B"), + ] + r = _run("--expand", "all", input_text=json.dumps(findings)) + assert r.returncode == 0 + assert "First distinct message." in r.stdout + assert "Second distinct message." in r.stdout + + def test_cli_findings_from_file(self, tmp_path): + findings_file = tmp_path / "findings.json" + findings_file.write_text(json.dumps([_finding()])) + r = _run("--findings", str(findings_file)) + assert r.returncode == 0 + assert "structure-review:src/auth/login.ts:42:warning" in r.stdout From b26203e57fa35cc87a6ab9345db4538ec49be13c Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 21:40:05 +0000 Subject: [PATCH 11/14] feat(code-review): render Tier-1 findings by default, Tier-2 via --expand Wires render_tiered_findings.py (Step 3.1) into /code-review step 7's prose-mode path: the per-finding listing now renders as compact Tier-1 lines plus an expansion hint by default, with Tier-2 (full message + suggested fix) rendered for a finding-id, or every finding, via --expand |all. --json and corrections/*.json are untouched -- both already read/write the full finding objects independently, and --expand is a structural no-op under --json since that branch never reaches the tiered-rendering path. The old [auto-fix failed] bracket tag is dropped from the default listing (it's step 6a fix-loop runtime state, not a finding property, and stays visible in the iteration log); [confidence: none]/ [suggestion] are now carried by each Tier-1 line's own severity/confidence field instead of a separate bracket tag. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- plugins/dev-team/docs/skills.md | 2 +- plugins/dev-team/skills/code-review/SKILL.md | 12 +- .../skills/code-review/output-format.md | 60 ++++- .../test_code_review_tiered_output_marker.py | 212 ++++++++++++++++++ 4 files changed, 273 insertions(+), 13 deletions(-) create mode 100644 plugins/dev-team/tests/skills/test_code_review_tiered_output_marker.py diff --git a/plugins/dev-team/docs/skills.md b/plugins/dev-team/docs/skills.md index a2fffb01c..62236fb09 100644 --- a/plugins/dev-team/docs/skills.md +++ b/plugins/dev-team/docs/skills.md @@ -46,7 +46,7 @@ Most skills are **user-invocable** as slash commands — shown as `/name`; run t | Skill | Options | File | Description | | --- | --- | --- | --- | | `/apply-fixes` | [--dry] [--skip-tests] [--skip-build] [--skip-lint] | [`apply-fixes/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/apply-fixes/SKILL.md) | Apply correction prompts generated by /code-review. Use this whenever the user wants to apply, fix, or action the results of a code review — phrases like "apply the fixes", "fix the issues", "apply corrections", or after /code-review has run and produced a corrections/ directory. | -| `/code-review` | [--agent ] [--since ] [--path ] [--all] [--json] [--internal] [--force --reason ""] [--static-analysis\|--no-static-analysis] [--init-risks] [--background] [--pdf] | [`code-review/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/code-review/SKILL.md) | Run all enabled review agents against target files. Use this whenever the user asks for a code review, wants feedback on their code, says "review my code", "check this before I PR", "what's wrong with this", "run the agents", or has just finished implementing a feature. Use proactively before commits and pull requests. | +| `/code-review` | [--agent ] [--since ] [--path ] [--all] [--json] [--expand \|all] [--internal] [--force --reason ""] [--static-analysis\|--no-static-analysis] [--init-risks] [--background] [--pdf] | [`code-review/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/code-review/SKILL.md) | Run all enabled review agents against target files. Use this whenever the user asks for a code review, wants feedback on their code, says "review my code", "check this before I PR", "what's wrong with this", "run the agents", or has just finished implementing a feature. Use proactively before commits and pull requests. | | `/frontend-architecture` | [--path ] [--since ] [--all] [--json] | [`frontend-architecture/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/frontend-architecture/SKILL.md) | Frontend component architecture review — dispatch the component-architecture-review agent over the frontend component files to catch reusable components that should be extracted, duplicated UI patterns, prop drilling, component-granularity problems, and inconsistent component APIs as a frontend evolves. Use when the user says "review the frontend architecture", "are my components reusable", "is this UI duplicated", "should this be a shared component", "check for prop drilling", or before extracting a component library. Advisory — it recommends, it does not edit. | | `/review` | [--agent ] [--since ] [--path ] [--all] [--json] [--internal] [--force --reason ""] [--static-analysis\|--no-static-analysis] [--init-risks] [--background] | [`review/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/review/SKILL.md) | Alias for /code-review. Run all enabled review agents against target files. Use this whenever the user asks for a code review, wants feedback on their code, says "review my code", "check this before I PR", "what's wrong with this", "run the agents", or has just finished implementing a feature. | | `/review-agent` | [--since ] [--path ] [--internal] [--json] | [`review-agent/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/review-agent/SKILL.md) | Run a single named review agent against target files. Use this when the user names a specific agent (e.g. "run security-review", "check for test issues", "run js-fp-review on this file") rather than wanting the full suite. Prefer this over /code-review when only one concern is relevant or speed matters. Also used by the orchestrator for inline review checkpoints during Phase 3 implementation. | diff --git a/plugins/dev-team/skills/code-review/SKILL.md b/plugins/dev-team/skills/code-review/SKILL.md index 30fb35935..d383502f5 100644 --- a/plugins/dev-team/skills/code-review/SKILL.md +++ b/plugins/dev-team/skills/code-review/SKILL.md @@ -8,6 +8,7 @@ description: >- before commits and pull requests. argument-hint: >- [--agent ] [--since ] [--path ] [--all] [--json] + [--expand |all] [--internal] [--force --reason ""] [--static-analysis|--no-static-analysis] [--init-risks] [--background] [--pdf] @@ -64,6 +65,7 @@ Arguments: $ARGUMENTS | `--resume` | Resume a sliced run — skip slices whose section artifact already exists on disk. See [`sliced-mode.md`](sliced-mode.md). | | `--no-slice` | Escape hatch — force the legacy single-pass review even on a large full-repo scope that would otherwise auto-engage sliced mode. | | `--json` | Output aggregated JSON to **stdout** instead of prose. Contractually non-interactive (for CI): never prompts; defaults to report-only (no code modified). | +| `--expand |all` | Prose-mode only (step 7): render Tier-2 (full message + suggested fix) for the named finding-id, or for every finding with `all`, after the Tier-1 report — see step 7. A no-op under `--json` (see step 7's `--json` branch). | | `--pdf` | After the durable report is written, also render it to a sibling PDF via `hooks/lib/report_pdf.py`. See `knowledge/report-pdf-integration.md`. No-op with a message when no report file is written (`--json` or `--internal`); under `--json`, that status goes to **stderr** so stdout stays pure JSON. Additive: never changes the review's own output or exit status. | | `--internal` | This is an orchestrator-internal dispatch (`/build`'s Step 6 backstop review, `/test-improve`'s Phase 4/5 end-of-phase review loop) — skip the `.dev-team-reports/code-review.md` report write in step 7. Orthogonal to `--json`: `--internal` alone still runs the prose/fix-loop path; both sanctioned callers use `--internal` without `--json` specifically to keep the fix loop. `/build` and `/test-improve` are the only sanctioned callers of this flag today — see `knowledge/report-output-location.md` for `/ship`'s deliberate exception (writes the report by default, no `--internal`). | | `--init-risks` | Scaffold `ACCEPTED-RISKS.md` from `templates/ACCEPTED-RISKS.md.tmpl` if absent. Exits non-zero without overwriting if present. Schema: `knowledge/accepted-risks-schema.md`. | @@ -880,7 +882,15 @@ Read `knowledge/review-template.md` for the structure. **A sentence describing the JSON is not the JSON.** A completed run whose final text reads like "Aggregated JSON emitted to stdout per `--json` contract; run stops here" — with no `{...}` object actually present anywhere in that text — is a contract violation, not compliance, even though it correctly stopped rather than proceeding further. The literal final output of the turn must be the JSON object itself, not a narration of having produced it. If the next action being considered is a summary sentence announcing that the JSON was (or is about to be) emitted, that is the signal to emit the actual object instead — there is no valid end state for a `--json` run that consists of prose alone. -Otherwise (no `--json`): emit the prose summary using the Code Review Summary template in [`output-format.md`](output-format.md#code-review-summary-report-step-7-prose-mode). Append the iteration table. +Otherwise (no `--json`): emit the prose summary using the Code Review Summary template in [`output-format.md`](output-format.md#code-review-summary-report-step-7-prose-mode). For that template's per-finding listing, render this round's aggregated finding list (the same list already assembled for the `--json` branch above and for step 8 — not re-derived) with `render_tiered_findings.py` (#2170) instead of listing each finding's full message inline: + +```bash +python3 "$CLAUDE_PLUGIN_ROOT/skills/code-review/scripts/render_tiered_findings.py" --findings [--expand |all] +``` + +Pass `--expand` through exactly as the caller supplied it (omit the flag entirely when the caller did not pass one): Tier-1 lines plus the expansion hint by default; the matching Tier-2 block(s) appended after the Tier-1 report when `--expand` was given. An unknown `--expand` id: relay the script's non-zero exit and "finding-id not found" message to the user rather than silently rendering nothing or crashing. Append the iteration table. + +**Scope of this wiring: the prose-mode path only.** `--json` (this step's branch above) and `./corrections/*.json` (step 8) already read and write the full finding objects independently of this rendering path — neither branch calls `render_tiered_findings.py`, and this change does not touch either of them. In particular, **`--expand` is a no-op under `--json`**: the `--json` branch above is unconditional ("the JSON object is the ONLY thing printed to stdout... non-negotiable") and must never call `render_tiered_findings.py`, so under `--json` there is nothing for `--expand` to act on. This is enforced structurally — by the `--json` branch never reaching the tiered-rendering code path described here — not by a check inside `render_tiered_findings.py` or inside the `--json` branch itself. **Write the durable report (skip when `--internal`).** See `knowledge/report-output-location.md` for the shared write-scope convention diff --git a/plugins/dev-team/skills/code-review/output-format.md b/plugins/dev-team/skills/code-review/output-format.md index 6c097a5f6..3296e78db 100644 --- a/plugins/dev-team/skills/code-review/output-format.md +++ b/plugins/dev-team/skills/code-review/output-format.md @@ -389,17 +389,55 @@ non-empty (issue #1752) — omit it entirely on a clean run, but never omit it when there is at least one entry, regardless of how the rest of the panel scored. -After the summary, list remaining issues grouped by file, sorted by severity. Mark each with: `[confidence: none]`, `[auto-fix failed]`, or `[suggestion]`. Append the iteration table above. - -`scripts/render_tiered_findings.py` (#2170) renders that per-finding listing -in two tiers from the same in-memory finding list, no re-dispatch: Tier-1 (one -line per finding — file, line, agent, severity/confidence, first sentence of -the message, finding-id — plus an expansion hint) by default, Tier-2 (full -message + suggestedFix) for a finding-id via `--expand |all`. -Wiring it into this step's default rendering path, and documenting `--expand` -as a `/code-review` flag, is a separate step (issue #2170 Step 3.2) — this -paragraph exists so the script itself is discoverable from the skill's own -docs. +After the summary, render remaining issues (those not auto-fixed — no +confidence, auto-fix failed, or suggestion-only) with +`scripts/render_tiered_findings.py` (#2170) against this round's aggregated +finding list — the same in-memory list already assembled for the `--json` +branch and for step 8, no re-dispatch, no I/O beyond that list. By default +this renders **Tier-1 only**: one line per finding (file, line, agent, +severity/confidence, first sentence of the message, finding-id), followed by +a trailing expansion hint. This is the canonical example for this template, +superseding the old full-message-per-issue listing: + +```text +src/db/query.ts:42 [security-review] error/high — SQL injection via unescaped input. (security-review:src/db/query.ts:42:error) +src/api/handler.ts:15 [domain-review] warning/none — Abstraction leak in handler. (domain-review:src/api/handler.ts:15:warning) +Expand a finding with --expand , or --expand all for every finding. +``` + +`--expand |all` (a `/code-review` flag — see Parse Arguments and +step 7 in `skills/code-review/SKILL.md`, issue #2170 Step 3.2) additionally +renders the matching finding's (or, with `all`, every finding's) **Tier-2** +block — the full message and suggested fix — appended after the Tier-1 +report above: + +```text +=== security-review:src/db/query.ts:42:error === +SQL injection via unescaped input. User-controlled `id` is concatenated +directly into the query string without parameterization. +Suggested fix: Use a parameterized query / prepared statement instead of +string concatenation. +``` + +A zero-finding round renders a clean-pass summary line instead +(`render_tiered_findings.CLEAN_PASS_SUMMARY`), with no per-finding lines and +no expansion hint — there is nothing to expand. Confidence and +suggestion-vs-error status are visible directly in each Tier-1 line's +`severity/confidence` field (e.g. `warning/none`, `suggestion/medium`); the +old separate `[confidence: none]`/`[suggestion]` bracket tags are retired +along with the full-message listing they annotated. `[auto-fix failed]` is +not carried into this rendering — it is a step 6a fix-loop runtime outcome, +not a property of the finding itself — and remains visible in the step +6a-iv iteration log above when the loop did not converge. Append the +iteration table above. + +`--json` (step 7's other branch) and `./corrections/*.json` (step 8) are +unaffected by this rendering path — both already read/write the full +finding objects independently of `render_tiered_findings.py`, which is +prose-path only, and neither branch is touched by it. `--expand` is +therefore a no-op under `--json` (nothing there ever calls this script); see +`skills/code-review/SKILL.md` step 7 for the exact non-interference +statement. ## Override audit log entry (step 2, `--force` path) diff --git a/plugins/dev-team/tests/skills/test_code_review_tiered_output_marker.py b/plugins/dev-team/tests/skills/test_code_review_tiered_output_marker.py new file mode 100644 index 000000000..6a35e8a38 --- /dev/null +++ b/plugins/dev-team/tests/skills/test_code_review_tiered_output_marker.py @@ -0,0 +1,212 @@ +"""Content guard for code-review/SKILL.md + output-format.md's tiered +findings wiring (#2170 Step 3.2). + +`render_tiered_findings.py` (Step 3.1) is only useful once `/code-review` +step 7's prose-mode path actually calls it instead of listing each finding's +full message inline, and once `--expand` is a documented, user-facing flag. +This module pins three things mechanically: + +1. `--expand |all` is documented in the Parse Arguments table. +2. Step 7's prose-mode path (only) is wired to `render_tiered_findings.py`, + and states the `--json`/`corrections/` non-interference guarantee — + including the `--expand`-under-`--json` no-op rule — explicitly. +3. The golden-file proof required by the plan: `/code-review` has no + standalone script that builds the aggregated `--json` object or + `./corrections/*.json` for the non-sliced path (steps 5/7/8 are prose in + SKILL.md, executed by the orchestrating agent in-context, not a callable + Python module — `consolidate.py` only serves sliced-mode aggregation, a + different code path entirely). There is therefore nothing to invoke + before/after this change and diff byte-for-byte. Per the task's own + fallback instruction, the mechanical proof here is content-guard level + instead: the `--json` branch (step 7) and step 8's corrections-writing + text contain zero references to `render_tiered_findings.py` and zero + references to `--expand` -- i.e. nothing in this change touches the code + *path* those branches describe, and no flag combination involving + `--expand` can alter what either branch does. `--json --expand ` + therefore renders identically to plain `--json` by construction, not by + inspection of output bytes that don't exist to compare. +""" + +from __future__ import annotations + +import sys + +from _repo_root import REPO_ROOT + +sys.path.insert( + 0, + str(REPO_ROOT / "plugins" / "dev-team" / "skills" / "code-review" / "scripts"), +) + +import render_tiered_findings as rtf + +SKILL = REPO_ROOT / "plugins" / "dev-team" / "skills" / "code-review" / "SKILL.md" +OUTPUT_FORMAT = ( + REPO_ROOT / "plugins" / "dev-team" / "skills" / "code-review" / "output-format.md" +) + +_PARSE_ARGS_HEADING = "## Parse Arguments" +_STEP_7_HEADING = "### 7. Generate report" +_STEP_8_HEADING = "### 8. Save correction prompts for remaining issues" +_STEP_9_HEADING = "### 9. Write pre-commit gate file" +_PROSE_BRANCH_MARKER = "Otherwise (no `--json`):" + + +def _skill_text() -> str: + return SKILL.read_text(encoding="utf-8") + + +def _section(text: str, start: str, end: str) -> str: + assert start in text, f"heading not found: {start!r}" + after = text.split(start, 1)[1] + assert end in after, f"heading not found after {start!r}: {end!r}" + return after.split(end, 1)[0] + + +def _parse_arguments_section() -> str: + text = _skill_text() + return _section(text, _PARSE_ARGS_HEADING, "## Progress tracking") + + +def _step_7_section() -> str: + text = _skill_text() + return _section(text, _STEP_7_HEADING, _STEP_8_HEADING) + + +def _step_8_section() -> str: + text = _skill_text() + return _section(text, _STEP_8_HEADING, _STEP_9_HEADING) + + +def _step_7_json_branch() -> str: + """Step 7's `--json` branch only -- everything before the prose-mode + marker. This is the text `/pr --json` and any other `--json` caller's + behavior is governed by.""" + section = _step_7_section() + assert _PROSE_BRANCH_MARKER in section + return section.split(_PROSE_BRANCH_MARKER, 1)[0] + + +def _step_7_prose_branch() -> str: + section = _step_7_section() + assert _PROSE_BRANCH_MARKER in section + return section.split(_PROSE_BRANCH_MARKER, 1)[1] + + +# --- 1. --expand documented in Parse Arguments ----------------------------- + + +def test_expand_flag_documented_in_parse_arguments_table() -> None: + section = _parse_arguments_section() + assert "`--expand |all`" in section + assert "Tier-2" in section + assert "no-op under `--json`" in section + + +def test_expand_flag_in_argument_hint_frontmatter() -> None: + text = _skill_text() + frontmatter = text.split("---", 2)[1] + assert "--expand" in frontmatter + + +# --- 2. Step 7 prose-mode path wired to render_tiered_findings.py ---------- + + +def test_prose_branch_invokes_render_tiered_findings_script() -> None: + prose = _step_7_prose_branch() + assert "render_tiered_findings.py" in prose + assert "--expand" in prose + + +def test_prose_branch_passes_expand_through_unchanged() -> None: + prose = _step_7_prose_branch() + assert "Pass `--expand` through exactly as the caller supplied it" in prose + + +def test_prose_branch_handles_unknown_expand_id() -> None: + prose = _step_7_prose_branch() + assert "finding-id not found" in prose + + +# --- 3. Explicit --json / corrections/ non-interference statement ---------- + + +def test_skill_states_json_and_corrections_are_untouched() -> None: + prose = _step_7_prose_branch() + assert "Scope of this wiring: the prose-mode path only." in prose + assert "already read and write the full finding objects independently" in prose + assert "neither branch calls `render_tiered_findings.py`" in prose + + +def test_skill_states_expand_is_a_noop_under_json() -> None: + prose = _step_7_prose_branch() + assert "**`--expand` is a no-op under `--json`**" in prose + assert "must never call `render_tiered_findings.py`" in prose + assert "enforced structurally" in prose + + +# --- 4. Golden-file-equivalent proof: --json branch and step 8 untouched --- + + +def test_json_branch_never_mentions_render_tiered_findings() -> None: + """The `--json` branch (this step's other branch) must never call into + the tiered-rendering script -- this is what makes `--expand` a no-op + under `--json` *structurally*, not by a flag check anywhere.""" + json_branch = _step_7_json_branch() + assert "render_tiered_findings" not in json_branch + + +def test_json_branch_never_mentions_expand() -> None: + """No reference to `--expand` anywhere in the `--json` branch's own + text -- proof by construction that `--json --expand ` and plain + `--json` are governed by the exact same branch text, hence identical + output, since there is no unconsumed script or callable module for this + non-sliced path to diff byte-for-byte (see module docstring).""" + json_branch = _step_7_json_branch() + assert "--expand" not in json_branch + + +def test_json_branch_still_states_json_is_the_only_stdout_output() -> None: + """Pin that this step's pre-existing `--json` contract text survived + this change unweakened.""" + json_branch = _step_7_json_branch() + assert "the JSON object is the ONLY thing printed to stdout" in json_branch + assert "non-negotiable" in json_branch + + +def test_step_8_never_mentions_render_tiered_findings_or_expand() -> None: + """Step 8 (`./corrections/*.json`) is untouched by this change: no + reference to the tiered-rendering script or to `--expand` anywhere in + its text.""" + step_8 = _step_8_section() + assert "render_tiered_findings" not in step_8 + assert "--expand" not in step_8 + + +def test_step_8_still_skips_entirely_under_json() -> None: + step_8 = _step_8_section() + assert "Skip this entire step if `--json` was set." in step_8 + + +# --- output-format.md: tiered shape is the new canonical template ---------- + + +def test_output_format_shows_tier1_example_using_real_expand_hint_constant() -> None: + """Tie the doc's example directly to the script's own `EXPAND_HINT` + constant, rather than a hand-typed string that could silently drift + from the actual rendered output.""" + text = OUTPUT_FORMAT.read_text(encoding="utf-8") + assert rtf.EXPAND_HINT in text + + +def test_output_format_names_clean_pass_summary_behavior() -> None: + text = OUTPUT_FORMAT.read_text(encoding="utf-8") + assert "CLEAN_PASS_SUMMARY" in text + assert "render_tiered_findings.py" in text + + +def test_output_format_states_json_and_corrections_unaffected() -> None: + text = OUTPUT_FORMAT.read_text(encoding="utf-8") + section = text.split("## Code Review Summary report", 1)[1] + assert "no-op under `--json`" in section + assert "corrections/*.json" in section From 236d40df7a64c7fd24b7d131006c90262207cac9 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 21:58:10 +0000 Subject: [PATCH 12/14] fix(code-review): print Tier-1 alongside --expand, mirror category fallback chain Slice 3's slice-boundary review checkpoint (7 agents) found --expand printed only the Tier-2 block, dropping the Tier-1 report entirely -- contradicting three explicit statements in SKILL.md/output-format.md that Tier-2 is appended after Tier-1, not a replacement for it. main() now prints the Tier-1 report before Tier-2 content in both --expand branches; the CLI test that missed this only asserted Tier-2 content was present, closed with two new tests asserting both tiers render. base_id()'s taxonomy-tag resolution now mirrors finding_signature.py's full category -> smell -> rule -> ruleId fallback chain (was category-only), so a test-smell-review finding's id carries its smell tag instead of falling back to a bare ordinal. Also: null-normalization in render_tier1_line now matches base_id's (an explicit JSON null no longer renders as the literal string "None"), first_sentence collapses embedded whitespace/newlines so a Tier-1 entry is always exactly one line, the ordinal-suffix separator changed from ":N" to "#N" so it can't collide with a base-id segment, an Eager Test was split into two single-focus tests, and coverage gaps closed for an absent/empty suggestedFix, a dict-wrapped {"findings": [...]} input shape, and malformed JSON input. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_016uBnw1i52qEik2k9LSHa4n --- .../scripts/render_tiered_findings.py | 71 ++++++++++----- .../scripts/test_render_tiered_findings.py | 87 +++++++++++++++++-- 2 files changed, 132 insertions(+), 26 deletions(-) diff --git a/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py b/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py index 1c77af2de..c88409663 100755 --- a/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py +++ b/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py @@ -19,11 +19,13 @@ ## Finding-id scheme `agent:file:line:severity`, plus `:category` appended when the finding -carries a truthy `category`. Two or more findings in the same run that still -land on an identical base id after that get a `:0`, `:1`, ... ordinal suffix -in list order — deterministic and collision-free within a single run because -the caller's list order is already fixed for that run. IDs are not meant to -be stable *across* runs (a re-dispatched round may reorder or drop findings). +carries a truthy taxonomy tag. Two or more findings in the same run that +still land on an identical base id after that get a `#0`, `#1`, ... ordinal +suffix in list order — deterministic and collision-free within a single run +because the caller's list order is already fixed for that run (the `#` +separator, rather than `:`, is used for the ordinal so it can never collide +with a base-id segment, which is `:`-delimited). IDs are not meant to be +stable *across* runs (a re-dispatched round may reorder or drop findings). `agent` is read as `finding.get("agent")` first (the aggregated/flattened finding shape `consolidate.py` produces, `agent`-tagged per finding) falling @@ -32,6 +34,13 @@ uses for the identical purpose, so this module reads either the pre- or post-flattening shape without extra glue. +The taxonomy tag itself mirrors `finding_signature.py`'s `signature()` +fallback chain exactly: `category` → `smell` → `rule` → `ruleId`, first +truthy wins. `smell` is `test-smell-review`'s taxonomy field per +`knowledge/review-agent-output-contract.md`'s "Documented per-agent +extensions" section — without this fallback a `test-smell-review` finding's +id loses its taxonomy segment and falls back to a bare ordinal suffix. + Stdlib-only. See docs/python-hook-contract.md. """ @@ -64,14 +73,15 @@ def first_sentence(message) -> str: """The first sentence of `message`, or the whole (stripped) string when - no sentence terminator is found.""" + no sentence terminator is found. Always a single line: internal + whitespace runs (including embedded newlines) are collapsed to a single + space, so a Tier-1 entry built from this is always exactly one line.""" text = str(message or "").strip() if not text: return "" match = _SENTENCE_END_RE.search(text) - if match is None: - return text - return text[: match.start() + 1] + result = text if match is None else text[: match.start() + 1] + return " ".join(result.split()) def _finding_agent(finding: dict) -> str: @@ -83,19 +93,32 @@ def _finding_line(finding: dict) -> str: return "" if line is None else str(line) +def _finding_category(finding: dict) -> str: + """The taxonomy tag for this finding, mirroring `finding_signature.py`'s + `signature()` fallback chain exactly: `category` -> `smell` -> `rule` -> + `ruleId`, first truthy wins.""" + return str( + finding.get("category") + or finding.get("smell") + or finding.get("rule") + or finding.get("ruleId") + or "" + ) + + def base_id(finding: dict) -> str: """The finding-id before ordinal-suffix collision resolution: - `agent:file:line:severity`, plus `:category` when `category` is a - truthy value on this finding.""" + `agent:file:line:severity`, plus `:category` when the taxonomy tag + (see `_finding_category`) is truthy on this finding.""" parts = [ _finding_agent(finding), str(finding.get("file") or ""), _finding_line(finding), str(finding.get("severity") or ""), ] - category = finding.get("category") + category = _finding_category(finding) if category: - parts.append(str(category)) + parts.append(category) return ":".join(parts) @@ -103,9 +126,11 @@ def compute_ids(findings: list[dict]) -> list[str]: """Finding-id for each entry in `findings`, in list order. A base id unique in this round is used as-is. A base id shared by two or - more findings gets a `:0`, `:1`, ... ordinal suffix, assigned in list + more findings gets a `#0`, `#1`, ... ordinal suffix, assigned in list order — deterministic because the input order is already fixed for a - given run. + given run. `#` (rather than `:`, which the base id itself uses as its + segment separator) guarantees the suffix can never collide with a + base-id segment. """ bases = [base_id(f) for f in findings] counts = Counter(bases) @@ -115,7 +140,7 @@ def compute_ids(findings: list[dict]) -> list[str]: if counts[base] > 1: ordinal = next_ordinal.get(base, 0) next_ordinal[base] = ordinal + 1 - ids.append(f"{base}:{ordinal}") + ids.append(f"{base}#{ordinal}") else: ids.append(base) return ids @@ -124,11 +149,11 @@ def compute_ids(findings: list[dict]) -> list[str]: def render_tier1_line(finding: dict, finding_id: str) -> str: """One Tier-1 line: `file:line [agent] severity/confidence — ()`.""" - file_ = finding.get("file", "") + file_ = str(finding.get("file") or "") line = _finding_line(finding) agent = _finding_agent(finding) - severity = finding.get("severity", "") - confidence = finding.get("confidence", "") + severity = str(finding.get("severity") or "") + confidence = str(finding.get("confidence") or "") sentence = first_sentence(finding.get("message", "")) return f"{file_}:{line} [{agent}] {severity}/{confidence} — {sentence} ({finding_id})" @@ -198,11 +223,15 @@ def main(argv: list[str] | None = None) -> int: findings = _load_findings(args.findings) ids = compute_ids(findings) + tier1 = render_tier1_report(findings, ids) + if args.expand is None: - print(render_tier1_report(findings, ids)) + print(tier1) return 0 if args.expand == "all": + print(tier1) + print() print(render_tier2_report(findings, ids)) return 0 @@ -212,6 +241,8 @@ def main(argv: list[str] | None = None) -> int: f"render_tiered_findings: finding-id not found in this round's findings: {args.expand!r}\n" ) return 1 + print(tier1) + print() print(block) return 0 diff --git a/plugins/dev-team/tests/scripts/test_render_tiered_findings.py b/plugins/dev-team/tests/scripts/test_render_tiered_findings.py index 3aa75bf88..3094c08cf 100644 --- a/plugins/dev-team/tests/scripts/test_render_tiered_findings.py +++ b/plugins/dev-team/tests/scripts/test_render_tiered_findings.py @@ -79,6 +79,17 @@ def test_agentname_fallback(self): finding["agentName"] = "structure-review" assert rtf.base_id(finding) == "structure-review:src/auth/login.ts:42:warning" + def test_smell_field_used_as_taxonomy_tag_when_no_category(self): + """test-smell-review carries its taxonomy in `smell`, not `category` + (knowledge/review-agent-output-contract.md) — base_id must mirror + finding_signature.py's signature() fallback chain so this finding + gets a semantically-named id segment, not a bare ordinal.""" + finding = _finding(agent="test-smell-review", smell="eager-test") + assert ( + rtf.base_id(finding) + == "test-smell-review:src/auth/login.ts:42:warning:eager-test" + ) + class TestComputeIds: def test_unique_findings_keep_base_ids(self): @@ -96,17 +107,17 @@ def test_colliding_base_ids_get_ordinal_suffixes_in_list_order(self): ] ids = rtf.compute_ids(findings) assert ids == [ - "structure-review:src/auth/login.ts:42:warning:0", - "structure-review:src/auth/login.ts:42:warning:1", + "structure-review:src/auth/login.ts:42:warning#0", + "structure-review:src/auth/login.ts:42:warning#1", ] def test_three_way_collision_gets_three_ordinals(self): findings = [_finding(message=f"Message {i}.") for i in range(3)] ids = rtf.compute_ids(findings) assert ids == [ - "structure-review:src/auth/login.ts:42:warning:0", - "structure-review:src/auth/login.ts:42:warning:1", - "structure-review:src/auth/login.ts:42:warning:2", + "structure-review:src/auth/login.ts:42:warning#0", + "structure-review:src/auth/login.ts:42:warning#1", + "structure-review:src/auth/login.ts:42:warning#2", ] @@ -147,7 +158,7 @@ def test_zero_findings_renders_clean_pass_with_no_hint(self): class TestRenderExpand: - def test_expand_one_renders_only_that_findings_tier2_content(self): + def test_expand_one_renders_only_the_first_findings_tier2_content(self): findings = [ _finding(message="First distinct message.", suggestedFix="Fix A"), _finding(message="Second distinct message.", suggestedFix="Fix B"), @@ -160,12 +171,32 @@ def test_expand_one_renders_only_that_findings_tier2_content(self): assert "Second distinct message." not in block assert "Fix B" not in block + def test_expand_one_renders_only_the_second_findings_tier2_content(self): + findings = [ + _finding(message="First distinct message.", suggestedFix="Fix A"), + _finding(message="Second distinct message.", suggestedFix="Fix B"), + ] + ids = rtf.compute_ids(findings) + block_other = rtf.render_expand_one(findings, ids, ids[1]) assert "Second distinct message." in block_other assert "Fix B" in block_other assert "First distinct message." not in block_other assert "Fix A" not in block_other + def test_expand_block_omits_suggested_fix_line_when_absent(self): + finding = _finding() + del finding["suggestedFix"] + ids = rtf.compute_ids([finding]) + block = rtf.render_expand_one([finding], ids, ids[0]) + assert "Suggested fix:" not in block + + def test_expand_block_omits_suggested_fix_line_when_empty_string(self): + finding = _finding(suggestedFix="") + ids = rtf.compute_ids([finding]) + block = rtf.render_expand_one([finding], ids, ids[0]) + assert "Suggested fix:" not in block + def test_expand_all_renders_every_findings_tier2_content(self): findings = [ _finding(message="First distinct message.", suggestedFix="Fix A"), @@ -215,6 +246,16 @@ def test_cli_expand_known_id_exits_zero_with_tier2_content(self): assert r.returncode == 0 assert "Split into LoginController" in r.stdout + def test_cli_expand_known_id_also_includes_tier1_report(self): + """--expand appends Tier-2 AFTER the Tier-1 report — it does not + replace it (skills/code-review/SKILL.md step 7, output-format.md).""" + findings = [_finding()] + ids = rtf.compute_ids(findings) + r = _run("--expand", ids[0], input_text=json.dumps(findings)) + assert r.returncode == 0 + assert ids[0] in r.stdout.splitlines()[0] + assert rtf.EXPAND_HINT in r.stdout + def test_cli_expand_unknown_id_fails_clearly(self): findings = [_finding()] r = _run("--expand", "no-such-id", input_text=json.dumps(findings)) @@ -232,9 +273,43 @@ def test_cli_expand_all(self): assert "First distinct message." in r.stdout assert "Second distinct message." in r.stdout + def test_cli_expand_all_also_includes_tier1_report(self): + """--expand all appends Tier-2 AFTER the Tier-1 report — it does not + replace it (skills/code-review/SKILL.md step 7, output-format.md).""" + findings = [ + _finding(message="First distinct message.", suggestedFix="Fix A"), + _finding(message="Second distinct message.", suggestedFix="Fix B"), + ] + ids = rtf.compute_ids(findings) + r = _run("--expand", "all", input_text=json.dumps(findings)) + assert r.returncode == 0 + assert ids[0] in r.stdout.splitlines()[0] + assert ids[1] in r.stdout.splitlines()[1] + assert rtf.EXPAND_HINT in r.stdout + def test_cli_findings_from_file(self, tmp_path): findings_file = tmp_path / "findings.json" findings_file.write_text(json.dumps([_finding()])) r = _run("--findings", str(findings_file)) assert r.returncode == 0 assert "structure-review:src/auth/login.ts:42:warning" in r.stdout + + def test_cli_accepts_dict_wrapped_findings_shape(self): + """`_load_findings` accepts both a bare list and a `{"findings": + [...]}` dict-wrapped shape — the dict-wrapped shape must render + identically to the equivalent bare list.""" + findings = [_finding()] + wrapped = {"findings": findings} + r = _run(input_text=json.dumps(wrapped)) + r_bare = _run(input_text=json.dumps(findings)) + assert r.returncode == 0 + assert r.stdout == r_bare.stdout + + def test_cli_malformed_json_propagates_uncaught_error(self): + """Pins current behavior (matches the sibling script + finding_signature.py, no documented contract requires catching this): + malformed JSON is not caught — json.loads's exception propagates, + producing a non-zero exit and a traceback on stderr.""" + r = _run(input_text="{not valid json") + assert r.returncode != 0 + assert "JSONDecodeError" in r.stderr From 27d0c7191e5d3d0bc593cca851796f6bb7cc99b5 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 22:38:53 +0000 Subject: [PATCH 13/14] fix(code-review): fix backstop review findings across #2168/#2169/#2170 /code-review --internal backstop review found and fixed: - test_review_mechanics.py: multi-line Python signature parsing dropped a test's body (false no-assertion error); JS test-call regex matched RegExp .test( member access; internal-collaborator doubles missing waiver markers. - render_tiered_findings.py: unrecognized --findings shapes (topFindings, issues) silently rendered a false clean pass instead of erroring. - checkpoint_abort.py: severity-floor duplicated finding_signature.py's is_actionable with a case-sensitivity mismatch; CLI mode scaffolding deduplicated via a shared _load_json_arg helper. - finding_signature.py: extracted finding_agent/finding_category helpers, now shared with render_tiered_findings.py instead of duplicated. - compare_eval_results.py: moved to repo-root scripts/ per ADR 0032 (monorepo-dev-only tooling, not a shipped script). - build/SKILL.md: clarified the abort-check merge must use the fix loop's post-convergence findings, not the original pre-fix triggering set. - test-review.md / code-review/SKILL.md: resolved a "do not re-report" vs "report findings" wording conflict, and aligned the Hunt-scope prose with its actual test-files-only wiring. - Added regression tests and coverage-gap fixtures the panel flagged across checkpoint_abort.py, test_review_mechanics.py, and compare_eval_results.py. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01KGHpm7gtqNb8NM2Smf76u7 --- plugins/dev-team/agents/test-review.md | 5 +- plugins/dev-team/docs/eval-maintenance.md | 2 +- plugins/dev-team/scripts/checkpoint_abort.py | 120 ++++---- .../dev-team/scripts/test_review_mechanics.py | 103 ++++--- plugins/dev-team/skills/build/SKILL.md | 2 +- plugins/dev-team/skills/code-review/SKILL.md | 18 +- .../code-review/scripts/finding_signature.py | 34 ++- .../scripts/render_tiered_findings.py | 97 ++++--- .../tests/scripts/test_checkpoint_abort.py | 86 ++++++ .../scripts/test_render_tiered_findings.py | 48 ++- .../scripts/test_test_review_mechanics.py | 274 ++++++++++++++++++ .../compare_eval_results.py | 22 +- .../repo}/test_compare_eval_results.py | 69 ++++- tests/repo/test_python_floor.py | 1 - 14 files changed, 717 insertions(+), 164 deletions(-) rename {plugins/dev-team/scripts => scripts}/compare_eval_results.py (93%) rename {plugins/dev-team/tests/scripts => tests/repo}/test_compare_eval_results.py (83%) diff --git a/plugins/dev-team/agents/test-review.md b/plugins/dev-team/agents/test-review.md index 14b210680..e0514d210 100644 --- a/plugins/dev-team/agents/test-review.md +++ b/plugins/dev-team/agents/test-review.md @@ -68,7 +68,7 @@ named smell test-smell-review could own instead. Run in three phases — mechanical pre-phase first, then enumerate, then classify. Phase 0 computes the `[MECHANICAL]` half by script instead of by prose judgment; Phases 1-2 stabilize the `[JUDGMENT]` half by forcing a full enumeration pass before applying it. -**Phase 0 — Mechanical pre-phase**: This agent has no `Bash` tool (like every `*-review.md` agent), so it never runs `test_review_mechanics.py` itself — never invent, approximate, or hand-simulate a result. The caller dispatching this agent computes each file's result first (`python3 "${CLAUDE_PLUGIN_ROOT}/scripts/test_review_mechanics.py" `, the same pre-pass architecture `/code-review`'s static-analysis pre-passes use) and supplies it as context — **detected by static analysis, do not re-report**; cite its counts and messages verbatim, including the Tolerated-Deviation Hunt categories below, which it now computes. +**Phase 0 — Mechanical pre-phase**: This agent has no `Bash` tool (like every `*-review.md` agent), so it never runs `test_review_mechanics.py` itself — never invent, approximate, or hand-simulate a result. The caller dispatching this agent computes each file's result first (`python3 "${CLAUDE_PLUGIN_ROOT}/scripts/test_review_mechanics.py" `, the same pre-pass architecture `/code-review`'s static-analysis pre-passes use) and supplies it as context — **detected by static analysis, do not re-derive** (never re-run the counting/pattern-matching this script already did); cite its counts and messages verbatim, including the Tolerated-Deviation Hunt categories below, which it now computes, and DO report them as this file's own findings per the bullets below — this pre-phase result is this agent's evidence, not a generic "already covered elsewhere" envelope to fold silently into context the way step 4's cross-cutting static-analysis findings are for every other lens. - **Result present, `mechanicalFail: true`** — report its findings as this file's issues, note in the summary that Phase 1/2 was skipped and why, and move on. - **Result present, `mechanicalFail: false`** — report its `warning`/`parse-failure` findings alongside the Phase 1/2 findings below, then run Phase 1/2 as usual. @@ -186,8 +186,7 @@ If a static-analysis pre-pass has already surfaced this exact finding (e.g. via ## Tolerated-Deviation Hunt -Computed by Phase 0's `test_review_mechanics.py` pass, not a separate manual grep — cite its `tolerated-deviation-consolidation` finding when present rather than re-deriving it. The categories below are the detection specification the script implements, kept here for reference. This pass covers every core-flow file in scope (non-test source files, not -third-party). Count tolerated-deviation artifacts from the following categories: +Computed by Phase 0's `test_review_mechanics.py` pass, not a separate manual grep — cite its `tolerated-deviation-consolidation` finding when present rather than re-deriving it. The categories below are the detection specification the script implements, kept here for reference. Phase 0's actual wiring (`skills/code-review/SKILL.md` step 2b, #2169) runs this pre-pass only for test files in scope, so this Hunt currently sees test files, not "every core-flow file" — non-test source files get no Phase 0 result at all and fall through to this file's own "No result supplied for a file" rule (run Phase 1/2 as usual; say nothing about Phase 0). Extending the pre-pass to non-test core-flow files is a separate wiring change, not made here. Count tolerated-deviation artifacts from the following categories: - **Disabled tests** — `@Ignore`, `@Disabled`, `xit(`, `xdescribe(`, `test.skip(`, `it.skip(`, `[Ignore]`, `[Skip]`, `pytest.mark.skip`, `pytest.mark.xfail` with no diff --git a/plugins/dev-team/docs/eval-maintenance.md b/plugins/dev-team/docs/eval-maintenance.md index a8670dbfe..22db88faf 100644 --- a/plugins/dev-team/docs/eval-maintenance.md +++ b/plugins/dev-team/docs/eval-maintenance.md @@ -11,7 +11,7 @@ for the operational run procedure see [`eval-running-guide.md`](eval-running-gui | Fixtures | `evals/fixtures/` | Input code (deliberately good or bad) the agents review. | | Expectations | `evals/expected/*.json` | The **contract**: what a correct verdict looks like per fixture/agent. | | Grader | `scripts/eval_grade.py` | Deterministic, model-free: compares recorded actuals to expectations. | -| Regression diff | `plugins/dev-team/scripts/compare_eval_results.py` | Diffs two `--actuals` result files against `evals/expected/*.json`, gating on true/false-positive-proxy count regression. Shipped-tree location mirrors `eval_ablation.py`'s existing precedent; both are monorepo-dev-only despite the placement. | +| Regression diff | `scripts/compare_eval_results.py` | Diffs two `--actuals` result files against `evals/expected/*.json`, gating on true/false-positive-proxy count regression. Repo-root placement (ADR 0032 category 2, monorepo-dev-only) next to `eval_grade.py` — not shipped, unlike `eval_ablation.py`, which ships only for its unrelated generic `--find-latest` reader mode. | | Variance | `scripts/eval_variance.py` | Aggregates K trials → pass@k, flap rate, quarantine. | | Trend | `.claude/metrics/eval-variance.jsonl` | Append-only stability history (metrics only). | | Semver contract | `scripts/eval_semver_classify.sh` | The eval corpus IS the version contract (#101). | diff --git a/plugins/dev-team/scripts/checkpoint_abort.py b/plugins/dev-team/scripts/checkpoint_abort.py index 9467a72df..4e9a9468a 100755 --- a/plugins/dev-team/scripts/checkpoint_abort.py +++ b/plugins/dev-team/scripts/checkpoint_abort.py @@ -35,6 +35,19 @@ import json import sys from pathlib import Path +from typing import Any + +# `finding_signature.py` lives under `skills/code-review/scripts/`; this +# script lives under `scripts/` — both are children of the plugin root +# (`plugins/dev-team/`). Imported (not re-typed) so `compute_round_outcome`'s +# severity floor can never silently diverge from `finding_signature.py`'s +# own `is_actionable` — see the comment on `_is_blocking_finding` below for +# the drift this closes. +_FINDING_SIGNATURE_DIR = Path(__file__).resolve().parents[1] / "skills" / "code-review" / "scripts" +if str(_FINDING_SIGNATURE_DIR) not in sys.path: + sys.path.insert(0, str(_FINDING_SIGNATURE_DIR)) + +from finding_signature import is_actionable # The bar this script's abort decision applies. Deliberately STRICTER than # `skills/code-review/SKILL.md` step 6a's "Severity floor (rounds >= 2)" rule @@ -50,16 +63,22 @@ QUALIFYING_SEVERITY = "error" QUALIFYING_CONFIDENCE = "high" -# The severity floor `compute_round_outcome` filters `findings` by — the same -# bar stated once in `skills/code-review/SKILL.md` step 6a and restated at +# `compute_round_outcome` filters `findings` by `finding_signature.py`'s +# `is_actionable` — the same severity floor stated once in +# `skills/code-review/SKILL.md` step 6a and restated at # `skills/build/SKILL.md`'s round-ledger section: only a finding at # `error`/`warning` severity AND `high`/`medium` confidence keeps a round # from converging. Suggestion-tier and low-confidence findings are "logged, -# never chased" and must not block. Deliberately looser than +# never chased" and must not block. This bar is deliberately looser than # QUALIFYING_SEVERITY/QUALIFYING_CONFIDENCE above, which is a different bar -# for a different decision (see the comment on those constants). -BLOCKING_SEVERITIES = frozenset({"error", "warning"}) -BLOCKING_CONFIDENCES = frozenset({"high", "medium"}) +# for a different decision (see the comment on those constants) — but it is +# the SAME bar as `is_actionable`'s, so it is imported rather than +# reimplemented as a second local constant pair. A prior local copy here +# (`BLOCKING_SEVERITIES`/`BLOCKING_CONFIDENCES`) compared case-sensitively +# while `is_actionable` lowercases first, so a differently-cased +# `"Error"`/`"High"` finding kept `finding_signature`'s fix loop going while +# this script's `compute_round_outcome` reported the round clean — the two +# copies had silently drifted. class CheckpointAbortError(ValueError): @@ -167,13 +186,10 @@ def decide_abort(cheap_results: list, ordered_lenses: list) -> dict: def _is_blocking_finding(finding) -> bool: """A finding counts toward `compute_round_outcome`'s blocked/pass verdict - only at `BLOCKING_SEVERITIES`/`BLOCKING_CONFIDENCES` — the same - severity-floor bar as `skills/code-review/SKILL.md` step 6a.""" - return ( - isinstance(finding, dict) - and finding.get("severity") in BLOCKING_SEVERITIES - and finding.get("confidence") in BLOCKING_CONFIDENCES - ) + only when `finding_signature.is_actionable` says so — the identical + severity-floor bar `skills/code-review/SKILL.md` step 6a already applies, + imported rather than duplicated (see the comment above the constants).""" + return isinstance(finding, dict) and is_actionable(finding) def compute_round_outcome(aborted: bool, redispatched: bool, findings: list) -> dict: @@ -192,8 +208,8 @@ def compute_round_outcome(aborted: bool, redispatched: bool, findings: list) -> round cannot report a clean pass while the lenses it deferred at abort time never actually ran. Otherwise, ``findings`` is first filtered down to the ones that clear the shared severity floor - (``BLOCKING_SEVERITIES``/``BLOCKING_CONFIDENCES`` — ``error``/``warning`` - severity at ``high``/``medium`` confidence): any such finding present -> + (``finding_signature.is_actionable`` — ``error``/``warning`` severity at + ``high``/``medium`` confidence): any such finding present -> ``"blocked"``; none -> ``"pass"``. Suggestion-tier and low-confidence findings never block on their own, matching ``skills/code-review/SKILL.md`` step 6a's floor. @@ -317,24 +333,36 @@ def _validate_merge_input(data) -> None: ) -def _run_abort_mode(args) -> int: +def _load_json_arg(path_or_dash: str, invalid_json_message: str) -> tuple[Any, int | None]: + """Shared read-then-parse step for every `--mode`'s CLI entry point: + read `path_or_dash` via `_read_text` (a file path or `-` for stdin), + then `json.loads` it. Prints a formatted `checkpoint_abort.py: ...` + error to stderr and returns `(None, 1)` on either an `OSError` (read + failure) or a `json.JSONDecodeError` (parse failure); returns + `(data, None)` on success. `invalid_json_message` is the full trailing + clause after `"checkpoint_abort.py: "` for a parse failure — each + `_run_*_mode` caller supplies its own wording (e.g. "cheap-tier results + are not valid JSON") so this shared helper doesn't have to guess a + caller's grammar. Extracted from three near-identical copies of this + same 2-step scaffold (structure-review, #2168 backstop review).""" try: - raw = _read_text(args.cheap_results_from) + raw = _read_text(path_or_dash) except OSError as exc: - print( - f"checkpoint_abort.py: cannot read {args.cheap_results_from}: {exc}", - file=sys.stderr, - ) - return 1 - + print(f"checkpoint_abort.py: cannot read {path_or_dash}: {exc}", file=sys.stderr) + return None, 1 try: - cheap_results = json.loads(raw) + return json.loads(raw), None except json.JSONDecodeError as exc: - print( - f"checkpoint_abort.py: cheap-tier results are not valid JSON: {exc}", - file=sys.stderr, - ) - return 1 + print(f"checkpoint_abort.py: {invalid_json_message}: {exc}", file=sys.stderr) + return None, 1 + + +def _run_abort_mode(args) -> int: + cheap_results, err = _load_json_arg( + args.cheap_results_from, "cheap-tier results are not valid JSON" + ) + if err is not None: + return err try: result = decide_abort(cheap_results, args.lenses) @@ -350,20 +378,9 @@ def _run_abort_mode(args) -> int: def _run_outcome_mode(args) -> int: - try: - raw = _read_text(args.from_path) - except OSError as exc: - print(f"checkpoint_abort.py: cannot read {args.from_path}: {exc}", file=sys.stderr) - return 1 - - try: - data = json.loads(raw) - except json.JSONDecodeError as exc: - print( - f"checkpoint_abort.py: --mode outcome input is not valid JSON: {exc}", - file=sys.stderr, - ) - return 1 + data, err = _load_json_arg(args.from_path, "--mode outcome input is not valid JSON") + if err is not None: + return err try: _validate_outcome_input(data) @@ -381,20 +398,9 @@ def _run_outcome_mode(args) -> int: def _run_merge_mode(args) -> int: - try: - raw = _read_text(args.from_path) - except OSError as exc: - print(f"checkpoint_abort.py: cannot read {args.from_path}: {exc}", file=sys.stderr) - return 1 - - try: - data = json.loads(raw) - except json.JSONDecodeError as exc: - print( - f"checkpoint_abort.py: --mode merge input is not valid JSON: {exc}", - file=sys.stderr, - ) - return 1 + data, err = _load_json_arg(args.from_path, "--mode merge input is not valid JSON") + if err is not None: + return err try: _validate_merge_input(data) diff --git a/plugins/dev-team/scripts/test_review_mechanics.py b/plugins/dev-team/scripts/test_review_mechanics.py index 8a1f21d3d..775d94db1 100755 --- a/plugins/dev-team/scripts/test_review_mechanics.py +++ b/plugins/dev-team/scripts/test_review_mechanics.py @@ -85,12 +85,15 @@ is a different list, not a duplicate. The Tolerated-Deviation Hunt (`_check_tolerated_deviation`) runs against -whichever single file the CLI is given, test or non-test — it does NOT -restrict itself to "core-flow, non-test source" the way `test-review.md`'s -prose scopes the hunt. Resolving that scoping gap (which files this check -should actually run against) is deferred to Step 2.3's wiring of this -script into test-review's protocol; this script's own per-file CLI shape -is unopinionated about which files it's called with. +whichever single file the CLI is given, test or non-test — this script's +own per-file CLI shape is unopinionated about which files it's called +with. Step 2.3's actual wiring (`skills/code-review/SKILL.md` step 2b) +calls this script only for test files in scope, so in practice the Hunt +currently sees test files, not the "core-flow, non-test source" scope +`test-review.md`'s prose describes — non-test files get no Phase 0 result +and fall through to that agent's own manual-judgment fallback. Extending +the wiring to also cover non-test core-flow files is a separate change, +tracked at the wiring layer (`test-review.md`'s Hunt section), not here. Stdlib-only (ADR 0014/0015). See docs/python-hook-contract.md. """ @@ -160,9 +163,13 @@ #: extraction downstream is identical for both forms. The `.each(...)` #: table itself may contain one level of nested parens (e.g. a function #: call inside the table) but no more — a reasonable approximation of the -#: common forms, not a full parser. +#: common forms, not a full parser. The leading `(? int: return text.count("\n", 0, idx) + 1 +def _consume_escape(out: list[str], text: str, j: int) -> int: + """Blank a backslash-escape pair (`text[j:j + 2]`, `text[j] == "\\\\"`) + in `out`, one character at a time so an embedded newline is preserved + (never blanked) and line numbers stay aligned with `text`. Returns the + index just past the pair. Split out of `_mask_code`'s quoted-string + branch to keep that branch's `while` body at one level of nesting.""" + if text[j] != "\n": + out[j] = " " + if text[j + 1] != "\n": + out[j + 1] = " " + return j + 2 + + def _mask_code(text: str) -> str: """Same-length copy of `text` with string/template/char literals and `//`/`/* */` comments blanked to spaces (newlines preserved), so @@ -236,11 +256,7 @@ def _mask_code(text: str) -> str: j = i + 1 while j < n: if text[j] == "\\" and j + 1 < n: - if text[j] != "\n": - out[j] = " " - if text[j + 1] != "\n": - out[j + 1] = " " - j += 2 + j = _consume_escape(out, text, j) continue closing = text[j] == quote if text[j] != "\n": @@ -382,18 +398,32 @@ def _extract_regions(text: str, masked: str, lang: str) -> list[dict]: return [] -def _python_test_regions(text: str) -> list[tuple[int, str]]: +def _python_test_regions(text: str, masked: str) -> list[tuple[int, str]]: """`(line_no, body_text)` for each `def test_*(...)`/`async def - test_*(...)` — body is every line after the `def` line indented deeper - than it, up to the first dedent or EOF (blank lines don't count as a - dedent).""" + test_*(...)` — body is every line after the parameter list's closing + `)` line, indented deeper than the `def` line, up to the first dedent + or EOF (blank lines don't count as a dedent). + + The closing paren is found by depth-counting over `masked` (like every + other region helper in this module) rather than assumed to be on the + `def` line itself — a black-formatted multi-line signature (`def + test_x(\\n tmp_path, cfg\\n):`) puts the body's first line right + after a `):` line whose OWN indent equals the `def` line's, which a + naive "next line" walk would misread as an immediate dedent and return + an empty body (a false no-assertion finding on a test that does + assert).""" lines = text.split("\n") regions = [] for m in _PY_TEST_DEF_RE.finditer(text): indent = m.group(1) + open_idx = m.end() - 1 # the `(` the regex itself just matched + close_idx = _matching_close_index(text, masked, open_idx, "(", ")") + if close_idx is None: + raise ParseFailure("unbalanced parens in a def test_*(...) signature") def_line_idx = text.count("\n", 0, m.start()) + sig_close_line_idx = text.count("\n", 0, close_idx) body_lines = [] - for line in lines[def_line_idx + 1 :]: + for line in lines[sig_close_line_idx + 1 :]: if line.strip() == "": body_lines.append(line) continue @@ -598,9 +628,9 @@ def _check_no_assertion(regions: list[dict], lang: str) -> list[dict]: return findings -def _check_no_assertion_python(text: str) -> list[dict]: +def _check_no_assertion_python(text: str, masked: str) -> list[dict]: findings = [] - for line_no, body in _python_test_regions(text): + for line_no, body in _python_test_regions(text, masked): if not _ASSERTION_RE.search(body): findings.append( _finding("no-assertion", "error", line_no, "Test has no assertion call — zero regression protection.") @@ -842,6 +872,21 @@ def _translate_double_detector_findings( _GATING_CATEGORIES = frozenset({"no-assertion", "internal-collaborator-doubling"}) +def _parse_failure_result(file_path: Path, message: str) -> dict: + """The shared `analyze_file` early-return shape for a file that could + not be read/decoded or whose test-region boundaries never closed + before EOF — a `parse-failure` finding, `mechanicalFail: False` (the + file falls through to the qualitative pass), and the doubling check + never ran.""" + return { + "file": str(file_path), + "mechanicalFail": False, + "findings": [_finding("parse-failure", None, _UNKNOWN_LINE, message)], + "skippedQualitative": False, + "doublingCheckRan": False, + } + + def analyze_file( root: Path, file_path: Path, @@ -862,20 +907,14 @@ def analyze_file( try: text = file_path.read_text(encoding="utf-8") except (OSError, UnicodeDecodeError) as exc: - return { - "file": str(file_path), - "mechanicalFail": False, - "findings": [_finding("parse-failure", None, _UNKNOWN_LINE, f"Could not read/decode file: {exc}")], - "skippedQualitative": False, - "doublingCheckRan": False, - } + return _parse_failure_result(file_path, f"Could not read/decode file: {exc}") findings: list[dict] = [] try: masked = _mask_code(text) regions = _extract_regions(text, masked, lang) if lang else [] if lang == "python": - findings += _check_no_assertion_python(text) + findings += _check_no_assertion_python(text, masked) elif lang in ("js_ts", "csharp", "java"): findings += _check_no_assertion(regions, lang) findings += _check_missing_await(regions, lang) @@ -885,13 +924,7 @@ def analyze_file( findings += _check_reflection_primary_strategy(text, lang) findings += _check_tolerated_deviation(text) except ParseFailure as exc: - return { - "file": str(file_path), - "mechanicalFail": False, - "findings": [_finding("parse-failure", None, _UNKNOWN_LINE, str(exc))], - "skippedQualitative": False, - "doublingCheckRan": False, - } + return _parse_failure_result(file_path, str(exc)) doubling_findings, doubling_check_ran = _translate_double_detector_findings(root, file_path, double_detector_runner) findings += doubling_findings diff --git a/plugins/dev-team/skills/build/SKILL.md b/plugins/dev-team/skills/build/SKILL.md index f00086c24..8ce3fbe4c 100644 --- a/plugins/dev-team/skills/build/SKILL.md +++ b/plugins/dev-team/skills/build/SKILL.md @@ -205,7 +205,7 @@ Work each step **one behavior at a time** — never all the code then all the te - **Clear freeze scope (issue #865).** When every step under the slice is `[x]` and freeze was engaged for it (dispatch bookkeeping above), run `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/build_slice_scope.py clear --hooks-dir /.claude/hooks` before starting the next slice. A slice that never engaged freeze has nothing to clear. 6. **Slice review checkpoint (batched).** **Dispatch-capability gate (re-confirm here — issue #1461):** before this checkpoint dispatches, re-verify the `Agent`/`Task` tool is present. If it is not, STOP per the Orchestrator constraints above — do not self-apply the batched checkpoint's checklist inline; report the missing capability and halt rather than checking off the slice. Otherwise, when every step under the current slice is `[x]` **and** the slice had any deferred `standard` (or unspecified) steps, run **one** review pass over the slice's accumulated changed files: the static self-heal pass first (`references/static-self-heal.md`), then `/review-agent spec-compliance-review --internal`, then **the quality lenses the resolver selects** — `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/select_lenses.py --files ` — dispatched **cheap-first (non-opus before opus-tier)**, **plus `refactor-opportunity-review` dispatched by name** (#1976). That lens declares `Scope: on-demand`, so the resolver no longer returns it; this once-per-slice dispatch is its post-GREEN home, and it is the ONLY place `/build` runs it — do not also dispatch it per behavior at the REFACTOR phase (that would spend more, not less, than the per-diff panel slot it replaced) and do not treat the resolver's silence as "this lens was dropped". Its subject is a slice's accumulated shape — semantic vs. structural duplication across everything the slice touched — which is visible here and not in any single behavior's diff. **Abort check (#2168).** Before dispatching the remaining opus-tier lenses, run `checkpoint_abort.py`'s abort check on the cheap-tier lenses' (from the cheap-first dispatch above) results — see the shared rule below (applies to both checkpoint fix loops above). Either way, apply the same review-fix loop (up to 5 iterations; escalate if it doesn't converge). `trivial`-only and all-`complex` slices have nothing to batch — skip this pass. Then **record the checkpoint outcome** (sub-step 7). - **Abort check (#2168) — applies to both checkpoint fix loops above (sub-steps 4 and 6).** Once the cheap-tier lenses return, invoke `checkpoint_abort.py` (`python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --cheap-results-from --lenses `) before dispatching the remaining opus-tier lenses. On `aborted: true`: skip dispatching the deferred opus-tier lenses this round, run the review-fix loop against the cheap-tier findings only, and once it converges, dispatch the deferred lenses and merge their findings into the round's finding set via `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --mode merge --from ` (`checkpoint_abort.merge_findings(existing, new)`'s CLI form) — the same dedup-by-`(agent, file, line, severity, message)` function Step 1.1 already proved order-independent. This checkpoint's report must name every deferred lens and the triggering finding (`checkpoint_abort.py`'s `deferredLenses`/`triggeringFinding`/`triggeringAgent`). The round's pass/blocked outcome always comes from calling `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --mode outcome --from ` (`compute_round_outcome(aborted, redispatched, findings)`'s CLI form) — never independently reimplemented — so a round that aborted and never re-dispatched its deferred lenses cannot report a pass. On `aborted: false`, dispatch every opus-tier lens normally. + **Abort check (#2168) — applies to both checkpoint fix loops above (sub-steps 4 and 6).** Once the cheap-tier lenses return, invoke `checkpoint_abort.py` (`python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --cheap-results-from --lenses `) before dispatching the remaining opus-tier lenses. On `aborted: true`: skip dispatching the deferred opus-tier lenses this round, run the review-fix loop against the cheap-tier findings only, and once it converges, dispatch the deferred lenses and merge findings into the round's finding set via `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --mode merge --from ` (`checkpoint_abort.merge_findings(existing, new)`'s CLI form) — the same dedup-by-`(agent, file, line, severity, message)` function Step 1.1 already proved order-independent. **`existing` here is the cheap-tier findings STILL OUTSTANDING after the fix loop converged (whatever the fix loop's own re-classification left as unresolved — typically none, if the fix loop actually fixed the triggering finding), never the original pre-fix-loop cheap-tier result set.** The triggering finding is `error`/`high` by construction (`checkpoint_abort.py`'s `QUALIFYING_SEVERITY`/`QUALIFYING_CONFIDENCE`), so passing the unfixed original set as `existing` would keep it in the merged set regardless of the fix loop's outcome, and `--mode outcome` below would then report `blocked` on every aborted round no matter what the deferred lenses found (backstop review, #2168) — the fix loop's own result is what `existing` must reflect. `new` is the deferred lenses' findings, dispatched fresh at the fix loop's final content. This checkpoint's report must name every deferred lens and the triggering finding (`checkpoint_abort.py`'s `deferredLenses`/`triggeringFinding`/`triggeringAgent`). The round's pass/blocked outcome always comes from calling `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/checkpoint_abort.py --mode outcome --from ` (`compute_round_outcome(aborted, redispatched, findings)`'s CLI form) — never independently reimplemented — so a round that aborted and never re-dispatched its deferred lenses cannot report a pass. On `aborted: false`, dispatch every opus-tier lens normally. **Verification-mode re-dispatch (#1628) — applies to both checkpoint fix loops above (sub-steps 4 and 6).** When a checkpoint re-dispatches an agent to CONFIRM a fix (rather than to discover new problems), send the narrowed verification payload — the finding, the fix diff hunks ± ~20 lines, and the agent's lens definition, never the full file set — with the mandatory `insufficient-context` escape, and resolve the agent's verification tier with `python3 "$CLAUDE_PLUGIN_ROOT/scripts/verify_tier.py" --agent `. The payload contract, the escape's escalation path, and the declared-never-inferred tier-down rule are stated once in [`../../knowledge/verification-mode.md`](../../knowledge/verification-mode.md) and are not restated here. diff --git a/plugins/dev-team/skills/code-review/SKILL.md b/plugins/dev-team/skills/code-review/SKILL.md index d383502f5..0b822ab95 100644 --- a/plugins/dev-team/skills/code-review/SKILL.md +++ b/plugins/dev-team/skills/code-review/SKILL.md @@ -65,7 +65,7 @@ Arguments: $ARGUMENTS | `--resume` | Resume a sliced run — skip slices whose section artifact already exists on disk. See [`sliced-mode.md`](sliced-mode.md). | | `--no-slice` | Escape hatch — force the legacy single-pass review even on a large full-repo scope that would otherwise auto-engage sliced mode. | | `--json` | Output aggregated JSON to **stdout** instead of prose. Contractually non-interactive (for CI): never prompts; defaults to report-only (no code modified). | -| `--expand |all` | Prose-mode only (step 7): render Tier-2 (full message + suggested fix) for the named finding-id, or for every finding with `all`, after the Tier-1 report — see step 7. A no-op under `--json` (see step 7's `--json` branch). | +| `--expand |all` | Prose-mode only (step 7): render Tier-2 (full message + suggested fix) for the named finding-id, or for every finding with `all`, after the Tier-1 report — see step 7. A no-op under `--json` (see step 7's `--json` branch). **Only meaningful within the SAME run that computed the ids** — pass it alongside `--since`/`--path`/etc. in one invocation once you already know a specific id, e.g. because the operating Claude session read the prior Tier-1 output and is now re-invoking this skill with the same scope plus `--expand ` still in the same conversation; that path never re-dispatches anything beyond what the scope would have dispatched anyway. A cold, separate `/code-review --expand ` run with no memory of where that id came from IS a full re-dispatch of the panel (steps 1-6 run in full, same as any other invocation) and the id is not guaranteed to still exist or mean the same finding — `render_tiered_findings.py`'s own docstring says ids are not stable across runs. `--expand` never triggers a SECOND panel dispatch on top of an already-running one; it only changes step 7's rendering of the one panel a given invocation already ran. | | `--pdf` | After the durable report is written, also render it to a sibling PDF via `hooks/lib/report_pdf.py`. See `knowledge/report-pdf-integration.md`. No-op with a message when no report file is written (`--json` or `--internal`); under `--json`, that status goes to **stderr** so stdout stays pure JSON. Additive: never changes the review's own output or exit status. | | `--internal` | This is an orchestrator-internal dispatch (`/build`'s Step 6 backstop review, `/test-improve`'s Phase 4/5 end-of-phase review loop) — skip the `.dev-team-reports/code-review.md` report write in step 7. Orthogonal to `--json`: `--internal` alone still runs the prose/fix-loop path; both sanctioned callers use `--internal` without `--json` specifically to keep the fix loop. `/build` and `/test-improve` are the only sanctioned callers of this flag today — see `knowledge/report-output-location.md` for `/ship`'s deliberate exception (writes the report by default, no `--internal`). | | `--init-risks` | Scaffold `ACCEPTED-RISKS.md` from `templates/ACCEPTED-RISKS.md.tmpl` if absent. Exits non-zero without overwriting if present. Schema: `knowledge/accepted-risks-schema.md`. | @@ -244,10 +244,18 @@ Protocol) needs each file's own `mechanicalFail`/findings result supplied as that file's context — the agent has no `Bash` tool and never runs this script itself. Keep the per-file results keyed by file path when assembling step 4's context so each file's `test-review` dispatch gets its own result, -not the whole batch's. Each result's `findings` array merges into step 4's -static-analysis context using the same envelope and the same "detected by -static analysis — do not re-report, focus on semantic concerns" framing as -the two pre-passes above. +not the whole batch's. Pass each result to `test-review` as its Phase 0 +input using `agents/test-review.md`'s own framing ("detected by static +analysis, do not re-derive" — the agent still reports it as this file's own +finding when `mechanicalFail` is true, per that file's Phase 0 bullets) — +**not** the generic "detected by static analysis — do not re-report, focus +on semantic concerns" envelope the two pre-passes above use for every other +agent. That generic framing is correct for `repo_invariants.py`/ +`internal_double_detector.py`'s findings, which every dispatched agent +receives as already-covered context to fold silently into a semantic +review; it would be wrong here, since `test-review` is this pre-pass's +sole intended reporter, not one of several agents absorbing someone else's +finding. **Pass `--files` (#1629).** Several checks are scoped to the changeset, because the conventions they enforce are "required going forward, do not diff --git a/plugins/dev-team/skills/code-review/scripts/finding_signature.py b/plugins/dev-team/skills/code-review/scripts/finding_signature.py index fb2791945..350af73ae 100755 --- a/plugins/dev-team/skills/code-review/scripts/finding_signature.py +++ b/plugins/dev-team/skills/code-review/scripts/finding_signature.py @@ -115,6 +115,30 @@ def _normalize_path(value) -> str: return text +def finding_agent(finding: dict) -> str: + """The reporting agent's name: `agent` (the aggregated/flattened finding + shape `consolidate.py` produces) falling back to `agentName` (the raw + per-agent-result field name). Shared by `signature()` below and by + `render_tiered_findings.py`'s finding-id scheme (#2170) — the same + fallback, one place, so a future field-name change can't silently + desync the two.""" + return str(finding.get("agent") or finding.get("agentName") or "") + + +def finding_category(finding: dict) -> str: + """The taxonomy tag for this finding: `category` -> `smell` -> `rule` -> + `ruleId`, first truthy wins (see `signature()`'s docstring for why each + fallback exists). Shared by `signature()` below and by + `render_tiered_findings.py`'s finding-id scheme (#2170).""" + return str( + finding.get("category") + or finding.get("smell") + or finding.get("rule") + or finding.get("ruleId") + or "" + ) + + def signature(finding: dict) -> str: """Stable identity hash for one finding: agent, file, category, and the normalized message. Deliberately excludes the line number. @@ -138,15 +162,9 @@ def signature(finding: dict) -> str: between `category` and `rule` in the fallback chain — after the genuinely canonical field, before the linter-style fallbacks. """ - agent = str(finding.get("agent") or finding.get("agentName") or "") + agent = finding_agent(finding) path = _normalize_path(finding.get("file")) - category = str( - finding.get("category") - or finding.get("smell") - or finding.get("rule") - or finding.get("ruleId") - or "" - ) + category = finding_category(finding) message = normalize_message(finding.get("message")) payload = f"{agent.lower()}\x1f{path}\x1f{category.lower()}\x1f{message}" return hashlib.sha256(payload.encode("utf-8")).hexdigest() diff --git a/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py b/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py index c88409663..3c19b00c4 100755 --- a/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py +++ b/plugins/dev-team/skills/code-review/scripts/render_tiered_findings.py @@ -27,19 +27,15 @@ with a base-id segment, which is `:`-delimited). IDs are not meant to be stable *across* runs (a re-dispatched round may reorder or drop findings). -`agent` is read as `finding.get("agent")` first (the aggregated/flattened -finding shape `consolidate.py` produces, `agent`-tagged per finding) falling -back to `finding.get("agentName")` (the raw per-agent-result field name) — -the same two-field fallback `finding_signature.py`'s `signature()` already -uses for the identical purpose, so this module reads either the pre- or -post-flattening shape without extra glue. - -The taxonomy tag itself mirrors `finding_signature.py`'s `signature()` -fallback chain exactly: `category` → `smell` → `rule` → `ruleId`, first -truthy wins. `smell` is `test-smell-review`'s taxonomy field per -`knowledge/review-agent-output-contract.md`'s "Documented per-agent -extensions" section — without this fallback a `test-smell-review` finding's -id loses its taxonomy segment and falls back to a bare ordinal suffix. +`agent` and the taxonomy tag are both read via `finding_signature.py`'s +`finding_agent()`/`finding_category()` — imported, not re-typed, so this +module's finding-id scheme can never silently drift from `signature()`'s own +identity hash (`agent`: the aggregated/flattened `agent` field falling back +to the raw per-agent-result `agentName`; taxonomy: `category` → `smell` → +`rule` → `ruleId`, first truthy wins — `smell` is `test-smell-review`'s +taxonomy field per `knowledge/review-agent-output-contract.md`'s "Documented +per-agent extensions" section, without which a `test-smell-review` finding's +id would lose its taxonomy segment and fall back to a bare ordinal suffix). Stdlib-only. See docs/python-hook-contract.md. """ @@ -53,6 +49,18 @@ from collections import Counter from pathlib import Path +# Reach the sibling finding_signature.py regardless of cwd or sys.path mode +# (same house pattern as consolidate.py's `from ledger import raw_dir`) so +# the agent/category fallback chains below are imported, not re-typed — +# `finding_signature.signature()` and this module's finding-id scheme must +# never drift apart (#2170 backstop review). +_HERE = Path(__file__).resolve().parent +if str(_HERE) not in sys.path: + sys.path.insert(0, str(_HERE)) + +from finding_signature import finding_agent as _finding_agent +from finding_signature import finding_category as _finding_category + #: Rendered instead of any per-finding line when a round has zero findings — #: there is nothing to list and nothing to expand. CLEAN_PASS_SUMMARY = "Clean pass: 0 findings this round — nothing to expand." @@ -84,28 +92,11 @@ def first_sentence(message) -> str: return " ".join(result.split()) -def _finding_agent(finding: dict) -> str: - return str(finding.get("agent") or finding.get("agentName") or "") - - def _finding_line(finding: dict) -> str: line = finding.get("line") return "" if line is None else str(line) -def _finding_category(finding: dict) -> str: - """The taxonomy tag for this finding, mirroring `finding_signature.py`'s - `signature()` fallback chain exactly: `category` -> `smell` -> `rule` -> - `ruleId`, first truthy wins.""" - return str( - finding.get("category") - or finding.get("smell") - or finding.get("rule") - or finding.get("ruleId") - or "" - ) - - def base_id(finding: dict) -> str: """The finding-id before ordinal-suffix collision resolution: `agent:file:line:severity`, plus `:category` when the taxonomy tag @@ -149,13 +140,13 @@ def compute_ids(findings: list[dict]) -> list[str]: def render_tier1_line(finding: dict, finding_id: str) -> str: """One Tier-1 line: `file:line [agent] severity/confidence — ()`.""" - file_ = str(finding.get("file") or "") + file_path = str(finding.get("file") or "") line = _finding_line(finding) agent = _finding_agent(finding) severity = str(finding.get("severity") or "") confidence = str(finding.get("confidence") or "") sentence = first_sentence(finding.get("message", "")) - return f"{file_}:{line} [{agent}] {severity}/{confidence} — {sentence} ({finding_id})" + return f"{file_path}:{line} [{agent}] {severity}/{confidence} — {sentence} ({finding_id})" def render_tier1_report(findings: list[dict], ids: list[str]) -> str: @@ -196,14 +187,42 @@ def render_expand_one(findings: list[dict], ids: list[str], finding_id: str) -> return render_tier2_block(findings[index], ids[index]) +class UnrecognizedFindingsShape(ValueError): + """Raised by `_load_findings` when the parsed JSON is neither a bare + list nor a dict carrying a recognized finding-list key. Regression + (backstop review, #2170): the previous version fell through to an + empty list for ANY unrecognized dict shape — including the full + aggregated `--json` object (`output-format.md`'s `topFindings` key, + not `findings`) and a raw per-agent `{status, issues, summary}` + result (`issues`, not `findings`) — which `main` then rendered as + `CLEAN_PASS_SUMMARY` even though real findings were present. A + misread shape must surface as an error, never as a silent clean + pass.""" + + +#: Recognized dict keys for a finding list, in preference order: the +#: documented bare shape this script's own CLI help describes +#: (`findings`), the actual aggregated `--json` object's consolidated list +#: (`topFindings`, `output-format.md`), and a raw per-agent +#: `{status, issues, summary}` result (`issues`, +#: `knowledge/review-agent-output-contract.md`). +_FINDING_LIST_KEYS = ("findings", "topFindings", "issues") + + def _load_findings(path: str) -> list[dict]: raw = sys.stdin.read() if path == "-" else Path(path).read_text(encoding="utf-8") data = json.loads(raw) if raw.strip() else [] + if isinstance(data, list): + return [f for f in data if isinstance(f, dict)] if isinstance(data, dict): - data = data.get("findings", []) - if not isinstance(data, list): - return [] - return [f for f in data if isinstance(f, dict)] + for key in _FINDING_LIST_KEYS: + value = data.get(key) + if isinstance(value, list): + return [f for f in value if isinstance(f, dict)] + raise UnrecognizedFindingsShape( + f"--findings input is a dict with none of {_FINDING_LIST_KEYS} as a list-valued key" + ) + raise UnrecognizedFindingsShape(f"--findings input is neither a list nor a dict (got {type(data).__name__})") def main(argv: list[str] | None = None) -> int: @@ -220,7 +239,11 @@ def main(argv: list[str] | None = None) -> int: ) args = parser.parse_args(argv) - findings = _load_findings(args.findings) + try: + findings = _load_findings(args.findings) + except (json.JSONDecodeError, UnrecognizedFindingsShape) as exc: + sys.stderr.write(f"render_tiered_findings: cannot interpret --findings input: {exc}\n") + return 1 ids = compute_ids(findings) tier1 = render_tier1_report(findings, ids) diff --git a/plugins/dev-team/tests/scripts/test_checkpoint_abort.py b/plugins/dev-team/tests/scripts/test_checkpoint_abort.py index 042b06882..547ac266a 100644 --- a/plugins/dev-team/tests/scripts/test_checkpoint_abort.py +++ b/plugins/dev-team/tests/scripts/test_checkpoint_abort.py @@ -197,6 +197,20 @@ def test_one_blocking_finding_among_suggestion_tier_blocks(self): assert out["outcome"] == "blocked" assert out["reason"] == "1 finding(s) remain" + def test_differently_cased_severity_and_confidence_still_block(self): + """Regression (backstop review, #2168): the severity floor is now + imported from `finding_signature.is_actionable`, which lowercases + `severity`/`confidence` before comparing — a prior local copy here + compared case-sensitively, so a finding tagged `"Error"`/`"High"` + (a differently-cased but semantically identical value) silently + never blocked while `finding_signature`'s own fix loop treated it + as fully actionable. Both modules must now agree on any casing.""" + out = checkpoint_abort.compute_round_outcome( + aborted=False, redispatched=False, findings=[_issue("Error", "High")] + ) + assert out["outcome"] == "blocked" + assert out["reason"] == "1 finding(s) remain" + class TestMergeFindings: def _finding(self, **kw): @@ -450,6 +464,43 @@ def test_from_file_happy_path(self, tmp_path): payload = json.loads(result.stdout) assert payload["outcome"] == "blocked" + def test_from_nonexistent_file_exits_nonzero(self, tmp_path): + """Regression (backstop review, #2168): `--mode outcome`'s + `--from`-file read shares `_run_abort_mode`'s exact error-handling + shape but had no test of its own for this branch.""" + missing = tmp_path / "does-not-exist.json" + result = subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "checkpoint_abort.py"), + "--mode", + "outcome", + "--from", + str(missing), + ], + capture_output=True, + text=True, + check=False, + ) + assert result.returncode != 0 + assert "cannot read" in result.stderr.lower() + + def test_invalid_json_exits_nonzero_with_clear_error(self): + """Regression (backstop review, #2168) — same rationale as + `test_from_nonexistent_file_exits_nonzero` above.""" + result = self._run_raw("not json") + assert result.returncode != 0 + assert "json" in result.stderr.lower() + + def _run_raw(self, raw_input, check=False): + return subprocess.run( + [sys.executable, str(_SCRIPTS_DIR / "checkpoint_abort.py"), "--mode", "outcome"], + input=raw_input, + capture_output=True, + text=True, + check=check, + ) + class TestCliModeMerge: def _run(self, payload, check=False): @@ -507,3 +558,38 @@ def test_from_file_happy_path(self, tmp_path): ) merged = json.loads(result.stdout) assert merged == [] + + def test_from_nonexistent_file_exits_nonzero(self, tmp_path): + """Regression (backstop review, #2168) — mirrors + `TestCliModeOutcome`'s equivalent test; `--mode merge`'s `--from` + read shares the same error-handling shape and had no test of its + own for this branch.""" + missing = tmp_path / "does-not-exist.json" + result = subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "checkpoint_abort.py"), + "--mode", + "merge", + "--from", + str(missing), + ], + capture_output=True, + text=True, + check=False, + ) + assert result.returncode != 0 + assert "cannot read" in result.stderr.lower() + + def test_invalid_json_exits_nonzero_with_clear_error(self): + """Regression (backstop review, #2168) — same rationale as + `test_from_nonexistent_file_exits_nonzero` above.""" + result = subprocess.run( + [sys.executable, str(_SCRIPTS_DIR / "checkpoint_abort.py"), "--mode", "merge"], + input="not json", + capture_output=True, + text=True, + check=False, + ) + assert result.returncode != 0 + assert "json" in result.stderr.lower() diff --git a/plugins/dev-team/tests/scripts/test_render_tiered_findings.py b/plugins/dev-team/tests/scripts/test_render_tiered_findings.py index 3094c08cf..eb02afb63 100644 --- a/plugins/dev-team/tests/scripts/test_render_tiered_findings.py +++ b/plugins/dev-team/tests/scripts/test_render_tiered_findings.py @@ -305,11 +305,47 @@ def test_cli_accepts_dict_wrapped_findings_shape(self): assert r.returncode == 0 assert r.stdout == r_bare.stdout - def test_cli_malformed_json_propagates_uncaught_error(self): - """Pins current behavior (matches the sibling script - finding_signature.py, no documented contract requires catching this): - malformed JSON is not caught — json.loads's exception propagates, - producing a non-zero exit and a traceback on stderr.""" + def test_cli_malformed_json_is_a_clean_error_not_a_traceback(self): + """Regression (backstop review, #2170): malformed JSON must produce + a clean, non-zero-exit error message on stderr — not an uncaught + traceback, and never a silent `CLEAN_PASS_SUMMARY` on stdout.""" r = _run(input_text="{not valid json") assert r.returncode != 0 - assert "JSONDecodeError" in r.stderr + assert "Traceback" not in r.stderr + assert "cannot interpret --findings input" in r.stderr + assert rtf.CLEAN_PASS_SUMMARY not in r.stdout + + def test_cli_aggregated_json_object_with_top_findings_key_is_recognized(self): + """Regression (backstop review, #2170): the actual aggregated + `--json` object (`output-format.md`) keys its consolidated list as + `topFindings`, not `findings` — the previous `_load_findings` fell + through to an empty list for this exact real-world shape and + rendered a false `CLEAN_PASS_SUMMARY`.""" + findings = [_finding()] + r = _run(input_text=json.dumps({"overall": "warn", "topFindings": findings})) + r_bare = _run(input_text=json.dumps(findings)) + assert r.returncode == 0 + assert r.stdout == r_bare.stdout + + def test_cli_per_agent_result_shape_with_issues_key_is_recognized(self): + """Regression (backstop review, #2170): a raw per-agent + `{status, issues, summary}` result + (`knowledge/review-agent-output-contract.md`) keys its list as + `issues`, not `findings` — same silent-empty failure mode as the + `topFindings` case above.""" + findings = [_finding()] + r = _run(input_text=json.dumps({"status": "warn", "issues": findings, "summary": "x"})) + r_bare = _run(input_text=json.dumps(findings)) + assert r.returncode == 0 + assert r.stdout == r_bare.stdout + + def test_cli_unrecognized_dict_shape_is_a_clean_error_not_a_silent_clean_pass(self): + """Regression (backstop review, #2170): a dict with none of + `findings`/`topFindings`/`issues` as a list-valued key (e.g. a raw + `{"agents": [...]}` object with no consolidated list at the top + level) must error loudly rather than silently rendering + `CLEAN_PASS_SUMMARY`.""" + r = _run(input_text=json.dumps({"agents": [{"agentName": "x", "issues": [_finding()]}]})) + assert r.returncode != 0 + assert "cannot interpret --findings input" in r.stderr + assert rtf.CLEAN_PASS_SUMMARY not in r.stdout diff --git a/plugins/dev-team/tests/scripts/test_test_review_mechanics.py b/plugins/dev-team/tests/scripts/test_test_review_mechanics.py index 9430d5a32..684770a08 100644 --- a/plugins/dev-team/tests/scripts/test_test_review_mechanics.py +++ b/plugins/dev-team/tests/scripts/test_test_review_mechanics.py @@ -53,6 +53,9 @@ def __init__(self, stdout: str, returncode: int = 0, stderr: str = "") -> None: self.returncode = returncode +# double-waiver: B1 — stubs the subprocess call to internal_double_detector.py +# (a first-party out-of-process collaborator) to avoid the real process-spawn +# cost on every fixture in this file unrelated to doubling detection itself. def _no_op_runner(*_args, **_kwargs) -> _StubCompleted: return _StubCompleted(json.dumps({"findings": []})) @@ -138,6 +141,101 @@ def test_python_no_assertion_branch_is_detected(self, tmp_path): assert len(hits) == 1 assert result["mechanicalFail"] is True + def test_python_multiline_signature_with_assertion_is_not_flagged(self, tmp_path): + """Regression (backstop review, #2169): a black-formatted + multi-line signature put the body's first real line right after a + `):` line whose own indent equals the `def` line's — a naive + next-line dedent check misread that as the body ending before it + started, silently dropping the whole body (including its assert) + and firing a false no-assertion error.""" + test_file = _tests_dir(tmp_path) / "test_widget.py" + test_file.write_text( + "def test_renders_without_crashing(\n" + " tmp_path, cfg\n" + "):\n" + " widget = Widget(tmp_path, cfg)\n" + " assert widget.render() is not None\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "no-assertion") == [] + assert result["mechanicalFail"] is False + + def test_python_multiline_signature_with_no_assertion_is_still_flagged(self, tmp_path): + """The multi-line-signature fix must not swallow a genuine + no-assertion defect — only the body boundary changes, not the + assertion search itself.""" + test_file = _tests_dir(tmp_path) / "test_widget.py" + test_file.write_text( + "def test_renders_without_crashing(\n" + " tmp_path, cfg\n" + "):\n" + " widget = Widget(tmp_path, cfg)\n" + " widget.render()\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "no-assertion")) == 1 + assert result["mechanicalFail"] is True + + +class TestJsTestCallMemberAccessRegression: + """Regression (backstop review, #2169): `_JS_TEST_CALL_RE`'s original + `\\b(?:it|test)` matched a member-access call like `pattern.test(...)` + or `/re/.test(...)` — ordinary RegExp usage, not a test declaration — + because `\\b` sits at a word boundary between `.` and `t` regardless of + what precedes the `.`. The fixed `(? {\n" + " expect(/^[a-z]+$/.test(value)).toBe(true);\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "no-assertion") == [] + assert result["mechanicalFail"] is False + + def test_dotted_test_call_on_a_custom_object_is_not_treated_as_a_test_region(self, tmp_path): + """A bare method call named `.test(` on any receiver — not just a + RegExp — must not be mistaken for an `it(`/`test(` declaration.""" + test_file = _tests_dir(tmp_path) / "widget.test.js" + test_file.write_text( + "it('checks the matcher', () => {\n" + " expect(matcher.test(value)).toBe(true);\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert _findings_by_category(result, "no-assertion") == [] + assert result["mechanicalFail"] is False + + def test_undotted_test_call_is_still_recognized_as_a_test_region(self, tmp_path): + """The lookbehind must exclude only a PRECEDING `.`/word-char/`$` + — a genuine top-level `test(` call is unaffected.""" + test_file = _tests_dir(tmp_path) / "widget.test.js" + test_file.write_text( + "test('renders without crashing', () => {\n" + " widget.render();\n" + "});\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + assert len(_findings_by_category(result, "no-assertion")) == 1 + assert result["mechanicalFail"] is True + class TestContentSliceRegression: """Fix 1 — the highest-priority finding: `_check_no_assertion`/ @@ -223,6 +321,50 @@ def test_async_test_body_with_no_await_is_warning_and_does_not_gate(self, tmp_pa assert hits[0]["severity"] == "warning" assert result["mechanicalFail"] is False + def test_csharp_async_task_with_no_await_is_warning(self, tmp_path): + """The C# branch of `_check_missing_await` had no dedicated fixture + (backstop review, #2169) — mirrors the JS/TS case above.""" + test_file = _tests_dir(tmp_path) / "WidgetTests.cs" + test_file.write_text( + "public class WidgetTests {\n" + " [Test]\n" + " public async Task FetchesDataAsync() {\n" + " var result = FetchData();\n" + " Assert.IsNotNull(result);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "missing-await") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + assert result["mechanicalFail"] is False + + def test_java_future_with_no_get_or_join_is_warning(self, tmp_path): + """The Java branch of `_check_missing_await` had no dedicated + fixture (backstop review, #2169).""" + test_file = _tests_dir(tmp_path) / "WidgetTest.java" + test_file.write_text( + "public class WidgetTest {\n" + " @Test\n" + " public void fetchesDataAsync() {\n" + " CompletableFuture future = fetchDataAsync();\n" + " assertNotNull(future);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "missing-await") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + assert result["mechanicalFail"] is False + class TestMockNotReset: def test_mock_construct_with_no_reset_call_is_warning(self, tmp_path): @@ -309,6 +451,30 @@ def test_java_tightened_marker_does_not_accept_unqualified_reset(self, tmp_path) assert len(_findings_by_category(result, "mock-not-reset")) == 1 + def test_csharp_mock_with_no_reset_or_reinstantiation_is_warning(self, tmp_path): + """The plain (non-suppressed) C# positive case had no dedicated + fixture (backstop review, #2169) — only the suppression path + (`test_csharp_setup_reinstantiation_suppresses_mock_not_reset` + above) was tested for C#.""" + test_file = _tests_dir(tmp_path) / "WidgetTests.cs" + test_file.write_text( + "public class WidgetTests {\n" + " [Test]\n" + " public void CallsGateway() {\n" + " var gateway = new Mock();\n" + " gateway.Object.Send();\n" + " Assert.IsTrue(true);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "mock-not-reset") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + class TestUnstubbedClockRngTimer: def test_unstubbed_new_date_is_warning(self, tmp_path): @@ -343,6 +509,48 @@ def test_fake_timers_marker_suppresses_unstubbed_clock_finding(self, tmp_path): assert _findings_by_category(result, "unstubbed-clock-rng-timer") == [] + def test_csharp_datetime_now_is_warning(self, tmp_path): + """The C# branch of `_check_unstubbed_clock_rng_timer` had no + dedicated fixture (backstop review, #2169).""" + test_file = _tests_dir(tmp_path) / "WidgetTests.cs" + test_file.write_text( + "public class WidgetTests {\n" + " [Test]\n" + " public void ChecksTimestamp() {\n" + " var now = DateTime.Now;\n" + " Assert.IsTrue(now.Year > 2000);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "unstubbed-clock-rng-timer") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + + def test_java_instant_now_is_warning(self, tmp_path): + """The Java branch of `_check_unstubbed_clock_rng_timer` had no + dedicated fixture (backstop review, #2169).""" + test_file = _tests_dir(tmp_path) / "WidgetTest.java" + test_file.write_text( + "public class WidgetTest {\n" + " @Test\n" + " public void checksTimestamp() {\n" + " Instant now = Instant.now();\n" + " assertNotNull(now);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "unstubbed-clock-rng-timer") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + class TestReflectionPrimaryStrategy: def test_reflection_alone_is_warning_and_never_gates(self, tmp_path): @@ -360,6 +568,66 @@ def test_reflection_alone_is_warning_and_never_gates(self, tmp_path): hits = _findings_by_category(result, "reflection-primary-strategy") assert len(hits) == 1 assert hits[0]["severity"] == "warning" + + def test_python_getattr_private_is_warning(self, tmp_path): + """The Python branch of `_check_reflection_primary_strategy` had no + dedicated fixture (backstop review, #2169).""" + test_file = _tests_dir(tmp_path) / "test_internals.py" + test_file.write_text( + "def test_accesses_private_state():\n" + " value = getattr(component, '_internal_state')\n" + " assert value is not None\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "reflection-primary-strategy") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + + def test_java_getdeclaredfield_is_warning(self, tmp_path): + """The Java branch of `_check_reflection_primary_strategy` had no + dedicated fixture (backstop review, #2169).""" + test_file = _tests_dir(tmp_path) / "WidgetTest.java" + test_file.write_text( + "public class WidgetTest {\n" + " @Test\n" + " public void accessesPrivateField() throws Exception {\n" + " Field field = Widget.class.getDeclaredField(\"internalState\");\n" + " assertNotNull(field);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "reflection-primary-strategy") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" + + def test_csharp_getmethod_nonpublic_is_warning(self, tmp_path): + """The C# branch of `_check_reflection_primary_strategy` had no + dedicated fixture (backstop review, #2169).""" + test_file = _tests_dir(tmp_path) / "WidgetTests.cs" + test_file.write_text( + "public class WidgetTests {\n" + " [Test]\n" + " public void AccessesPrivateMethod() {\n" + " var method = typeof(Widget).GetMethod(\"DoInternal\", " + "BindingFlags.NonPublic | BindingFlags.Instance);\n" + " Assert.IsNotNull(method);\n" + " }\n" + "}\n", + encoding="utf-8", + ) + + result = trm.analyze_file(tmp_path, test_file, double_detector_runner=_no_op_runner) + + hits = _findings_by_category(result, "reflection-primary-strategy") + assert len(hits) == 1 + assert hits[0]["severity"] == "warning" assert result["mechanicalFail"] is False assert result["skippedQualitative"] is False @@ -487,6 +755,9 @@ def test_runner_oserror_is_warning_severity_and_check_did_not_run(self, tmp_path encoding="utf-8", ) + # double-waiver: B1 — stubs the subprocess call to + # internal_double_detector.py to simulate a spawn failure without a + # real missing-executable environment. def _raising_runner(*_args, **_kwargs): raise OSError("no such executable") @@ -504,6 +775,9 @@ def test_runner_invalid_json_is_warning_severity_and_check_did_not_run(self, tmp encoding="utf-8", ) + # double-waiver: B1 — stubs the subprocess call to + # internal_double_detector.py to simulate a malformed response + # without depending on the real detector's actual output shape. def _garbage_runner(*_args, **_kwargs): return _StubCompleted("not json", returncode=1, stderr="boom") diff --git a/plugins/dev-team/scripts/compare_eval_results.py b/scripts/compare_eval_results.py similarity index 93% rename from plugins/dev-team/scripts/compare_eval_results.py rename to scripts/compare_eval_results.py index 4cb9875d7..7cca6c297 100755 --- a/plugins/dev-team/scripts/compare_eval_results.py +++ b/scripts/compare_eval_results.py @@ -81,15 +81,19 @@ detection count, not to re-enforce the corpus's own declared tolerance ranges (that remains `eval_grade.py`'s job). -Shipped-tree placement ------------------------- -This script lives in `plugins/dev-team/scripts/` (shipped) even though its -whole domain is the repo's own non-shipped `evals/` corpus. It mirrors the -existing precedent of `eval_ablation.py` (same directory, same repo-root -default) rather than introducing a new violation. It is monorepo-dev-only -tooling: useful only to a `test-review.md`/eval-corpus maintainer re-running -this exact regression check against this repo's own eval corpus, never -invoked by a downstream project that installs the plugin. +Repo-root placement (ADR 0032) +------------------------------- +This script lives at repo-root `scripts/`, next to `eval_grade.py`, not +under the shipped `plugins/dev-team/scripts/` tree — ADR 0032's category 2 +(monorepo-dev-only tooling whose whole domain is the repo's own non-shipped +`evals/` corpus). It is useful only to a `test-review.md`/eval-corpus +maintainer re-running this exact regression check against this repo's own +eval corpus, never invoked by a downstream project that installs the +plugin. `eval_ablation.py` is NOT the precedent for shipping a script like +this one: it ships specifically because its `--find-latest` mode is a +generic JSONL reader with no repo-specific behavior, invoked by the shipped +`harness-audit` skill via `${CLAUDE_PLUGIN_ROOT}` — this script has no such +shipped-skill caller or portable mode (backstop review, #2169). Exit codes ---------- diff --git a/plugins/dev-team/tests/scripts/test_compare_eval_results.py b/tests/repo/test_compare_eval_results.py similarity index 83% rename from plugins/dev-team/tests/scripts/test_compare_eval_results.py rename to tests/repo/test_compare_eval_results.py index a756a1832..931ddf54d 100644 --- a/plugins/dev-team/tests/scripts/test_compare_eval_results.py +++ b/tests/repo/test_compare_eval_results.py @@ -21,7 +21,7 @@ from _repo_root import REPO_ROOT as _REPO_ROOT -_SCRIPTS_DIR = _REPO_ROOT / "plugins" / "dev-team" / "scripts" +_SCRIPTS_DIR = _REPO_ROOT / "scripts" sys.path.insert(0, str(_SCRIPTS_DIR)) import compare_eval_results as cer @@ -190,6 +190,73 @@ def test_multi_fixture_input_produces_one_row_per_fixture(self): assert scope["compared"] == 3 +class TestLoadExpected: + """`_load_expected`'s deliberate, commented design decision — skip a + malformed/unreadable `expected/*.json` file rather than raising, since + this script's job is to compare result files, not re-run + `eval_grade.py --check-corpus` — had no dedicated fixture (backstop + review, #2169).""" + + def test_malformed_expected_file_is_skipped_and_valid_ones_still_load(self, tmp_path): + expected_dir = tmp_path / "expected" + expected_dir.mkdir() + (expected_dir / "bad.json").write_text("{not valid json", encoding="utf-8") + (expected_dir / "fixA.json").write_text( + json.dumps( + { + "fixture": "fixA", + "applicableAgents": ["test-review"], + "agents": {"test-review": {"issueCount": {"min": 1, "max": 2}}}, + } + ), + encoding="utf-8", + ) + + loaded = cer._load_expected(expected_dir) + + assert "bad" not in loaded + assert loaded["fixA"] == {"test-review": {"issueCount": {"min": 1, "max": 2}}} + + def test_cli_does_not_crash_with_a_malformed_expected_file_present(self, tmp_path): + """CLI-level companion to the unit test above: the malformed file + must not surface as an uncaught exception through the full CLI + path, and the valid fixture alongside it must still be compared.""" + expected_dir = tmp_path / "expected" + expected_dir.mkdir() + (expected_dir / "bad.json").write_text("{not valid json", encoding="utf-8") + (expected_dir / "fixA.json").write_text( + json.dumps( + { + "fixture": "fixA", + "applicableAgents": ["test-review"], + "agents": {"test-review": {"issueCount": {"min": 1, "max": 2}}}, + } + ), + encoding="utf-8", + ) + before_path = tmp_path / "before.json" + after_path = tmp_path / "after.json" + before_path.write_text(json.dumps(_actuals_block("fixA", "test-review", 1)), encoding="utf-8") + after_path.write_text(json.dumps(_actuals_block("fixA", "test-review", 1)), encoding="utf-8") + + result = subprocess.run( + [ + sys.executable, + str(_SCRIPTS_DIR / "compare_eval_results.py"), + str(before_path), + str(after_path), + "--expected-dir", + str(expected_dir), + ], + capture_output=True, + text=True, + check=False, + ) + + assert result.returncode == 0 + assert "Traceback" not in result.stderr + + class TestCli: def _write_json(self, path, data) -> None: path.write_text(json.dumps(data), encoding="utf-8") diff --git a/tests/repo/test_python_floor.py b/tests/repo/test_python_floor.py index 6b9ae76d4..8758c382d 100644 --- a/tests/repo/test_python_floor.py +++ b/tests/repo/test_python_floor.py @@ -182,7 +182,6 @@ "stdlib argparse/json/pathlib only; no floor-sensitive runtime API" ), "checkpoint_abort.py": "stdlib argparse/json/sys/pathlib only; no floor-sensitive runtime API", - "compare_eval_results.py": "stdlib argparse/json/sys/pathlib only; no floor-sensitive runtime API", "coverage_delta_steering.py": "stdlib argparse/json/pathlib only; no floor-sensitive runtime API", "mutation_yield_steering.py": ( "stdlib argparse/json/sys/pathlib only; no floor-sensitive runtime API " From 1878555db1392ff1cacf620bde177aae3a178e69 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 22:53:38 +0000 Subject: [PATCH 14/14] fix(agents): avoid bare skills/code-review/SKILL.md path reference test-review.md's Hunt-scope prose (added by the prior backstop-review fix commit) referenced skills/code-review/SKILL.md as a literal path, which the anchor-citation guard (tests/agents/test_agent_knowledge_anchor.py) requires to carry a knowledge/index.json anchor or a "Whole-file load:" token. Rephrase as /code-review, matching this file's existing convention elsewhere for referencing the same skill. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01KGHpm7gtqNb8NM2Smf76u7 --- plugins/dev-team/agents/test-review.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugins/dev-team/agents/test-review.md b/plugins/dev-team/agents/test-review.md index e0514d210..fd4721581 100644 --- a/plugins/dev-team/agents/test-review.md +++ b/plugins/dev-team/agents/test-review.md @@ -186,7 +186,7 @@ If a static-analysis pre-pass has already surfaced this exact finding (e.g. via ## Tolerated-Deviation Hunt -Computed by Phase 0's `test_review_mechanics.py` pass, not a separate manual grep — cite its `tolerated-deviation-consolidation` finding when present rather than re-deriving it. The categories below are the detection specification the script implements, kept here for reference. Phase 0's actual wiring (`skills/code-review/SKILL.md` step 2b, #2169) runs this pre-pass only for test files in scope, so this Hunt currently sees test files, not "every core-flow file" — non-test source files get no Phase 0 result at all and fall through to this file's own "No result supplied for a file" rule (run Phase 1/2 as usual; say nothing about Phase 0). Extending the pre-pass to non-test core-flow files is a separate wiring change, not made here. Count tolerated-deviation artifacts from the following categories: +Computed by Phase 0's `test_review_mechanics.py` pass, not a separate manual grep — cite its `tolerated-deviation-consolidation` finding when present rather than re-deriving it. The categories below are the detection specification the script implements, kept here for reference. Phase 0's actual wiring (`/code-review`'s step 2b, #2169) runs this pre-pass only for test files in scope, so this Hunt currently sees test files, not "every core-flow file" — non-test source files get no Phase 0 result at all and fall through to this file's own "No result supplied for a file" rule (run Phase 1/2 as usual; say nothing about Phase 0). Extending the pre-pass to non-test core-flow files is a separate wiring change, not made here. Count tolerated-deviation artifacts from the following categories: - **Disabled tests** — `@Ignore`, `@Disabled`, `xit(`, `xdescribe(`, `test.skip(`, `it.skip(`, `[Ignore]`, `[Skip]`, `pytest.mark.skip`, `pytest.mark.xfail` with no