Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
25 commits
Select commit Hold shift + click to select a range
25e1008
chore(metrics): record plan-approval audit entry for #2172 batch
claude Sep 21, 2026
fe46745
chore(metrics): record acceptance-criteria gate audit entry for #2172…
claude Sep 21, 2026
1d020aa
feat(hooks): resolve per-agent-type skill hints from frontmatter (#2187)
claude Sep 21, 2026
0f1aac0
feat(hooks): PreToolUse hook injects skill context into Agent/Task di…
claude Sep 21, 2026
41e7d3f
feat(hooks): register subagent_skill_context on Agent/Task PreToolUse…
claude Sep 21, 2026
503f1e0
feat(skills): claim extraction + output schema for source-verificatio…
claude Sep 21, 2026
9f5a631
feat(skills): add source-verification SKILL.md procedure (#2189)
claude Sep 21, 2026
c4f08f8
feat(skills): Hypothesis property scaffold for round-trip/invariant p…
claude Sep 21, 2026
75740ff
chore(skills): register source-verification in agent-registry and ski…
claude Sep 21, 2026
f891e0e
feat(skills): wire source-verification into doc-review (#2189)
claude Sep 21, 2026
7385ccc
feat(skills): add property-based-testing SKILL.md procedure (#2190)
claude Sep 21, 2026
7db7d55
chore(skills): register property-based-testing in agent-registry and …
claude Sep 21, 2026
3a03c50
feat(skills): fast-check (JS/TS) property generation path (#2190)
claude Sep 21, 2026
8f41080
feat(hooks): detect incomplete/malformed SubagentStop from transcript…
claude Sep 21, 2026
0c32a74
feat(hooks): emit boundary event on incomplete SubagentStop (#2188)
claude Sep 21, 2026
a48f384
fix(skills): wire property-based-testing's JS/TS path to Step 4.3's f…
claude Sep 21, 2026
4312c0f
feat(hooks): register subagent_completion_guard on SubagentStop (#2188)
claude Sep 21, 2026
bf6507f
docs(hooks): document subagent_completion_guard (#2188)
claude Sep 21, 2026
2751fc0
fix(tests): close full-suite gate findings surfaced by the #2172 batch
claude Sep 21, 2026
4997a1a
fix(hooks): address code-review Wave 1 correctness/test findings
claude Sep 21, 2026
89c5e32
fix(hooks): address arch-review findings on the #2172 batch
claude Sep 21, 2026
a804046
fix(hooks): address code-review Wave 2 security/domain findings
claude Sep 21, 2026
4508ddf
fix(ci): fix Python ceiling deps and skip hypothesis-scaffold tests g…
claude Sep 21, 2026
d716eee
Merge remote-tracking branch 'origin/main' into feat/2172-agent-lifec…
claude Sep 21, 2026
3cb486f
Merge branch 'main' into feat/2172-agent-lifecycle-improvements
bdfinst Sep 22, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .claude/metrics/config-changelog.jsonl
Original file line number Diff line number Diff line change
Expand Up @@ -82,3 +82,5 @@
{"timestamp": "2026-09-16T19:21:38.872232+00:00", "type": "approval", "proposed": "Acceptance-criteria set for the DeFOSPAM spec-review-gaps plan (14 criteria)", "evidence_shown": ["plans/defospam-spec-review-gaps.md", "spec-compliance-review criteria-verification report"], "risks_surfaced": ["C6 (error): four cross-cutting concerns unnamed; grouping requirement absent \u2014 revised", "C2: refusal branch for unresolved path/URL absent \u2014 revised", "C5: reason clause in the printed message absent \u2014 revised", "C9: recording on a passed check absent \u2014 revised", "C11: under-scoped to slice 5 \u2014 revised to slices 3 and 5 (reviewer said 2, 3, 5; slice 2 states no such boundary)", "C14: cumulative-regression guard absent \u2014 revised"], "description": "Criteria gate: 5 flagged, all revised rather than overridden. Reviewer's C11 scope corrected against the plan text."}
{"timestamp": "2026-09-17T15:34:39.329144+00:00", "type": "approval", "proposed": "plan status approved: plans/measure-rereview-duplication.md (issue #2165)", "evidence_shown": "plans/measure-rereview-duplication.md", "risks_surfaced": []}
{"timestamp": "2026-09-17T15:36:16.556298+00:00", "type": "approval", "proposed": "acceptance-criteria set: plans/measure-rereview-duplication.md (issue #2165)", "evidence_shown": "plans/measure-rereview-duplication.md", "risks_surfaced": ["AC6: cross-reference to step 1.4 checkpoint list not inlined", "AC8: additive-alias pattern not defined in AC text"]}
{"timestamp": "2026-09-21T18:49:13Z", "type": "approval", "proposed": "Approve plan for Agent lifecycle improvements (#2172 batch: #2187-#2190)", "evidence_shown": "plans/2172-agent-lifecycle-improvements.md", "risks_surfaced": ["Strategic critic recommended splitting into up to 3 PRs; acknowledged, not adopted (user pre-decided single-batch)", "Slice 2 malformed-hand-back scenario is provisional pending Step 2.1a transcript-shape confirmation", "Slice 4 JS/TS fast-check test may need network-exempt fallback if vendoring proves impractical"], "description": "Auto-approved (non-interactive) - no usable TTY in this remote session"}
{"timestamp": "2026-09-21T18:51:16Z", "type": "approval", "proposed": "Acceptance-criteria gate for plans/2172-agent-lifecycle-improvements.md", "evidence_shown": "plans/2172-agent-lifecycle-improvements.md", "risks_surfaced": ["Step 1.4 latency check has no quantitative SLA threshold", "Step 1.2 missing idempotent-double-fire test case", "Step 2.1a conditional AC pass/fail ambiguity if no hand-back signal found", "Step 2.1b missing invalid-JSON (not just unreadable) transcript test", "Step 3.1 claim-extraction heuristics under-specified with only 2 of 5 pinned", "Step 3.2 missing malformed-but-parseable WebFetch response case", "Step 3.3 doc-review integration test is structural-only, not behavioral", "Step 4.1 property-derivation edge cases (cross-module pair, ambiguous invariant) untested", "Step 4.2 missing language-specific runtime-failure cases (Hypothesis/fast-check install failure)", "Step 4.3 vendoring-impractical threshold undefined"], "description": "Acceptance-criteria gate auto-passed with 10 flagged (0 blocker, 2 warning, 8 suggestion/minor) criterion findings (non-interactive) - no human gate. Trigger: --yes flag. Findings will be resolved as implementation-time decisions during Step 4, per the plan own assumptions-recording convention."}
31 changes: 31 additions & 0 deletions plugins/dev-team/agents/doc-review.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@ tools: Read, Grep, Glob, mcp__codegraph__*, mcp__plugin_repowise_repowise__get_c
model: haiku
effort: medium
color: green
skills:
- source-verification
---

# Documentation Review
Expand Down Expand Up @@ -111,9 +113,38 @@ across any language and comment syntax (`//`, `#`, `/* */`, `--`):
- `docs/agent-architecture.md` references a configuration or governance detail that is no longer current
- Agent or skill files changed without corresponding update to `CLAUDE.md` registry tables

### External claim verification

- When reviewed content asserts specific behavior of an external API, tool, or
library (version numbers, endpoint behavior, config defaults, documented
flags), follow `${CLAUDE_PLUGIN_ROOT}/skills/source-verification/SKILL.md`'s
(Whole-file load: the full extraction/verification procedure, not one
section) claim-extraction/verification procedure over it. This agent has no
`Bash`/`Skill`/`WebFetch` grant, so it applies the procedure's heuristic
categories and its Read/Grep-driven internal-claim verification directly
(never invoking `../skills/source-verification/scripts/claim_extractor.py`
as a subprocess) and reports
an external-tool/spec claim it cannot fetch as `unverifiable` rather than
guessing at it.
- A claim it reports `contradicted` is `error` (documentation actively
misleads). A claim it reports `unverifiable` is `warning` (documentation is
stale or incomplete — no source could confirm it).

`get_why` (recorded decision rationale) is available to check whether stale-looking
code/docs still have a live rationale before flagging staleness.

## Skills

Whole-file load: each linked SKILL.md is loaded in full when invoked.

- [Source Verification](../skills/source-verification/SKILL.md) — invoke for
the External claim verification checklist above. This agent has no
`Bash`/`Skill`/`WebFetch` tool grant, so it applies the skill's
extraction/verification procedure by direct inspection (Read/Grep) rather
than running `../skills/source-verification/scripts/claim_extractor.py`,
and reports a claim it cannot
fetch a source for as `unverifiable`.

## Self-Challenge

After producing findings, run the shared challenger loop in `${CLAUDE_PLUGIN_ROOT}/knowledge/adversarial-review-protocol.md` (Whole-file load: the slim shared methodology — The Loop + Output format — read in full), then work these doc-review-specific challenges:
Expand Down
28 changes: 28 additions & 0 deletions plugins/dev-team/docs/developer-notes.md
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,34 @@ topic.
| Code knowledge graphs | [`codegraph-vs-graphify.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/knowledge/codegraph-vs-graphify.md) | When to use CodeGraph vs Graphify, how `/project-init` installs each, and the CLAUDE.md-preservation guard. |
| Script conventions | [ADR 0014](../../../docs/adr/0014-python-for-cross-os-scripts.md), [ADR 0015](../../../docs/adr/0015-bash-removal-complete.md), [ADR 0031](../../../docs/adr/0031-raise-shipped-python-floor-to-3-10.md) | Why every shipped script is Python 3.10+ stdlib-only, the completed bash removal, and the floor's move off EOL 3.8. |

## Hooks

`PreToolUse`/`PostToolUse` guard hooks are documented in
[`agent-architecture.md` § Governance](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/docs/agent-architecture.md#governance).
`SubagentStop` hooks aren't yet indexed there; this section is the entry
point for them.

### Subagent Completion Guard

A `SubagentStop` hook (`hooks/subagent_completion_guard.py`, #2188)
classifies each dispatched subagent's own transcript tail into `clean`,
`empty-final-turn`, `truncated-final-turn`, or `unreadable`, and emits a
`boundary-events.jsonl` "warn" record for the two non-clean, explainable
outcomes (`empty-final-turn`, `truncated-final-turn`) via
`hooks/lib/boundary_events.emit_boundary_event`; `clean` and `unreadable`
stay silent — record-only, never blocking. It runs alongside the plugin's
other `SubagentStop` hooks, `hooks/cost_meter.py` and
`hooks/task_completion_metrics.py` (both undocumented as of this entry).
Classification reads the last JSON row of the subagent's own transcript file
directly — no sidechain filtering needed, since every row in a subagent's
own `subagents/agent-<hash>.jsonl` file already belongs to it — and checks
`message.stop_reason == "max_tokens"` (→ `truncated-final-turn`, checked
first since a token-limit cutoff can itself produce empty content) before
falling back to empty/whitespace-only content (→ `empty-final-turn`).
Fail-open throughout: a missing/unreadable transcript, or a last row missing
`message`/`content` entirely, classifies as `unreadable` and emits nothing.
Tests: `tests/hooks/test_subagent_completion_guard.py`.

The rest of this page covers the one extension path that touches several of
these files at once and has no other single writeup: adding a new language to
the static-analysis suite.
Expand Down
2 changes: 2 additions & 0 deletions plugins/dev-team/docs/skills.md
Original file line number Diff line number Diff line change
Expand Up @@ -178,9 +178,11 @@ Most skills are **user-invocable** as slash commands — shown as `/name`; run t
| `/long-eval` | [status\|ensure-alive] --module <file> --out <dir> | [`long-eval/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/long-eval/SKILL.md) | Run an eval that takes longer than one cloud-session container lifetime — agent calibration, prompt A/B sweeps, judge-panel scoring — so it survives the frequent container recycles that kill in-process work. Use when the user says "run this long eval", "the eval keeps dying on restart", "make the eval survive restarts", "resume the eval", "keep the eval alive", or when a full-corpus calibration/benchmark will clearly outlast a single session. Ships a restart-durable engine + CLI so nothing is re-invented per eval. |
| `/mutation-night-watch` | no flags — run directly | [`mutation-night-watch/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/mutation-night-watch/SKILL.md) | Launch, schedule, and hand off an unattended, LLM-free overnight mutation night-watch run. Use when the user wants a mutation-score baseline waiting each morning without paying LLM cost or blocking a session overnight, says "run mutation testing overnight", "schedule a nightly mutation scan", "set up a mutation night watch", or asks how to get an unattended mutation baseline. Wraps mutation_nightwatch.py — report-only measurement, never generation. |
| `/orchestration-benchmark` | [--task-class <trivial\|standard\|complex>] [--runs <n>] [--dry-run] | [`orchestration-benchmark/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/orchestration-benchmark/SKILL.md) | Run the pre-registered solo-vs-coordinated A/B benchmark: three arms (solo session, current orchestration, delegation-only sweep) over the same task matrix at matched verification rigor, measuring dollar cost, token band shift, quality, rework, and wall-clock. Use when the user asks "is orchestration worth it", "benchmark the pipeline against a solo run", "measure delegation value", "orchestration benchmark", or wants the crossover threshold below which a solo session beats delegation. |
| `/property-based-testing` | <module_path> <function_name> | [`property-based-testing/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/property-based-testing/SKILL.md) | Generate a runnable property-based test for a target function from its signature/docstring — a round-trip test when an encode/decode pair exists, or an invariant test when the docstring documents a postcondition (e.g. "returns sorted", "is idempotent"). Detects the project's language via project-init's stack detection, uses Hypothesis for Python and fast-check for JavaScript/TypeScript, and recommends running /mutation-testing against the generated suite afterward. Use when the user says "generate a property test", "property-based test this function", "add Hypothesis/fast-check tests", or wants round-trip/invariant coverage for a pure function rather than hand-written example-based tests. |
| `/proxy-resilience` | no flags — run directly | [`proxy-resilience/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/proxy-resilience/SKILL.md) | Bounded backoff, retry ceiling, and escalation convention for repeated failures against a corporate Anthropic proxy. Use when you observe repeated HTTP 429 rate-limit responses or connection-refused errors that reference a proxy host, or the user says "proxy is rate-limiting", "429 from the proxy", "proxy connection refused", or "corporate proxy is flaky". |
| `/repo-review` | [--path <dir>] [--json] [--pdf] | [`repo-review/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/repo-review/SKILL.md) | Whole-repository drift review for the review agents that a per-diff /code-review pass cannot meaningfully evaluate — accumulated file/CLAUDE.md size drift, AI-provenance verification debt, harness-config completeness, and cross-file frontend component duplication. Use when the user asks for a "repo review", "drift review", "whole-tree review", wants to check accumulated size/token drift, verification debt, or duplicated frontend components across the WHOLE codebase rather than a single diff, or periodically (e.g. every N merged PRs) to catch drift no single diff-scoped review would surface. Report-only — never gates a commit. |
| `/report-pdf` | <path.md> [--out <path>] | [`report-pdf/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/report-pdf/SKILL.md) | Render a dev-team Markdown report to a polished, shareable PDF. Use when the user says "make a PDF of the report", "export the code-review report as PDF", "turn .dev-team-reports/code-review.md into a PDF", or wants any .dev-team-reports or reports Markdown file as a styled document to attach to a ticket or hand to a non-terminal stakeholder. |
| `/run-report` | [--session <id>] | [`run-report/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/run-report/SKILL.md) | Report one orchestrated run's timeline — per-state dwell time, rejection count, hook denials/bypasses grouped by cause, and cost — joined from boundary-events.jsonl, cost-metering.jsonl, and workflow-states.jsonl for a given session_id (default: most recent). Use when the user asks "how did that run go", "show the run report", "/run-report", or wants a single view of a `/ship`/`/autoship`/`/build` run instead of cross-referencing streams by hand. |
| `/source-verification` | no flags — run directly | [`source-verification/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/source-verification/SKILL.md) | Extract and verify factual claims in generated content (docs, diffs, review comments) against this repo's own code and, where needed, external sources. Use before publishing content that asserts specific behavior, version numbers, or API details — anywhere a wrong claim would mislead a reader. Flags every claim as verified, contradicted, or unverifiable; never silently drops one or defaults it to "verified". |
| `/stryker-xunit-v2-shim` | no flags — run directly | [`stryker-xunit-v2-shim/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/stryker-xunit-v2-shim/SKILL.md) | Build a xunit.v2 Stryker shim so Stryker.NET produces a valid mutation score for a xunit.v3 test project. Stryker.NET cannot observe mutant kills through xunit.v3 (it runs on the Microsoft Testing Platform), so a normal run reports a false near-zero score with almost every mutant reported Survived. Use this BEFORE running Stryker whenever the target .NET test project references xunit.v3 — including when mutation is enabled via /test-improve or /mutation-testing, or you are about to run dotnet-stryker — and as a rescue when a run already reported ~0% or everything Survived or the user says the score looks suspiciously low. When Stryker and xunit.v3 both appear, build the shim first. |
| `/test-improve` | <repo-path> [--parent <url>] [--analyze-only] [--from-phase [<n>]] [--stack <id>] | [`test-improve/SKILL.md`](https://github.com/bdfinst/agentic-dev-team/blob/main/plugins/dev-team/skills/test-improve/SKILL.md) | Consolidated analyze-then-improve test orchestrator. Defaults to lightweight ceremony; opts into heavier capabilities (Gherkin extraction, mutation testing, refactor-for-testability) only when the operator asks. Always baselines coverage (and mutation, when enabled) before any test change, runs the end-of-phase review loop after Phases 5 and 7, and produces a stable 10-section executive-summary report. Use when the user says "improve our tests", "modernize the test suite", "upgrade our tests", or runs /test-improve. |
8 changes: 8 additions & 0 deletions plugins/dev-team/hooks/hooks.json
Original file line number Diff line number Diff line change
Expand Up @@ -158,6 +158,10 @@
{
"type": "command",
"command": "sh \"${CLAUDE_PLUGIN_ROOT}/hooks/py.sh\" \"${CLAUDE_PLUGIN_ROOT}/hooks/agent_dispatch_ledger.py\""
},
{
"type": "command",
"command": "sh \"${CLAUDE_PLUGIN_ROOT}/hooks/py.sh\" \"${CLAUDE_PLUGIN_ROOT}/hooks/subagent_skill_context.py\""
}
]
}
Expand Down Expand Up @@ -285,6 +289,10 @@
{
"type": "command",
"command": "sh \"${CLAUDE_PLUGIN_ROOT}/hooks/py.sh\" \"${CLAUDE_PLUGIN_ROOT}/hooks/task_completion_metrics.py\""
},
{
"type": "command",
"command": "sh \"${CLAUDE_PLUGIN_ROOT}/hooks/py.sh\" \"${CLAUDE_PLUGIN_ROOT}/hooks/subagent_completion_guard.py\""
}
]
}
Expand Down
74 changes: 74 additions & 0 deletions plugins/dev-team/hooks/lib/agent_skill_hints.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,74 @@
"""hooks/lib/agent_skill_hints.py — resolve an agent type's declared skills
from its own frontmatter (#2187, Slice 1 Step 1.1).

`subagent_skill_context.py`'s `PreToolUse` hook (Step 1.2) needs to know
which skills a dispatched `subagent_type` is expected to load, so it can
inject a short reminder note into the dispatch prompt. Per this plan's
design note (ADR 0028), each agent's own frontmatter `skills:` field is the
single source of truth for that mapping — `knowledge/agent-registry.md`
does not mirror `skills:` per-agent, so it is not consulted here.

Reuses `minimal_yaml.py`'s `extract_frontmatter_block` + `parse_yaml`
directly rather than writing a second frontmatter parser — that module
already parses this exact SKILL.md/agent-frontmatter shape for
`build_skills_index.py`.

Stdlib only (ADR 0014).
"""

from __future__ import annotations

import re
import sys
from pathlib import Path

_LIB_DIR = Path(__file__).resolve().parent
if str(_LIB_DIR) not in sys.path:
sys.path.insert(0, str(_LIB_DIR))

from minimal_yaml import (
FrontmatterError,
YamlError,
extract_frontmatter_block,
parse_yaml,
)

#: `agent_type` ultimately comes from an Agent/Task dispatch's own
#: `subagent_type` (a PreToolUse hook payload field) — model-controlled
#: input, not a trusted registry lookup. A real agent stem is always a bare
#: identifier (`[A-Za-z0-9_-]+`), so anything else (path separators, `..`,
#: an absolute path) is rejected before it ever reaches the filesystem,
#: rather than relying on `.md`-suffix / read-only / best-effort framing to
#: make a traversal harmless.
_VALID_AGENT_STEM_RE = re.compile(r"^[A-Za-z0-9_-]+$")


def skills_for_agent_type(agent_type: str, agents_dir: Path) -> list[str]:
"""Return the `skills:` frontmatter list declared by
`<agents_dir>/<agent_type>.md`.

Returns `[]` when `agent_type` isn't a bare agent-name stem, the file
doesn't exist, has no `skills:` key, or its frontmatter can't be parsed
— this is a best-effort hint source, never a hard dependency, so every
failure mode degrades to "no hint" rather than raising. Only the one
matching agent file is read, never the whole `agents_dir`.
"""
if not _VALID_AGENT_STEM_RE.match(agent_type):
return []
try:
text = (agents_dir / f"{agent_type}.md").read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError):
return []
try:
frontmatter = parse_yaml(extract_frontmatter_block(text))
except (FrontmatterError, YamlError):
return []
if not isinstance(frontmatter, dict):
return []
skills = frontmatter.get("skills")
if not isinstance(skills, list):
return []
return [skill for skill in skills if isinstance(skill, str)]


__all__ = ("skills_for_agent_type",)
Loading
Loading