pi-dev-team 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (780) hide show
  1. package/LICENSE +21 -0
  2. package/PORTING.md +134 -0
  3. package/README.md +207 -0
  4. package/UPSTREAM.json +64 -0
  5. package/agents/Explore.md +15 -0
  6. package/agents/a11y-review.md +118 -0
  7. package/agents/adr-author.md +70 -0
  8. package/agents/ai-provenance-review.md +120 -0
  9. package/agents/angular-reactivity-review.md +95 -0
  10. package/agents/arch-review.md +135 -0
  11. package/agents/architect.md +78 -0
  12. package/agents/autoship-batch-proposer.md +69 -0
  13. package/agents/claude-setup-review.md +136 -0
  14. package/agents/codebase-recon.md +184 -0
  15. package/agents/component-architecture-review.md +119 -0
  16. package/agents/concurrency-review.md +109 -0
  17. package/agents/correctness-review.md +290 -0
  18. package/agents/data-flow-tracer.md +120 -0
  19. package/agents/doc-review.md +165 -0
  20. package/agents/domain-review.md +136 -0
  21. package/agents/general-purpose.md +10 -0
  22. package/agents/gherkin-quality-critic.md +113 -0
  23. package/agents/js-fp-review.md +114 -0
  24. package/agents/mutation-kill.md +684 -0
  25. package/agents/naming-review.md +142 -0
  26. package/agents/orchestrator.md +339 -0
  27. package/agents/performance-review.md +105 -0
  28. package/agents/plan-review-acceptance.md +115 -0
  29. package/agents/plan-review-design.md +90 -0
  30. package/agents/plan-review-parallelization.md +84 -0
  31. package/agents/plan-review-strategic.md +96 -0
  32. package/agents/plan-review-ux.md +110 -0
  33. package/agents/platform-engineer.md +64 -0
  34. package/agents/product-manager.md +68 -0
  35. package/agents/progress-guardian.md +79 -0
  36. package/agents/qa-engineer.md +289 -0
  37. package/agents/quality-reviewer.md +132 -0
  38. package/agents/react-reactivity-review.md +102 -0
  39. package/agents/refactor-opportunity-review.md +128 -0
  40. package/agents/security-engineer.md +60 -0
  41. package/agents/security-review.md +218 -0
  42. package/agents/session-analysis.md +95 -0
  43. package/agents/software-engineer.md +105 -0
  44. package/agents/spec-compliance-review.md +100 -0
  45. package/agents/spec-reviewer.md +114 -0
  46. package/agents/structure-review.md +146 -0
  47. package/agents/tech-writer.md +84 -0
  48. package/agents/test-review.md +246 -0
  49. package/agents/test-smell-review.md +188 -0
  50. package/agents/token-efficiency-review.md +139 -0
  51. package/agents/ui-ux-designer.md +54 -0
  52. package/agents/vue-reactivity-review.md +95 -0
  53. package/bin/__pycache__/claudecpython-314.pyc +0 -0
  54. package/bin/claude +258 -0
  55. package/docs/upstream/.pages +1 -0
  56. package/docs/upstream/CHANGELOG.md +2586 -0
  57. package/docs/upstream/README.md +155 -0
  58. package/docs/upstream/agent-architecture.md +214 -0
  59. package/docs/upstream/agent_info.md +187 -0
  60. package/docs/upstream/artifact-migration.md +124 -0
  61. package/docs/upstream/code-intelligence-nudge.md +149 -0
  62. package/docs/upstream/code-review-process.md +294 -0
  63. package/docs/upstream/concurrent-use.md +73 -0
  64. package/docs/upstream/context-management.md +111 -0
  65. package/docs/upstream/developer-notes.md +280 -0
  66. package/docs/upstream/diagrams/architecture-overview.svg +101 -0
  67. package/docs/upstream/diagrams/review-dispatch.svg +139 -0
  68. package/docs/upstream/diagrams/team-agents.svg +128 -0
  69. package/docs/upstream/diagrams/test-improve-flow.svg +166 -0
  70. package/docs/upstream/diagrams/workflow-linear.svg +66 -0
  71. package/docs/upstream/diagrams/workflow-three-phase.svg +200 -0
  72. package/docs/upstream/eval-maintenance.md +95 -0
  73. package/docs/upstream/eval-running-guide.md +147 -0
  74. package/docs/upstream/eval-system.md +291 -0
  75. package/docs/upstream/session-review-oss-complements.md +75 -0
  76. package/docs/upstream/session-review.md +212 -0
  77. package/docs/upstream/skills.md +188 -0
  78. package/docs/upstream/team-structure.md +21 -0
  79. package/docs/upstream/telemetry-ci-access.md +129 -0
  80. package/docs/upstream/telemetry-repo-security.md +120 -0
  81. package/docs/upstream/test-evaluation.md +277 -0
  82. package/docs/upstream/test-improve.md +154 -0
  83. package/docs/upstream/triage-workflow.md +282 -0
  84. package/docs/upstream/workflows.md +289 -0
  85. package/extensions/dev-team/index.ts +539 -0
  86. package/extensions/dev-team/lib/agents.ts +272 -0
  87. package/extensions/dev-team/lib/ai-credits.ts +92 -0
  88. package/extensions/dev-team/lib/autocompact.ts +81 -0
  89. package/extensions/dev-team/lib/child-run.ts +102 -0
  90. package/extensions/dev-team/lib/config.ts +236 -0
  91. package/extensions/dev-team/lib/gh-command.ts +103 -0
  92. package/extensions/dev-team/lib/github-style.ts +307 -0
  93. package/extensions/dev-team/lib/hooks.ts +350 -0
  94. package/extensions/dev-team/lib/metrics.ts +115 -0
  95. package/extensions/dev-team/lib/safe-read.ts +49 -0
  96. package/extensions/dev-team/lib/session-files.ts +57 -0
  97. package/extensions/dev-team/lib/session-spend.ts +123 -0
  98. package/extensions/dev-team/lib/shell-scan.ts +205 -0
  99. package/extensions/dev-team/lib/skills.ts +213 -0
  100. package/extensions/dev-team/lib/subagent-render.ts +245 -0
  101. package/extensions/dev-team/lib/subagent-types.ts +164 -0
  102. package/extensions/dev-team/lib/subagent.ts +596 -0
  103. package/extensions/dev-team/lib/terminal-text.ts +54 -0
  104. package/extensions/dev-team/lib/tools-misc.ts +152 -0
  105. package/extensions/dev-team/lib/transcript.ts +110 -0
  106. package/extensions/dev-team/lib/trust.ts +52 -0
  107. package/extensions/dev-team/lib/usage-breakdown.ts +176 -0
  108. package/extensions/dev-team/lib/usage-chart.ts +153 -0
  109. package/extensions/dev-team/lib/usage-command.ts +107 -0
  110. package/extensions/dev-team/lib/usage-history.ts +203 -0
  111. package/extensions/dev-team/lib/usage-render.ts +225 -0
  112. package/extensions/dev-team/lib/usage-split-bar.ts +127 -0
  113. package/extensions/dev-team/lib/usage-state.ts +116 -0
  114. package/extensions/dev-team/lib/usage-text.ts +159 -0
  115. package/extensions/dev-team/lib/usage-view.ts +109 -0
  116. package/hooks/__pycache__/refactor_test_freeze_guard.cpython-314.pyc +0 -0
  117. package/hooks/agent_dispatch_ledger.py +190 -0
  118. package/hooks/autocompact_setup_nudge.py +99 -0
  119. package/hooks/bash_retry_guard.py +228 -0
  120. package/hooks/boundary_events_write_guard.py +352 -0
  121. package/hooks/code_intelligence_nudge.py +293 -0
  122. package/hooks/code_intelligence_turn_mark.py +317 -0
  123. package/hooks/codegraph_bootstrap.py +139 -0
  124. package/hooks/contract_version_guard.py +362 -0
  125. package/hooks/cost_meter.py +106 -0
  126. package/hooks/destructive-commands.json +62 -0
  127. package/hooks/destructive_guard.py +477 -0
  128. package/hooks/eval_compliance_check.py +440 -0
  129. package/hooks/guards.json +17 -0
  130. package/hooks/hooks.json +323 -0
  131. package/hooks/internal_double_gate.py +296 -0
  132. package/hooks/js_fp_review.py +212 -0
  133. package/hooks/knowledge_index.py +119 -0
  134. package/hooks/lib/__pycache__/artifact_paths.cpython-314.pyc +0 -0
  135. package/hooks/lib/__pycache__/atomic_state.cpython-314.pyc +0 -0
  136. package/hooks/lib/__pycache__/autocompact_config.cpython-314.pyc +0 -0
  137. package/hooks/lib/__pycache__/boundary_events.cpython-314.pyc +0 -0
  138. package/hooks/lib/__pycache__/doc_classification.cpython-314.pyc +0 -0
  139. package/hooks/lib/__pycache__/gh_pr_create_detect.cpython-314.pyc +0 -0
  140. package/hooks/lib/__pycache__/git_safe_diff.cpython-314.pyc +0 -0
  141. package/hooks/lib/__pycache__/instrument_log.cpython-314.pyc +0 -0
  142. package/hooks/lib/__pycache__/metrics_query.cpython-314.pyc +0 -0
  143. package/hooks/lib/__pycache__/plugin_version.cpython-314.pyc +0 -0
  144. package/hooks/lib/__pycache__/pre_commit_doc_classifier.cpython-314.pyc +0 -0
  145. package/hooks/lib/__pycache__/review_agent_registry.cpython-314.pyc +0 -0
  146. package/hooks/lib/__pycache__/review_gate_corroboration.cpython-314.pyc +0 -0
  147. package/hooks/lib/__pycache__/review_gate_hash.cpython-314.pyc +0 -0
  148. package/hooks/lib/__pycache__/review_verdicts.cpython-314.pyc +0 -0
  149. package/hooks/lib/__pycache__/stdin_json.cpython-314.pyc +0 -0
  150. package/hooks/lib/__pycache__/stryker_invocation.cpython-314.pyc +0 -0
  151. package/hooks/lib/__pycache__/telemetry_consent.cpython-314.pyc +0 -0
  152. package/hooks/lib/__pycache__/test_file_classify.cpython-314.pyc +0 -0
  153. package/hooks/lib/__pycache__/token_efficiency_limits.cpython-314.pyc +0 -0
  154. package/hooks/lib/__pycache__/verify_guard_state.cpython-314.pyc +0 -0
  155. package/hooks/lib/__pycache__/xunit_v3_operator_gate.cpython-314.pyc +0 -0
  156. package/hooks/lib/agent_skill_hints.py +74 -0
  157. package/hooks/lib/artifact_paths.py +263 -0
  158. package/hooks/lib/atomic_state.py +557 -0
  159. package/hooks/lib/autocompact_config.py +103 -0
  160. package/hooks/lib/autoship_log.py +106 -0
  161. package/hooks/lib/banned_scripts_policy.py +51 -0
  162. package/hooks/lib/boundary_events.py +436 -0
  163. package/hooks/lib/build_knowledge_index.py +504 -0
  164. package/hooks/lib/build_skills_index.py +361 -0
  165. package/hooks/lib/build_state.py +116 -0
  166. package/hooks/lib/classify_ship_outcome.py +126 -0
  167. package/hooks/lib/config_changelog_schema.py +115 -0
  168. package/hooks/lib/cost_meter.py +955 -0
  169. package/hooks/lib/doc_classification.py +116 -0
  170. package/hooks/lib/gh_pr_create_detect.py +136 -0
  171. package/hooks/lib/git_safe_diff.py +123 -0
  172. package/hooks/lib/instrument_log.py +66 -0
  173. package/hooks/lib/iteration_journal_gate.py +197 -0
  174. package/hooks/lib/knowledge_index_paths.py +88 -0
  175. package/hooks/lib/mcp_json_repowise.py +177 -0
  176. package/hooks/lib/metrics_query.py +202 -0
  177. package/hooks/lib/minimal_yaml.py +434 -0
  178. package/hooks/lib/plugin_version.py +142 -0
  179. package/hooks/lib/pre_commit_detect.py +537 -0
  180. package/hooks/lib/pre_commit_doc_classifier.py +126 -0
  181. package/hooks/lib/pricing.py +118 -0
  182. package/hooks/lib/report_pdf.py +371 -0
  183. package/hooks/lib/review_agent_registry.py +142 -0
  184. package/hooks/lib/review_dispatch_ledger.py +101 -0
  185. package/hooks/lib/review_gate_corroboration.py +521 -0
  186. package/hooks/lib/review_gate_hash.py +252 -0
  187. package/hooks/lib/review_gate_normalized_hash.py +1115 -0
  188. package/hooks/lib/review_verdicts.py +301 -0
  189. package/hooks/lib/run_report.py +160 -0
  190. package/hooks/lib/skill_categories.yaml +125 -0
  191. package/hooks/lib/stdin_json.py +57 -0
  192. package/hooks/lib/stryker_invocation.py +102 -0
  193. package/hooks/lib/telemetry_consent.py +41 -0
  194. package/hooks/lib/telemetry_report.py +108 -0
  195. package/hooks/lib/test_file_classify.py +160 -0
  196. package/hooks/lib/token_efficiency_limits.py +51 -0
  197. package/hooks/lib/turn_identity.py +77 -0
  198. package/hooks/lib/verify_guard_state.py +110 -0
  199. package/hooks/lib/workflow_state.py +206 -0
  200. package/hooks/lib/xunit_v3_operator_gate.py +596 -0
  201. package/hooks/mcp_json_repowise_nudge.py +74 -0
  202. package/hooks/mutation_adapters/__init__.py +7 -0
  203. package/hooks/mutation_adapters/__pycache__/__init__.cpython-314.pyc +0 -0
  204. package/hooks/mutation_adapters/__pycache__/lib.cpython-314.pyc +0 -0
  205. package/hooks/mutation_adapters/__pycache__/mutmut.cpython-314.pyc +0 -0
  206. package/hooks/mutation_adapters/__pycache__/pitest.cpython-314.pyc +0 -0
  207. package/hooks/mutation_adapters/__pycache__/stryker.cpython-314.pyc +0 -0
  208. package/hooks/mutation_adapters/__pycache__/stryker_net.cpython-314.pyc +0 -0
  209. package/hooks/mutation_adapters/lib.py +478 -0
  210. package/hooks/mutation_adapters/mutmut.py +188 -0
  211. package/hooks/mutation_adapters/pitest.py +266 -0
  212. package/hooks/mutation_adapters/stryker.py +157 -0
  213. package/hooks/mutation_adapters/stryker_net.py +264 -0
  214. package/hooks/mutation_gate.py +193 -0
  215. package/hooks/mutation_testing_smoke_gate.py +371 -0
  216. package/hooks/pending_review_notify.py +121 -0
  217. package/hooks/phase_marker.py +138 -0
  218. package/hooks/post_compact_state_reinject.py +180 -0
  219. package/hooks/post_format.py +115 -0
  220. package/hooks/pre_commit_knowledge_index.py +128 -0
  221. package/hooks/pre_commit_review.py +66 -0
  222. package/hooks/pre_pr_review.py +694 -0
  223. package/hooks/pre_tool_guard.py +405 -0
  224. package/hooks/py.sh +73 -0
  225. package/hooks/refactor-bash-write-patterns.json +29 -0
  226. package/hooks/refactor_test_bash_guard.py +253 -0
  227. package/hooks/refactor_test_freeze_guard.py +139 -0
  228. package/hooks/refactor_test_revert_guard.py +186 -0
  229. package/hooks/repo_review_nudge.py +287 -0
  230. package/hooks/review_verdict_recorder.py +464 -0
  231. package/hooks/scan_bash_command_for_banned_scripts.py +428 -0
  232. package/hooks/scan_worktree_for_banned_scripts.py +238 -0
  233. package/hooks/session_learning_trigger.py +248 -0
  234. package/hooks/skills_index.py +126 -0
  235. package/hooks/stryker_xunit_shim_guard.py +571 -0
  236. package/hooks/subagent_completion_guard.py +309 -0
  237. package/hooks/subagent_skill_context.py +139 -0
  238. package/hooks/task_completion_metrics.py +216 -0
  239. package/hooks/tdd_guard.py +229 -0
  240. package/hooks/telemetry.py +341 -0
  241. package/hooks/token_efficiency_review.py +194 -0
  242. package/hooks/verify_guard.py +183 -0
  243. package/hooks/verify_guard_edit_marker.py +73 -0
  244. package/hooks/version_check.py +173 -0
  245. package/knowledge/accepted-risks-schema.md +98 -0
  246. package/knowledge/adr-decision-criteria.md +64 -0
  247. package/knowledge/adversarial-review-protocol.md +139 -0
  248. package/knowledge/agent-registry.md +228 -0
  249. package/knowledge/agent-review-methodology.md +80 -0
  250. package/knowledge/ai-friendly-repo-guidelines.md +67 -0
  251. package/knowledge/architecture-assessment.md +96 -0
  252. package/knowledge/artifact-lifecycle.md +57 -0
  253. package/knowledge/cd-maturity-model.md +82 -0
  254. package/knowledge/cd-test-architecture.md +190 -0
  255. package/knowledge/ci-cd-file-scope.md +24 -0
  256. package/knowledge/codegraph-vs-graphify.md +192 -0
  257. package/knowledge/component-test-patterns.md +139 -0
  258. package/knowledge/database-change-management.md +80 -0
  259. package/knowledge/database-test-patterns.md +79 -0
  260. package/knowledge/decision-defaults.md +88 -0
  261. package/knowledge/dependency-breaking-techniques.md +116 -0
  262. package/knowledge/deployment-pipeline.md +86 -0
  263. package/knowledge/design-smells.md +122 -0
  264. package/knowledge/directory-enumeration.md +38 -0
  265. package/knowledge/domain-modeling.md +123 -0
  266. package/knowledge/evidence-bundle.md +90 -0
  267. package/knowledge/exploratory-testing-field-guide.md +122 -0
  268. package/knowledge/failure-routing.md +28 -0
  269. package/knowledge/fixture-construction.md +56 -0
  270. package/knowledge/frontend-component-architecture.md +139 -0
  271. package/knowledge/gherkin-quality-review-dispatch.md +135 -0
  272. package/knowledge/index.json +6766 -0
  273. package/knowledge/internal-collaborator-doubling.md +101 -0
  274. package/knowledge/legacy-test-strategy.md +71 -0
  275. package/knowledge/long-run-waiting.md +66 -0
  276. package/knowledge/microservice-testing.md +71 -0
  277. package/knowledge/model-pricing.json +23 -0
  278. package/knowledge/mutation-score-formulas.md +60 -0
  279. package/knowledge/object-calisthenics.md +147 -0
  280. package/knowledge/oracle-provenance.md +94 -0
  281. package/knowledge/orchestrator-script-implementation.md +185 -0
  282. package/knowledge/owasp-detection.md +148 -0
  283. package/knowledge/plan-review-rubric.md +56 -0
  284. package/knowledge/proxy-connectivity.md +62 -0
  285. package/knowledge/reactive-effect-patterns.md +73 -0
  286. package/knowledge/recon-inventory-excludes.txt +32 -0
  287. package/knowledge/references/bdd-value-guide.md +61 -0
  288. package/knowledge/references/csharp-http-client-testing.md +264 -0
  289. package/knowledge/release-strategies.md +74 -0
  290. package/knowledge/report-output-location.md +117 -0
  291. package/knowledge/report-pdf-integration.md +63 -0
  292. package/knowledge/report-print.css +129 -0
  293. package/knowledge/report-template.md +114 -0
  294. package/knowledge/report-to-pdf.md +69 -0
  295. package/knowledge/request-processing-flow.md +63 -0
  296. package/knowledge/result-verification.md +52 -0
  297. package/knowledge/review-agent-output-contract.md +121 -0
  298. package/knowledge/review-lens-classification.md +113 -0
  299. package/knowledge/review-rubric.md +62 -0
  300. package/knowledge/review-template.md +104 -0
  301. package/knowledge/rule-fixtures/A02.insecure-random-js/negative.js +1 -0
  302. package/knowledge/rule-fixtures/A02.insecure-random-js/positive.js +1 -0
  303. package/knowledge/rule-fixtures/A02.weak-hashing-md5/negative.py +1 -0
  304. package/knowledge/rule-fixtures/A02.weak-hashing-md5/positive.py +1 -0
  305. package/knowledge/rule-fixtures/A03.command-injection/negative.js +1 -0
  306. package/knowledge/rule-fixtures/A03.command-injection/positive.js +1 -0
  307. package/knowledge/rule-fixtures/A03.sql-injection/negative.js +1 -0
  308. package/knowledge/rule-fixtures/A03.sql-injection/positive.js +1 -0
  309. package/knowledge/rule-fixtures/A03.xss-innerhtml/negative.js +1 -0
  310. package/knowledge/rule-fixtures/A03.xss-innerhtml/positive.js +1 -0
  311. package/knowledge/rule-fixtures/A05.cors-wildcard/negative.js +1 -0
  312. package/knowledge/rule-fixtures/A05.cors-wildcard/positive.js +1 -0
  313. package/knowledge/rule-fixtures/A05.default-credentials/negative.js +1 -0
  314. package/knowledge/rule-fixtures/A05.default-credentials/positive.js +1 -0
  315. package/knowledge/rule-fixtures/A07.jwt-alg-none/negative.js +1 -0
  316. package/knowledge/rule-fixtures/A07.jwt-alg-none/positive.js +1 -0
  317. package/knowledge/rule-fixtures/A08.binary-formatter/negative.cs +1 -0
  318. package/knowledge/rule-fixtures/A08.binary-formatter/positive.cs +1 -0
  319. package/knowledge/rule-fixtures/A08.js-eval/negative.js +1 -0
  320. package/knowledge/rule-fixtures/A08.js-eval/positive.js +1 -0
  321. package/knowledge/rule-fixtures/A08.object-input-stream/negative.java +1 -0
  322. package/knowledge/rule-fixtures/A08.object-input-stream/positive.java +1 -0
  323. package/knowledge/schemas/disposition-register-v1.json +65 -0
  324. package/knowledge/schemas/recon-envelope-v1.json +198 -0
  325. package/knowledge/schemas/unified-finding-v1.json +72 -0
  326. package/knowledge/security-primitives-contract.md +301 -0
  327. package/knowledge/security-review-rule-map.yaml +107 -0
  328. package/knowledge/skills-registry.md +72 -0
  329. package/knowledge/task-size-classifier.md +103 -0
  330. package/knowledge/telemetry-schema.md +881 -0
  331. package/knowledge/test-automation-maturity.md +56 -0
  332. package/knowledge/test-automation-principles.md +71 -0
  333. package/knowledge/test-cadence-tradeoffs.md +68 -0
  334. package/knowledge/test-doubles.md +105 -0
  335. package/knowledge/test-file-indicators.md +22 -0
  336. package/knowledge/test-layer-gates.md +35 -0
  337. package/knowledge/test-matrix-examples/django-batch.md +24 -0
  338. package/knowledge/test-matrix-examples/dotnet-grpc-fronting-api.md +90 -0
  339. package/knowledge/test-matrix-examples/dotnet-http-consumer.md +131 -0
  340. package/knowledge/test-matrix-examples/react-node-spa.md +24 -0
  341. package/knowledge/test-matrix-examples/spring-boot-service.md +25 -0
  342. package/knowledge/test-matrix-examples/ssr-htmx.md +24 -0
  343. package/knowledge/test-organization.md +70 -0
  344. package/knowledge/test-pyramid.md +84 -0
  345. package/knowledge/test-refactoring.md +67 -0
  346. package/knowledge/test-review-division-of-labor.md +85 -0
  347. package/knowledge/test-smells.md +80 -0
  348. package/knowledge/test-stack-profiles/bdd-frameworks.md +235 -0
  349. package/knowledge/test-stack-profiles/django.md +13 -0
  350. package/knowledge/test-stack-profiles/dotnet.md +18 -0
  351. package/knowledge/test-stack-profiles/go.md +16 -0
  352. package/knowledge/test-stack-profiles/node.md +16 -0
  353. package/knowledge/test-stack-profiles/react.md +12 -0
  354. package/knowledge/test-stack-profiles/spring-boot.md +16 -0
  355. package/knowledge/test-stack-profiles/ssr-htmx.md +14 -0
  356. package/knowledge/test-stack-profiles/vue.md +12 -0
  357. package/knowledge/test-strategy.md +70 -0
  358. package/knowledge/testability-patterns.md +240 -0
  359. package/knowledge/testing-quadrants.md +44 -0
  360. package/knowledge/testing-techniques/approval.md +15 -0
  361. package/knowledge/testing-techniques/chaos.md +17 -0
  362. package/knowledge/testing-techniques/fuzz.md +15 -0
  363. package/knowledge/testing-techniques/property-based.md +15 -0
  364. package/knowledge/testing-techniques/schema-validation.md +15 -0
  365. package/knowledge/testing-techniques/screenshot.md +15 -0
  366. package/knowledge/three-phase-workflow.md +198 -0
  367. package/knowledge/value-patterns.md +55 -0
  368. package/knowledge/verification-mode.md +116 -0
  369. package/knowledge/virtual-service-libraries.md +75 -0
  370. package/knowledge/wave-consolidation-guidance.md +21 -0
  371. package/overrides/agents/Explore.md +15 -0
  372. package/overrides/agents/general-purpose.md +10 -0
  373. package/overrides/notes/autoship.md +6 -0
  374. package/overrides/notes/issues-from-assessment.md +3 -0
  375. package/overrides/notes/issues-from-plan.md +3 -0
  376. package/overrides/notes/mutation-night-watch.md +3 -0
  377. package/overrides/notes/mutation-testing.md +3 -0
  378. package/overrides/notes/pr.md +7 -0
  379. package/overrides/notes/project-init.md +6 -0
  380. package/overrides/notes/setup.md +13 -0
  381. package/overrides/notes/specs.md +3 -0
  382. package/overrides/skills/headless-run/SKILL.md +45 -0
  383. package/overrides/skills/upgrade/SKILL.md +30 -0
  384. package/overrides/skills/version/SKILL.md +25 -0
  385. package/package.json +36 -0
  386. package/scripts/authoring_digest.py +93 -0
  387. package/scripts/autoship_discover.py +121 -0
  388. package/scripts/autoship_group.py +409 -0
  389. package/scripts/autoship_proposals.py +494 -0
  390. package/scripts/autoship_queue.py +291 -0
  391. package/scripts/autoship_reclaim.py +495 -0
  392. package/scripts/build_jobs.py +108 -0
  393. package/scripts/build_rollback_point.py +240 -0
  394. package/scripts/build_slice_scope.py +157 -0
  395. package/scripts/build_wave.py +109 -0
  396. package/scripts/build_wave_reconcile.py +252 -0
  397. package/scripts/build_worktree_baseref.py +113 -0
  398. package/scripts/check_agent_scope.py +117 -0
  399. package/scripts/check_agent_tool_mapping.py +213 -0
  400. package/scripts/check_review_agent_mcp_tools.py +317 -0
  401. package/scripts/check_security_assessment_mcp_tools.py +165 -0
  402. package/scripts/checkpoint_abort.py +502 -0
  403. package/scripts/claude_setup_review.py +438 -0
  404. package/scripts/codebase_recon.py +556 -0
  405. package/scripts/coverage_config.py +623 -0
  406. package/scripts/coverage_delta_steering.py +330 -0
  407. package/scripts/coverage_discovery_dotnet.py +315 -0
  408. package/scripts/coverage_discovery_java.py +742 -0
  409. package/scripts/coverage_discovery_js.py +546 -0
  410. package/scripts/coverage_gap_ranking.py +556 -0
  411. package/scripts/coverage_readiness.py +455 -0
  412. package/scripts/coverage_report_parse.py +521 -0
  413. package/scripts/detect_bdd_convention.py +252 -0
  414. package/scripts/eval_ablation.py +376 -0
  415. package/scripts/gherkin_analysis_coverage_gate.py +306 -0
  416. package/scripts/gherkin_cross_feature_duplicate_titles_gate.py +173 -0
  417. package/scripts/gherkin_effectiveness_rollup.py +238 -0
  418. package/scripts/gherkin_failure_path_gate.py +206 -0
  419. package/scripts/gherkin_feature_merge.py +720 -0
  420. package/scripts/gherkin_stub_gate.py +163 -0
  421. package/scripts/gherkin_stub_merge.py +479 -0
  422. package/scripts/git_origin_host.py +88 -0
  423. package/scripts/install-java-static-analysis.py +110 -0
  424. package/scripts/issue_deps.py +74 -0
  425. package/scripts/lib/_bdd_markers.py +28 -0
  426. package/scripts/lib/_gherkin_text.py +93 -0
  427. package/scripts/lib/_vendored_tree.py +70 -0
  428. package/scripts/lib/autoship_state.py +397 -0
  429. package/scripts/lib/claude_md_guard.py +226 -0
  430. package/scripts/lib/deterministic_recon.py +446 -0
  431. package/scripts/lib/mcp_tool_grants.py +211 -0
  432. package/scripts/lib/plan_parse.py +386 -0
  433. package/scripts/lib/review_result.py +84 -0
  434. package/scripts/lib/review_roster.py +86 -0
  435. package/scripts/lib/session_log/__init__.py +34 -0
  436. package/scripts/lib/session_log/__pycache__/__init__.cpython-314.pyc +0 -0
  437. package/scripts/lib/session_log/__pycache__/records.cpython-314.pyc +0 -0
  438. package/scripts/lib/session_log/classify.py +231 -0
  439. package/scripts/lib/session_log/corrections.py +194 -0
  440. package/scripts/lib/session_log/discovery.py +108 -0
  441. package/scripts/lib/session_log/records.py +218 -0
  442. package/scripts/lib/session_log/redact.py +76 -0
  443. package/scripts/lib/session_log/signals.py +373 -0
  444. package/scripts/lib/session_report_downstream.py +614 -0
  445. package/scripts/lib/session_report_maintainer.py +1273 -0
  446. package/scripts/lib/session_report_shared.py +262 -0
  447. package/scripts/lib/settings_hook_guard.py +157 -0
  448. package/scripts/lib/slug.py +33 -0
  449. package/scripts/lib/stub_extractors/__init__.py +82 -0
  450. package/scripts/lib/stub_extractors/_common.py +328 -0
  451. package/scripts/lib/stub_extractors/csharp.py +19 -0
  452. package/scripts/lib/stub_extractors/go.py +173 -0
  453. package/scripts/lib/stub_extractors/java.py +18 -0
  454. package/scripts/lib/stub_extractors/jsts.py +126 -0
  455. package/scripts/mutation_stack_sections.py +149 -0
  456. package/scripts/mutation_yield_steering.py +345 -0
  457. package/scripts/orchestrator.py +895 -0
  458. package/scripts/plan_gherkin_export.py +227 -0
  459. package/scripts/plan_waves.py +208 -0
  460. package/scripts/pr_close_keyword_lint.py +108 -0
  461. package/scripts/progress_guardian.py +888 -0
  462. package/scripts/recon_inventory.py +273 -0
  463. package/scripts/review_findings_log.py +93 -0
  464. package/scripts/run_invariants.py +124 -0
  465. package/scripts/select_lenses.py +640 -0
  466. package/scripts/session_report.py +486 -0
  467. package/scripts/set_autocompact_env.py +221 -0
  468. package/scripts/ship_resume_guard.py +135 -0
  469. package/scripts/ship_review_gate.py +63 -0
  470. package/scripts/specs_convention_marker.py +103 -0
  471. package/scripts/test_improve_resume.py +277 -0
  472. package/scripts/test_review_mechanics.py +958 -0
  473. package/scripts/token_efficiency_review.py +322 -0
  474. package/scripts/verdict_scope.py +285 -0
  475. package/scripts/verify_gherkin_quality_critic_isolation.py +296 -0
  476. package/scripts/verify_tier.py +157 -0
  477. package/skills/adr-tools/SKILL.md +118 -0
  478. package/skills/agent-readiness/SKILL.md +105 -0
  479. package/skills/agent-readiness/ai_friendly_analyzers.py +326 -0
  480. package/skills/agent-readiness/scanner.py +441 -0
  481. package/skills/agent-readiness/scorecard.yaml +88 -0
  482. package/skills/api-design/SKILL.md +115 -0
  483. package/skills/apply-fixes/SKILL.md +171 -0
  484. package/skills/apply-test-doubles/SKILL.md +321 -0
  485. package/skills/artifact-lifecycle/SKILL.md +127 -0
  486. package/skills/autoship/SKILL.md +1124 -0
  487. package/skills/benchmark/SKILL.md +105 -0
  488. package/skills/branch-workflow/SKILL.md +89 -0
  489. package/skills/browse/SKILL.md +184 -0
  490. package/skills/browser-testing/SKILL.md +62 -0
  491. package/skills/browser-testing/references/playwright-patterns.md +216 -0
  492. package/skills/build/SKILL.md +422 -0
  493. package/skills/build/references/static-self-heal.md +245 -0
  494. package/skills/careful/SKILL.md +72 -0
  495. package/skills/cd-test-architecture/SKILL.md +371 -0
  496. package/skills/ci-debugging/SKILL.md +105 -0
  497. package/skills/co-evolution-audit/SKILL.md +269 -0
  498. package/skills/code-review/SKILL.md +1015 -0
  499. package/skills/code-review/examples/aggregated-sample.json +56 -0
  500. package/skills/code-review/examples/sample-report.md +41 -0
  501. package/skills/code-review/output-format.md +478 -0
  502. package/skills/code-review/scripts/activation.py +86 -0
  503. package/skills/code-review/scripts/change_impact.py +357 -0
  504. package/skills/code-review/scripts/change_shape.py +372 -0
  505. package/skills/code-review/scripts/change_size.py +212 -0
  506. package/skills/code-review/scripts/changed_file_list.py +141 -0
  507. package/skills/code-review/scripts/closing_pass.py +187 -0
  508. package/skills/code-review/scripts/consolidate.py +277 -0
  509. package/skills/code-review/scripts/contract_failure_report.py +185 -0
  510. package/skills/code-review/scripts/dispatch_reconcile.py +66 -0
  511. package/skills/code-review/scripts/dispatch_waves.py +164 -0
  512. package/skills/code-review/scripts/finding_signature.py +446 -0
  513. package/skills/code-review/scripts/ledger.py +283 -0
  514. package/skills/code-review/scripts/partition.py +169 -0
  515. package/skills/code-review/scripts/render_tiered_findings.py +274 -0
  516. package/skills/code-review/scripts/repo_invariants.py +1066 -0
  517. package/skills/code-review/scripts/review_context_pack.py +306 -0
  518. package/skills/code-review/scripts/review_round_log.py +345 -0
  519. package/skills/code-review/scripts/review_value_coverage.py +297 -0
  520. package/skills/code-review/scripts/validate_review_output.py +467 -0
  521. package/skills/code-review/sliced-mode.md +205 -0
  522. package/skills/competitive-analysis/SKILL.md +191 -0
  523. package/skills/context-loading-protocol/SKILL.md +157 -0
  524. package/skills/continue/SKILL.md +90 -0
  525. package/skills/cost-report/SKILL.md +178 -0
  526. package/skills/coverage-baseline/SKILL.md +335 -0
  527. package/skills/coverage-baseline/references/multi-project-discovery.md +202 -0
  528. package/skills/coverage-delta/SKILL.md +181 -0
  529. package/skills/coverage-delta/references/mutation-gate.md +70 -0
  530. package/skills/design-doc/SKILL.md +95 -0
  531. package/skills/design-interrogation/SKILL.md +89 -0
  532. package/skills/design-it-twice/SKILL.md +91 -0
  533. package/skills/docker-image-audit/SKILL.md +108 -0
  534. package/skills/docker-image-audit/references/install-guide.md +64 -0
  535. package/skills/docker-image-audit/references/report-template.md +73 -0
  536. package/skills/docker-image-create/SKILL.md +185 -0
  537. package/skills/domain-analysis/SKILL.md +183 -0
  538. package/skills/domain-driven-design/SKILL.md +194 -0
  539. package/skills/exploratory-testing/SKILL.md +108 -0
  540. package/skills/explore/SKILL.md +51 -0
  541. package/skills/farley-score/SKILL.md +165 -0
  542. package/skills/feature-file-validation/SKILL.md +78 -0
  543. package/skills/feature-file-validation/references/validation-rules.md +115 -0
  544. package/skills/feedback-learning/SKILL.md +414 -0
  545. package/skills/fix/SKILL.md +450 -0
  546. package/skills/freeze/SKILL.md +68 -0
  547. package/skills/frontend-architecture/SKILL.md +113 -0
  548. package/skills/gherkin-derive/SKILL.md +630 -0
  549. package/skills/gherkin-public/SKILL.md +266 -0
  550. package/skills/governance-compliance/SKILL.md +150 -0
  551. package/skills/guard/SKILL.md +75 -0
  552. package/skills/handoff/SKILL.md +139 -0
  553. package/skills/handoff/references/summary-templates.md +242 -0
  554. package/skills/harness-audit/SKILL.md +751 -0
  555. package/skills/harness-audit/scripts/lesson_validate.py +386 -0
  556. package/skills/harness-audit/scripts/redundancy_criterion.py +188 -0
  557. package/skills/headless-run/SKILL.md +45 -0
  558. package/skills/headless-run/scripts/isolated_dispatch.py +381 -0
  559. package/skills/help/SKILL.md +72 -0
  560. package/skills/hexagonal-architecture/SKILL.md +85 -0
  561. package/skills/human-oversight-protocol/SKILL.md +224 -0
  562. package/skills/issues-from-assessment/SKILL.md +223 -0
  563. package/skills/issues-from-plan/SKILL.md +133 -0
  564. package/skills/legacy-code/SKILL.md +132 -0
  565. package/skills/mermaid-diagramming/SKILL.md +120 -0
  566. package/skills/mutation-night-watch/SKILL.md +154 -0
  567. package/skills/mutation-night-watch/references/scheduling.md +135 -0
  568. package/skills/mutation-testing/SKILL.md +396 -0
  569. package/skills/mutation-testing/references/languages/csharp-stryker-net.md +676 -0
  570. package/skills/mutation-testing/references/languages/go-go-mutesting.md +95 -0
  571. package/skills/mutation-testing/references/languages/java-pitest.md +77 -0
  572. package/skills/mutation-testing/references/languages/javascript-stryker.md +188 -0
  573. package/skills/mutation-testing/references/languages/python-mutmut.md +97 -0
  574. package/skills/mutation-testing/references/time-estimation.md +34 -0
  575. package/skills/mutation-testing/references/tool-detection.md +15 -0
  576. package/skills/mutation-testing/references/workflow-callers.md +23 -0
  577. package/skills/mutation-testing/scripts/__pycache__/xunit_v3_feature_detector.cpython-314.pyc +0 -0
  578. package/skills/mutation-testing/scripts/csharp_stryker_net_slice_runner.py +635 -0
  579. package/skills/mutation-testing/scripts/csharp_stryker_net_status_loop.py +525 -0
  580. package/skills/mutation-testing/scripts/csharp_stryker_net_wrapper.py +681 -0
  581. package/skills/mutation-testing/scripts/mutation_baseline_reuse.py +292 -0
  582. package/skills/mutation-testing/scripts/mutation_exclude_policy.py +268 -0
  583. package/skills/mutation-testing/scripts/mutation_feasibility_gate.py +463 -0
  584. package/skills/mutation-testing/scripts/mutation_kill_headless.py +331 -0
  585. package/skills/mutation-testing/scripts/mutation_kill_insert.py +199 -0
  586. package/skills/mutation-testing/scripts/mutation_kill_insert_python.py +150 -0
  587. package/skills/mutation-testing/scripts/mutation_kill_loop.py +869 -0
  588. package/skills/mutation-testing/scripts/mutation_kill_loop_python.py +949 -0
  589. package/skills/mutation-testing/scripts/mutation_kill_retry.py +592 -0
  590. package/skills/mutation-testing/scripts/mutation_kill_shared.py +620 -0
  591. package/skills/mutation-testing/scripts/mutation_nightwatch.py +462 -0
  592. package/skills/mutation-testing/scripts/mutation_nightwatch_stacks.py +425 -0
  593. package/skills/mutation-testing/scripts/mutation_report.py +743 -0
  594. package/skills/mutation-testing/scripts/mutation_report_cli.py +175 -0
  595. package/skills/mutation-testing/scripts/mutation_safety_gate.py +69 -0
  596. package/skills/mutation-testing/scripts/stryker_shard_pipeline.py +847 -0
  597. package/skills/mutation-testing/scripts/stryker_shard_setup.py +440 -0
  598. package/skills/mutation-testing/scripts/stryker_timeout_retry.py +142 -0
  599. package/skills/mutation-testing/scripts/xunit_v3_feature_detector.py +341 -0
  600. package/skills/performance-benchmark/SKILL.md +174 -0
  601. package/skills/performance-benchmark/examples/report-format.md +43 -0
  602. package/skills/performance-benchmark/references/benchmark-script.md +169 -0
  603. package/skills/performance-metrics/SKILL.md +265 -0
  604. package/skills/plan/SKILL.md +199 -0
  605. package/skills/plan/references/gherkin-persistence.md +43 -0
  606. package/skills/plan/references/plan-template.md +182 -0
  607. package/skills/pr/SKILL.md +289 -0
  608. package/skills/pr/scripts/gate_retry_state.py +368 -0
  609. package/skills/project-init/README.md +141 -0
  610. package/skills/project-init/SKILL.md +1197 -0
  611. package/skills/project-init/evals/evals.json +200 -0
  612. package/skills/project-init/references/capability-tools.md +55 -0
  613. package/skills/project-init/references/configs.md +221 -0
  614. package/skills/property-based-testing/SKILL.md +121 -0
  615. package/skills/property-based-testing/fixtures/invariant_fixture.py +15 -0
  616. package/skills/property-based-testing/fixtures/js-roundtrip/README.md +42 -0
  617. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/LICENSE +21 -0
  618. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/README.md +263 -0
  619. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/fast-check.d.ts +5165 -0
  620. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/fast-check.js +12147 -0
  621. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/package.json +3 -0
  622. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/types57/fast-check.d.ts +5165 -0
  623. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/fast-check.d.ts +5165 -0
  624. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/fast-check.js +12011 -0
  625. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/rolldown-runtime-D7D4PA-g.js +13 -0
  626. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/types57/fast-check.d.ts +5165 -0
  627. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/package.json +94 -0
  628. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/LICENSE +21 -0
  629. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/README.md +168 -0
  630. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/RandomGenerator-DcXj09Ch.d.ts +14 -0
  631. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformBigInt.d.ts +15 -0
  632. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformBigInt.js +38 -0
  633. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat32.d.ts +15 -0
  634. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat32.js +18 -0
  635. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat64.d.ts +15 -0
  636. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat64.js +22 -0
  637. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformInt.d.ts +15 -0
  638. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformInt.js +134 -0
  639. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/RandomGenerator-DcXj09Ch.d.ts +14 -0
  640. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformBigInt.d.ts +15 -0
  641. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformBigInt.js +37 -0
  642. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat32.d.ts +15 -0
  643. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat32.js +17 -0
  644. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat64.d.ts +15 -0
  645. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat64.js +21 -0
  646. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformInt.d.ts +15 -0
  647. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformInt.js +133 -0
  648. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/congruential32.d.ts +7 -0
  649. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/congruential32.js +44 -0
  650. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/mersenne.d.ts +7 -0
  651. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/mersenne.js +90 -0
  652. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xoroshiro128plus.d.ts +7 -0
  653. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xoroshiro128plus.js +80 -0
  654. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xorshift128plus.d.ts +7 -0
  655. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xorshift128plus.js +78 -0
  656. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/package.json +3 -0
  657. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/JumpableRandomGenerator.d.ts +16 -0
  658. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/JumpableRandomGenerator.js +0 -0
  659. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/RandomGenerator.d.ts +2 -0
  660. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/RandomGenerator.js +0 -0
  661. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/generateN.d.ts +6 -0
  662. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/generateN.js +8 -0
  663. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/purify.d.ts +12 -0
  664. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/purify.js +9 -0
  665. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/skipN.d.ts +6 -0
  666. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/skipN.js +6 -0
  667. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/congruential32.d.ts +7 -0
  668. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/congruential32.js +46 -0
  669. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/mersenne.d.ts +7 -0
  670. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/mersenne.js +92 -0
  671. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xoroshiro128plus.d.ts +7 -0
  672. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xoroshiro128plus.js +82 -0
  673. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xorshift128plus.d.ts +7 -0
  674. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xorshift128plus.js +80 -0
  675. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/JumpableRandomGenerator.d.ts +16 -0
  676. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/JumpableRandomGenerator.js +0 -0
  677. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/RandomGenerator.d.ts +2 -0
  678. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/RandomGenerator.js +0 -0
  679. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/generateN.d.ts +6 -0
  680. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/generateN.js +9 -0
  681. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/purify.d.ts +12 -0
  682. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/purify.js +10 -0
  683. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/skipN.d.ts +6 -0
  684. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/skipN.js +7 -0
  685. package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/package.json +133 -0
  686. package/skills/property-based-testing/fixtures/js-roundtrip/package-lock.json +1179 -0
  687. package/skills/property-based-testing/fixtures/js-roundtrip/package.json +14 -0
  688. package/skills/property-based-testing/fixtures/js-roundtrip/roundtrip.js +29 -0
  689. package/skills/property-based-testing/fixtures/js-roundtrip/roundtrip.properties.test.js +16 -0
  690. package/skills/property-based-testing/fixtures/no_property_fixture.py +10 -0
  691. package/skills/property-based-testing/fixtures/roundtrip_fixture.py +16 -0
  692. package/skills/property-based-testing/references/languages/javascript.md +54 -0
  693. package/skills/property-based-testing/scripts/detect_and_dispatch.py +80 -0
  694. package/skills/property-based-testing/scripts/hypothesis_scaffold.py +276 -0
  695. package/skills/proxy-resilience/SKILL.md +84 -0
  696. package/skills/quality-gate-pipeline/SKILL.md +184 -0
  697. package/skills/quality-targets-converge/SKILL.md +254 -0
  698. package/skills/repo-review/SKILL.md +159 -0
  699. package/skills/report-pdf/SKILL.md +66 -0
  700. package/skills/review/SKILL.md +47 -0
  701. package/skills/review-agent/SKILL.md +152 -0
  702. package/skills/review-summary/SKILL.md +73 -0
  703. package/skills/run-report/SKILL.md +70 -0
  704. package/skills/semantic-duplication-scan/SKILL.md +337 -0
  705. package/skills/semantic-scan/SKILL.md +53 -0
  706. package/skills/semgrep-analyze/SKILL.md +139 -0
  707. package/skills/setup/SKILL.md +1122 -0
  708. package/skills/ship/SKILL.md +240 -0
  709. package/skills/source-verification/SKILL.md +210 -0
  710. package/skills/source-verification/scripts/claim_extractor.py +155 -0
  711. package/skills/specs/.size-baseline.json +4 -0
  712. package/skills/specs/SKILL.md +243 -0
  713. package/skills/specs/references/completeness-checklist.md +83 -0
  714. package/skills/specs/references/extraction.md +58 -0
  715. package/skills/specs/references/glossary.md +59 -0
  716. package/skills/specs/references/persistence.md +115 -0
  717. package/skills/specs/references/predictability-check.md +77 -0
  718. package/skills/static-analysis-integration/SKILL.md +235 -0
  719. package/skills/static-analysis-integration/adapters/_envelope.py +26 -0
  720. package/skills/static-analysis-integration/adapters/jscpd-adapter.py +66 -0
  721. package/skills/static-analysis-integration/adapters/lizard-adapter.py +81 -0
  722. package/skills/static-analysis-integration/adapters/mypy-adapter.py +50 -0
  723. package/skills/static-analysis-integration/adapters/mypy-src-layout.py +93 -0
  724. package/skills/static-analysis-integration/adapters/security-review-adapter.py +212 -0
  725. package/skills/static-analysis-integration/maintenance.md +23 -0
  726. package/skills/static-analysis-integration/references/language-setup.md +228 -0
  727. package/skills/static-analysis-integration/references/sarif-parser.md +124 -0
  728. package/skills/static-analysis-integration/references/security-review-adapter.md +118 -0
  729. package/skills/static-analysis-integration/references/tool-configs.md +617 -0
  730. package/skills/static-analysis-integration/rulesets/pmd-quickstart.xml +24 -0
  731. package/skills/stryker-xunit-v2-shim/SKILL.md +274 -0
  732. package/skills/stryker-xunit-v2-shim/references/shim-howto.md +256 -0
  733. package/skills/stryker-xunit-v2-shim/scripts/generate_shim.py +143 -0
  734. package/skills/systematic-debugging/SKILL.md +130 -0
  735. package/skills/telemetry/SKILL.md +75 -0
  736. package/skills/test-audit-disable/SKILL.md +129 -0
  737. package/skills/test-design/SKILL.md +177 -0
  738. package/skills/test-design/scripts/__pycache__/internal_double_detector.cpython-314.pyc +0 -0
  739. package/skills/test-design/scripts/internal_double_detector.py +631 -0
  740. package/skills/test-design-advisor/SKILL.md +166 -0
  741. package/skills/test-driven-development/SKILL.md +169 -0
  742. package/skills/test-health/SKILL.md +262 -0
  743. package/skills/test-improve/SKILL.md +239 -0
  744. package/skills/test-improve/references/phase-0-approach-contract.md +228 -0
  745. package/skills/test-improve/references/phase-1-analyze.md +131 -0
  746. package/skills/test-improve/references/phase-2-baseline.md +121 -0
  747. package/skills/test-improve/references/phase-3-derive-gherkin.md +53 -0
  748. package/skills/test-improve/references/phase-4-plan-fixes.md +34 -0
  749. package/skills/test-improve/references/phase-5-improve.md +215 -0
  750. package/skills/test-improve/references/phase-6-refactor-decision.md +45 -0
  751. package/skills/test-improve/references/phase-7-refactor.md +44 -0
  752. package/skills/test-improve/references/phase-8-validate.md +66 -0
  753. package/skills/test-improve/references/phase-9-close-out-prompt.md +11 -0
  754. package/skills/test-improve/references/phase-9-report.md +62 -0
  755. package/skills/test-improve/references/review-loop.md +92 -0
  756. package/skills/test-improve/templates/executive-summary.md +123 -0
  757. package/skills/threat-modeling/SKILL.md +108 -0
  758. package/skills/triage/SKILL.md +211 -0
  759. package/skills/ubiquitous-language/SKILL.md +192 -0
  760. package/skills/ubiquitous-language/scripts/collect_domain_signals.py +300 -0
  761. package/skills/unfreeze/SKILL.md +37 -0
  762. package/skills/upgrade/SKILL.md +31 -0
  763. package/skills/upgrade/scripts/check_version_drift.py +113 -0
  764. package/skills/upgrade/scripts/enable_autoupdate.py +149 -0
  765. package/skills/version/SKILL.md +25 -0
  766. package/sync/__pycache__/sync_upstream.cpython-314.pyc +0 -0
  767. package/sync/sync_upstream.py +293 -0
  768. package/templates/ACCEPTED-RISKS.md.tmpl +46 -0
  769. package/templates/agents/agent-template.md +151 -0
  770. package/templates/agents/angular-testing.md +66 -0
  771. package/templates/agents/csharp-quality.md +63 -0
  772. package/templates/agents/esm-enforcer.md +52 -0
  773. package/templates/agents/front-end-testing.md +65 -0
  774. package/templates/agents/go-quality.md +65 -0
  775. package/templates/agents/python-quality.md +62 -0
  776. package/templates/agents/react-testing.md +61 -0
  777. package/templates/agents/ts-enforcer.md +60 -0
  778. package/templates/agents/twelve-factor-audit.md +49 -0
  779. package/tools/entropy-check.py +250 -0
  780. package/tools/model-hash-verify.py +213 -0
@@ -0,0 +1,751 @@
1
+ ---
2
+ name: harness-audit
3
+ description: >-
4
+ Analyze review agent effectiveness, model routing, and orchestration complexity
5
+ against actual usage data. Produces a report of harness components that may be
6
+ candidates for simplification or removal. Use periodically to prevent harness
7
+ staleness as model capabilities improve. Audits the dev-team plugin's OWN
8
+ harness from runtime metrics — not your project repo's readiness (for that,
9
+ use /agent-readiness).
10
+ argument-hint: "[--output <path>] [--pdf]"
11
+ user-invocable: true
12
+ allowed-tools: Read, Glob, Grep, Bash(date *, python3 *, jq *), Write
13
+ ---
14
+
15
+ # Harness Audit
16
+
17
+ Role: orchestrator. This command analyzes harness effectiveness — it does not modify agents or configuration.
18
+
19
+ You have been invoked with the `/harness-audit` command.
20
+
21
+ > **Not `/agent-readiness`.** This audits the **dev-team plugin's own harness**
22
+ > (review-agent effectiveness, model tiers, orchestration) from accumulated
23
+ > runtime metrics in `metrics/`. `/agent-readiness` scores the **subject
24
+ > repository's** readiness for AI-assisted development from a static checkout.
25
+ > Different subject, different input, different output. The inward-facing
26
+ > companion here is `/session-review`, whose `session-digest.jsonl` this command
27
+ > consumes (Step 1).
28
+
29
+ **Portable executable, repo-relative data (#1653).** `eval_ablation.py` now ships under `plugins/dev-team/scripts/` — its `--find-latest`/`--jsonl` mode is a generic JSONL reader with no repo-shape assumption, unlike the module's `--mode knowledge`/`--mode agent` paths, which stay monorepo-only (they read `evals/` fixtures never shipped). The call below is therefore `${CLAUDE_PLUGIN_ROOT}`-qualified like any other shipped script; only the `--jsonl` argument stays bare — `metrics/eval-ablation.jsonl` is genuinely *this repo's own* runtime metrics stream, consistent with the note above that this command audits the dev-team plugin's own harness.
30
+
31
+ ## Orchestrator constraints
32
+
33
+ 1. **Do not modify agents or configuration.** Produce a report only. All remediation requires human action.
34
+ 2. **Write the report to a file.** Present only the summary table and next-steps in chat — do not repeat the full report.
35
+ 3. **Be concise.** Use tables and short sentences. No preambles, no filler.
36
+
37
+ ## Parse Arguments
38
+
39
+ Arguments: $ARGUMENTS
40
+
41
+ - `--output <path>`: Write report to a specific path. Default: `.dev-team-reports/harness-audit-<date>.md`
42
+ - `--pdf`: After writing the report, render it to a sibling PDF via `hooks/lib/report_pdf.py`, resolving against the **actual** report path written this run (the `--output` override when given, else the default). See `knowledge/report-pdf-integration.md`. Additive; non-fatal if no engine is available.
43
+
44
+ ## Steps
45
+
46
+ ### 1. Check for metrics data
47
+
48
+ Read metrics JSONL files from `metrics/`. Full field reference for every
49
+ stream below: `${CLAUDE_PLUGIN_ROOT}/knowledge/telemetry-schema.md` — read it
50
+ instead of re-deriving a schema from the emitter. Five complementary streams
51
+ exist:
52
+
53
+ - `metrics/*-task-log.jsonl` — **self-reported** task logs (whatever the model
54
+ chose to record about itself).
55
+ - `metrics/session-digest.jsonl` — **ground-truth** real-session digests from
56
+ `/session-review` (#129): token/cost trends, `rework`/`accuracy` counts, and
57
+ `utilization.never_observed_*`. Prefer this where it disagrees with the
58
+ self-reports, and use `never_observed_*` to corroborate stale-component
59
+ flags. Schema + join: see `docs/eval-system.md` → "Session-review trend
60
+ digest".
61
+ - `~/.claude/metrics/artifact-usage.json` — **per-artifact usage index** written by the
62
+ telemetry hook on each Skill invocation. Use `last_used_at` to identify
63
+ artifacts that have never been observed (absent from the index) or are stale
64
+ (absent or `last_used_at` > 30 days ago). Cross-reference with
65
+ `never_observed_*` in `session-digest.jsonl` for corroboration. See
66
+ `knowledge/artifact-lifecycle.md` for the lifecycle threshold definitions.
67
+ - `metrics/boundary-events.jsonl` — **boundary-level (policy-gateway) events**
68
+ (#859): every guard hook's `block`/`warn`/`bypass` decision plus
69
+ `intervention` keywords, each with the emitting `hook` and a `matched_rule`
70
+ rule ID. Where `session-digest.jsonl`'s `rework` counts show outcomes
71
+ without causes, join on `session_id` (when present on both streams) to
72
+ attribute friction to a specific hook/rule instead of reasoning from counts
73
+ alone.
74
+ - `metrics/eval-ablation.jsonl` — **causal** per-agent ablation evidence from
75
+ `/agent-eval --ablation <agent>` (#868): a controlled baseline-vs-ablated
76
+ integration-tier delta, not accumulated usage data. When a record exists
77
+ for a drop-candidate agent, Step 3 cites its measured delta/verdict instead
78
+ of relying on `review-value.jsonl` alone.
79
+
80
+ If no metrics data exists or insufficient data is available (fewer than 10 review runs logged), report:
81
+
82
+ > "Insufficient metrics data for a meaningful audit. Run the system for a period to accumulate review data, then re-run `/harness-audit`. Minimum: 10 logged review runs."
83
+
84
+ List what data is missing and exit.
85
+
86
+ ### 2. Check for a stale baseline (re-baseline detection, #860)
87
+
88
+ Report-only — this step never edits `evals/baseline.json` or re-runs evals
89
+ itself; it only decides whether the report needs a **Re-baseline Required**
90
+ section.
91
+
92
+ 1. Read `evals/baseline.json`. If the file is absent, skip this step
93
+ entirely (nothing to compare).
94
+ 2. Read its `model` field (written by `scripts/eval_variance.py
95
+ --write-baseline --model <name>`, per the change-contract flow in
96
+ `skills/feedback-learning/SKILL.md`). **Absent field = pre-migration
97
+ baseline — do not prompt.** This is deliberate: a baseline recorded
98
+ before the `model` field existed carries no false signal either way.
99
+ 3. Read the current session's model from `metrics/session-digest.jsonl`
100
+ (the most recent record's model field) or session metadata.
101
+ 4. Compare. **On mismatch**, the report (Step 8) gains a **Re-baseline
102
+ Required** section instructing the operator to re-run the eval suite and
103
+ re-write the baseline (`/agent-eval` full suite + `eval_variance.py
104
+ --write-baseline --model <current-model>`) before trusting any pre/post
105
+ comparison elsewhere in this report or in a feedback-learning change
106
+ contract. Flag explicitly that scaffolding kept alive by old-model scores
107
+ (e.g. a removal candidate from Step 3 that "still fails" on the old
108
+ model) may now be re-evaluable and possibly removable.
109
+ 5. **On match** (or the field absent), no section is added — this is silent
110
+ success, not a finding.
111
+
112
+ ### 3. Analyze review agent effectiveness
113
+
114
+ For each review agent in the registry (`knowledge/agent-registry.md`):
115
+
116
+ 1. **Finding rate**: How often does this agent produce findings (fail or warn) vs. pass?
117
+ 2. **Zero-fail agents**: Flag agents that have never returned `fail` across all logged reviews. These are removal candidates — they may not be catching real issues.
118
+ 3. **False positive rate**: If correction data exists (from `/apply-fixes`), check how often findings were dismissed vs. applied. Agents with >50% dismissed findings have a high false positive rate.
119
+ 4. **Finding severity distribution**: Is the agent producing mostly minor findings? If >80% of findings are minor severity, consider whether the agent justifies its token cost. Compute this from the `severity_breakdown` object on `metrics/review-value.jsonl` rows (`{errors, warnings, suggestions}`, added in #1256) — aggregate per `agents_run` and treat `suggestions` as the minor bucket. Rows written before #1256 lack the field; count them as "no severity data" and exclude them from the ratio rather than assuming a mix (small-N honesty, consistent with Step 5). If **no** row carries `severity_breakdown`, report this analysis as dark ("severity breakdown unavailable — pre-#1256 metrics") rather than fabricating a distribution.
120
+
121
+ ```bash
122
+ log=".claude/metrics/review-value.jsonl"; [ -f "$log" ] || log="metrics/review-value.jsonl"
123
+ [ -f "$log" ] && jq -s '
124
+ map(select(.severity_breakdown != null))
125
+ | group_by(.agents_run | sort | join(","))
126
+ | map({
127
+ agents: (.[0].agents_run | sort | join(", ")),
128
+ errors: (map(.severity_breakdown.errors) | add // 0),
129
+ warnings: (map(.severity_breakdown.warnings) | add // 0),
130
+ suggestions: (map(.severity_breakdown.suggestions) | add // 0)
131
+ })
132
+ | map(. + {total: (.errors + .warnings + .suggestions)})
133
+ | map(select(.total > 0) | . + {minor_pct: (.suggestions / .total * 100 | round)})' \
134
+ "$log"
135
+ ```
136
+
137
+ ### 4. Analyze review-value fix rates
138
+
139
+ **Check the sample's validity first (#2019).** Run this before reading a single rate, and report its verdict at the top of every section that cites per-lens value:
140
+
141
+ ```bash
142
+ python3 "$CLAUDE_PLUGIN_ROOT/skills/code-review/scripts/review_value_coverage.py" --json
143
+ ```
144
+
145
+ Both writers of `review-value.jsonl` are triggered by agent instruction, not by mechanism, so the rows collected skew toward rounds that found something — an agent that found nothing is markedly less likely to run a "record the review value" step. #1512 measured this directly: ~100% "found something" across a 10-record sample, which nearly justified pruning lenses that were working fine.
146
+
147
+ The script reconciles the value rows against `agent_dispatch_ledger`'s `boundary-events.jsonl` records, which a hook writes whether or not a round found anything. A verdict other than `usable` means the numbers below describe the *collection*, not the lenses:
148
+
149
+ | verdict | what it means for this section |
150
+ |---|---|
151
+ | `no-data` | no rows; report the analysis as dark |
152
+ | `unverifiable` | no dispatch records, so completeness cannot be checked |
153
+ | `undercollected` | rows are a non-random subset; rates are not attributable to lenses |
154
+ | `insufficient` | below the 100-row floor #1512 set |
155
+ | `biased` | quiet rounds are not reaching the log; no-op rates are unreadable |
156
+ | `usable` | rates may be cited |
157
+
158
+ **Never emit a drop candidate, tier-down, or gating recommendation from a non-`usable` sample.** Report the verdict and what would make the sample usable instead. This is the check that would have stopped #1512's ten biased records from being read as evidence.
159
+
160
+ Read `metrics/review-value.jsonl` (written by `/build` per #348, schema in `performance-metrics`). If the file is absent, note it and continue — this section is skippable.
161
+
162
+ For each **checkpoint type** (the `checkpoint` field: `step` or `slice`) and each **agent combination** (`agents_run` list, treated as a set-key), compute:
163
+
164
+ **Exclude read-only rows first (#1257).** Fix-rate ROI is only meaningful for
165
+ fix-applying `/build` checkpoints. A read-only review (`source: "code-review"`)
166
+ never applies fixes, so **every** row it produces has `issues_fixed: 0` and a 0%
167
+ fix rate — feeding those to the drop-candidate logic falsely flags a whole panel
168
+ that may be surfacing real defects (the 2026-07-20 run mislabeled all 7 agents
169
+ this way). The `jq` filters to the **fix-applying** sources — `build-checkpoint`
170
+ and, since #1962, `build-backstop` (both run the review-fix loop and can move
171
+ `issues_fixed`), treating an absent `source` as `build-checkpoint` for
172
+ back-compat — and drops `outcome: "skipped"` rows (a backstop suppressed by
173
+ `--backstop-review=skip` never ran, so counting it would dilute every rate with
174
+ a non-event), **before** grouping:
175
+
176
+ ```bash
177
+ log=".claude/metrics/review-value.jsonl"; [ -f "$log" ] || log="metrics/review-value.jsonl"
178
+ [ -f "$log" ] && jq -s '
179
+ map(select((.source // "build-checkpoint") | . == "build-checkpoint" or . == "build-backstop"))
180
+ | map(select(.outcome != "skipped"))
181
+ | group_by(.checkpoint + "|" + (.agents_run | sort | join(",")))
182
+ | map({
183
+ checkpoint: .[0].checkpoint,
184
+ agents: (.[0].agents_run | sort | join(", ")),
185
+ total: length,
186
+ no_op: (map(select(.outcome=="no-op")) | length),
187
+ fixed: (map(select(.outcome=="fixed")) | length),
188
+ escalated: (map(select(.outcome=="escalated")) | length),
189
+ fix_rate: ((map(select(.outcome=="fixed")) | length) / length * 100 | round),
190
+ issues_found: (map(.issues_found) | add // 0),
191
+ issues_fixed: (map(.issues_fixed) | add // 0),
192
+ fix_iterations:(map(.fix_iterations)| add // 0)
193
+ })' \
194
+ "$log"
195
+ ```
196
+
197
+ #### Per-lens outcomes split by diff shape — the measurement a test-only gate waits on (#1964)
198
+
199
+ `/code-review`'s existing cost gates narrow by file *type* (change-shape),
200
+ diff *size* (change-size), and architectural *signal* (change-impact). None of
201
+ them exploits a fourth, structurally-guaranteed shape: under `/test-improve`'s
202
+ default `refactor-mode: no-refactor`, Phase 5's diff **cannot** contain
203
+ production code, because `/build` rejects it — yet the four opus-tier
204
+ `Scope: always` lenses run on it anyway, per Story and again at phase end.
205
+
206
+ Whether any of them can be dropped there is an empirical question, so answer
207
+ it before touching a gate. Group `diff_shape: "test-only"` rows by lens and
208
+ report the outcome split against the same lens's `mixed` rows:
209
+
210
+ ```bash
211
+ log=".claude/metrics/review-value.jsonl"; [ -f "$log" ] || log="metrics/review-value.jsonl"
212
+ [ -f "$log" ] && jq -s '
213
+ map(select((.source // "build-checkpoint") | . == "build-checkpoint" or . == "build-backstop"))
214
+ | map(select(.outcome != "skipped" and .diff_shape != null))
215
+ | map({shape: .diff_shape, outcome, lens: .agents_run[]})
216
+ | group_by(.lens)
217
+ | map({lens: .[0].lens,
218
+ test_only: (map(select(.shape=="test-only")) | length),
219
+ test_only_no_op: (map(select(.shape=="test-only" and .outcome=="no-op")) | length),
220
+ mixed: (map(select(.shape=="mixed")) | length),
221
+ mixed_no_op:(map(select(.shape=="mixed" and .outcome=="no-op")) | length)})
222
+ | sort_by(-.test_only)' \
223
+ "$log"
224
+ ```
225
+
226
+ Report `test_only_no_op / test_only` per lens, alongside that lens's `mixed`
227
+ rate as the control — a lens that no-ops at the same rate on *both* shapes is
228
+ simply a quiet lens, not one this diff shape defeats, and gating it on shape
229
+ would be reading noise as signal. Only a lens that no-ops on test-only diffs
230
+ **and** earns its keep on mixed ones is a candidate for
231
+ `change_shape.py`'s `TEST_ONLY_SKIP_LENSES`, and each entry lands in its own
232
+ PR citing these numbers. State the row count: with few rows, the honest
233
+ finding is "not enough data yet", not a recommendation.
234
+
235
+ Two lenses are **not** candidates regardless of what the split shows, and the
236
+ report should say so rather than proposing them: `security-review` (tests
237
+ routinely embed credentials and injection payloads) and `correctness-review`
238
+ (an inverted assertion is exactly its subject).
239
+
240
+ #### Backstop redundancy — the measurement that gates `--backstop-review=skip` (#1962)
241
+
242
+ `/build`'s Step-6 backstop reviews files an inline checkpoint (sub-steps 4/6)
243
+ already reviewed in the same run, and under an enclosing orchestrator (e.g.
244
+ `/test-improve` Phase 5, which runs its own end-of-phase panel over the
245
+ cumulative diff) it is the third review layer over the same test code. Whether
246
+ that layer earns its cost is an empirical question, and `source:
247
+ "build-backstop"` rows are the answer. Report the backstop's own outcome split
248
+ next to the checkpoint split, restricted to builds where a checkpoint actually
249
+ ran first — a backstop on a `trivial`-only slice reviewed something nothing else
250
+ did, and including it would understate redundancy:
251
+
252
+ ```bash
253
+ log=".claude/metrics/review-value.jsonl"; [ -f "$log" ] || log="metrics/review-value.jsonl"
254
+ [ -f "$log" ] && jq -s '
255
+ (map(select((.source // "build-checkpoint") == "build-checkpoint"))
256
+ | map(.plan + "|" + (.slice // "")) | unique) as $reviewed
257
+ | map(select(.source == "build-backstop" and .outcome != "skipped"))
258
+ | map(select(((.plan + "|" + (.slice // "")) | IN($reviewed[]))))
259
+ | {backstop_runs_after_a_checkpoint: length,
260
+ no_op: (map(select(.outcome=="no-op")) | length),
261
+ fixed: (map(select(.outcome=="fixed")) | length),
262
+ escalated: (map(select(.outcome=="escalated")) | length),
263
+ issues_found: (map(.issues_found) | add // 0)}' \
264
+ "$log"
265
+ ```
266
+
267
+ Read it honestly, and state the sample size in the report: a backstop that is
268
+ **~all `no-op` after a checkpoint already ran** is the evidence that lets a
269
+ caller pass `--backstop-review=skip`; any non-trivial `fixed` count is evidence
270
+ it is catching what the checkpoints miss, and the flag should stay unused. A
271
+ handful of rows is not a finding — say so rather than recommending a flip off
272
+ noise. This is the same evidence-first discipline the architectural-impact gate
273
+ applies to widening `GATED_LENSES`: measure the lens, then narrow it.
274
+
275
+ For **read-only `code-review` rows**, report **finding-rate** (how often the
276
+ panel surfaced any issue) instead of fix-rate, and state plainly in the report
277
+ that these rows are excluded from the fix-rate drop-candidate logic because they
278
+ apply no fixes by design — a 0% fix rate there is expected, not a signal:
279
+
280
+ ```bash
281
+ log=".claude/metrics/review-value.jsonl"; [ -f "$log" ] || log="metrics/review-value.jsonl"
282
+ [ -f "$log" ] && jq -s '
283
+ map(select(.source == "code-review"))
284
+ | map(. + {found: ((.issues_found // .findings_new) // 0)})
285
+ | if length == 0 then "no read-only rows" else
286
+ group_by(.agents_run | sort | join(","))
287
+ | map({
288
+ agents: (.[0].agents_run | sort | join(", ")),
289
+ total: length,
290
+ found_issues: (map(select(.found > 0)) | length),
291
+ finding_rate: ((map(select(.found > 0)) | length) / length * 100 | round)
292
+ })
293
+ end' \
294
+ "$log"
295
+ ```
296
+
297
+ `(.issues_found // .findings_new)` covers both read-only row shapes: the
298
+ original whole-run `code-review` row (`issues_found`) and the per-round row
299
+ `/code-review` writes from #1624 (`findings_new`). Both are `source:
300
+ "code-review"`; only the round rows carry `round`/`dispatch_purpose`.
301
+
302
+ ### 4a. Analyze re-review churn (#1624)
303
+
304
+ Rows carrying a `round` field are `/code-review`'s per-round instrumentation
305
+ (#1624) — one per dispatch round, with `fix_provenance_new` counting how many
306
+ of that round's new findings landed inside the previous round's fix delta.
307
+ **If no row carries `round`, report "no round data yet" and skip this
308
+ section** — do not infer churn from the dispatch ledger's frequency counts
309
+ alone, which is exactly the inference #1623 documents as unavailable.
310
+
311
+ **Churn ratio** — of the rounds that exist only because an earlier round's fix
312
+ was applied (`round >= 2`), what fraction found *nothing but* problems that
313
+ fix created? A high ratio means the loop is chasing its own tail:
314
+
315
+ ```bash
316
+ log=".claude/metrics/review-value.jsonl"; [ -f "$log" ] || log="metrics/review-value.jsonl"
317
+ [ -f "$log" ] && jq -s '
318
+ map(select(.round != null and .round >= 2))
319
+ | if length == 0 then "no round data yet" else
320
+ {
321
+ rounds_after_first: length,
322
+ pure_churn_rounds: (map(select(.findings_new > 0 and .fix_provenance_new == .findings_new)) | length),
323
+ churn_ratio: ((map(select(.findings_new > 0 and .fix_provenance_new == .findings_new)) | length) / length * 100 | round),
324
+ max_round_reached: (map(.round) | max)
325
+ }
326
+ end' \
327
+ "$log"
328
+ ```
329
+
330
+ **Per-agent discovery-vs-verification split** — how much of each agent's
331
+ dispatch cost is spent confirming fixes rather than finding new problems.
332
+ Cross-reference against the `agent_dispatch_ledger` frequency table from
333
+ Step 3: an agent near the top of that table whose rounds are mostly
334
+ `verification` is a tier-down candidate for #1628's opt-in `verify_model:`,
335
+ not evidence of high discovery value:
336
+
337
+ ```bash
338
+ log=".claude/metrics/review-value.jsonl"; [ -f "$log" ] || log="metrics/review-value.jsonl"
339
+ [ -f "$log" ] && jq -s '
340
+ map(select(.dispatch_purpose != null))
341
+ | map({purpose: .dispatch_purpose, agent: .agents_run[]})
342
+ | group_by(.agent)
343
+ | map({
344
+ agent: .[0].agent,
345
+ discovery: (map(select(.purpose=="discovery")) | length),
346
+ verification: (map(select(.purpose=="verification")) | length),
347
+ closing: (map(select(.purpose=="closing")) | length)
348
+ })' \
349
+ "$log"
350
+ ```
351
+
352
+ **Gate recidivism** — how often the review-corroboration gate blocked
353
+ because a fix invalidated the prior round's corroboration. This needs no new
354
+ stream; `boundary-events.jsonl` already records it. A session with repeated
355
+ blocks is the operator-visible symptom of the same churn.
356
+
357
+ #1886 moved this gate from `git commit` (`hook == "pre_commit_review"`) to
358
+ `gh pr create` (`hook == "pre_pr_review"`) — query BOTH hook names so a
359
+ session that spans the migration (or a fork still running the older cached
360
+ plugin version) is not silently undercounted:
361
+
362
+ ```bash
363
+ log=".claude/metrics/boundary-events.jsonl"
364
+ [ -f "$log" ] && jq -rs '
365
+ map(select((.hook == "pre_commit_review" or .hook == "pre_pr_review") and (.matched_rule | startswith("dispatch-evidence-"))))
366
+ | group_by(.session_id)
367
+ | map({session: (.[0].session_id // "unknown"), blocks: length, rules: (map(.matched_rule) | unique)})
368
+ | sort_by(-.blocks)' \
369
+ "$log"
370
+ ```
371
+
372
+ Report all three together. State the sample size next to each number —
373
+ small-N honesty, consistent with Step 5. These metrics exist to tell whether
374
+ #1623's churn-reduction slices worked; a ratio computed from three rounds is
375
+ not evidence either way, and should be reported as such.
376
+
377
+ Flag **drop candidates**: any checkpoint+agents combination (from the
378
+ fix-applying rows only) with `fix_rate == 0` across **N ≥ 5** logged runs is a
379
+ drop candidate — it consistently adds overhead without catching defects.
380
+
381
+ Flag **high-value checkpoints**: `fix_rate ≥ 50%` — these are earning their cost and should be retained.
382
+
383
+ **Drop-candidate recommendations** (P2-S3):
384
+ For each drop candidate emit a recommendation in this form:
385
+ > `<checkpoint>/<agents>` fixed 0/<N> runs (fix rate 0%) — candidate to drop. To act: remove this checkpoint type from the relevant `/build` step-complexity tier or exclude these agents from the checkpoint's dispatch list. Do not auto-edit skills; present for human decision.
386
+
387
+ **Cite ablation evidence when available (#868).** `review-value.jsonl` alone is
388
+ observational — a zero fix-rate agent might have been shielded by another
389
+ agent, dispatched against the wrong changesets, or never given a defect to
390
+ catch. Before finalizing each per-agent drop-candidate recommendation, check
391
+ for causal evidence:
392
+
393
+ ```bash
394
+ for agent in <each single-agent drop candidate>; do
395
+ python3 "$CLAUDE_PLUGIN_ROOT/scripts/eval_ablation.py" --find-latest "$agent" \
396
+ --jsonl metrics/eval-ablation.jsonl
397
+ done
398
+ ```
399
+
400
+ - **Record found** — cite it in the recommendation instead of (or alongside)
401
+ the fix-rate line: `<agent> — ablation run <recorded_at> (model
402
+ <model>): delta {issues_caught: <n>, test_commands_passed: <n>, tokens:
403
+ <n>}, verdict "<verdict>". <If verdict is "baseline failed —
404
+ inconclusive": state the evidence is unusable and the fix-rate signal
405
+ above is the only basis for this recommendation.>`
406
+ - **No record found** — state the evidence is correlational-only and name
407
+ the exact command that would upgrade it: `No ablation evidence for
408
+ <agent> — this recommendation is based on correlational usage data only.
409
+ Run \`/agent-eval --ablation <agent>\` to get a controlled baseline-vs-
410
+ ablated delta before acting.`
411
+
412
+ This applies only to drop candidates that resolve to a **single** review
413
+ agent (multi-agent checkpoint combinations have no single-agent ablation
414
+ record to cite — note that explicitly rather than guessing which member
415
+ agent a record might apply to).
416
+
417
+ Do not modify any skill or agent file. The report is the only artifact.
418
+
419
+ ### 4b. Harness-sourced rows — a separate, labelled population (#2051)
420
+
421
+ `evals/code-review-benchmark/` (#821) is a **replay harness**, not a live
422
+ session — it dispatches `/code-review --json` against real Defects4J/BugsJS
423
+ defects and recorded diffs, entirely outside any `/build`/`/code-review`
424
+ session an operator ran. Its runner (`runner.emit_review_value_rows()`)
425
+ writes rows in the same `review-value.jsonl` shape as the live writers
426
+ above, but with `source: "harness"` — a value distinct from
427
+ `build-checkpoint` / `build-backstop` / `code-review`, and written to the
428
+ harness's **own** results directory (`evals/code-review-benchmark/results/
429
+ review-value.jsonl` by default, `--results-dir` elsewhere), never
430
+ `.claude/metrics/`. Both are structural, not just a labelling convention:
431
+ even a caller that pointed this step at the wrong file could not
432
+ accidentally pool harness rows into the live population, because they
433
+ don't live in the same file.
434
+
435
+ **Read the harness stream separately, and never merge it into Step 3/4's
436
+ live-population computations above.** It answers a different question —
437
+ per-lens yield replayed against a fixed, real-defect corpus — not "how did
438
+ this operator's own sessions go":
439
+
440
+ ```bash
441
+ harness_log="evals/code-review-benchmark/results/review-value.jsonl"
442
+ [ -f "$harness_log" ] && jq -s '
443
+ map(select(.source == "harness"))
444
+ | group_by(.agents_run | sort | join(","))
445
+ | map({
446
+ lens: (.[0].agents_run | sort | join(", ")),
447
+ dispatches: length,
448
+ no_op: (map(select(.issues_found == 0)) | length),
449
+ found_issues: (map(select(.issues_found > 0)) | length),
450
+ issues_found: (map(.issues_found) | add // 0)
451
+ })' \
452
+ "$harness_log"
453
+ ```
454
+
455
+ Report this as its own labelled table (`### Harness-Sourced Rows
456
+ (source: "harness")`), separate from Step 4's live tables, and state the
457
+ corpus it came from (Defects4J / BugsJS / recorded-diff, per the row's
458
+ `dataset` field) and the row count. Absent = no harness sweep has run yet;
459
+ report that plainly rather than treating a missing file as zero evidence
460
+ either way.
461
+
462
+ ### 4c. Agent-vs-tool redundancy criterion (#1983 Part 2)
463
+
464
+ A seam distinct from Step 4's fix-rate/finding-rate analysis: those measure
465
+ agents against *usage*, but nothing before this caught "this lens's
466
+ findings are a subset of what an already-running deterministic tool
467
+ reported for the same rounds" — the gap the #1974 pre-pass spike
468
+ identified. `skills/harness-audit/scripts/redundancy_criterion.py`
469
+ implements this as a real, computable comparison (not prose): for one
470
+ lens's findings in a round, `is_round_redundant()` checks whether every
471
+ `{file, line}` finding it applied is within a line-tolerance of some
472
+ finding the deterministic static-analysis pre-pass
473
+ (`skills/static-analysis-integration/SKILL.md`, including the
474
+ `lizard`/`jscpd` metric adapters — #1974) already reported for the same
475
+ file that round. `classify_lens_redundancy()` aggregates that verdict
476
+ across a lens's rounds into `redundant-candidate` / `not-redundant` /
477
+ `insufficient-data` (same `N >= 5` small-N-honesty floor as Step 4's own
478
+ drop-candidate rule), excluding no-op rounds (a lens that found nothing
479
+ proves nothing about redundancy either way).
480
+
481
+ **Data sources, named precisely — read this before running it.**
482
+ `review-value.jsonl` rows alone cannot drive this criterion: the schema is
483
+ explicit that rows carry "counts and outcomes only, never code or file
484
+ content" (`knowledge/telemetry-schema.md`), so there is no `file`/`line` on
485
+ a row to compare. This criterion instead needs, for the SAME round:
486
+
487
+ 1. **A lens's applied findings** (`{file, line}` pairs) — `/code-review
488
+ --json`'s aggregated payload, `agents[].issues[]`
489
+ (`skills/code-review/output-format.md`). Available today from a saved
490
+ raw `--json` dispatch artifact — e.g. the code-review-benchmark
491
+ harness's `results/raw/*.txt` (§4b above; the harness's own dispatches
492
+ are real `/code-review --json` rounds), or an operator's own saved
493
+ `--json` capture.
494
+ 2. **The deterministic pre-pass envelope for that same round** — the
495
+ unified finding envelope `static-analysis-integration`'s Step 6 returns
496
+ (`findings[]`, each `{file, line, rule_id, metadata: {source}}` —
497
+ `knowledge/security-primitives-contract.md`).
498
+
499
+ Neither is persisted as a committed metrics stream today, so running this
500
+ step means collecting a handful of real rounds' raw `/code-review --json`
501
+ output plus their pre-pass envelope by hand (or from the harness's `raw/`
502
+ directory), building the `[{"applied_findings": [...], "pretool_findings":
503
+ [...]}, ...]` shape `classify_lens_redundancy()` expects, and running:
504
+
505
+ ```bash
506
+ python3 "${CLAUDE_PLUGIN_ROOT}/skills/harness-audit/scripts/redundancy_criterion.py" \
507
+ --rounds <path-to-rounds.json>
508
+ ```
509
+
510
+ Report a **Agent-vs-Tool Redundancy** section: one row per lens with its
511
+ verdict, `rounds_with_findings`, and `rounds_fully_subsumed`. A
512
+ `redundant-candidate` verdict is evidence for a human tier-down/removal
513
+ decision, never an automatic one — same "report only, human acts" rule as
514
+ every other section in this skill.
515
+
516
+ ### 5. Lesson Validation — validated-outcome weighting (#866)
517
+
518
+ Close the loop on `/feedback-learning` lessons: does an adopted lesson
519
+ measurably help, or should it become a rollback candidate? This step is
520
+ **report-only**, consistent with the orchestrator constraints above — it
521
+ never edits an agent, skill, or CLAUDE.md file, and a `harmful` verdict is
522
+ always a *proposal*, never an automatic rollback.
523
+
524
+ Reads `metrics/config-changelog.jsonl` (written by `/feedback-learning`,
525
+ schema in [feedback-learning](../feedback-learning/SKILL.md) → Audit Trail)
526
+ and `metrics/session-digest.jsonl` (this command's existing Step 1 input).
527
+ Both are metrics-only — no prompt or code content, consistent with the
528
+ session-review privacy boundary.
529
+
530
+ Run the deterministic helper (pure stdlib, zero model tokens for the
531
+ computation):
532
+
533
+ ```bash
534
+ changelog=".claude/metrics/config-changelog.jsonl"; [ -f "$changelog" ] || changelog="metrics/config-changelog.jsonl"
535
+ python3 "${CLAUDE_PLUGIN_ROOT}/skills/harness-audit/scripts/lesson_validate.py" \
536
+ --changelog "$changelog" \
537
+ --digest metrics/session-digest.jsonl \
538
+ --apply -o memory/lesson-validation.json
539
+ ```
540
+
541
+ - **`--apply`** appends new `type: "validation"` entries to
542
+ `metrics/config-changelog.jsonl` for every newly-judged lesson — this is an
543
+ **append-only** write (new lines only); it never rewrites or deletes an
544
+ existing line. Verify this yourself if in doubt: a byte-for-byte diff of the
545
+ file before and after the run must show only appended lines.
546
+ - Every **adopted lesson with structured evidence** (`amend`/`learn`/`remember`
547
+ entries whose `evidence` field is an object, not the literal string
548
+ `"unmeasurable"` and not absent) whose observation window has elapsed gets a
549
+ verdict:
550
+ - **validated** — the watched metric moved in the expected `direction`.
551
+ - **neutral** — the window elapsed, adequate data exists, no meaningful
552
+ movement either way.
553
+ - **harmful** — the watched metric moved against the expected `direction`.
554
+ - **insufficient data** — fewer than `window_sessions` digest records exist
555
+ on either side of adoption. This is a data condition, **never** reported
556
+ as `neutral` — small-N honesty over a false-precision judgment.
557
+ - Comparison is **direction-only** on window means (v1 — no significance
558
+ testing; the digest carries small-N aggregate counts where formal testing
559
+ would be false precision).
560
+ - Entries marked `"unmeasurable"` and **legacy** entries (written before the
561
+ `evidence` field existed, so the key is absent) are **surfaced as counts
562
+ only** — they never receive a verdict and are never proposed for rollback
563
+ on evidence grounds.
564
+ - Each **harmful** verdict emits a **rollback proposal** carrying the
565
+ original entry's `timestamp`, `file_modified`, `section_modified`, and
566
+ `previous_value` — enough for `/feedback-learning`'s existing
567
+ [Rollback](../feedback-learning/SKILL.md#rollback) flow to act on it after
568
+ a human approves. Never auto-apply.
569
+
570
+ Include a **Lesson Validation** section in the report (Step 8) summarizing
571
+ verdict counts, the unmeasurable/legacy counts, and the full list of rollback
572
+ proposals.
573
+
574
+ ### 6. Analyze model routing
575
+
576
+ For each agent listed in `knowledge/agent-registry.md` (with its `model:`/`effort:` frontmatter, resolved natively by the harness per `agents/orchestrator.md` → Model/Effort Resolution — ADR 0026):
577
+
578
+ 1. **Over-tiered agents**: Agents assigned to opus that consistently produce simple pattern-match findings may work equally well on sonnet or haiku.
579
+ 2. **Under-tiered agents**: Agents on haiku that frequently miss issues caught by human review may need a higher tier.
580
+ 3. **Cost distribution**: Which agents consume the most tokens? Are the most expensive agents also the most valuable?
581
+
582
+ ### 7. Analyze orchestration complexity
583
+
584
+ Review the current pipeline for components that may be unnecessary overhead:
585
+
586
+ 1. **Phase count**: Are all three phases (Research, Plan, Implement) needed for the types of tasks being run? If most tasks are simple, suggest a fast path.
587
+ 2. **Review checkpoint frequency**: Are inline reviews running on every step? If most steps are trivial, the complexity classification (see `skills/plan/SKILL.md` § Complexity Classification) should be catching this.
588
+ 3. **Unused skills**: Skills loaded but never applied in logged sessions.
589
+ 4. **Context pollution per phase (#1520)**: Read the per-phase resident-vs-spend ratios from the phase markers (`phase-report`) and flag phases whose context lingered rather than being one-time cost — candidates for earlier mid-phase compaction or narrower subagent scoping. Skip if the log is absent.
590
+
591
+ ```bash
592
+ log=".claude/metrics/phase-markers.jsonl"; [ -f "$log" ] || log="metrics/phase-markers.jsonl"
593
+ [ -f "$log" ] && python3 "${CLAUDE_PLUGIN_ROOT}/hooks/lib/cost_meter.py" phase-report --log "$log" --json
594
+ ```
595
+
596
+ A phase with a high `resident_to_spent_ratio` spent proportionally little fresh generation while carrying a large resident context — cite it in § Orchestration Simplification Opportunities as a compaction/scoping candidate. This is a session-scoped proxy (resident is sampled at the `/handoff` boundary — see `skills/cost-report/SKILL.md` § Context pollution), not exact per-phase accounting.
597
+
598
+ ### 8. Produce report
599
+
600
+ When `--pdf` was passed, after writing the report render **the actual output
601
+ path** (the `--output` override when given, else `.dev-team-reports/harness-audit-<date>.md`)
602
+ to a sibling PDF per `knowledge/report-pdf-integration.md` (additive; non-fatal
603
+ if no engine):
604
+
605
+ ```bash
606
+ sh "$CLAUDE_PLUGIN_ROOT/hooks/py.sh" "$CLAUDE_PLUGIN_ROOT/hooks/lib/report_pdf.py" <the-output-path>
607
+ ```
608
+
609
+ Write the report to the output path using this structure:
610
+
611
+ ```markdown
612
+ # Harness Audit Report
613
+
614
+ **Date**: <date>
615
+ **Metrics period**: <earliest to latest logged review>
616
+ **Review runs analyzed**: <count>
617
+
618
+ ## Re-baseline Required
619
+
620
+ > Only present when Step 2 detects a model mismatch between
621
+ > `evals/baseline.json`'s `model` field and the current session's model.
622
+ > Omit this section entirely on a match or an absent/pre-migration field.
623
+
624
+ - **Baseline model**: <model recorded in evals/baseline.json>
625
+ - **Current session model**: <current model>
626
+ - **Action**: Re-run the eval suite and re-write the baseline
627
+ (`/agent-eval` full suite, then `eval_variance.py --write-baseline
628
+ --model <current-model>`) before trusting any pre/post comparison in this
629
+ report or in a feedback-learning change contract.
630
+ - **Possibly stale scaffolding**: <any removal candidate below whose
631
+ "zero fail" or "high false positive" verdict was measured on the old
632
+ model — flag for re-evaluation, not automatic removal>
633
+
634
+ ## Review Agent Effectiveness
635
+
636
+ ### Removal Candidates (zero fail findings)
637
+ | Agent | Reviews | Pass rate | Recommendation |
638
+ |-------|---------|-----------|----------------|
639
+
640
+ ### High False Positive Rate (>50% dismissed)
641
+ | Agent | Findings | Dismissed | Rate | Recommendation |
642
+ |-------|----------|-----------|------|----------------|
643
+
644
+ ### Low-Value Agents (>80% minor severity)
645
+ | Agent | Findings | Minor % | Recommendation |
646
+ |-------|----------|---------|----------------|
647
+
648
+ ## Review-Value Fix Rates (inline checkpoint ROI)
649
+
650
+ > Source: `metrics/review-value.jsonl`. Absent = no `/build` runs logged yet.
651
+
652
+ ### Per-Checkpoint-Type Fix Rates
653
+ | Checkpoint | Agents | Runs | No-op | Fixed | Escalated | Fix rate |
654
+ |------------|--------|------|-------|-------|-----------|----------|
655
+
656
+ ### Drop Candidates (fix rate 0%, N ≥ 5 runs)
657
+ | Checkpoint | Agents | Runs | Ablation evidence | Recommendation |
658
+ |------------|--------|------|--------------------|-----------------|
659
+
660
+ > To act on a drop candidate: remove the checkpoint type from the relevant `/build`
661
+ > step-complexity tier or exclude the agents from that checkpoint's dispatch list.
662
+ > Requires human decision — do not auto-edit skills.
663
+ >
664
+ > "Ablation evidence" column: the cited `metrics/eval-ablation.jsonl` verdict +
665
+ > date for single-agent candidates, or "correlational only — run
666
+ > `/agent-eval --ablation <agent>`" when no record exists.
667
+
668
+ ### High-Value Checkpoints (fix rate ≥ 50%)
669
+ | Checkpoint | Agents | Runs | Fix rate | Issues fixed |
670
+ |------------|--------|------|----------|--------------|
671
+
672
+ ## Harness-Sourced Rows (source: "harness", #2051)
673
+
674
+ > Source: `evals/code-review-benchmark/results/review-value.jsonl` — a
675
+ > **separate, labelled population**, never pooled with the live rows above.
676
+ > Absent = no harness sweep has run yet.
677
+
678
+ | Lens | Dataset(s) | Dispatches | No-op | Found issues |
679
+ |------|------------|------------|-------|---------------|
680
+
681
+ ## Agent-vs-Tool Redundancy (#1983 Part 2)
682
+
683
+ > Source: per-round applied findings (`/code-review --json` `agents[].issues[]`)
684
+ > paired with the static-analysis pre-pass envelope for the same round — see
685
+ > §4c for why `review-value.jsonl` alone cannot drive this. Absent = no
686
+ > rounds' data collected yet for this criterion.
687
+
688
+ | Lens | Verdict | Rounds w/ findings | Rounds fully subsumed |
689
+ |------|---------|---------------------|-------------------------|
690
+
691
+ > A `redundant-candidate` verdict is evidence for a human tier-down/removal
692
+ > decision — never acted on automatically.
693
+
694
+ ## Lesson Validation (validated-outcome weighting, #866)
695
+
696
+ > Source: `metrics/config-changelog.jsonl` × `metrics/session-digest.jsonl`.
697
+ > Report-only — verdicts are appended as new `type: "validation"` entries;
698
+ > harmful verdicts are rollback *proposals*, never automatic.
699
+
700
+ ### Verdicts
701
+ | Lesson (`timestamp`) | Metric | Direction | Verdict |
702
+ |---|---|---|---|
703
+
704
+ ### Rollback Proposals (harmful verdicts — human approval required)
705
+ | Lesson (`timestamp`) | File | Section | Recommendation |
706
+ |---|---|---|---|
707
+
708
+ > To act on a rollback proposal: run `/feedback-learning` and confirm the
709
+ > rollback against the `timestamp` above. Never applied automatically.
710
+
711
+ ### Unmeasurable / Legacy (surfaced, not judged)
712
+ - Unmeasurable lessons: <count>
713
+ - Legacy lessons (no `evidence` field): <count>
714
+
715
+ ## Model Routing Recommendations
716
+
717
+ | Agent | Current tier | Suggested tier | Rationale |
718
+ |-------|-------------|----------------|-----------|
719
+
720
+ ## Orchestration Simplification Opportunities
721
+
722
+ - <Finding and recommendation>
723
+ - <Context-pollution candidates (#1520): any phase with a high resident/spent ratio from `phase-report` — recommend earlier mid-phase compaction or narrower subagent scoping. Omit if no phase markers logged.>
724
+
725
+ ## Summary
726
+
727
+ - Agents to consider removing: <count>
728
+ - Model tier changes suggested: <count>
729
+ - Orchestration simplifications: <count>
730
+ - Review-value drop candidates: <count>
731
+ - Review-value high-value checkpoints: <count>
732
+ - Harness-sourced rows analyzed (source: "harness"): <count>
733
+ - Agent-vs-tool redundancy candidates: <count>
734
+ - Re-baseline required: <yes/no>
735
+ - Lessons validated / neutral / harmful / insufficient data: <count> / <count> / <count> / <count>
736
+ - Rollback proposals (harmful verdicts): <count>
737
+
738
+ ## Next Steps
739
+
740
+ <Actionable recommendations prioritized by impact>
741
+ ```
742
+
743
+ ### 9. Present results
744
+
745
+ Display a summary of the report and the file path. Do not repeat the full report in chat — the file is the artifact.
746
+
747
+ ## Error Handling
748
+
749
+ - Missing metrics files: Report what's missing, suggest how to generate data
750
+ - Incomplete agent registry: Flag agents found in metrics but missing from the registry
751
+ - No actionable findings: Report that the harness appears well-calibrated — this is a valid outcome