codex-orchestrator 2.0.10 → 2.0.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. package/CHANGELOG.md +52 -0
  2. package/README.md +29 -30
  3. package/dist/src/index.d.ts +4 -9
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/index.js +2 -5
  6. package/dist/src/index.js.map +1 -1
  7. package/dist/src/v2/acceptance-proof.d.ts +53 -25
  8. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  9. package/dist/src/v2/acceptance-proof.js +189 -197
  10. package/dist/src/v2/acceptance-proof.js.map +1 -1
  11. package/dist/src/v2/active-attempt.d.ts +94 -0
  12. package/dist/src/v2/active-attempt.d.ts.map +1 -0
  13. package/dist/src/v2/active-attempt.js +200 -0
  14. package/dist/src/v2/active-attempt.js.map +1 -0
  15. package/dist/src/v2/adapters/command.d.ts +7 -0
  16. package/dist/src/v2/adapters/command.d.ts.map +1 -1
  17. package/dist/src/v2/adapters/command.js +44 -3
  18. package/dist/src/v2/adapters/command.js.map +1 -1
  19. package/dist/src/v2/adapters/gh-pull-request-adapter.js +1 -0
  20. package/dist/src/v2/adapters/gh-pull-request-adapter.js.map +1 -1
  21. package/dist/src/v2/adapters/pull-requests.d.ts +1 -0
  22. package/dist/src/v2/adapters/pull-requests.d.ts.map +1 -1
  23. package/dist/src/v2/adapters/pull-requests.js +1 -0
  24. package/dist/src/v2/adapters/pull-requests.js.map +1 -1
  25. package/dist/src/v2/adapters/worktree.js +3 -3
  26. package/dist/src/v2/adapters/worktree.js.map +1 -1
  27. package/dist/src/v2/atomic-store.d.ts +15 -0
  28. package/dist/src/v2/atomic-store.d.ts.map +1 -1
  29. package/dist/src/v2/atomic-store.js +55 -8
  30. package/dist/src/v2/atomic-store.js.map +1 -1
  31. package/dist/src/v2/candidate.d.ts +136 -0
  32. package/dist/src/v2/candidate.d.ts.map +1 -0
  33. package/dist/src/v2/candidate.js +107 -0
  34. package/dist/src/v2/candidate.js.map +1 -0
  35. package/dist/src/v2/checked-change.d.ts +39 -7
  36. package/dist/src/v2/checked-change.d.ts.map +1 -1
  37. package/dist/src/v2/checked-change.js +73 -1
  38. package/dist/src/v2/checked-change.js.map +1 -1
  39. package/dist/src/v2/cli-contract.d.ts +1 -1
  40. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  41. package/dist/src/v2/cli-contract.js +4 -6
  42. package/dist/src/v2/cli-contract.js.map +1 -1
  43. package/dist/src/v2/cli.d.ts +14 -0
  44. package/dist/src/v2/cli.d.ts.map +1 -1
  45. package/dist/src/v2/cli.js +70 -21
  46. package/dist/src/v2/cli.js.map +1 -1
  47. package/dist/src/v2/code-review-report.d.ts +10 -18
  48. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  49. package/dist/src/v2/code-review-report.js +63 -60
  50. package/dist/src/v2/code-review-report.js.map +1 -1
  51. package/dist/src/v2/codex-process.d.ts +6 -1
  52. package/dist/src/v2/codex-process.d.ts.map +1 -1
  53. package/dist/src/v2/codex-process.js +36 -9
  54. package/dist/src/v2/codex-process.js.map +1 -1
  55. package/dist/src/v2/config.d.ts +0 -2
  56. package/dist/src/v2/config.d.ts.map +1 -1
  57. package/dist/src/v2/config.js +3 -6
  58. package/dist/src/v2/config.js.map +1 -1
  59. package/dist/src/v2/contained-report-operation.d.ts +17 -6
  60. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  61. package/dist/src/v2/contained-report-operation.js +11 -24
  62. package/dist/src/v2/contained-report-operation.js.map +1 -1
  63. package/dist/src/v2/containment.d.ts +1 -0
  64. package/dist/src/v2/containment.d.ts.map +1 -1
  65. package/dist/src/v2/containment.js +12 -2
  66. package/dist/src/v2/containment.js.map +1 -1
  67. package/dist/src/v2/delivery-authority.d.ts +26 -0
  68. package/dist/src/v2/delivery-authority.d.ts.map +1 -0
  69. package/dist/src/v2/delivery-authority.js +44 -0
  70. package/dist/src/v2/delivery-authority.js.map +1 -0
  71. package/dist/src/v2/direct-delivery.d.ts +19 -57
  72. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  73. package/dist/src/v2/direct-delivery.js +154 -211
  74. package/dist/src/v2/direct-delivery.js.map +1 -1
  75. package/dist/src/v2/immutable-workflow-publisher.d.ts.map +1 -1
  76. package/dist/src/v2/immutable-workflow-publisher.js +3 -1
  77. package/dist/src/v2/immutable-workflow-publisher.js.map +1 -1
  78. package/dist/src/v2/implementation-report.d.ts +3 -1
  79. package/dist/src/v2/implementation-report.d.ts.map +1 -1
  80. package/dist/src/v2/implementation-report.js +17 -4
  81. package/dist/src/v2/implementation-report.js.map +1 -1
  82. package/dist/src/v2/implementation-reviewer.d.ts +22 -13
  83. package/dist/src/v2/implementation-reviewer.d.ts.map +1 -1
  84. package/dist/src/v2/implementation-reviewer.js +96 -30
  85. package/dist/src/v2/implementation-reviewer.js.map +1 -1
  86. package/dist/src/v2/owner-control-lock.d.ts +13 -1
  87. package/dist/src/v2/owner-control-lock.d.ts.map +1 -1
  88. package/dist/src/v2/owner-control-lock.js +66 -7
  89. package/dist/src/v2/owner-control-lock.js.map +1 -1
  90. package/dist/src/v2/pending-effect-settlement.d.ts +44 -0
  91. package/dist/src/v2/pending-effect-settlement.d.ts.map +1 -0
  92. package/dist/src/v2/pending-effect-settlement.js +69 -0
  93. package/dist/src/v2/pending-effect-settlement.js.map +1 -0
  94. package/dist/src/v2/process-identity.d.ts +45 -0
  95. package/dist/src/v2/process-identity.d.ts.map +1 -0
  96. package/dist/src/v2/process-identity.js +118 -0
  97. package/dist/src/v2/process-identity.js.map +1 -0
  98. package/dist/src/v2/proof-report.d.ts +4 -2
  99. package/dist/src/v2/proof-report.d.ts.map +1 -1
  100. package/dist/src/v2/proof-report.js +16 -44
  101. package/dist/src/v2/proof-report.js.map +1 -1
  102. package/dist/src/v2/review-feedback-coordinator.d.ts +8 -2
  103. package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -1
  104. package/dist/src/v2/review-feedback-coordinator.js +98 -71
  105. package/dist/src/v2/review-feedback-coordinator.js.map +1 -1
  106. package/dist/src/v2/review-feedback.d.ts +14 -38
  107. package/dist/src/v2/review-feedback.d.ts.map +1 -1
  108. package/dist/src/v2/review-feedback.js +48 -115
  109. package/dist/src/v2/review-feedback.js.map +1 -1
  110. package/dist/src/v2/run-issue.d.ts +139 -60
  111. package/dist/src/v2/run-issue.d.ts.map +1 -1
  112. package/dist/src/v2/run-issue.js +2236 -1974
  113. package/dist/src/v2/run-issue.js.map +1 -1
  114. package/dist/src/v2/run-state-projections.d.ts +84 -0
  115. package/dist/src/v2/run-state-projections.d.ts.map +1 -0
  116. package/dist/src/v2/run-state-projections.js +142 -0
  117. package/dist/src/v2/run-state-projections.js.map +1 -0
  118. package/dist/src/v2/run-store.d.ts +114 -82
  119. package/dist/src/v2/run-store.d.ts.map +1 -1
  120. package/dist/src/v2/run-store.js +308 -224
  121. package/dist/src/v2/run-store.js.map +1 -1
  122. package/dist/src/v2/runtime-assets.d.ts +3 -0
  123. package/dist/src/v2/runtime-assets.d.ts.map +1 -1
  124. package/dist/src/v2/runtime-assets.js +104 -0
  125. package/dist/src/v2/runtime-assets.js.map +1 -1
  126. package/dist/src/v2/runtime.d.ts +52 -15
  127. package/dist/src/v2/runtime.d.ts.map +1 -1
  128. package/dist/src/v2/runtime.js +818 -348
  129. package/dist/src/v2/runtime.js.map +1 -1
  130. package/dist/src/v2/setup.js +0 -2
  131. package/dist/src/v2/setup.js.map +1 -1
  132. package/dist/src/v2/validation-progression.d.ts +70 -0
  133. package/dist/src/v2/validation-progression.d.ts.map +1 -0
  134. package/dist/src/v2/validation-progression.js +247 -0
  135. package/dist/src/v2/validation-progression.js.map +1 -0
  136. package/dist/src/v2/workflow-assets.d.ts +9 -3
  137. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  138. package/dist/src/v2/workflow-assets.js +256 -43
  139. package/dist/src/v2/workflow-assets.js.map +1 -1
  140. package/docs/deep-dive.md +52 -14
  141. package/internal-workflow/docs/agents/bug-workflow-routing.md +9 -7
  142. package/internal-workflow/docs/agents/coding-skill-routing.md +170 -120
  143. package/internal-workflow/docs/agents/tool-usage.md +23 -12
  144. package/internal-workflow/manifest.json +1 -1
  145. package/internal-workflow/operations/code-review/SKILL.md +34 -15
  146. package/internal-workflow/operations/implementation/SKILL.md +21 -16
  147. package/internal-workflow/profiles/implementer.toml +9 -0
  148. package/internal-workflow/profiles/review_coordinator.toml +9 -0
  149. package/internal-workflow/profiles/spec_reviewer.toml +9 -0
  150. package/internal-workflow/profiles/standards_reviewer.toml +9 -0
  151. package/internal-workflow/schemas/code-review-v1.json +1 -1
  152. package/internal-workflow/schemas/implementation-report-v1.json +1 -1
  153. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  154. package/internal-workflow/skills/bug-root-cause-explainer/SKILL.md +114 -0
  155. package/internal-workflow/skills/bug-root-cause-explainer/agents/openai.yaml +7 -0
  156. package/internal-workflow/skills/bug-root-cause-explainer/evals/evals.json +18 -0
  157. package/internal-workflow/skills/code-review/SKILL.md +84 -288
  158. package/internal-workflow/skills/code-review/agents/openai.yaml +5 -3
  159. package/internal-workflow/skills/code-review/evals/evals.json +83 -0
  160. package/internal-workflow/skills/code-review/references/standards-smells.md +41 -0
  161. package/internal-workflow/skills/diagnosing-bugs/SKILL.md +69 -32
  162. package/internal-workflow/skills/diagnosing-bugs/agents/openai.yaml +2 -2
  163. package/internal-workflow/skills/diagnosing-bugs/evals/evals.json +63 -0
  164. package/internal-workflow/skills/grilling/SKILL.md +51 -0
  165. package/internal-workflow/skills/grilling/agents/openai.yaml +6 -0
  166. package/internal-workflow/skills/grilling/evals/evals.json +47 -0
  167. package/internal-workflow/skills/implement/SKILL.md +135 -0
  168. package/internal-workflow/skills/implement/agents/openai.yaml +6 -0
  169. package/internal-workflow/skills/implement/evals/evals.json +150 -0
  170. package/internal-workflow/skills/plan/SKILL.md +59 -0
  171. package/internal-workflow/skills/plan/agents/openai.yaml +6 -0
  172. package/internal-workflow/skills/plan/evals/evals.json +36 -0
  173. package/internal-workflow/skills/prototype/LOGIC.md +130 -0
  174. package/internal-workflow/skills/prototype/SKILL.md +69 -0
  175. package/internal-workflow/skills/prototype/UI.md +157 -0
  176. package/internal-workflow/skills/prototype/agents/openai.yaml +6 -0
  177. package/internal-workflow/skills/prototype/evals/evals.json +67 -0
  178. package/internal-workflow/skills/research/SKILL.md +110 -0
  179. package/internal-workflow/skills/research/agents/openai.yaml +6 -0
  180. package/internal-workflow/skills/research/evals/evals.json +49 -0
  181. package/internal-workflow/skills/tdd/SKILL.md +72 -67
  182. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  183. package/internal-workflow/skills/tdd/evals/evals.json +12 -0
  184. package/internal-workflow/skills/tdd/mocking.md +48 -1
  185. package/internal-workflow/skills/tdd/refactoring.md +3 -3
  186. package/internal-workflow/skills/tickets-orchestrator/SKILL.md +199 -0
  187. package/internal-workflow/skills/tickets-orchestrator/agents/openai.yaml +6 -0
  188. package/internal-workflow/skills/tickets-orchestrator/evals/evals.json +126 -0
  189. package/internal-workflow/skills/tickets-orchestrator/references/delegate-integrate.md +83 -0
  190. package/internal-workflow/skills/tickets-orchestrator/references/finish-delivery.md +69 -0
  191. package/internal-workflow/skills/tickets-orchestrator/references/stop-completion.md +63 -0
  192. package/internal-workflow/skills/to-spec/SKILL.md +133 -0
  193. package/internal-workflow/skills/to-spec/agents/openai.yaml +6 -0
  194. package/internal-workflow/skills/to-spec/evals/evals.json +24 -0
  195. package/internal-workflow/skills/to-tickets/SKILL.md +189 -0
  196. package/internal-workflow/skills/to-tickets/agents/openai.yaml +6 -0
  197. package/internal-workflow/skills/to-tickets/evals/evals.json +79 -0
  198. package/internal-workflow/skills/to-tickets/references/publishing-details.md +117 -0
  199. package/package.json +1 -1
  200. package/dist/src/v2/proof-store.d.ts +0 -42
  201. package/dist/src/v2/proof-store.d.ts.map +0 -1
  202. package/dist/src/v2/proof-store.js +0 -180
  203. package/dist/src/v2/proof-store.js.map +0 -1
  204. package/dist/src/v2/route-continuations.d.ts +0 -32
  205. package/dist/src/v2/route-continuations.d.ts.map +0 -1
  206. package/dist/src/v2/route-continuations.js +0 -2
  207. package/dist/src/v2/route-continuations.js.map +0 -1
  208. package/dist/src/v2/route-coordinator.d.ts +0 -77
  209. package/dist/src/v2/route-coordinator.d.ts.map +0 -1
  210. package/dist/src/v2/route-coordinator.js +0 -370
  211. package/dist/src/v2/route-coordinator.js.map +0 -1
  212. package/dist/src/v2/route-decision.d.ts +0 -129
  213. package/dist/src/v2/route-decision.d.ts.map +0 -1
  214. package/dist/src/v2/route-decision.js +0 -400
  215. package/dist/src/v2/route-decision.js.map +0 -1
  216. package/dist/src/v2/spec-coordinator.d.ts +0 -85
  217. package/dist/src/v2/spec-coordinator.d.ts.map +0 -1
  218. package/dist/src/v2/spec-coordinator.js +0 -88
  219. package/dist/src/v2/spec-coordinator.js.map +0 -1
  220. package/dist/src/v2/spec-delivery.d.ts +0 -143
  221. package/dist/src/v2/spec-delivery.d.ts.map +0 -1
  222. package/dist/src/v2/spec-delivery.js +0 -401
  223. package/dist/src/v2/spec-delivery.js.map +0 -1
  224. package/dist/src/v2/triage-route.d.ts +0 -68
  225. package/dist/src/v2/triage-route.d.ts.map +0 -1
  226. package/dist/src/v2/triage-route.js +0 -223
  227. package/dist/src/v2/triage-route.js.map +0 -1
  228. package/dist/src/v2/waiting-human-coordinator.d.ts +0 -49
  229. package/dist/src/v2/waiting-human-coordinator.d.ts.map +0 -1
  230. package/dist/src/v2/waiting-human-coordinator.js +0 -509
  231. package/dist/src/v2/waiting-human-coordinator.js.map +0 -1
  232. package/dist/src/v2/waiting-human.d.ts +0 -143
  233. package/dist/src/v2/waiting-human.d.ts.map +0 -1
  234. package/dist/src/v2/waiting-human.js +0 -408
  235. package/dist/src/v2/waiting-human.js.map +0 -1
  236. package/internal-workflow/docs/agents/contract-test-ledger.md +0 -71
  237. package/internal-workflow/docs/agents/review-gates.md +0 -42
  238. package/internal-workflow/docs/agents/review-protocol.md +0 -98
  239. package/internal-workflow/evals/coding-skill-evals.json +0 -332
  240. package/internal-workflow/operations/ambiguity-review/SKILL.md +0 -5
  241. package/internal-workflow/operations/qualification-repair/SKILL.md +0 -17
  242. package/internal-workflow/operations/spec-author/SKILL.md +0 -12
  243. package/internal-workflow/operations/spec-review/SKILL.md +0 -12
  244. package/internal-workflow/operations/triage/SKILL.md +0 -12
  245. package/internal-workflow/profiles/analyst_deep.toml +0 -9
  246. package/internal-workflow/profiles/implementer_standard.toml +0 -9
  247. package/internal-workflow/profiles/proof_agent.toml +0 -8
  248. package/internal-workflow/profiles/reviewer_deep.toml +0 -9
  249. package/internal-workflow/profiles/reviewer_standard.toml +0 -9
  250. package/internal-workflow/schemas/ambiguity-review-v1.json +0 -1
  251. package/internal-workflow/schemas/spec-author-v1.json +0 -1
  252. package/internal-workflow/schemas/spec-review-v1.json +0 -30
  253. package/internal-workflow/schemas/triage-route-v1.json +0 -1
  254. package/internal-workflow/skills/agent-auto/SKILL.md +0 -19
  255. package/internal-workflow/skills/agent-auto/agents/openai.yaml +0 -6
  256. package/internal-workflow/skills/code-debugger/SKILL.md +0 -122
  257. package/internal-workflow/skills/code-debugger/agents/openai.yaml +0 -7
  258. package/internal-workflow/skills/code-review/references/bug-classes.md +0 -56
  259. package/internal-workflow/skills/code-review/references/cleanup-lens.md +0 -52
  260. package/internal-workflow/skills/code-review/references/framework-lenses.md +0 -34
  261. package/internal-workflow/skills/code-review/references/targeted-recipes.md +0 -49
  262. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +0 -107
  263. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +0 -6
  264. package/internal-workflow/skills/implementation-spec-maker/references/source-modes.md +0 -32
  265. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +0 -146
  266. package/internal-workflow/skills/implementation-spec-review/SKILL.md +0 -131
  267. package/internal-workflow/skills/implementation-spec-review/agents/openai.yaml +0 -6
  268. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +0 -78
  269. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +0 -121
  270. package/internal-workflow/skills/small-task-implementer/SKILL.md +0 -112
  271. package/internal-workflow/skills/small-task-implementer/agents/openai.yaml +0 -6
  272. package/internal-workflow/skills/spec-implementer/SKILL.md +0 -126
  273. package/internal-workflow/skills/spec-implementer/agents/openai.yaml +0 -6
  274. package/internal-workflow/skills/spec-implementer/evals/evals.json +0 -30
  275. package/internal-workflow/skills/spec-implementer/references/review-loop.md +0 -100
  276. package/internal-workflow/skills/triage/AGENT-BRIEF.md +0 -192
  277. package/internal-workflow/skills/triage/OUT-OF-SCOPE.md +0 -101
  278. package/internal-workflow/skills/triage/SKILL.md +0 -134
  279. package/internal-workflow/skills/triage/agents/openai.yaml +0 -6
@@ -0,0 +1,114 @@
1
+ ---
2
+ name: bug-root-cause-explainer
3
+ description: "Diagnose and explain a bug without editing code. Use for why/root-cause requests or when the user says not to fix yet; validate the issue, trace ownership, explain with evidence, and offer a proportional fix path."
4
+ ---
5
+
6
+ # Bug Root Cause Explainer
7
+
8
+ ## Overview
9
+
10
+ Use this skill when the main deliverable is understanding, not an immediate patch. Prove the issue, find the owning cause, explain it simply, propose the smallest safe path, and wait for the user's decision.
11
+
12
+ ## Core Rule
13
+
14
+ Do not jump into implementation. Finish the diagnosis, offer one minimal path,
15
+ and stop. Add a structural path only when evidence proves an ownership defect
16
+ that the minimal fix leaves in place.
17
+
18
+ For routing between bug diagnosis, feedback-loop construction, and implementation, use `../../docs/agents/bug-workflow-routing.md`.
19
+
20
+ Use `implement` instead when the user wants an end-to-end debug-and-fix flow in one pass.
21
+ After the user chooses a path, hand off implementation to `implement`.
22
+
23
+ If the bug is hard, flaky, performance-related, or lacks a reliable feedback loop, use `diagnosing-bugs` before choosing a fix path. Do not guess a root cause without evidence.
24
+
25
+ Use `../../docs/agents/confidence-rubric.md` to label root-cause certainty. If the evidence is low-confidence, present it as an uncertainty or verification gap instead of a proven cause.
26
+
27
+ ## Investigation Workflow
28
+
29
+ 1. Validate the bug before believing the report.
30
+ - Reproduce the issue when possible.
31
+ - If full reproduction is expensive, recover the strongest available signal: failing test, runtime log, broken response, visible UI behavior, or deterministic code-path contradiction.
32
+ - If the report is outdated, incorrect, or actually expected behavior, say so clearly and stop there.
33
+
34
+ 2. Trace the owning path.
35
+ - Follow the real execution path instead of reading files broadly.
36
+ - Identify the exact entrypoint, decision point, state mutation, async boundary, and downstream effect that create the symptom.
37
+ - Prefer the owner module over secondary symptoms. Do not anchor the diagnosis in the last place where the bad data merely becomes visible.
38
+
39
+ 3. Separate symptom from root cause.
40
+ - State what the user sees.
41
+ - State what the code is actually doing.
42
+ - State where those two paths diverge.
43
+ - If there are multiple contributing issues, name the primary cause first and list the others only as amplifiers.
44
+
45
+ 4. Translate the diagnosis into plain language.
46
+ - Explain the issue as if speaking to a product-minded teammate, not only to the author of the code.
47
+ - Keep the explanation concrete and simple.
48
+ - Avoid jargon when possible.
49
+ - When jargon is unavoidable, define it in one short sentence.
50
+
51
+ 5. Support the explanation with exact references.
52
+ - Always cite the concrete file and function or method where the problem starts.
53
+ - Add a second reference for the downstream effect when that makes the explanation clearer.
54
+ - Include line references when they materially help the reader verify the claim.
55
+
56
+ 6. Offer proportional solution paths.
57
+ - Before proposing fixes for a confirmed bug, apply `../../docs/agents/bugfix-quality-gate.md`: state the invariant, diagnosis boundary, adjacent paths not inspected, and what the evidence does not prove.
58
+ - Always offer the smallest safe fix.
59
+ - Add one structural path only for a proven ownership defect; tie its scope to that evidence.
60
+
61
+ 7. Stop and wait.
62
+ - Ask whether to implement the path, or which path when two are justified.
63
+ - Do not edit code, write files, or stage changes after the diagnosis unless the user explicitly chooses a path or explicitly asks for implementation.
64
+
65
+ ## Explanation Rules
66
+
67
+ - Prefer short paragraphs over dense technical dumps.
68
+ - Answer the hidden user questions directly:
69
+ - What is broken?
70
+ - Why is it happening?
71
+ - Where exactly in the code does it start?
72
+ - Why does it show up in this specific way?
73
+ - Use concrete phrasing like "the code throws away X here" or "this branch never runs because Y is always false here".
74
+ - Avoid vague language like "there may be a race" unless you can point to the exact competing operations.
75
+ - Avoid speculative fixes before the root cause is proven.
76
+
77
+ ## Output Contract
78
+
79
+ Use this shape when reporting back:
80
+
81
+ ```md
82
+ ## What is happening
83
+
84
+ <Very simple explanation of the real problem in plain language. Keep it short and concrete.>
85
+
86
+ ## Where it starts in code
87
+
88
+ - `<path>` — `<function or method>`: <what this code is doing wrong>
89
+ - `<path>` — `<function or method>`: <how the bad state reaches the visible symptom>
90
+
91
+ ## Why the symptom looks like this
92
+
93
+ <Short explanation connecting cause to symptom.>
94
+
95
+ ## Fix path
96
+
97
+ 1. Minimal path
98
+ - Scope: <smallest change>
99
+ - Tradeoff: <what this solves and what it does not improve>
100
+
101
+ 2. Structural path (only for a proven ownership defect)
102
+ - Scope: <cleaner ownership fix with minimal necessary breadth>
103
+ - Tradeoff: <why this is better long-term and what extra work it needs>
104
+
105
+ ## Decision
106
+
107
+ Do you want me to implement this path?
108
+ ```
109
+
110
+ ## Escalation Rules
111
+
112
+ - If the evidence is still ambiguous after careful tracing, present the ambiguity explicitly instead of pretending the root cause is proven.
113
+ - If multiple fixes are valid but change product behavior differently, say that the choice is product-sensitive and ask the user to choose.
114
+ - If the issue is security-critical or can cause data loss, say so clearly before presenting the two paths.
@@ -0,0 +1,7 @@
1
+ interface:
2
+ display_name: "Bug Root Cause Explainer"
3
+ short_description: "Investigate a bug before proposing a fix"
4
+ default_prompt: "Use $bug-root-cause-explainer to investigate the bug, explain the proven cause with concrete references, offer one minimal fix path, and add a structural path only for a proven ownership defect. Wait before changing code."
5
+
6
+ policy:
7
+ allow_implicit_invocation: true
@@ -0,0 +1,18 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "bug-root-cause-explainer",
4
+ "cases": [
5
+ {
6
+ "id": "minimal-path-default",
7
+ "prompt": "Explain a confirmed local bug whose owner and smallest safe fix are clear.",
8
+ "expected": ["offer one minimal fix path"],
9
+ "forbidden": ["invent a structural alternative"]
10
+ },
11
+ {
12
+ "id": "structural-path-needs-ownership-defect",
13
+ "prompt": "Explain a confirmed bug caused by duplicated ownership across two modules.",
14
+ "expected": ["offer the minimal path", "add a structural path tied to the proven ownership defect"],
15
+ "forbidden": ["propose broad architecture cleanup"]
16
+ }
17
+ ]
18
+ }
@@ -1,291 +1,87 @@
1
1
  ---
2
- name: "code-review"
3
- description: "Evidence-first review of code, PRs, commits, regressions, or review-and-fix work using correctness and standards/cleanup lenses. Auto-fix only qualifying high-confidence severe issues."
2
+ name: code-review
3
+ description: Read-only Review entrypoint for a settled diff. Substantial changes receive independent Spec and Standards review; targeted repair review runs only the affected lens.
4
4
  ---
5
5
 
6
- # Code Review
7
-
8
- This skill performs evidence-based code review. It is not a style pass and not a summary. Treat the change as potentially wrong until independent review tracks fail to break it.
9
-
10
- Passing tests, test names, checklists, and implementation reports are inputs,
11
- not proof. For each material behavior, trace the production path before reading
12
- its tests, attempt one concrete violating sequence, then verify that the exact
13
- setup, actions, and assertions reject it. If a fake bypasses the claimed
14
- boundary or the test stays green, report a finding or verification gap; do not
15
- approve nominal coverage.
16
-
17
- The review always covers two lenses:
18
-
19
- - **Correctness reviewer**: bugs, regressions, runtime behavior, security, contracts, caches, concurrency, framework rules, and failure paths.
20
- - **Spec & standards reviewer**: requested behavior, documented repo standards, architecture fit, duplication, cleanup, and workaround-shaped implementation.
21
-
22
- The main agent is the coordinator. It pins the review target, assigns both
23
- lenses to one reviewer for `simple` and `medium`, and splits them across two
24
- independent reviewers only for `high`. It verifies the strongest findings,
25
- applies only safe fixes, and returns a concise findings-first report.
26
-
27
- Full review is bounded to the settled diff, its authority, changed owners, and
28
- callers or contracts with plausible fan-out. `Full` means complete coverage of
29
- that assigned scope once; it does not mean a repository-wide audit. Do not load
30
- unrelated modules or activate optional lenses without a diff signal or mandatory
31
- Review Focus.
32
-
33
- ## When To Use
34
-
35
- Use this skill when the user asks for:
36
-
37
- - code review, PR review, commit audit, regression scan, or bug hunt
38
- - review since a branch, commit, tag, merge-base, or working tree state
39
- - review and fix of critical or high-severity high-confidence defects
40
- - framework-focused review such as `NestJS`, `Next.js`, `Flutter`, or `Dart`
41
-
42
- For every implementation profile, the spec/standards lens includes bounded
43
- cleanup for duplication, obsolete paths, workaround branches, and unjustified
44
- abstractions. High-risk work assigns that lens to its own reviewer in the same
45
- parallel final wave. A concrete evidenced simplification risk named by the
46
- user, approved source, or repo policy amplifies this lens inside the same review
47
- activation; it never creates a separate cleanup gate.
48
-
49
- Exception for approved spec execution: follow
50
- `../spec-implementer/references/review-loop.md`. Intermediate code-review
51
- checkpoints activate only their assigned Review Focus; final cleanup coverage
52
- uses the durable Review Plan and canonical Defect Ledger.
53
-
54
- ## Implementation Review Adapter
55
-
56
- When this skill is called from `$spec-implementer`:
57
-
58
- - read `../spec-implementer/references/review-loop.md` and the persisted
59
- `## Implementation Review State`
60
- - recheck the scheduled profile against the settled diff; return an
61
- underclassified profile to the executor before launching reviewers
62
- - accept the scheduled mode, session, revision, and lenses after that check
63
- - pin the target and give reviewers the owner-defined capsule
64
- - return the usable result and stable defect updates to the executor
65
- - keep cleanup findings in the spec/standards lineage and canonical Defect Ledger
66
-
67
- Do not infer a fresh review loop, choose another mode, or make the owner's
68
- terminal decision inside this Adapter.
69
-
70
- ## Progressive References
71
-
72
- Read only the references the current review needs:
73
-
74
- - Detailed bug classes: `references/bug-classes.md`
75
- - Cleanup lens method: `references/cleanup-lens.md`
76
- - Framework lenses for Next.js, NestJS, Flutter, and Dart: `references/framework-lenses.md`
77
- - Targeted recipes for recurring diff shapes: `references/targeted-recipes.md`
78
- - Contract test ledger: `../../docs/agents/contract-test-ledger.md`
79
- - Shared confidence rubric: `../../docs/agents/confidence-rubric.md`
80
-
81
- Load `references/framework-lenses.md` when the user names a framework or files/configs strongly imply one. Load `references/targeted-recipes.md` when the diff shape matches them. Load `../../docs/agents/contract-test-ledger.md` only for a material contract delta with a named failure ordinary targeted proof could miss. Load `references/bug-classes.md` for substantial reviews or broad bug hunts.
82
- Load `references/cleanup-lens.md` when the spec/standards lens is assigned. Use
83
- its bounded method by default and its amplified method only for a concrete
84
- evidenced simplification risk supplied as mandatory Review Focus.
85
-
86
- ## Coordinator Workflow
87
-
88
- ### 1. Pin The Review Target
89
-
90
- Identify exactly what is being reviewed.
91
-
92
- - If the user gave a fixed point, use it directly: branch, commit SHA, tag, `main`, `HEAD~5`, etc.
93
- - If they did not, infer from context:
94
- - current uncommitted work: `git diff` plus staged diff if relevant
95
- - branch review: `git diff <base>...HEAD`
96
- - commit review: `git show <commit>`
97
- - If there is no safe inference, ask one short question: "Review against which branch or commit?"
98
-
99
- Capture:
100
-
101
- - `git status --short`
102
- - diff stat
103
- - the exact diff command used
104
- - commit list when reviewing a branch range
105
- - changed files and nearest related tests/docs
106
-
107
- Use three-dot diff for branch/base reviews: `git diff <fixed-point>...HEAD`.
108
-
109
- ### 2. Discover Spec And Standards
110
-
111
- Do this before reviewer tracks so both tracks receive bounded inputs.
112
-
113
- Spec sources, in priority order:
114
-
115
- 1. Issue or PR references in commit messages, branch names, PR metadata, or user prompt.
116
- 2. A spec/PRD/plan path supplied by the user.
117
- 3. Matching files under `docs/`, `specs/`, `.scratch/`, or local issue folders.
118
- 4. If none exists, continue and mark the spec axis as "no spec found" instead of inventing requirements.
119
-
120
- Standards sources:
121
-
122
- - `AGENTS.md`, `CLAUDE.md`, `CONTRIBUTING.md`
123
- - `CONTEXT.md`, context maps, domain docs, ADRs
124
- - `STYLE.md`, `STANDARDS.md`, style guides, review checklists
125
- - `.editorconfig`, ESLint, Biome, Prettier, TypeScript, analyzer, or framework configs
126
- - relevant test files and existing examples in the touched area
127
-
128
- Machine-enforced config matters as context, but do not spend review findings on issues a required formatter/linter would already catch unless the tool is absent or failing.
129
-
130
- ### 3. Activate Lenses
131
-
132
- Before deep review, decide which lenses apply.
133
-
134
- - Always activate the general correctness and spec/standards lenses.
135
- - Always apply bounded cleanup inside spec/standards; amplify it only for a
136
- concrete evidenced Review Focus, never from size or risk labels alone.
137
- - Add framework lenses when explicit or strongly implied by files/configs.
138
- - Add targeted recipes when the diff shape matches them.
139
- - If the user, plan, or implementation spec provides `Review Focus`, treat each listed lens, targeted recipe, invariant, and risk as mandatory. Do not replace it with a generic review; report any focus item that cannot be verified as a verification gap.
140
- - If inference is uncertain, say so and continue with the best-supported review instead of pretending certainty.
141
-
142
- Mandatory delta lenses:
143
-
144
- - **Contract Delta Review:** If the diff expands a shared contract, enum/status, schema, DTO, permission, capability, token, or scope, search old consumers and report `changed contract -> searched call sites -> risky fallback/default branches -> tests or fix`.
145
- - **Backend Trust Boundary Review:** If the diff adds a credential, OAuth scope, permission, role-sensitive write, or external capability, verify the backend-side guard. UI controls, query/state flags, and caller intent are not authorization.
146
-
147
- Common framework signals:
148
-
149
- - **Next.js**: `next.config.*`, `app/**`, `pages/**`, route handlers, server actions, `use server`, `use client`, revalidation APIs.
150
- - **NestJS**: `@nestjs/*`, controllers, providers, modules, guards, pipes, interceptors, DTOs, `nest-cli.json`.
151
- - **Flutter**: `pubspec.yaml`, `lib/**`, widgets, navigation, bloc/provider/riverpod/notifiers.
152
- - **Dart**: `.dart`, `Future`, `Stream`, isolates, generated serializers, null-safety constructs.
153
-
154
- ### 4. Run Profile-Selected Reviewers
155
-
156
- For `simple`, run one `reviewer_fast`; for `medium`, run one
157
- `reviewer_standard`. Assign that child both correctness and spec/standards
158
- lenses in one bounded Full review. For `high`, run two `reviewer_deep` children
159
- in parallel with one disjoint lens each. Invoking `$code-review` authorizes
160
- these reviewers. Preserve every fulfilled launch handle if a parallel peer
161
- fails, then close all launched children. Tell every reviewer not to edit or
162
- revert unrelated work.
163
-
164
- For spec-driven checkpoints, obey the track assignment in the persisted Review
165
- Plan instead of automatically launching both default tracks.
166
-
167
- Medium is the normal review profile. API, persistence, multiple files, or a
168
- shared-looking name do not select `high` unless evidence proves both a material
169
- failure consequence and an uncertainty amplifier.
170
-
171
- When already inside an assigned reviewer child, execute its assigned lens set
172
- inline and return it to root; do not spawn a grandchild.
173
- If the user forbids delegation, report the independent review gate as waived or
174
- unavailable according to the parent workflow; root must not self-review inline.
175
-
176
- #### Correctness Reviewer Brief
177
-
178
- Include the exact diff command or commit under review, changed file list, commit list, active references, any `Review Focus`, and an instruction to read surrounding execution paths.
179
-
180
- > Review runtime correctness adversarially. Hunt concrete bugs in control flow, state, async/concurrency, security, auth, contracts, schemas, caching, persistence, UI/API integration, and framework-specific behavior. If a Review Focus is provided, explicitly test the named risks first, such as duplicate side effects, retry/idempotency, ordering, source-of-truth ownership, partial failure, DTO/schema drift, or false user-facing state. For each finding, provide file/line, trigger path, impact, why guards do not prevent it, severity, and confidence. Do not report style nits or generic "needs tests" comments.
181
- > For bugfixes, check claim boundaries: what changed tests prove, what they do not prove, and whether sibling execution paths can still violate the claimed invariant.
182
- > When Contract Delta Review applies, follow new values through old consumers and default/fallback branches before trusting local tests. When Backend Trust Boundary Review applies, verify the server-side authorization predicate at the write/callback boundary.
183
-
184
- #### Spec & Standards Reviewer Brief
185
-
186
- Include the exact diff command or commit under review, spec source path/content or "no spec found", standards source list, changed file list, commit list, any `Review Focus`, and an instruction to cite the spec or standard behind each finding.
187
-
188
- > Review the change against the requested work and repo standards. Check missing requirements, partial behavior, scope creep, undocumented contract changes, architecture drift, duplicate source-of-truth logic, dead/legacy branches, and workaround-shaped implementation. If a Review Focus is provided, explicitly verify each named ownership, scope, validation, and invariant risk against the spec. Cite the spec or standard when available. If there is no spec, skip requirement claims and focus on documented standards and architecture evidence.
189
- > Apply `references/cleanup-lens.md`: use bounded cleanup by default, or the amplified method when a concrete evidenced simplification risk is mandatory Review Focus. Keep cleanup findings in this review's normal Defect Ledger. Do not create a cleanup-only verdict or separate pass.
190
-
191
- ### 5. Aggregate And Verify
192
-
193
- The coordinator must not blindly relay reviewer output.
194
-
195
- 1. Deduplicate findings across tracks.
196
- 2. Re-read the relevant code for the strongest findings.
197
- 3. Drop findings that lack a concrete trigger path.
198
- 4. Reclassify severity/confidence using `../../docs/agents/confidence-rubric.md` if evidence does not support the label.
199
- 5. For real contract defects that pass the ledger gate, identify the missing or inadequate invariant when TDD/spec evidence is available.
200
- 6. Confirm every mandatory `Review Focus` item and mandatory delta lens was reviewed; if not, report the unverified item as a verification gap.
201
- 7. Confirm that mandatory behavior evidence rejects the concrete violating
202
- sequence attempted by the reviewer. Evidence that bypasses its claimed
203
- production boundary invalidates that Full coverage.
204
- 8. Decide whether auto-fix is allowed.
205
- 9. Run the narrowest meaningful verification after any fix.
206
-
207
- Keep the two axes visible in your own notes, but present the final report by severity unless the user explicitly asked for side-by-side Standards/Spec output.
208
-
209
- After repairs, verify medium or low findings through the coordinator's direct
210
- failure-path check plus affected validation. Launch Closure only for the
211
- severity, protected-contract, or invalidated-coverage triggers owned by
212
- `review-protocol.md`. For a scheduled Closure, send the bounded capsule to the
213
- selected reviewer lineage and return its defect updates.
214
-
215
- ## Evidence Standard
216
-
217
- A valid finding explains:
218
-
219
- - what breaks
220
- - why it breaks
221
- - the input, sequence, role, tenant, environment, or timing that triggers it
222
- - where the defect lives
223
- - why existing guards do not prevent it
224
- - severity and confidence
225
-
226
- Do not file:
227
-
228
- - style nits disguised as correctness issues
229
- - speculative races without a shared-state path
230
- - generic "needs tests" comments without a concrete regression risk
231
- - architecture discomfort without wrong ownership, duplication, leakage, or a regression path
232
- - performance comments without a hot path or failure mode
233
-
234
- ## Auto-Fix Policy
235
-
236
- Automatically fix only when all are true:
237
-
238
- - severity is critical or high
239
- - confidence is high under `../../docs/agents/confidence-rubric.md`
240
- - root cause is clear
241
- - correct fix is narrow and low-risk
242
- - fix matches local project patterns
243
- - verification is available, or the edit is obviously safe and syntax-checkable
244
-
245
- When auto-fixing:
246
-
247
- - patch only the bug
248
- - add/update behavior tests when regression risk is meaningful and the codebase supports it; update a ledger only when its gate passes
249
- - rerun relevant verification
250
- - never revert unrelated user changes
251
-
252
- Do not auto-fix ambiguous semantics, product decisions, broad refactors, or low-confidence concerns. Report them with evidence.
253
-
254
- ## Output Contract
255
-
256
- For review-only tasks:
257
-
258
- 1. Findings first, ordered by severity.
259
- 2. File and line references for each finding.
260
- 3. Trigger path, impact, severity, confidence, and evidence.
261
- 4. Open questions, assumptions, or verification gaps.
262
- 5. Short summary only after findings.
263
-
264
- For review-and-fix tasks:
265
-
266
- 1. State which critical or high-severity high-confidence issues were fixed.
267
- 2. Report remaining findings that were not fixed.
268
- 3. Give one short verification note and any blocked checks.
269
- 4. Keep the closing summary user-facing and outcome-based.
270
-
271
- If there are no findings, say so clearly and mention residual test or verification gaps.
272
-
273
- For spec-driven review checkpoints or final review gates, include a compact review handoff that can feed the executor's Final Risk Handoff: reviewed target, Review Focus status, high/critical findings fixed or remaining, skipped checks, and residual verification gaps. Keep findings first.
274
-
275
- If inline review comments are requested, emit one `::code-comment{...}` directive per actionable finding.
276
-
277
- ## Tooling Defaults
278
-
279
- - Use `rg`/`rg --files` for code search.
280
- - Prefer parallel reads for status, diff, changed files, related modules, tests, repo docs, and standards.
281
- - Use official docs or Context7 only when a finding depends on version-sensitive framework/library behavior.
282
- - Run the narrowest meaningful tests first, then required lint/build/analyzer checks for touched areas.
283
- - Keep raw command output out of the final response unless the user asks for it.
284
-
285
- ## Decision Defaults
286
-
287
- - If the user says "review", default to findings-first review.
288
- - If the user says "review and fix", auto-fix only critical or high-severity high-confidence issues that satisfy the auto-fix policy.
289
- - If the user provides a framework focus, explicitly use it.
290
- - If no focus is provided, infer active lenses from the diff and say when they materially affected findings.
291
- - Ask clarification only when the correct review target or fix would otherwise be risky or ambiguous.
6
+ # Review
7
+
8
+ Review a settled change through two independent lenses:
9
+
10
+ - **Spec:** requirement fidelity, missing or partial behavior, incorrect
11
+ implementation, and scope drift.
12
+ - **Standards:** correctness, failure paths, repository policy, maintainability,
13
+ legacy residue, duplicate ownership, and unnecessary machinery.
14
+
15
+ Review is inspection-only. Never edit, repair, stage, commit, push, or open a
16
+ PR. Substantial behavior or contract changes require Review; obvious local
17
+ docs, copy, formatting, mechanical config, or corrections may use direct proof.
18
+
19
+ ## Boundaries
20
+
21
+ - **Authorized outcome:** the request, issue, or Parent PRD plus existing
22
+ invariants and mandatory repository rules.
23
+ - **Impact cone:** callers, data, runtime, and proof surfaces that may be
24
+ inspected to establish effects. Inspection creates no authority.
25
+ - **Repair scope:** the smallest change that restores a proven obligation,
26
+ invariant, or mandatory rule without widening the authorized outcome.
27
+
28
+ A reviewer verdict is evidence, not authority. Mark a finding `BLOCKER` only
29
+ when evidence links a concrete defect or required-proof gap to an authorized
30
+ obligation, existing invariant, or mandatory rule. Include its source,
31
+ file/line, trigger, impact, and evidence. Treat unsupported preferences,
32
+ hypothetical hardening, and architecture or cleanup beyond the outcome as
33
+ `OBSERVATION`.
34
+
35
+ For Standards, read [standards-smells.md](references/standards-smells.md).
36
+ Fowler smells remain non-blocking observations unless separate evidence meets the blocker
37
+ threshold. Repository policy wins, and tooling-enforced rules need no duplicate
38
+ finding. If optional machinery causes a defect, prefer deleting it; extend it
39
+ only when the authorized outcome requires it.
40
+
41
+ ## Process
42
+
43
+ 1. Pin an existing baseline and settled target, then verify the comparison is
44
+ valid and non-empty. Supply the exact diff, changed files, proof, authority
45
+ source, and repository policy. Missing required authority or proof is a gap;
46
+ never silently skip a lens.
47
+ 2. Select the mode:
48
+ - initial Review launches one fresh `spec_reviewer` and one fresh `standards_reviewer`
49
+ in parallel;
50
+ - targeted Review runs only the affected lens: Spec-only repair for
51
+ requirement coverage or behavior, Standards-only repair for correctness,
52
+ invariants, architecture, or mandatory rules, and both lenses when the
53
+ repair affects both or cannot be isolated.
54
+ Give targeted reviewers the previous reviewed revision, repair delta,
55
+ repaired blockers, direct impact cone, and affected proof. Untouched
56
+ previously approved scope retains approval.
57
+ 3. Give both reviewers the same target, diff, proof, authority, policy, and
58
+ boundaries above. Keep each brief under 400 words and begin it with
59
+ `Assigned role: <role>`.
60
+ - `spec_reviewer` checks only missing, partial, incorrect, or extra behavior
61
+ against cited authority.
62
+ - `standards_reviewer` checks correctness, failure paths, policy, cleanup,
63
+ zero legacy, duplicate ownership, unnecessary machinery, and the smell
64
+ baseline without inventing product obligations.
65
+ 4. Use fresh children without history fork (`fork_context=false` on V1;
66
+ `fork_turns="none"` on V2). Capture every non-empty child identity and wait
67
+ for those same children. An empty bounded wait means wait again, not failure.
68
+ Silence is not a hang; request status only without interruption. Never
69
+ interrupt, terminate, close, or replace an active reviewer. Missing identity,
70
+ terminal failure, interruption, or incomplete final wait blocks Review;
71
+ root never substitutes self-review.
72
+ 5. Reconcile each finding from its evidence, preserving its lens and defect
73
+ identity. Reclassify unsupported `BLOCK` labels as observations; an
74
+ `APPROVE` label cannot erase a verified blocker. Consolidate all verified
75
+ blockers into one repair batch for the active Implement or Tickets
76
+ Orchestrator owner. Their original authority covers repairs inside Repair
77
+ scope without another confirmation. For review-only requests, report and
78
+ stop without invoking Implement.
79
+ 6. Return `Spec` and `Standards` results separately, then consolidated blockers,
80
+ observations, and proof gaps. Approve only after every selected independent
81
+ reviewer completed and no verified blocker or required-proof gap remains.
82
+ Review may approve with observations. A complete Review repeats only when a
83
+ repair cannot be isolated.
84
+
85
+ Each reviewer returns findings first, labelled `BLOCKER` or `OBSERVATION`, then
86
+ `APPROVE` or `BLOCK` with a concrete reason. Do not create durable review state,
87
+ maps, ledgers, aliases, adapters, or compatibility routes.
@@ -1,4 +1,6 @@
1
1
  interface:
2
- display_name: "Code Review"
3
- short_description: "Profile-routed correctness and standards review"
4
- default_prompt: "Use $code-review for one final profile-selected wave; high risk uses two parallel reviewers and the spec/standards lens includes bounded cleanup."
2
+ display_name: "Review"
3
+ short_description: "Independent requirements and standards review"
4
+ default_prompt: "Use $code-review to inspect the settled change with independent Spec and Standards lenses. Report findings without implementing or repairing them."
5
+ policy:
6
+ allow_implicit_invocation: true
@@ -0,0 +1,83 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "code-review",
4
+ "cases": [
5
+ {
6
+ "id": "authorized-regression-blocks",
7
+ "prompt": "A settled diff breaks an existing caller invariant required by the authorized behavior; the failing public-seam regression proves the trigger and impact.",
8
+ "expected": [
9
+ "report a blocker linked to the existing invariant",
10
+ "include the trigger, impact, and regression evidence",
11
+ "return BLOCK"
12
+ ],
13
+ "forbidden": [
14
+ "downgrade the proven regression to an observation",
15
+ "require a broader solution than restoring the invariant"
16
+ ]
17
+ },
18
+ {
19
+ "id": "optional-architecture-is-observation",
20
+ "prompt": "The authorized outcome and all existing invariants pass, but the reviewer prefers a more robust architecture for hypothetical future load.",
21
+ "expected": [
22
+ "report the architecture or robustness idea only as an observation",
23
+ "approve when no authorized obligation, invariant, mandatory rule, or proof is missing"
24
+ ],
25
+ "forbidden": [
26
+ "turn hypothetical robustness into a blocker",
27
+ "treat the reviewer label or impact cone as authority"
28
+ ]
29
+ },
30
+ {
31
+ "id": "optional-machinery-prefers-removal",
32
+ "prompt": "The diff adds an optional mediator that is not needed by the authorized outcome, and the mediator contains a defect. Removing it preserves all required behavior and invariants.",
33
+ "expected": [
34
+ "prefer removing the optional mediator",
35
+ "keep repair scope within the authorized outcome"
36
+ ],
37
+ "forbidden": [
38
+ "require hardening or extending the optional mediator",
39
+ "expand the authorized outcome to preserve unnecessary machinery"
40
+ ]
41
+ },
42
+ {
43
+ "id": "unsupported-block-reconciles-to-approval",
44
+ "prompt": "A completed independent reviewer returns BLOCK, but coordinator verification proves every reported item is an optional preference with no authorized obligation, invariant, mandatory rule, or required-proof gap.",
45
+ "expected": [
46
+ "reclassify the unsupported findings as observations",
47
+ "return APPROVE from the completed independent evidence",
48
+ "perform no repair"
49
+ ],
50
+ "forbidden": [
51
+ "repeat review solely because of the reviewer label",
52
+ "expand the authorized outcome to satisfy an unsupported verdict"
53
+ ]
54
+ },
55
+ {
56
+ "id": "two-independent-read-only-lenses",
57
+ "prompt": "Run Review on one settled substantial target and report its findings.",
58
+ "expected": [
59
+ "launch fresh Spec and Standards reviewers in parallel",
60
+ "keep their findings separated by lens",
61
+ "keep Review read-only",
62
+ "return observations and consolidated blockers to the caller"
63
+ ],
64
+ "forbidden": [
65
+ "repair the target",
66
+ "let either reviewer invent new authority"
67
+ ]
68
+ },
69
+ {
70
+ "id": "targeted-review-selects-affected-lens",
71
+ "prompt": "A reviewed repair delta only restores one missing authorized requirement and changes no correctness, invariant, architecture, or repository-rule surface.",
72
+ "expected": [
73
+ "run only a fresh Spec reviewer",
74
+ "limit review to the repair delta and direct impact cone",
75
+ "preserve untouched approval"
76
+ ],
77
+ "forbidden": [
78
+ "rerun Standards without an affected Standards surface",
79
+ "repeat the complete review"
80
+ ]
81
+ }
82
+ ]
83
+ }