codex-orchestrator 2.0.11 → 2.0.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (262) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/README.md +25 -51
  3. package/dist/src/index.d.ts +2 -8
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/index.js +1 -4
  6. package/dist/src/index.js.map +1 -1
  7. package/dist/src/v2/acceptance-proof.d.ts +46 -31
  8. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  9. package/dist/src/v2/acceptance-proof.js +157 -195
  10. package/dist/src/v2/acceptance-proof.js.map +1 -1
  11. package/dist/src/v2/active-attempt.d.ts +94 -0
  12. package/dist/src/v2/active-attempt.d.ts.map +1 -0
  13. package/dist/src/v2/active-attempt.js +200 -0
  14. package/dist/src/v2/active-attempt.js.map +1 -0
  15. package/dist/src/v2/adapters/command.d.ts +6 -0
  16. package/dist/src/v2/adapters/command.d.ts.map +1 -1
  17. package/dist/src/v2/adapters/command.js +43 -2
  18. package/dist/src/v2/adapters/command.js.map +1 -1
  19. package/dist/src/v2/candidate.d.ts +15 -31
  20. package/dist/src/v2/candidate.d.ts.map +1 -1
  21. package/dist/src/v2/candidate.js +7 -29
  22. package/dist/src/v2/candidate.js.map +1 -1
  23. package/dist/src/v2/checked-change.d.ts +3 -2
  24. package/dist/src/v2/checked-change.d.ts.map +1 -1
  25. package/dist/src/v2/checked-change.js +4 -3
  26. package/dist/src/v2/checked-change.js.map +1 -1
  27. package/dist/src/v2/cli-contract.d.ts +1 -1
  28. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  29. package/dist/src/v2/cli-contract.js +4 -6
  30. package/dist/src/v2/cli-contract.js.map +1 -1
  31. package/dist/src/v2/cli.d.ts +8 -0
  32. package/dist/src/v2/cli.d.ts.map +1 -1
  33. package/dist/src/v2/cli.js +13 -0
  34. package/dist/src/v2/cli.js.map +1 -1
  35. package/dist/src/v2/code-review-report.d.ts +10 -18
  36. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  37. package/dist/src/v2/code-review-report.js +63 -60
  38. package/dist/src/v2/code-review-report.js.map +1 -1
  39. package/dist/src/v2/codex-process.d.ts +6 -2
  40. package/dist/src/v2/codex-process.d.ts.map +1 -1
  41. package/dist/src/v2/codex-process.js +25 -9
  42. package/dist/src/v2/codex-process.js.map +1 -1
  43. package/dist/src/v2/config.d.ts +0 -2
  44. package/dist/src/v2/config.d.ts.map +1 -1
  45. package/dist/src/v2/config.js +3 -6
  46. package/dist/src/v2/config.js.map +1 -1
  47. package/dist/src/v2/contained-report-operation.d.ts +41 -196
  48. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  49. package/dist/src/v2/contained-report-operation.js +139 -466
  50. package/dist/src/v2/contained-report-operation.js.map +1 -1
  51. package/dist/src/v2/containment.d.ts +1 -0
  52. package/dist/src/v2/containment.d.ts.map +1 -1
  53. package/dist/src/v2/containment.js +12 -2
  54. package/dist/src/v2/containment.js.map +1 -1
  55. package/dist/src/v2/delivery-authority.d.ts +26 -0
  56. package/dist/src/v2/delivery-authority.d.ts.map +1 -0
  57. package/dist/src/v2/delivery-authority.js +44 -0
  58. package/dist/src/v2/delivery-authority.js.map +1 -0
  59. package/dist/src/v2/direct-delivery.d.ts +16 -36
  60. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  61. package/dist/src/v2/direct-delivery.js +135 -122
  62. package/dist/src/v2/direct-delivery.js.map +1 -1
  63. package/dist/src/v2/immutable-workflow-publisher.d.ts.map +1 -1
  64. package/dist/src/v2/immutable-workflow-publisher.js +3 -1
  65. package/dist/src/v2/immutable-workflow-publisher.js.map +1 -1
  66. package/dist/src/v2/implementation-report.d.ts +3 -1
  67. package/dist/src/v2/implementation-report.d.ts.map +1 -1
  68. package/dist/src/v2/implementation-report.js +28 -10
  69. package/dist/src/v2/implementation-report.js.map +1 -1
  70. package/dist/src/v2/implementation-reviewer.d.ts +41 -12
  71. package/dist/src/v2/implementation-reviewer.d.ts.map +1 -1
  72. package/dist/src/v2/implementation-reviewer.js +114 -42
  73. package/dist/src/v2/implementation-reviewer.js.map +1 -1
  74. package/dist/src/v2/pending-effect-settlement.d.ts +44 -0
  75. package/dist/src/v2/pending-effect-settlement.d.ts.map +1 -0
  76. package/dist/src/v2/pending-effect-settlement.js +69 -0
  77. package/dist/src/v2/pending-effect-settlement.js.map +1 -0
  78. package/dist/src/v2/process-identity.d.ts +45 -0
  79. package/dist/src/v2/process-identity.d.ts.map +1 -0
  80. package/dist/src/v2/process-identity.js +118 -0
  81. package/dist/src/v2/process-identity.js.map +1 -0
  82. package/dist/src/v2/proof-report.d.ts +2 -1
  83. package/dist/src/v2/proof-report.d.ts.map +1 -1
  84. package/dist/src/v2/proof-report.js +10 -4
  85. package/dist/src/v2/proof-report.js.map +1 -1
  86. package/dist/src/v2/review-feedback-coordinator.d.ts +1 -1
  87. package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -1
  88. package/dist/src/v2/review-feedback-coordinator.js +1 -1
  89. package/dist/src/v2/review-feedback-coordinator.js.map +1 -1
  90. package/dist/src/v2/review-feedback.d.ts +14 -20
  91. package/dist/src/v2/review-feedback.d.ts.map +1 -1
  92. package/dist/src/v2/review-feedback.js +45 -87
  93. package/dist/src/v2/review-feedback.js.map +1 -1
  94. package/dist/src/v2/run-issue.d.ts +129 -88
  95. package/dist/src/v2/run-issue.d.ts.map +1 -1
  96. package/dist/src/v2/run-issue.js +1965 -2381
  97. package/dist/src/v2/run-issue.js.map +1 -1
  98. package/dist/src/v2/run-state-projections.d.ts +84 -0
  99. package/dist/src/v2/run-state-projections.d.ts.map +1 -0
  100. package/dist/src/v2/run-state-projections.js +142 -0
  101. package/dist/src/v2/run-state-projections.js.map +1 -0
  102. package/dist/src/v2/run-store.d.ts +99 -81
  103. package/dist/src/v2/run-store.d.ts.map +1 -1
  104. package/dist/src/v2/run-store.js +245 -542
  105. package/dist/src/v2/run-store.js.map +1 -1
  106. package/dist/src/v2/runtime-assets.d.ts +3 -0
  107. package/dist/src/v2/runtime-assets.d.ts.map +1 -1
  108. package/dist/src/v2/runtime-assets.js +104 -0
  109. package/dist/src/v2/runtime-assets.js.map +1 -1
  110. package/dist/src/v2/runtime.d.ts +56 -44
  111. package/dist/src/v2/runtime.d.ts.map +1 -1
  112. package/dist/src/v2/runtime.js +383 -503
  113. package/dist/src/v2/runtime.js.map +1 -1
  114. package/dist/src/v2/setup.js +0 -2
  115. package/dist/src/v2/setup.js.map +1 -1
  116. package/dist/src/v2/validation-progression.d.ts +70 -0
  117. package/dist/src/v2/validation-progression.d.ts.map +1 -0
  118. package/dist/src/v2/validation-progression.js +247 -0
  119. package/dist/src/v2/validation-progression.js.map +1 -0
  120. package/dist/src/v2/workflow-assets.d.ts +9 -3
  121. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  122. package/dist/src/v2/workflow-assets.js +256 -43
  123. package/dist/src/v2/workflow-assets.js.map +1 -1
  124. package/internal-workflow/docs/agents/bug-workflow-routing.md +9 -7
  125. package/internal-workflow/docs/agents/coding-skill-routing.md +170 -120
  126. package/internal-workflow/docs/agents/tool-usage.md +23 -12
  127. package/internal-workflow/manifest.json +1 -1
  128. package/internal-workflow/operations/code-review/SKILL.md +34 -15
  129. package/internal-workflow/operations/implementation/SKILL.md +21 -16
  130. package/internal-workflow/profiles/implementer.toml +9 -0
  131. package/internal-workflow/profiles/review_coordinator.toml +9 -0
  132. package/internal-workflow/profiles/spec_reviewer.toml +9 -0
  133. package/internal-workflow/profiles/standards_reviewer.toml +9 -0
  134. package/internal-workflow/schemas/code-review-v1.json +1 -1
  135. package/internal-workflow/schemas/implementation-report-v1.json +1 -1
  136. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  137. package/internal-workflow/skills/bug-root-cause-explainer/SKILL.md +114 -0
  138. package/internal-workflow/skills/bug-root-cause-explainer/agents/openai.yaml +7 -0
  139. package/internal-workflow/skills/bug-root-cause-explainer/evals/evals.json +18 -0
  140. package/internal-workflow/skills/code-review/SKILL.md +84 -306
  141. package/internal-workflow/skills/code-review/agents/openai.yaml +5 -3
  142. package/internal-workflow/skills/code-review/evals/evals.json +83 -0
  143. package/internal-workflow/skills/code-review/references/standards-smells.md +41 -0
  144. package/internal-workflow/skills/diagnosing-bugs/SKILL.md +69 -32
  145. package/internal-workflow/skills/diagnosing-bugs/agents/openai.yaml +2 -2
  146. package/internal-workflow/skills/diagnosing-bugs/evals/evals.json +63 -0
  147. package/internal-workflow/skills/grilling/SKILL.md +51 -0
  148. package/internal-workflow/skills/grilling/agents/openai.yaml +6 -0
  149. package/internal-workflow/skills/grilling/evals/evals.json +47 -0
  150. package/internal-workflow/skills/implement/SKILL.md +135 -0
  151. package/internal-workflow/skills/implement/agents/openai.yaml +6 -0
  152. package/internal-workflow/skills/implement/evals/evals.json +150 -0
  153. package/internal-workflow/skills/plan/SKILL.md +59 -0
  154. package/internal-workflow/skills/plan/agents/openai.yaml +6 -0
  155. package/internal-workflow/skills/plan/evals/evals.json +36 -0
  156. package/internal-workflow/skills/prototype/LOGIC.md +130 -0
  157. package/internal-workflow/skills/prototype/SKILL.md +69 -0
  158. package/internal-workflow/skills/prototype/UI.md +157 -0
  159. package/internal-workflow/skills/prototype/agents/openai.yaml +6 -0
  160. package/internal-workflow/skills/prototype/evals/evals.json +67 -0
  161. package/internal-workflow/skills/research/SKILL.md +110 -0
  162. package/internal-workflow/skills/research/agents/openai.yaml +6 -0
  163. package/internal-workflow/skills/research/evals/evals.json +49 -0
  164. package/internal-workflow/skills/tdd/SKILL.md +72 -67
  165. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  166. package/internal-workflow/skills/tdd/evals/evals.json +12 -0
  167. package/internal-workflow/skills/tdd/mocking.md +48 -1
  168. package/internal-workflow/skills/tdd/refactoring.md +3 -3
  169. package/internal-workflow/skills/tickets-orchestrator/SKILL.md +199 -0
  170. package/internal-workflow/skills/tickets-orchestrator/agents/openai.yaml +6 -0
  171. package/internal-workflow/skills/tickets-orchestrator/evals/evals.json +126 -0
  172. package/internal-workflow/skills/tickets-orchestrator/references/delegate-integrate.md +83 -0
  173. package/internal-workflow/skills/tickets-orchestrator/references/finish-delivery.md +69 -0
  174. package/internal-workflow/skills/tickets-orchestrator/references/stop-completion.md +63 -0
  175. package/internal-workflow/skills/to-spec/SKILL.md +133 -0
  176. package/internal-workflow/skills/to-spec/agents/openai.yaml +6 -0
  177. package/internal-workflow/skills/to-spec/evals/evals.json +24 -0
  178. package/internal-workflow/skills/to-tickets/SKILL.md +189 -0
  179. package/internal-workflow/skills/to-tickets/agents/openai.yaml +6 -0
  180. package/internal-workflow/skills/to-tickets/evals/evals.json +79 -0
  181. package/internal-workflow/skills/to-tickets/references/publishing-details.md +117 -0
  182. package/package.json +1 -1
  183. package/dist/src/v2/proof-store.d.ts +0 -54
  184. package/dist/src/v2/proof-store.d.ts.map +0 -1
  185. package/dist/src/v2/proof-store.js +0 -301
  186. package/dist/src/v2/proof-store.js.map +0 -1
  187. package/dist/src/v2/route-continuations.d.ts +0 -32
  188. package/dist/src/v2/route-continuations.d.ts.map +0 -1
  189. package/dist/src/v2/route-continuations.js +0 -2
  190. package/dist/src/v2/route-continuations.js.map +0 -1
  191. package/dist/src/v2/route-coordinator.d.ts +0 -72
  192. package/dist/src/v2/route-coordinator.d.ts.map +0 -1
  193. package/dist/src/v2/route-coordinator.js +0 -275
  194. package/dist/src/v2/route-coordinator.js.map +0 -1
  195. package/dist/src/v2/route-decision.d.ts +0 -120
  196. package/dist/src/v2/route-decision.d.ts.map +0 -1
  197. package/dist/src/v2/route-decision.js +0 -380
  198. package/dist/src/v2/route-decision.js.map +0 -1
  199. package/dist/src/v2/spec-coordinator.d.ts +0 -73
  200. package/dist/src/v2/spec-coordinator.d.ts.map +0 -1
  201. package/dist/src/v2/spec-coordinator.js +0 -126
  202. package/dist/src/v2/spec-coordinator.js.map +0 -1
  203. package/dist/src/v2/spec-delivery.d.ts +0 -112
  204. package/dist/src/v2/spec-delivery.d.ts.map +0 -1
  205. package/dist/src/v2/spec-delivery.js +0 -336
  206. package/dist/src/v2/spec-delivery.js.map +0 -1
  207. package/dist/src/v2/triage-route.d.ts +0 -68
  208. package/dist/src/v2/triage-route.d.ts.map +0 -1
  209. package/dist/src/v2/triage-route.js +0 -223
  210. package/dist/src/v2/triage-route.js.map +0 -1
  211. package/dist/src/v2/waiting-human-coordinator.d.ts +0 -49
  212. package/dist/src/v2/waiting-human-coordinator.d.ts.map +0 -1
  213. package/dist/src/v2/waiting-human-coordinator.js +0 -509
  214. package/dist/src/v2/waiting-human-coordinator.js.map +0 -1
  215. package/dist/src/v2/waiting-human.d.ts +0 -143
  216. package/dist/src/v2/waiting-human.d.ts.map +0 -1
  217. package/dist/src/v2/waiting-human.js +0 -408
  218. package/dist/src/v2/waiting-human.js.map +0 -1
  219. package/internal-workflow/docs/agents/contract-test-ledger.md +0 -71
  220. package/internal-workflow/docs/agents/review-gates.md +0 -42
  221. package/internal-workflow/docs/agents/review-protocol.md +0 -98
  222. package/internal-workflow/evals/coding-skill-evals.json +0 -373
  223. package/internal-workflow/operations/ambiguity-review/SKILL.md +0 -5
  224. package/internal-workflow/operations/qualification-repair/SKILL.md +0 -17
  225. package/internal-workflow/operations/spec-author/SKILL.md +0 -12
  226. package/internal-workflow/operations/spec-review/SKILL.md +0 -12
  227. package/internal-workflow/operations/triage/SKILL.md +0 -12
  228. package/internal-workflow/profiles/analyst_deep.toml +0 -9
  229. package/internal-workflow/profiles/implementer_standard.toml +0 -9
  230. package/internal-workflow/profiles/proof_agent.toml +0 -8
  231. package/internal-workflow/profiles/reviewer_deep.toml +0 -9
  232. package/internal-workflow/profiles/reviewer_standard.toml +0 -9
  233. package/internal-workflow/schemas/ambiguity-review-v1.json +0 -1
  234. package/internal-workflow/schemas/spec-author-v1.json +0 -1
  235. package/internal-workflow/schemas/spec-review-v1.json +0 -30
  236. package/internal-workflow/schemas/triage-route-v1.json +0 -1
  237. package/internal-workflow/skills/agent-auto/SKILL.md +0 -19
  238. package/internal-workflow/skills/agent-auto/agents/openai.yaml +0 -6
  239. package/internal-workflow/skills/code-debugger/SKILL.md +0 -122
  240. package/internal-workflow/skills/code-debugger/agents/openai.yaml +0 -7
  241. package/internal-workflow/skills/code-review/references/bug-classes.md +0 -56
  242. package/internal-workflow/skills/code-review/references/cleanup-lens.md +0 -52
  243. package/internal-workflow/skills/code-review/references/framework-lenses.md +0 -34
  244. package/internal-workflow/skills/code-review/references/targeted-recipes.md +0 -49
  245. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +0 -107
  246. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +0 -6
  247. package/internal-workflow/skills/implementation-spec-maker/references/source-modes.md +0 -32
  248. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +0 -146
  249. package/internal-workflow/skills/implementation-spec-review/SKILL.md +0 -131
  250. package/internal-workflow/skills/implementation-spec-review/agents/openai.yaml +0 -6
  251. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +0 -78
  252. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +0 -121
  253. package/internal-workflow/skills/small-task-implementer/SKILL.md +0 -112
  254. package/internal-workflow/skills/small-task-implementer/agents/openai.yaml +0 -6
  255. package/internal-workflow/skills/spec-implementer/SKILL.md +0 -133
  256. package/internal-workflow/skills/spec-implementer/agents/openai.yaml +0 -6
  257. package/internal-workflow/skills/spec-implementer/evals/evals.json +0 -30
  258. package/internal-workflow/skills/spec-implementer/references/review-loop.md +0 -100
  259. package/internal-workflow/skills/triage/AGENT-BRIEF.md +0 -192
  260. package/internal-workflow/skills/triage/OUT-OF-SCOPE.md +0 -101
  261. package/internal-workflow/skills/triage/SKILL.md +0 -134
  262. package/internal-workflow/skills/triage/agents/openai.yaml +0 -6
@@ -1,309 +1,87 @@
1
1
  ---
2
- name: "code-review"
3
- description: "Evidence-first review of code, PRs, commits, regressions, or review-and-fix work using correctness and standards/cleanup lenses. Auto-fix only qualifying high-confidence severe issues."
2
+ name: code-review
3
+ description: Read-only Review entrypoint for a settled diff. Substantial changes receive independent Spec and Standards review; targeted repair review runs only the affected lens.
4
4
  ---
5
5
 
6
- # Code Review
7
-
8
- This skill performs evidence-based code review. It is not a style pass and not a summary. Treat the change as potentially wrong until independent review tracks fail to break it.
9
-
10
- Passing tests, test names, checklists, and implementation reports are inputs,
11
- not proof. For each material behavior, trace the production path before reading
12
- its tests, attempt one concrete violating sequence, then verify that the exact
13
- setup, actions, and assertions reject it. If a fake bypasses the claimed
14
- boundary or the test stays green, report a finding or verification gap; do not
15
- approve nominal coverage.
16
-
17
- The review always covers two lenses:
18
-
19
- - **Correctness reviewer**: bugs, regressions, runtime behavior, security, contracts, caches, concurrency, framework rules, and failure paths.
20
- - **Spec & standards reviewer**: requested behavior, documented repo standards, architecture fit, duplication, cleanup, and workaround-shaped implementation.
21
-
22
- The main agent is the coordinator. It pins the review target, assigns both
23
- lenses to one reviewer for `simple` and `medium`, and splits them across two
24
- independent reviewers only for `high`. It verifies the strongest findings,
25
- applies only safe fixes, and returns a concise findings-first report.
26
-
27
- Full review is bounded to the settled diff, its authority, changed owners, and
28
- callers or contracts with plausible fan-out. `Full` means complete coverage of
29
- that assigned scope once; it does not mean a repository-wide audit. Do not load
30
- unrelated modules or activate optional lenses without a diff signal or mandatory
31
- Review Focus.
32
-
33
- ## When To Use
34
-
35
- Use this skill when the user asks for:
36
-
37
- - code review, PR review, commit audit, regression scan, or bug hunt
38
- - review since a branch, commit, tag, merge-base, or working tree state
39
- - review and fix of critical or high-severity high-confidence defects
40
- - framework-focused review such as `NestJS`, `Next.js`, `Flutter`, or `Dart`
41
-
42
- For every implementation profile, the spec/standards lens includes bounded
43
- cleanup for duplication, obsolete paths, workaround branches, and unjustified
44
- abstractions. High-risk work assigns that lens to its own reviewer in the same
45
- parallel final wave. A concrete evidenced simplification risk named by the
46
- user, approved source, or repo policy amplifies this lens inside the same review
47
- activation; it never creates a separate cleanup gate.
48
-
49
- Exception for approved spec execution: follow
50
- `../spec-implementer/references/review-loop.md`. Intermediate code-review
51
- checkpoints activate only their assigned Review Focus; final cleanup coverage
52
- uses the durable Review Plan and canonical Defect Ledger.
53
-
54
- ## Implementation Review Adapter
55
-
56
- When this skill is called from `$spec-implementer`:
57
-
58
- - read `../spec-implementer/references/review-loop.md` and the persisted
59
- `## Implementation Review State`
60
- - recheck the scheduled profile against the settled diff; return an
61
- underclassified profile to the executor before launching reviewers
62
- - accept the scheduled mode, session, revision, and lenses after that check
63
- - pin the target and give reviewers the owner-defined capsule
64
- - return the usable result and stable defect updates to the executor
65
- - keep cleanup findings in the spec/standards lineage and canonical Defect Ledger
66
-
67
- Do not infer a fresh review loop, choose another mode, or make the owner's
68
- terminal decision inside this Adapter.
69
-
70
- ## Progressive References
71
-
72
- Read only the references the current review needs:
73
-
74
- - Detailed bug classes: `references/bug-classes.md`
75
- - Cleanup lens method: `references/cleanup-lens.md`
76
- - Framework lenses for Next.js, NestJS, Flutter, and Dart: `references/framework-lenses.md`
77
- - Targeted recipes for recurring diff shapes: `references/targeted-recipes.md`
78
- - Contract test ledger: `../../docs/agents/contract-test-ledger.md`
79
- - Shared confidence rubric: `../../docs/agents/confidence-rubric.md`
80
-
81
- Load `references/framework-lenses.md` when the user names a framework or files/configs strongly imply one. Load `references/targeted-recipes.md` when the diff shape matches them. Load `../../docs/agents/contract-test-ledger.md` only for a material contract delta with a named failure ordinary targeted proof could miss. Load `references/bug-classes.md` for substantial reviews or broad bug hunts.
82
- Load `references/cleanup-lens.md` when the spec/standards lens is assigned. Use
83
- its bounded method by default and its amplified method only for a concrete
84
- evidenced simplification risk supplied as mandatory Review Focus.
85
-
86
- ## Coordinator Workflow
87
-
88
- ### 1. Pin The Review Target
89
-
90
- Identify exactly what is being reviewed.
91
-
92
- - If the user gave a fixed point, use it directly: branch, commit SHA, tag, `main`, `HEAD~5`, etc.
93
- - If they did not, infer from context:
94
- - current uncommitted work: `git diff` plus staged diff if relevant
95
- - branch review: `git diff <base>...HEAD`
96
- - commit review: `git show <commit>`
97
- - If there is no safe inference, ask one short question: "Review against which branch or commit?"
98
-
99
- Capture:
100
-
101
- - `git status --short`
102
- - diff stat
103
- - the exact diff command used
104
- - commit list when reviewing a branch range
105
- - changed files and nearest related tests/docs
106
-
107
- Use three-dot diff for branch/base reviews: `git diff <fixed-point>...HEAD`.
108
-
109
- ### 2. Discover Spec And Standards
110
-
111
- Do this before reviewer tracks so both tracks receive bounded inputs.
112
-
113
- Spec sources, in priority order:
114
-
115
- 1. Issue or PR references in commit messages, branch names, PR metadata, or user prompt.
116
- 2. A spec/PRD/plan path supplied by the user.
117
- 3. Matching files under `docs/`, `specs/`, `.scratch/`, or local issue folders.
118
- 4. If none exists, continue and mark the spec axis as "no spec found" instead of inventing requirements.
119
-
120
- Standards sources:
121
-
122
- - `AGENTS.md`, `CLAUDE.md`, `CONTRIBUTING.md`
123
- - `CONTEXT.md`, context maps, domain docs, ADRs
124
- - `STYLE.md`, `STANDARDS.md`, style guides, review checklists
125
- - `.editorconfig`, ESLint, Biome, Prettier, TypeScript, analyzer, or framework configs
126
- - relevant test files and existing examples in the touched area
127
-
128
- Machine-enforced config matters as context, but do not spend review findings on issues a required formatter/linter would already catch unless the tool is absent or failing.
129
-
130
- ### 3. Activate Lenses
131
-
132
- Before deep review, decide which lenses apply.
133
-
134
- - Always activate the general correctness and spec/standards lenses.
135
- - Always apply bounded cleanup inside spec/standards; amplify it only for a
136
- concrete evidenced Review Focus, never from size or risk labels alone.
137
- - Add framework lenses when explicit or strongly implied by files/configs.
138
- - Add targeted recipes when the diff shape matches them.
139
- - If the user, plan, or implementation spec provides `Review Focus`, treat each listed lens, targeted recipe, invariant, and risk as mandatory. Do not replace it with a generic review; report any focus item that cannot be verified as a verification gap.
140
- - If inference is uncertain, say so and continue with the best-supported review instead of pretending certainty.
141
-
142
- Mandatory delta lenses:
143
-
144
- - **Contract Delta Review:** If the diff expands a shared contract, enum/status, schema, DTO, permission, capability, token, or scope, search old consumers and report `changed contract -> searched call sites -> risky fallback/default branches -> tests or fix`.
145
- - **Backend Trust Boundary Review:** If the diff adds a credential, OAuth scope, permission, role-sensitive write, or external capability, verify the backend-side guard. UI controls, query/state flags, and caller intent are not authorization.
146
-
147
- Common framework signals:
148
-
149
- - **Next.js**: `next.config.*`, `app/**`, `pages/**`, route handlers, server actions, `use server`, `use client`, revalidation APIs.
150
- - **NestJS**: `@nestjs/*`, controllers, providers, modules, guards, pipes, interceptors, DTOs, `nest-cli.json`.
151
- - **Flutter**: `pubspec.yaml`, `lib/**`, widgets, navigation, bloc/provider/riverpod/notifiers.
152
- - **Dart**: `.dart`, `Future`, `Stream`, isolates, generated serializers, null-safety constructs.
153
-
154
- ### 4. Run Profile-Selected Reviewers
155
-
156
- For `simple`, run one `reviewer_fast`; for `medium`, run one
157
- `reviewer_standard`. Assign that child both correctness and spec/standards
158
- lenses in one bounded Full review. For `high`, run two `reviewer_deep` children
159
- in parallel with one disjoint lens each. Invoking `$code-review` authorizes
160
- these reviewers. Preserve every fulfilled launch handle if a parallel peer
161
- fails, then close all launched children. Tell every reviewer not to edit or
162
- revert unrelated work.
163
-
164
- Begin every reviewer launch brief with an exact `Assigned role: <role>` line,
165
- using `reviewer_fast`, `reviewer_standard`, or `reviewer_deep`. Keep that role
166
- explicit for Full and Closure launches even when the profile or existing
167
- lineage already implies it; a generic child prompt is not evidence that the
168
- profile-selected reviewer topology was executed.
169
-
170
- A reviewer is launched only after the child-launch tool returns a non-empty
171
- handle for that brief. Wait only on returned handles, then close every launched
172
- child. Never treat an intended role, an empty wait, coordinator analysis, or a
173
- final-answer claim as reviewer execution. If launch is unavailable or returns
174
- no handle, report the independent review gate as unavailable and do not
175
- self-review or claim that the reviewer completed.
176
-
177
- A recorded reviewer role or lineage from an earlier Full review identifies
178
- which role must own Closure; it is never a live child handle. Every Closure
179
- activation must launch a new child in that same role for the current session,
180
- capture the new non-empty handle, and wait only on that handle.
181
-
182
- For spec-driven checkpoints, obey the track assignment in the persisted Review
183
- Plan instead of automatically launching both default tracks.
184
-
185
- Medium is the normal review profile. API, persistence, multiple files, or a
186
- shared-looking name do not select `high` unless evidence proves both a material
187
- failure consequence and an uncertainty amplifier.
188
-
189
- When already inside an assigned reviewer child, execute its assigned lens set
190
- inline and return it to root; do not spawn a grandchild.
191
- If the user forbids delegation, report the independent review gate as waived or
192
- unavailable according to the parent workflow; root must not self-review inline.
193
-
194
- #### Correctness Reviewer Brief
195
-
196
- Include the exact diff command or commit under review, changed file list, commit list, active references, any `Review Focus`, and an instruction to read surrounding execution paths.
197
-
198
- > Review runtime correctness adversarially. Hunt concrete bugs in control flow, state, async/concurrency, security, auth, contracts, schemas, caching, persistence, UI/API integration, and framework-specific behavior. If a Review Focus is provided, explicitly test the named risks first, such as duplicate side effects, retry/idempotency, ordering, source-of-truth ownership, partial failure, DTO/schema drift, or false user-facing state. For each finding, provide file/line, trigger path, impact, why guards do not prevent it, severity, and confidence. Do not report style nits or generic "needs tests" comments.
199
- > For bugfixes, check claim boundaries: what changed tests prove, what they do not prove, and whether sibling execution paths can still violate the claimed invariant.
200
- > When Contract Delta Review applies, follow new values through old consumers and default/fallback branches before trusting local tests. When Backend Trust Boundary Review applies, verify the server-side authorization predicate at the write/callback boundary.
201
-
202
- #### Spec & Standards Reviewer Brief
203
-
204
- Include the exact diff command or commit under review, spec source path/content or "no spec found", standards source list, changed file list, commit list, any `Review Focus`, and an instruction to cite the spec or standard behind each finding.
205
-
206
- > Review the change against the requested work and repo standards. Check missing requirements, partial behavior, scope creep, undocumented contract changes, architecture drift, duplicate source-of-truth logic, dead/legacy branches, and workaround-shaped implementation. If a Review Focus is provided, explicitly verify each named ownership, scope, validation, and invariant risk against the spec. Cite the spec or standard when available. If there is no spec, skip requirement claims and focus on documented standards and architecture evidence.
207
- > Apply `references/cleanup-lens.md`: use bounded cleanup by default, or the amplified method when a concrete evidenced simplification risk is mandatory Review Focus. Keep cleanup findings in this review's normal Defect Ledger. Do not create a cleanup-only verdict or separate pass.
208
-
209
- ### 5. Aggregate And Verify
210
-
211
- The coordinator must not blindly relay reviewer output.
212
-
213
- 1. Deduplicate findings across tracks.
214
- 2. Re-read the relevant code for the strongest findings.
215
- 3. Drop findings that lack a concrete trigger path.
216
- 4. Reclassify severity/confidence using `../../docs/agents/confidence-rubric.md` if evidence does not support the label.
217
- 5. For real contract defects that pass the ledger gate, identify the missing or inadequate invariant when TDD/spec evidence is available.
218
- 6. Confirm every mandatory `Review Focus` item and mandatory delta lens was reviewed; if not, report the unverified item as a verification gap.
219
- 7. Confirm that mandatory behavior evidence rejects the concrete violating
220
- sequence attempted by the reviewer. Evidence that bypasses its claimed
221
- production boundary invalidates that Full coverage.
222
- 8. Decide whether auto-fix is allowed.
223
- 9. Run the narrowest meaningful verification after any fix.
224
-
225
- Keep the two axes visible in your own notes, but present the final report by severity unless the user explicitly asked for side-by-side Standards/Spec output.
226
-
227
- After repairs, verify medium or low findings through the coordinator's direct
228
- failure-path check plus affected validation. Launch Closure only for the
229
- severity, protected-contract, or invalidated-coverage triggers owned by
230
- `review-protocol.md`. For a scheduled Closure, send the bounded capsule to the
231
- selected reviewer lineage and return its defect updates.
232
-
233
- ## Evidence Standard
234
-
235
- A valid finding explains:
236
-
237
- - what breaks
238
- - why it breaks
239
- - the input, sequence, role, tenant, environment, or timing that triggers it
240
- - where the defect lives
241
- - why existing guards do not prevent it
242
- - severity and confidence
243
-
244
- Do not file:
245
-
246
- - style nits disguised as correctness issues
247
- - speculative races without a shared-state path
248
- - generic "needs tests" comments without a concrete regression risk
249
- - architecture discomfort without wrong ownership, duplication, leakage, or a regression path
250
- - performance comments without a hot path or failure mode
251
-
252
- ## Auto-Fix Policy
253
-
254
- Automatically fix only when all are true:
255
-
256
- - severity is critical or high
257
- - confidence is high under `../../docs/agents/confidence-rubric.md`
258
- - root cause is clear
259
- - correct fix is narrow and low-risk
260
- - fix matches local project patterns
261
- - verification is available, or the edit is obviously safe and syntax-checkable
262
-
263
- When auto-fixing:
264
-
265
- - patch only the bug
266
- - add/update behavior tests when regression risk is meaningful and the codebase supports it; update a ledger only when its gate passes
267
- - rerun relevant verification
268
- - never revert unrelated user changes
269
-
270
- Do not auto-fix ambiguous semantics, product decisions, broad refactors, or low-confidence concerns. Report them with evidence.
271
-
272
- ## Output Contract
273
-
274
- For review-only tasks:
275
-
276
- 1. Findings first, ordered by severity.
277
- 2. File and line references for each finding.
278
- 3. Trigger path, impact, severity, confidence, and evidence.
279
- 4. Open questions, assumptions, or verification gaps.
280
- 5. Short summary only after findings.
281
-
282
- For review-and-fix tasks:
283
-
284
- 1. State which critical or high-severity high-confidence issues were fixed.
285
- 2. Report remaining findings that were not fixed.
286
- 3. Give one short verification note and any blocked checks.
287
- 4. Keep the closing summary user-facing and outcome-based.
288
-
289
- If there are no findings, say so clearly and mention residual test or verification gaps.
290
-
291
- For spec-driven review checkpoints or final review gates, include a compact review handoff that can feed the executor's Final Risk Handoff: reviewed target, Review Focus status, high/critical findings fixed or remaining, skipped checks, and residual verification gaps. Keep findings first.
292
-
293
- If inline review comments are requested, emit one `::code-comment{...}` directive per actionable finding.
294
-
295
- ## Tooling Defaults
296
-
297
- - Use `rg`/`rg --files` for code search.
298
- - Prefer parallel reads for status, diff, changed files, related modules, tests, repo docs, and standards.
299
- - Use official docs or Context7 only when a finding depends on version-sensitive framework/library behavior.
300
- - Run the narrowest meaningful tests first, then required lint/build/analyzer checks for touched areas.
301
- - Keep raw command output out of the final response unless the user asks for it.
302
-
303
- ## Decision Defaults
304
-
305
- - If the user says "review", default to findings-first review.
306
- - If the user says "review and fix", auto-fix only critical or high-severity high-confidence issues that satisfy the auto-fix policy.
307
- - If the user provides a framework focus, explicitly use it.
308
- - If no focus is provided, infer active lenses from the diff and say when they materially affected findings.
309
- - Ask clarification only when the correct review target or fix would otherwise be risky or ambiguous.
6
+ # Review
7
+
8
+ Review a settled change through two independent lenses:
9
+
10
+ - **Spec:** requirement fidelity, missing or partial behavior, incorrect
11
+ implementation, and scope drift.
12
+ - **Standards:** correctness, failure paths, repository policy, maintainability,
13
+ legacy residue, duplicate ownership, and unnecessary machinery.
14
+
15
+ Review is inspection-only. Never edit, repair, stage, commit, push, or open a
16
+ PR. Substantial behavior or contract changes require Review; obvious local
17
+ docs, copy, formatting, mechanical config, or corrections may use direct proof.
18
+
19
+ ## Boundaries
20
+
21
+ - **Authorized outcome:** the request, issue, or Parent PRD plus existing
22
+ invariants and mandatory repository rules.
23
+ - **Impact cone:** callers, data, runtime, and proof surfaces that may be
24
+ inspected to establish effects. Inspection creates no authority.
25
+ - **Repair scope:** the smallest change that restores a proven obligation,
26
+ invariant, or mandatory rule without widening the authorized outcome.
27
+
28
+ A reviewer verdict is evidence, not authority. Mark a finding `BLOCKER` only
29
+ when evidence links a concrete defect or required-proof gap to an authorized
30
+ obligation, existing invariant, or mandatory rule. Include its source,
31
+ file/line, trigger, impact, and evidence. Treat unsupported preferences,
32
+ hypothetical hardening, and architecture or cleanup beyond the outcome as
33
+ `OBSERVATION`.
34
+
35
+ For Standards, read [standards-smells.md](references/standards-smells.md).
36
+ Fowler smells remain non-blocking observations unless separate evidence meets the blocker
37
+ threshold. Repository policy wins, and tooling-enforced rules need no duplicate
38
+ finding. If optional machinery causes a defect, prefer deleting it; extend it
39
+ only when the authorized outcome requires it.
40
+
41
+ ## Process
42
+
43
+ 1. Pin an existing baseline and settled target, then verify the comparison is
44
+ valid and non-empty. Supply the exact diff, changed files, proof, authority
45
+ source, and repository policy. Missing required authority or proof is a gap;
46
+ never silently skip a lens.
47
+ 2. Select the mode:
48
+ - initial Review launches one fresh `spec_reviewer` and one fresh `standards_reviewer`
49
+ in parallel;
50
+ - targeted Review runs only the affected lens: Spec-only repair for
51
+ requirement coverage or behavior, Standards-only repair for correctness,
52
+ invariants, architecture, or mandatory rules, and both lenses when the
53
+ repair affects both or cannot be isolated.
54
+ Give targeted reviewers the previous reviewed revision, repair delta,
55
+ repaired blockers, direct impact cone, and affected proof. Untouched
56
+ previously approved scope retains approval.
57
+ 3. Give both reviewers the same target, diff, proof, authority, policy, and
58
+ boundaries above. Keep each brief under 400 words and begin it with
59
+ `Assigned role: <role>`.
60
+ - `spec_reviewer` checks only missing, partial, incorrect, or extra behavior
61
+ against cited authority.
62
+ - `standards_reviewer` checks correctness, failure paths, policy, cleanup,
63
+ zero legacy, duplicate ownership, unnecessary machinery, and the smell
64
+ baseline without inventing product obligations.
65
+ 4. Use fresh children without history fork (`fork_context=false` on V1;
66
+ `fork_turns="none"` on V2). Capture every non-empty child identity and wait
67
+ for those same children. An empty bounded wait means wait again, not failure.
68
+ Silence is not a hang; request status only without interruption. Never
69
+ interrupt, terminate, close, or replace an active reviewer. Missing identity,
70
+ terminal failure, interruption, or incomplete final wait blocks Review;
71
+ root never substitutes self-review.
72
+ 5. Reconcile each finding from its evidence, preserving its lens and defect
73
+ identity. Reclassify unsupported `BLOCK` labels as observations; an
74
+ `APPROVE` label cannot erase a verified blocker. Consolidate all verified
75
+ blockers into one repair batch for the active Implement or Tickets
76
+ Orchestrator owner. Their original authority covers repairs inside Repair
77
+ scope without another confirmation. For review-only requests, report and
78
+ stop without invoking Implement.
79
+ 6. Return `Spec` and `Standards` results separately, then consolidated blockers,
80
+ observations, and proof gaps. Approve only after every selected independent
81
+ reviewer completed and no verified blocker or required-proof gap remains.
82
+ Review may approve with observations. A complete Review repeats only when a
83
+ repair cannot be isolated.
84
+
85
+ Each reviewer returns findings first, labelled `BLOCKER` or `OBSERVATION`, then
86
+ `APPROVE` or `BLOCK` with a concrete reason. Do not create durable review state,
87
+ maps, ledgers, aliases, adapters, or compatibility routes.
@@ -1,4 +1,6 @@
1
1
  interface:
2
- display_name: "Code Review"
3
- short_description: "Profile-routed correctness and standards review"
4
- default_prompt: "Use $code-review for one final profile-selected wave; high risk uses two parallel reviewers and the spec/standards lens includes bounded cleanup."
2
+ display_name: "Review"
3
+ short_description: "Independent requirements and standards review"
4
+ default_prompt: "Use $code-review to inspect the settled change with independent Spec and Standards lenses. Report findings without implementing or repairing them."
5
+ policy:
6
+ allow_implicit_invocation: true
@@ -0,0 +1,83 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "code-review",
4
+ "cases": [
5
+ {
6
+ "id": "authorized-regression-blocks",
7
+ "prompt": "A settled diff breaks an existing caller invariant required by the authorized behavior; the failing public-seam regression proves the trigger and impact.",
8
+ "expected": [
9
+ "report a blocker linked to the existing invariant",
10
+ "include the trigger, impact, and regression evidence",
11
+ "return BLOCK"
12
+ ],
13
+ "forbidden": [
14
+ "downgrade the proven regression to an observation",
15
+ "require a broader solution than restoring the invariant"
16
+ ]
17
+ },
18
+ {
19
+ "id": "optional-architecture-is-observation",
20
+ "prompt": "The authorized outcome and all existing invariants pass, but the reviewer prefers a more robust architecture for hypothetical future load.",
21
+ "expected": [
22
+ "report the architecture or robustness idea only as an observation",
23
+ "approve when no authorized obligation, invariant, mandatory rule, or proof is missing"
24
+ ],
25
+ "forbidden": [
26
+ "turn hypothetical robustness into a blocker",
27
+ "treat the reviewer label or impact cone as authority"
28
+ ]
29
+ },
30
+ {
31
+ "id": "optional-machinery-prefers-removal",
32
+ "prompt": "The diff adds an optional mediator that is not needed by the authorized outcome, and the mediator contains a defect. Removing it preserves all required behavior and invariants.",
33
+ "expected": [
34
+ "prefer removing the optional mediator",
35
+ "keep repair scope within the authorized outcome"
36
+ ],
37
+ "forbidden": [
38
+ "require hardening or extending the optional mediator",
39
+ "expand the authorized outcome to preserve unnecessary machinery"
40
+ ]
41
+ },
42
+ {
43
+ "id": "unsupported-block-reconciles-to-approval",
44
+ "prompt": "A completed independent reviewer returns BLOCK, but coordinator verification proves every reported item is an optional preference with no authorized obligation, invariant, mandatory rule, or required-proof gap.",
45
+ "expected": [
46
+ "reclassify the unsupported findings as observations",
47
+ "return APPROVE from the completed independent evidence",
48
+ "perform no repair"
49
+ ],
50
+ "forbidden": [
51
+ "repeat review solely because of the reviewer label",
52
+ "expand the authorized outcome to satisfy an unsupported verdict"
53
+ ]
54
+ },
55
+ {
56
+ "id": "two-independent-read-only-lenses",
57
+ "prompt": "Run Review on one settled substantial target and report its findings.",
58
+ "expected": [
59
+ "launch fresh Spec and Standards reviewers in parallel",
60
+ "keep their findings separated by lens",
61
+ "keep Review read-only",
62
+ "return observations and consolidated blockers to the caller"
63
+ ],
64
+ "forbidden": [
65
+ "repair the target",
66
+ "let either reviewer invent new authority"
67
+ ]
68
+ },
69
+ {
70
+ "id": "targeted-review-selects-affected-lens",
71
+ "prompt": "A reviewed repair delta only restores one missing authorized requirement and changes no correctness, invariant, architecture, or repository-rule surface.",
72
+ "expected": [
73
+ "run only a fresh Spec reviewer",
74
+ "limit review to the repair delta and direct impact cone",
75
+ "preserve untouched approval"
76
+ ],
77
+ "forbidden": [
78
+ "rerun Standards without an affected Standards surface",
79
+ "repeat the complete review"
80
+ ]
81
+ }
82
+ ]
83
+ }
@@ -0,0 +1,41 @@
1
+ # Standards Smell Baseline
2
+
3
+ Two rules bind this baseline:
4
+
5
+ - **The repo overrides.** A documented repo standard always wins; where it
6
+ endorses something the baseline would flag, suppress the smell.
7
+ - **Always a judgement call.** Each smell is a labelled heuristic, never a hard
8
+ violation — and, like any standard here, skip anything tooling already
9
+ enforces.
10
+
11
+ Each smell reads *what it is* -> *how to fix*:
12
+
13
+ - **Mysterious Name** — a function, variable, or type whose name doesn't reveal
14
+ what it does or holds. -> rename it; if no honest name comes, the design's
15
+ murky.
16
+ - **Duplicated Code** — the same logic shape appears in more than one hunk or
17
+ file in the change. -> extract the shared shape, call it from both.
18
+ - **Feature Envy** — a method that reaches into another object's data more than
19
+ its own. -> move the method onto the data it envies.
20
+ - **Data Clumps** — the same few fields or params keep travelling together (a
21
+ type wanting to be born). -> bundle them into one type, pass that.
22
+ - **Primitive Obsession** — a primitive or string standing in for a domain
23
+ concept that deserves its own type. -> give the concept its own small type.
24
+ - **Repeated Switches** — the same `switch`/`if`-cascade on the same type recurs
25
+ across the change. -> replace with polymorphism, or one map both sites share.
26
+ - **Shotgun Surgery** — one logical change forces scattered edits across many
27
+ files in the diff. -> gather what changes together into one module.
28
+ - **Divergent Change** — one file or module is edited for several unrelated
29
+ reasons. -> split so each module changes for one reason.
30
+ - **Speculative Generality** — abstraction, parameters, or hooks added for needs
31
+ the spec doesn't have. -> delete it; inline back until a real need shows.
32
+ - **Message Chains** — long `a.b().c().d()` navigation the caller shouldn't
33
+ depend on. -> hide the walk behind one method on the first object.
34
+ - **Middle Man** — a class or function that mostly just delegates onward. -> cut
35
+ it, call the real target direct.
36
+ - **Refused Bequest** — a subclass or implementer that ignores or overrides most
37
+ of what it inherits. -> drop the inheritance, use composition.
38
+
39
+ A smell alone is a non-blocking observation. Escalate only when evidence
40
+ demonstrates a concrete defect or proof gap causally linked to an explicit
41
+ obligation, existing invariant, or mandatory repository rule.