codex-orchestrator 2.0.11 → 2.0.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (262) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/README.md +25 -51
  3. package/dist/src/index.d.ts +2 -8
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/index.js +1 -4
  6. package/dist/src/index.js.map +1 -1
  7. package/dist/src/v2/acceptance-proof.d.ts +46 -31
  8. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  9. package/dist/src/v2/acceptance-proof.js +157 -195
  10. package/dist/src/v2/acceptance-proof.js.map +1 -1
  11. package/dist/src/v2/active-attempt.d.ts +94 -0
  12. package/dist/src/v2/active-attempt.d.ts.map +1 -0
  13. package/dist/src/v2/active-attempt.js +200 -0
  14. package/dist/src/v2/active-attempt.js.map +1 -0
  15. package/dist/src/v2/adapters/command.d.ts +6 -0
  16. package/dist/src/v2/adapters/command.d.ts.map +1 -1
  17. package/dist/src/v2/adapters/command.js +43 -2
  18. package/dist/src/v2/adapters/command.js.map +1 -1
  19. package/dist/src/v2/candidate.d.ts +15 -31
  20. package/dist/src/v2/candidate.d.ts.map +1 -1
  21. package/dist/src/v2/candidate.js +7 -29
  22. package/dist/src/v2/candidate.js.map +1 -1
  23. package/dist/src/v2/checked-change.d.ts +3 -2
  24. package/dist/src/v2/checked-change.d.ts.map +1 -1
  25. package/dist/src/v2/checked-change.js +4 -3
  26. package/dist/src/v2/checked-change.js.map +1 -1
  27. package/dist/src/v2/cli-contract.d.ts +1 -1
  28. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  29. package/dist/src/v2/cli-contract.js +4 -6
  30. package/dist/src/v2/cli-contract.js.map +1 -1
  31. package/dist/src/v2/cli.d.ts +8 -0
  32. package/dist/src/v2/cli.d.ts.map +1 -1
  33. package/dist/src/v2/cli.js +13 -0
  34. package/dist/src/v2/cli.js.map +1 -1
  35. package/dist/src/v2/code-review-report.d.ts +10 -18
  36. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  37. package/dist/src/v2/code-review-report.js +63 -60
  38. package/dist/src/v2/code-review-report.js.map +1 -1
  39. package/dist/src/v2/codex-process.d.ts +6 -2
  40. package/dist/src/v2/codex-process.d.ts.map +1 -1
  41. package/dist/src/v2/codex-process.js +25 -9
  42. package/dist/src/v2/codex-process.js.map +1 -1
  43. package/dist/src/v2/config.d.ts +0 -2
  44. package/dist/src/v2/config.d.ts.map +1 -1
  45. package/dist/src/v2/config.js +3 -6
  46. package/dist/src/v2/config.js.map +1 -1
  47. package/dist/src/v2/contained-report-operation.d.ts +41 -196
  48. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  49. package/dist/src/v2/contained-report-operation.js +139 -466
  50. package/dist/src/v2/contained-report-operation.js.map +1 -1
  51. package/dist/src/v2/containment.d.ts +1 -0
  52. package/dist/src/v2/containment.d.ts.map +1 -1
  53. package/dist/src/v2/containment.js +12 -2
  54. package/dist/src/v2/containment.js.map +1 -1
  55. package/dist/src/v2/delivery-authority.d.ts +26 -0
  56. package/dist/src/v2/delivery-authority.d.ts.map +1 -0
  57. package/dist/src/v2/delivery-authority.js +44 -0
  58. package/dist/src/v2/delivery-authority.js.map +1 -0
  59. package/dist/src/v2/direct-delivery.d.ts +16 -36
  60. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  61. package/dist/src/v2/direct-delivery.js +135 -122
  62. package/dist/src/v2/direct-delivery.js.map +1 -1
  63. package/dist/src/v2/immutable-workflow-publisher.d.ts.map +1 -1
  64. package/dist/src/v2/immutable-workflow-publisher.js +3 -1
  65. package/dist/src/v2/immutable-workflow-publisher.js.map +1 -1
  66. package/dist/src/v2/implementation-report.d.ts +3 -1
  67. package/dist/src/v2/implementation-report.d.ts.map +1 -1
  68. package/dist/src/v2/implementation-report.js +28 -10
  69. package/dist/src/v2/implementation-report.js.map +1 -1
  70. package/dist/src/v2/implementation-reviewer.d.ts +41 -12
  71. package/dist/src/v2/implementation-reviewer.d.ts.map +1 -1
  72. package/dist/src/v2/implementation-reviewer.js +114 -42
  73. package/dist/src/v2/implementation-reviewer.js.map +1 -1
  74. package/dist/src/v2/pending-effect-settlement.d.ts +44 -0
  75. package/dist/src/v2/pending-effect-settlement.d.ts.map +1 -0
  76. package/dist/src/v2/pending-effect-settlement.js +69 -0
  77. package/dist/src/v2/pending-effect-settlement.js.map +1 -0
  78. package/dist/src/v2/process-identity.d.ts +45 -0
  79. package/dist/src/v2/process-identity.d.ts.map +1 -0
  80. package/dist/src/v2/process-identity.js +118 -0
  81. package/dist/src/v2/process-identity.js.map +1 -0
  82. package/dist/src/v2/proof-report.d.ts +2 -1
  83. package/dist/src/v2/proof-report.d.ts.map +1 -1
  84. package/dist/src/v2/proof-report.js +10 -4
  85. package/dist/src/v2/proof-report.js.map +1 -1
  86. package/dist/src/v2/review-feedback-coordinator.d.ts +1 -1
  87. package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -1
  88. package/dist/src/v2/review-feedback-coordinator.js +1 -1
  89. package/dist/src/v2/review-feedback-coordinator.js.map +1 -1
  90. package/dist/src/v2/review-feedback.d.ts +14 -20
  91. package/dist/src/v2/review-feedback.d.ts.map +1 -1
  92. package/dist/src/v2/review-feedback.js +45 -87
  93. package/dist/src/v2/review-feedback.js.map +1 -1
  94. package/dist/src/v2/run-issue.d.ts +129 -88
  95. package/dist/src/v2/run-issue.d.ts.map +1 -1
  96. package/dist/src/v2/run-issue.js +1965 -2381
  97. package/dist/src/v2/run-issue.js.map +1 -1
  98. package/dist/src/v2/run-state-projections.d.ts +84 -0
  99. package/dist/src/v2/run-state-projections.d.ts.map +1 -0
  100. package/dist/src/v2/run-state-projections.js +142 -0
  101. package/dist/src/v2/run-state-projections.js.map +1 -0
  102. package/dist/src/v2/run-store.d.ts +99 -81
  103. package/dist/src/v2/run-store.d.ts.map +1 -1
  104. package/dist/src/v2/run-store.js +245 -542
  105. package/dist/src/v2/run-store.js.map +1 -1
  106. package/dist/src/v2/runtime-assets.d.ts +3 -0
  107. package/dist/src/v2/runtime-assets.d.ts.map +1 -1
  108. package/dist/src/v2/runtime-assets.js +104 -0
  109. package/dist/src/v2/runtime-assets.js.map +1 -1
  110. package/dist/src/v2/runtime.d.ts +56 -44
  111. package/dist/src/v2/runtime.d.ts.map +1 -1
  112. package/dist/src/v2/runtime.js +383 -503
  113. package/dist/src/v2/runtime.js.map +1 -1
  114. package/dist/src/v2/setup.js +0 -2
  115. package/dist/src/v2/setup.js.map +1 -1
  116. package/dist/src/v2/validation-progression.d.ts +70 -0
  117. package/dist/src/v2/validation-progression.d.ts.map +1 -0
  118. package/dist/src/v2/validation-progression.js +247 -0
  119. package/dist/src/v2/validation-progression.js.map +1 -0
  120. package/dist/src/v2/workflow-assets.d.ts +9 -3
  121. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  122. package/dist/src/v2/workflow-assets.js +256 -43
  123. package/dist/src/v2/workflow-assets.js.map +1 -1
  124. package/internal-workflow/docs/agents/bug-workflow-routing.md +9 -7
  125. package/internal-workflow/docs/agents/coding-skill-routing.md +170 -120
  126. package/internal-workflow/docs/agents/tool-usage.md +23 -12
  127. package/internal-workflow/manifest.json +1 -1
  128. package/internal-workflow/operations/code-review/SKILL.md +34 -15
  129. package/internal-workflow/operations/implementation/SKILL.md +21 -16
  130. package/internal-workflow/profiles/implementer.toml +9 -0
  131. package/internal-workflow/profiles/review_coordinator.toml +9 -0
  132. package/internal-workflow/profiles/spec_reviewer.toml +9 -0
  133. package/internal-workflow/profiles/standards_reviewer.toml +9 -0
  134. package/internal-workflow/schemas/code-review-v1.json +1 -1
  135. package/internal-workflow/schemas/implementation-report-v1.json +1 -1
  136. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  137. package/internal-workflow/skills/bug-root-cause-explainer/SKILL.md +114 -0
  138. package/internal-workflow/skills/bug-root-cause-explainer/agents/openai.yaml +7 -0
  139. package/internal-workflow/skills/bug-root-cause-explainer/evals/evals.json +18 -0
  140. package/internal-workflow/skills/code-review/SKILL.md +84 -306
  141. package/internal-workflow/skills/code-review/agents/openai.yaml +5 -3
  142. package/internal-workflow/skills/code-review/evals/evals.json +83 -0
  143. package/internal-workflow/skills/code-review/references/standards-smells.md +41 -0
  144. package/internal-workflow/skills/diagnosing-bugs/SKILL.md +69 -32
  145. package/internal-workflow/skills/diagnosing-bugs/agents/openai.yaml +2 -2
  146. package/internal-workflow/skills/diagnosing-bugs/evals/evals.json +63 -0
  147. package/internal-workflow/skills/grilling/SKILL.md +51 -0
  148. package/internal-workflow/skills/grilling/agents/openai.yaml +6 -0
  149. package/internal-workflow/skills/grilling/evals/evals.json +47 -0
  150. package/internal-workflow/skills/implement/SKILL.md +135 -0
  151. package/internal-workflow/skills/implement/agents/openai.yaml +6 -0
  152. package/internal-workflow/skills/implement/evals/evals.json +150 -0
  153. package/internal-workflow/skills/plan/SKILL.md +59 -0
  154. package/internal-workflow/skills/plan/agents/openai.yaml +6 -0
  155. package/internal-workflow/skills/plan/evals/evals.json +36 -0
  156. package/internal-workflow/skills/prototype/LOGIC.md +130 -0
  157. package/internal-workflow/skills/prototype/SKILL.md +69 -0
  158. package/internal-workflow/skills/prototype/UI.md +157 -0
  159. package/internal-workflow/skills/prototype/agents/openai.yaml +6 -0
  160. package/internal-workflow/skills/prototype/evals/evals.json +67 -0
  161. package/internal-workflow/skills/research/SKILL.md +110 -0
  162. package/internal-workflow/skills/research/agents/openai.yaml +6 -0
  163. package/internal-workflow/skills/research/evals/evals.json +49 -0
  164. package/internal-workflow/skills/tdd/SKILL.md +72 -67
  165. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  166. package/internal-workflow/skills/tdd/evals/evals.json +12 -0
  167. package/internal-workflow/skills/tdd/mocking.md +48 -1
  168. package/internal-workflow/skills/tdd/refactoring.md +3 -3
  169. package/internal-workflow/skills/tickets-orchestrator/SKILL.md +199 -0
  170. package/internal-workflow/skills/tickets-orchestrator/agents/openai.yaml +6 -0
  171. package/internal-workflow/skills/tickets-orchestrator/evals/evals.json +126 -0
  172. package/internal-workflow/skills/tickets-orchestrator/references/delegate-integrate.md +83 -0
  173. package/internal-workflow/skills/tickets-orchestrator/references/finish-delivery.md +69 -0
  174. package/internal-workflow/skills/tickets-orchestrator/references/stop-completion.md +63 -0
  175. package/internal-workflow/skills/to-spec/SKILL.md +133 -0
  176. package/internal-workflow/skills/to-spec/agents/openai.yaml +6 -0
  177. package/internal-workflow/skills/to-spec/evals/evals.json +24 -0
  178. package/internal-workflow/skills/to-tickets/SKILL.md +189 -0
  179. package/internal-workflow/skills/to-tickets/agents/openai.yaml +6 -0
  180. package/internal-workflow/skills/to-tickets/evals/evals.json +79 -0
  181. package/internal-workflow/skills/to-tickets/references/publishing-details.md +117 -0
  182. package/package.json +1 -1
  183. package/dist/src/v2/proof-store.d.ts +0 -54
  184. package/dist/src/v2/proof-store.d.ts.map +0 -1
  185. package/dist/src/v2/proof-store.js +0 -301
  186. package/dist/src/v2/proof-store.js.map +0 -1
  187. package/dist/src/v2/route-continuations.d.ts +0 -32
  188. package/dist/src/v2/route-continuations.d.ts.map +0 -1
  189. package/dist/src/v2/route-continuations.js +0 -2
  190. package/dist/src/v2/route-continuations.js.map +0 -1
  191. package/dist/src/v2/route-coordinator.d.ts +0 -72
  192. package/dist/src/v2/route-coordinator.d.ts.map +0 -1
  193. package/dist/src/v2/route-coordinator.js +0 -275
  194. package/dist/src/v2/route-coordinator.js.map +0 -1
  195. package/dist/src/v2/route-decision.d.ts +0 -120
  196. package/dist/src/v2/route-decision.d.ts.map +0 -1
  197. package/dist/src/v2/route-decision.js +0 -380
  198. package/dist/src/v2/route-decision.js.map +0 -1
  199. package/dist/src/v2/spec-coordinator.d.ts +0 -73
  200. package/dist/src/v2/spec-coordinator.d.ts.map +0 -1
  201. package/dist/src/v2/spec-coordinator.js +0 -126
  202. package/dist/src/v2/spec-coordinator.js.map +0 -1
  203. package/dist/src/v2/spec-delivery.d.ts +0 -112
  204. package/dist/src/v2/spec-delivery.d.ts.map +0 -1
  205. package/dist/src/v2/spec-delivery.js +0 -336
  206. package/dist/src/v2/spec-delivery.js.map +0 -1
  207. package/dist/src/v2/triage-route.d.ts +0 -68
  208. package/dist/src/v2/triage-route.d.ts.map +0 -1
  209. package/dist/src/v2/triage-route.js +0 -223
  210. package/dist/src/v2/triage-route.js.map +0 -1
  211. package/dist/src/v2/waiting-human-coordinator.d.ts +0 -49
  212. package/dist/src/v2/waiting-human-coordinator.d.ts.map +0 -1
  213. package/dist/src/v2/waiting-human-coordinator.js +0 -509
  214. package/dist/src/v2/waiting-human-coordinator.js.map +0 -1
  215. package/dist/src/v2/waiting-human.d.ts +0 -143
  216. package/dist/src/v2/waiting-human.d.ts.map +0 -1
  217. package/dist/src/v2/waiting-human.js +0 -408
  218. package/dist/src/v2/waiting-human.js.map +0 -1
  219. package/internal-workflow/docs/agents/contract-test-ledger.md +0 -71
  220. package/internal-workflow/docs/agents/review-gates.md +0 -42
  221. package/internal-workflow/docs/agents/review-protocol.md +0 -98
  222. package/internal-workflow/evals/coding-skill-evals.json +0 -373
  223. package/internal-workflow/operations/ambiguity-review/SKILL.md +0 -5
  224. package/internal-workflow/operations/qualification-repair/SKILL.md +0 -17
  225. package/internal-workflow/operations/spec-author/SKILL.md +0 -12
  226. package/internal-workflow/operations/spec-review/SKILL.md +0 -12
  227. package/internal-workflow/operations/triage/SKILL.md +0 -12
  228. package/internal-workflow/profiles/analyst_deep.toml +0 -9
  229. package/internal-workflow/profiles/implementer_standard.toml +0 -9
  230. package/internal-workflow/profiles/proof_agent.toml +0 -8
  231. package/internal-workflow/profiles/reviewer_deep.toml +0 -9
  232. package/internal-workflow/profiles/reviewer_standard.toml +0 -9
  233. package/internal-workflow/schemas/ambiguity-review-v1.json +0 -1
  234. package/internal-workflow/schemas/spec-author-v1.json +0 -1
  235. package/internal-workflow/schemas/spec-review-v1.json +0 -30
  236. package/internal-workflow/schemas/triage-route-v1.json +0 -1
  237. package/internal-workflow/skills/agent-auto/SKILL.md +0 -19
  238. package/internal-workflow/skills/agent-auto/agents/openai.yaml +0 -6
  239. package/internal-workflow/skills/code-debugger/SKILL.md +0 -122
  240. package/internal-workflow/skills/code-debugger/agents/openai.yaml +0 -7
  241. package/internal-workflow/skills/code-review/references/bug-classes.md +0 -56
  242. package/internal-workflow/skills/code-review/references/cleanup-lens.md +0 -52
  243. package/internal-workflow/skills/code-review/references/framework-lenses.md +0 -34
  244. package/internal-workflow/skills/code-review/references/targeted-recipes.md +0 -49
  245. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +0 -107
  246. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +0 -6
  247. package/internal-workflow/skills/implementation-spec-maker/references/source-modes.md +0 -32
  248. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +0 -146
  249. package/internal-workflow/skills/implementation-spec-review/SKILL.md +0 -131
  250. package/internal-workflow/skills/implementation-spec-review/agents/openai.yaml +0 -6
  251. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +0 -78
  252. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +0 -121
  253. package/internal-workflow/skills/small-task-implementer/SKILL.md +0 -112
  254. package/internal-workflow/skills/small-task-implementer/agents/openai.yaml +0 -6
  255. package/internal-workflow/skills/spec-implementer/SKILL.md +0 -133
  256. package/internal-workflow/skills/spec-implementer/agents/openai.yaml +0 -6
  257. package/internal-workflow/skills/spec-implementer/evals/evals.json +0 -30
  258. package/internal-workflow/skills/spec-implementer/references/review-loop.md +0 -100
  259. package/internal-workflow/skills/triage/AGENT-BRIEF.md +0 -192
  260. package/internal-workflow/skills/triage/OUT-OF-SCOPE.md +0 -101
  261. package/internal-workflow/skills/triage/SKILL.md +0 -134
  262. package/internal-workflow/skills/triage/agents/openai.yaml +0 -6
@@ -1,15 +1,15 @@
1
1
  ---
2
2
  name: diagnosing-bugs
3
- description: Debug hard, flaky, unclear, or performance bugs through reproduce, minimize, hypothesize, instrument, fix, and regression-test. Trigger for nondeterministic failures, unclear breakage, or performance regressions.
3
+ description: Diagnose hard, flaky, unclear, or performance bugs through reproduction, minimisation, ranked hypotheses, and targeted instrumentation. Stop after proving the root cause and hand the proven signal to Implement; do not fix the bug.
4
4
  ---
5
5
 
6
6
  # Diagnosing Bugs
7
7
 
8
- A discipline for hard bugs. Skip phases only when explicitly justified.
8
+ A diagnosis-only discipline for hard bugs. Skip phases only when explicitly justified. End with a proven root cause or an honest blocker.
9
9
 
10
- Routing precedence: use `$CODEX_ORCHESTRATOR_WORKFLOW_ROOT/docs/agents/bug-workflow-routing.md`. This skill owns the feedback loop; after the loop proves the bug, return to the original intent: diagnosis-only output or implementation through `code-debugger`.
10
+ Routing precedence: use [`../../docs/agents/bug-workflow-routing.md`](../../docs/agents/bug-workflow-routing.md). This skill owns reproduction and root-cause proof. It does not own the regression test or fix, post-fix verification, production mutation, or Git lifecycle. Once the cause is proven, hand the reproducible signal and evidence to `implement` and stop.
11
11
 
12
- Use `$CODEX_ORCHESTRATOR_WORKFLOW_ROOT/docs/agents/confidence-rubric.md` when deciding whether a hypothesis, root cause, or fix is high-confidence enough to act on. Low-confidence concerns are questions or verification gaps, not proven causes.
12
+ Use [`../../docs/agents/confidence-rubric.md`](../../docs/agents/confidence-rubric.md) when deciding whether a hypothesis or root cause is high-confidence enough to report or hand off. Low-confidence concerns are questions or verification gaps, not proven causes.
13
13
 
14
14
  When exploring the codebase, read `CONTEXT.md` (if it exists) to get a clear mental model of the relevant modules, and check ADRs in the area you're touching.
15
15
 
@@ -21,18 +21,18 @@ Spend disproportionate effort here. **Be aggressive. Be creative. Refuse to give
21
21
 
22
22
  ### Ways to construct one — try them in roughly this order
23
23
 
24
- 1. **Failing test** at whatever seam reaches the bug — unit, integration, e2e.
24
+ 1. **Existing failing test** at whatever seam reaches the bug — unit, integration, e2e. Do not turn the repro into a new durable regression test in this skill.
25
25
  2. **Curl / HTTP script** against a running dev server.
26
26
  3. **CLI invocation** with a fixture input, diffing stdout against a known-good snapshot.
27
27
  4. **Headless browser script** (Playwright / Puppeteer) — drives the UI, asserts on DOM/console/network.
28
28
  5. **Replay a captured trace.** Save a real network request / payload / event log to disk; replay it through the code path in isolation.
29
29
  6. **Throwaway harness.** Spin up a minimal subset of the system (one service, mocked deps) that exercises the bug code path with a single function call.
30
30
  7. **Property / fuzz loop.** If the bug is "sometimes wrong output", run 1000 random inputs and look for the failure mode.
31
- 8. **Bisection harness.** If the bug appeared between two known states (commit, dataset, version), automate "boot at state X, check, repeat" so you can `git bisect run` it.
31
+ 8. **Bisection harness.** If the bug appeared between two known states (snapshot, dataset, version), automate "boot at state X, check, repeat" over read-only states supplied by the active driver. Diagnosis performs no checkout or other Git action.
32
32
  9. **Differential loop.** Run the same input through old-version vs new-version (or two configs) and diff outputs.
33
33
  10. **HITL bash script.** Last resort. If a human must click, drive _them_ with `scripts/hitl-loop.template.sh` so the loop is still structured. Captured output feeds back to you.
34
34
 
35
- Build the right feedback loop, and the bug is 90% fixed.
35
+ Build the right feedback loop, and most of the diagnostic uncertainty is gone.
36
36
 
37
37
  ### Tighten the loop
38
38
 
@@ -50,13 +50,13 @@ The goal is not a clean repro but a **higher reproduction rate**. Loop the trigg
50
50
 
51
51
  ### When you genuinely cannot build a loop
52
52
 
53
- Stop and say so explicitly. List what you tried. Ask the user for: (a) access to whatever environment reproduces it, (b) a captured artifact (HAR file, log dump, core dump, screen recording with timestamps), or (c) permission to add temporary production instrumentation. Do **not** proceed to hypothesise without a loop.
53
+ Stop and say so explicitly. List what you tried. Ask the user for: (a) access to whatever environment reproduces it, (b) a captured artifact (HAR file, log dump, core dump, screen recording with timestamps), or (c) an authorized Implement change that adds the missing diagnostic seam. Do **not** proceed to hypothesise without a loop, and do not edit production source or runtime state from this skill.
54
54
 
55
55
  ### Completion criterion — a tight loop that goes red
56
56
 
57
57
  Phase 1 is done when the loop is **tight** and **red-capable**: you can name **one command** — a script path, a test invocation, a curl — that you have **already run at least once** (paste the invocation and its output), and that is:
58
58
 
59
- - [ ] **Red-capable** — it drives the actual bug code path and asserts the **user's exact symptom**, so it can go red on this bug and green once fixed. Not "runs without erroring" — it must be able to _catch this specific bug_.
59
+ - [ ] **Red-capable** — it drives the actual bug code path and asserts the **user's exact symptom**, producing a signal that `implement` can later use for red/green proof. Not "runs without erroring" — it must be able to _catch this specific bug_.
60
60
  - [ ] **Deterministic** — same verdict every run (flaky bugs: a pinned, high reproduction rate, per above).
61
61
  - [ ] **Fast** — seconds, not minutes.
62
62
  - [ ] **Agent-runnable** — you can run it unattended; a human in the loop only via `scripts/hitl-loop.template.sh`.
@@ -69,15 +69,15 @@ Run the loop. Watch it go red — the bug appears.
69
69
 
70
70
  Confirm:
71
71
 
72
- - [ ] The loop produces the failure mode the **user** described — not a different failure that happens to be nearby. Wrong bug = wrong fix.
72
+ - [ ] The loop produces the failure mode the **user** described — not a different failure that happens to be nearby. Wrong bug = wrong diagnosis.
73
73
  - [ ] The failure is reproducible across multiple runs (or, for non-deterministic bugs, reproducible at a high enough rate to debug against).
74
- - [ ] You have captured the exact symptom (error message, wrong output, slow timing) so later phases can verify the fix actually addresses it.
74
+ - [ ] You have captured the exact symptom (error message, wrong output, slow timing) so later phases can prove what produces it.
75
75
 
76
76
  ### Minimise
77
77
 
78
78
  Once it's red, shrink the repro to the **smallest scenario that still goes red**. Cut inputs, callers, config, data, and steps **one at a time**, re-running the loop after each cut — keep only what's load-bearing for the failure.
79
79
 
80
- Why bother: a minimal repro shrinks the hypothesis space in Phase 3 (fewer moving parts left to suspect) and becomes the clean regression test in Phase 5.
80
+ Why bother: a minimal repro shrinks the hypothesis space in Phase 3 and gives `implement` a precise signal to preserve while fixing the bug.
81
81
 
82
82
  Done when **every remaining element is load-bearing** — removing any one of them makes the loop go green.
83
83
 
@@ -102,37 +102,74 @@ Each probe must map to a specific prediction from Phase 3. **Change one variable
102
102
  Tool preference:
103
103
 
104
104
  1. **Debugger / REPL inspection** if the env supports it. One breakpoint beats ten logs.
105
- 2. **Targeted logs** at the boundaries that distinguish hypotheses.
105
+ 2. **Targeted logs** in a throwaway harness or an already-authorized diagnostic surface at the boundaries that distinguish hypotheses.
106
106
  3. Never "log everything and grep".
107
107
 
108
108
  **Tag every debug log** with a unique prefix, e.g. `[DEBUG-a4f2]`. Cleanup at the end becomes a single grep. Untagged logs survive; tagged logs die.
109
109
 
110
- **Perf branch.** For performance regressions, logs are usually wrong. Instead: establish a baseline measurement (timing harness, `performance.now()`, profiler, query plan), then bisect. Measure first, fix second.
110
+ If distinguishing the hypotheses requires changing production source or runtime state, stop with that missing instrumentation seam and hand it to `implement`. Diagnosis may describe the smallest observation point, but it does not apply that mutation.
111
111
 
112
- ## Phase 5 — Fix + regression test
112
+ **Perf branch.** For performance regressions, logs are usually wrong. Instead: establish a baseline measurement (timing harness, `performance.now()`, profiler, query plan), then bisect. Measure first, isolate the cause second.
113
113
 
114
- Write the regression test **before the fix** — but only if there is a **correct seam** for it.
114
+ ## Phase 5 — Prove the root cause and hand off
115
115
 
116
- A correct seam is one where the test exercises the **real bug pattern** as it occurs at the call site. If the only available seam is too shallow (single-caller test when the bug needs multiple callers, unit test that can't replicate the chain that triggered the bug), a regression test there gives false confidence.
116
+ Prove the winning hypothesis against the tight loop before reporting it as the root cause:
117
117
 
118
- **If no correct seam exists, that itself is the finding.** Note it. The codebase architecture is preventing the bug from being locked down. Flag this for the next phase.
118
+ 1. Name the exact trigger, owning code path, and mechanism that produces the observed symptom.
119
+ 2. Show the probe result that confirmed the winning prediction and the result that ruled out the strongest alternative.
120
+ 3. Re-run the unchanged repro after removing or disabling the probe. It must still produce the original symptom; a probe-induced failure is not proof.
121
+ 4. Classify confidence with the confidence rubric. If an assumption still changes the conclusion, report the remaining verification gap instead of claiming a proven cause.
122
+ 5. Remove all `[DEBUG-...]` instrumentation and unrelated throwaway
123
+ prototypes. When the red signal depends on a throwaway harness, keep the
124
+ minimal repro harness and fixture available with the handoff until Implement
125
+ confirms the same RED and creates the durable regression test. This narrow
126
+ executable handoff artifact is not production code or a Git action; after
127
+ confirmation, the active root removes it unless the user authorized it to
128
+ remain.
119
129
 
120
- If a correct seam exists:
130
+ Then stop. Do not author a regression test, apply or recommend a speculative patch, run post-fix checks, commit, push, open a PR, or update a tracker. Hand `implement`:
121
131
 
122
- 1. Turn the minimised repro into a failing test at that seam.
123
- 2. Watch it fail.
124
- 3. Apply the fix.
125
- 4. Watch it pass.
126
- 5. Re-run the Phase 1 feedback loop against the original (un-minimised) scenario.
132
+ - the exact unchanged reproduction command and captured pre-fix failing output;
133
+ - the minimal load-bearing fixture, inputs, state, and steps;
134
+ - a signal digest covering the command, fixture bytes, and expected failing verdict;
135
+ - the proven trigger, owner, mechanism, and supporting probe evidence;
136
+ - the strongest alternative ruled out and how it was falsified;
137
+ - any missing test seam, environment dependency, or residual uncertainty that constrains implementation proof;
138
+ - post-mortem observations about what could have prevented the bug, clearly separated from the proven cause and without starting architecture or production work.
127
139
 
128
- ## Phase 6 — Cleanup + post-mortem
140
+ The handoff is executable evidence, not narration. Implement must rerun the
141
+ unchanged signal before editing and get the same diagnosed failure. A missing
142
+ fixture, changed digest, different symptom, or unexpectedly green result
143
+ returns to Diagnosis or blocks the fix; Implement must not silently substitute
144
+ a nearby test.
129
145
 
130
- Required before declaring done:
146
+ If the root cause is not proven, do not hand off a guess as implementation input. Return the best red-capable signal, tested hypotheses, and the concrete evidence or access still required.
131
147
 
132
- - [ ] Original repro no longer reproduces (re-run the Phase 1 loop)
133
- - [ ] Regression test passes (or absence of seam is documented)
134
- - [ ] All `[DEBUG-...]` instrumentation removed (`grep` the prefix)
135
- - [ ] Throwaway prototypes deleted (or moved to a clearly-marked debug location)
136
- - [ ] The hypothesis that turned out correct is stated in the commit / PR message — so the next debugger learns
148
+ ## Output Contract
137
149
 
138
- **Then ask: what would have prevented this bug?** If the answer involves architectural change (no good test seam, tangled callers, hidden coupling) hand off to the `$improve-codebase-architecture` skill with the specifics. Make the recommendation **after** the fix is in, not before — you have more information now than when you started.
150
+ Use this shape when reporting back:
151
+
152
+ ```md
153
+ ## Reproduction
154
+
155
+ - Command: `<exact command already run>`
156
+ - Signal: <captured symptom and repeatability>
157
+ - Minimal scenario: <load-bearing inputs, state, and steps>
158
+
159
+ ## Proven root cause
160
+
161
+ - Trigger: <what activates the bug>
162
+ - Owner: `<path>` — `<function or method>`
163
+ - Mechanism: <how the owner produces the symptom>
164
+ - Evidence: <confirming probe and falsified alternative>
165
+ - Confidence: <high, medium, or low, with any remaining gap>
166
+
167
+ ## Implement handoff
168
+
169
+ - Preserve this signal: <exact unchanged reproduction command, captured pre-fix failing output, minimal load-bearing fixture, signal digest, and expected failing verdict>
170
+ - Constraints: <missing test seam, environment dependency, or residual uncertainty>
171
+ - Post-mortem observations: <prevention or architecture evidence for later authorized work>
172
+ - Next owner: `implement`
173
+ ```
174
+
175
+ When the cause is not proven, replace `Proven root cause` and `Implement handoff` with `Blocked diagnosis`; list tested hypotheses and the exact missing evidence. Do not route implementation from an unproven diagnosis.
@@ -1,6 +1,6 @@
1
1
  interface:
2
2
  display_name: "Diagnosing Bugs"
3
- short_description: "Run a disciplined hard-bug investigation loop"
4
- default_prompt: "Use $diagnosing-bugs to reproduce, minimise, diagnose, and verify this difficult bug."
3
+ short_description: "Prove hard-bug root causes before implementation"
4
+ default_prompt: "Use $diagnosing-bugs to reproduce and minimise this difficult bug, prove its root cause with targeted evidence, hand the proven signal to $implement, and stop before any regression test or fix."
5
5
  policy:
6
6
  allow_implicit_invocation: true
@@ -0,0 +1,63 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "diagnosing-bugs",
4
+ "cases": [
5
+ {
6
+ "id": "proven-cause-handoff",
7
+ "prompt": "A deterministic two-second CLI repro is red. Diagnose the hard bug and the strongest competing hypothesis, but do not fix it.",
8
+ "expected": [
9
+ "run and minimise the red-capable repro",
10
+ "test ranked falsifiable hypotheses with targeted probes",
11
+ "prove the trigger, owner, mechanism, and falsified alternative",
12
+ "hand the unchanged failing signal to Implement and stop"
13
+ ],
14
+ "forbidden": [
15
+ "author a regression test",
16
+ "apply the fix",
17
+ "perform a Git action"
18
+ ]
19
+ },
20
+ {
21
+ "id": "blocked-without-red-capable-loop",
22
+ "prompt": "The reported production-only failure has no local repro, captured artifact, or safe observation surface.",
23
+ "expected": [
24
+ "list the feedback loops attempted",
25
+ "request the exact environment access, captured artifact, or authorized diagnostic seam still required",
26
+ "return an honest blocked diagnosis without guessing"
27
+ ],
28
+ "forbidden": [
29
+ "declare a speculative root cause",
30
+ "edit production instrumentation",
31
+ "route an unproven guess to Implement"
32
+ ]
33
+ },
34
+ {
35
+ "id": "missing-seam-postmortem-handoff",
36
+ "prompt": "The root cause is proven, but the real caller has no durable test seam and the diagnosis reveals one prevention opportunity.",
37
+ "expected": [
38
+ "report the missing test seam as an implementation constraint",
39
+ "separate the post-mortem observation from the proven cause",
40
+ "hand both observations and the failing signal to Implement"
41
+ ],
42
+ "forbidden": [
43
+ "create a test-only production abstraction",
44
+ "start an architecture refactor",
45
+ "commit diagnostic work"
46
+ ]
47
+ },
48
+ {
49
+ "id": "red-capable-implement-handoff",
50
+ "prompt": "The cause is proven and Implement is about to receive the diagnosis.",
51
+ "expected": [
52
+ "hand off the exact unchanged command, captured failing output, minimal load-bearing fixture, and signal digest",
53
+ "retain a required throwaway repro harness until Implement confirms the same RED",
54
+ "require Implement to reproduce the same diagnosed failure before editing",
55
+ "block or return to Diagnosis if the signal changed or is unexpectedly green"
56
+ ],
57
+ "forbidden": [
58
+ "replace the signal with a nearby passing test",
59
+ "start the fix inside Diagnosis"
60
+ ]
61
+ }
62
+ ]
63
+ }
@@ -0,0 +1,51 @@
1
+ ---
2
+ name: grilling
3
+ description: Grill the user about a plan, decision, or idea through dependency-aware frontier rounds. Use when the user wants to stress-test their thinking or uses a grill trigger phrase.
4
+ ---
5
+
6
+ # Grilling
7
+
8
+ Interview the user until you reach a shared understanding. Model the subject as
9
+ a **design tree**: each material decision branches into the decisions that
10
+ depend on it.
11
+
12
+ Root owns every user-facing question, recommendation, wait for feedback, and
13
+ accepted decision. Never delegate or impersonate the dialogue. A bounded
14
+ `explorer` may gather read-only evidence when a fact depends on a cross-module
15
+ path; it returns evidence to root and makes no product decision.
16
+
17
+ ## Frontier rounds
18
+
19
+ The **frontier** is every unresolved material decision whose prerequisites are
20
+ settled. Ask the whole frontier in one numbered round. For each question:
21
+
22
+ - state the decision and enough context to answer it;
23
+ - give two or three concrete options when multiple paths are plausible;
24
+ - mark one recommended answer and explain the trade-off briefly;
25
+ - say plainly when evidence leaves only one valid option.
26
+
27
+ Then wait for the user's answers. Each answer reshapes the design tree: record
28
+ the settled decision, recompute the frontier, and ask the next numbered round.
29
+ A question whose answer depends on a decision still open in the current round
30
+ belongs to a later round. Do not ask it early or guess its prerequisite.
31
+
32
+ Finding facts is the agent's job, never the user's. Look up facts in the
33
+ environment, codebase, current docs, or authorized external sources. If a fact
34
+ lookup is still running, treat it as an unsettled prerequisite: defer only its
35
+ downstream questions and ask the rest of the current frontier now. The
36
+ decisions remain the user's; put every material decision to them and use their
37
+ answers to resolve only minor follow-ons.
38
+
39
+ The interview is complete only when the frontier is empty: every material
40
+ branch is settled or explicitly ruled out and no decision is silently assumed.
41
+ Summarize the resulting shared understanding and ask the user to confirm it.
42
+ Do not act on the result until the user explicitly confirms the shared
43
+ understanding. A product plan produced here is input to the active driver; this
44
+ skill does not implement it.
45
+
46
+ ## Mutation boundary
47
+
48
+ Grilling is write-free. It may read repository domain language to phrase
49
+ questions consistently, but it never creates or edits `CONTEXT.md`,
50
+ `CONTEXT-MAP.md`, or ADRs. Domain-document mutation belongs only to
51
+ `$domain-modeling`, when that discipline is separately active.
@@ -0,0 +1,6 @@
1
+ interface:
2
+ display_name: "Grilling"
3
+ short_description: "Stress-test decisions in frontier rounds"
4
+ default_prompt: "Use $grilling on root to map the dependency tree, ask each currently unblocked frontier in a numbered round, and stop for explicit shared-understanding confirmation without writing domain docs."
5
+ policy:
6
+ allow_implicit_invocation: true
@@ -0,0 +1,47 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "grilling",
4
+ "cases": [
5
+ {
6
+ "id": "independent-frontier-round",
7
+ "prompt": "Three material decisions are independent and answerable now. Start the grilling session.",
8
+ "expected": [
9
+ "map all three as the current frontier and ask them in one numbered round",
10
+ "give a recommendation for each decision and wait for the user's answers",
11
+ "keep every user-facing question on root"
12
+ ],
13
+ "forbidden": [
14
+ "ask only one independent question",
15
+ "write domain documents",
16
+ "act on the resulting plan"
17
+ ]
18
+ },
19
+ {
20
+ "id": "dependent-frontier-deferral",
21
+ "prompt": "Decision B depends on decision A, while decision C is independent and a repository fact for D is still being looked up.",
22
+ "expected": [
23
+ "ask A and C in the current numbered frontier round",
24
+ "defer B until A settles and defer only D while its fact prerequisite is unresolved",
25
+ "recompute the frontier after answers and finish only when it is empty"
26
+ ],
27
+ "forbidden": [
28
+ "ask B in the same round as A",
29
+ "ask the user to find the repository fact",
30
+ "claim completion before explicit shared-understanding confirmation"
31
+ ]
32
+ },
33
+ {
34
+ "id": "implicit-grilling-remains-write-free",
35
+ "prompt": "Without naming a wrapper, grill this terminology choice and tell me whether it deserves an ADR.",
36
+ "expected": [
37
+ "run a write-free frontier interview",
38
+ "discuss terminology and ADR suitability without changing files",
39
+ "leave domain-document mutation to a separately active domain-modeling discipline"
40
+ ],
41
+ "forbidden": [
42
+ "edit CONTEXT.md",
43
+ "create an ADR"
44
+ ]
45
+ }
46
+ ]
47
+ }
@@ -0,0 +1,135 @@
1
+ ---
2
+ name: implement
3
+ description: Implement a clear authorized feature, fix, obvious local edit, or executable ticket through observable proof and applicable Review. This is the single coding execution owner.
4
+ ---
5
+
6
+ # Implement
7
+
8
+ Implement the authorized outcome. Do not reopen settled product decisions or
9
+ create a replacement plan, spec, ticket, workflow state, or compatibility path.
10
+
11
+ ## Kernel
12
+
13
+ - **Authority:** perform only the requested outcome and authorized delivery
14
+ actions. Tracker publication is not implementation authority. Normal direct
15
+ and single-ticket Implement authority includes one scoped local commit after
16
+ proof and applicable Review unless user or repository policy explicitly
17
+ forbids or reserves Git. Push and PR are separate and never implicit.
18
+ - **Preservation:** read repository policy and current status before edits.
19
+ Preserve unrelated work and user-owned runtimes. Stop on overlapping dirty
20
+ scope or a decision that changes behavior, ownership, or boundaries.
21
+ - **Proof:** prove the final observable outcome through the real caller seam.
22
+ Authority-defined proof cannot be replaced by weaker tests or a completion
23
+ claim.
24
+
25
+ ## Context Ownership
26
+
27
+ Direct non-ticket work remains in the current root context.
28
+
29
+ For one executable ticket, root launches exactly one fresh `implementer` child.
30
+ The assignment must include:
31
+
32
+ - the complete ticket;
33
+ - the complete Parent PRD;
34
+ - applicable repository policy;
35
+ - bounded write scope and exclusions;
36
+ - required proof and explicit delivery/Git boundaries.
37
+
38
+ The assignment begins with `Assigned role: implementer` so the requested role,
39
+ fresh child identity, and completed wait can be verified independently.
40
+
41
+ The worker performs no Git actions and returns changed files, observable proof,
42
+ skipped checks, risks, decision deltas, overlap, and blockers. Root verifies a
43
+ non-empty fresh child identity, waits for that same child to complete, checks
44
+ the assignment inputs, and integrates only isolated output. A single ticket
45
+ never activates the graph coordinator.
46
+
47
+ ## Execution
48
+
49
+ 1. Confirm authority, repository policy, current status, owner code, caller
50
+ seam, and the smallest credible proof. Record pre-existing dirty paths; an
51
+ overlap with owned scope blocks edits, while disjoint dirty paths remain
52
+ untouched and unstaged.
53
+ 2. Use `$tdd` where possible. Otherwise establish direct pre-change evidence
54
+ when useful and apply the narrowest observable proof after the edit.
55
+ When a diagnosis handoff exists, rerun the unchanged reproduction command
56
+ before editing and require the same diagnosed failure, fixture digest, and
57
+ symptom. Then make the first durable regression test at the natural public
58
+ seam fail for the diagnosed symptom before applying the fix. If the handoff
59
+ is stale, incomplete, unexpectedly green, or reproduces another symptom,
60
+ stop and return the evidence gap to Diagnosis instead of substituting a
61
+ nearby signal.
62
+ 3. Implement the smallest complete change in the existing owner. Delete
63
+ superseded paths in scope; do not add aliases, wrappers, adapters, fallback
64
+ routes, or speculative layers.
65
+ 4. Run targeted proof and the smallest affected integration check.
66
+ During substantial work, run the relevant typecheck or single test file as
67
+ the seam settles. Run a full suite once at the end only when repository
68
+ policy, risk, or the changed shared contract makes it proportionate.
69
+ 5. Classify the settled result by content:
70
+ - substantial: behavior or contract beyond an obvious local edit, including
71
+ public API, persistence, auth/payment, concurrency/shared state, or
72
+ cross-module interaction;
73
+ - obvious local: docs, copy, formatting, mechanical config, or an obvious
74
+ local correction with direct proof.
75
+ A public returned-record shape change remains a substantial contract change
76
+ even when its implementation is one line or arrives through one ticket.
77
+ 6. For substantial work, invoke `$code-review` on the settled proof. The first
78
+ invocation reviews the complete authorized result with one fresh Spec
79
+ reviewer and one fresh Standards reviewer in parallel. Obvious local work
80
+ may skip Review.
81
+ 7. Handle each reviewer result:
82
+ - reviewer findings are evidence, not implementation authority. Before any
83
+ repair, independently verify **Authority** (the existing authorized
84
+ obligation being restored), **Trigger** (the concrete failing path or
85
+ unmet requirement), **Impact** (the observable defect or proof gap in the
86
+ authorized result), and **Minimal repair** (the least change that restores
87
+ that obligation). State one short evidence-backed sentence connecting
88
+ those four facts before editing;
89
+ - classify the result without collapsing failure paths: a new obligation,
90
+ changed authorized outcome, ownership or architecture decision, or repair
91
+ of a neighboring problem is a Decision Delta and stops for authority. A
92
+ missing Authority, Trigger, or Impact makes the finding unverified and it
93
+ receives no repair; classify it as an observation unless required proof
94
+ cannot establish the authorized result, in which case the proof gap blocks
95
+ completion without authorizing code. Only a finding with all four facts
96
+ established is a verified blocker eligible for repair. Apply the same gate
97
+ to UI, backend, persistence, concurrency, infrastructure, and every other
98
+ change type;
99
+ - if optional machinery added by the current diff causes the problem,
100
+ remove it instead of expanding the outcome to support it;
101
+ - consolidate every verified blocker from the current review into one
102
+ repair batch. The repair may widen the investigated impact cone and touch
103
+ necessary neighboring files within explicit exclusions and preservation
104
+ boundaries, but it must not widen the authorized outcome. The active
105
+ owner applies it: root for direct work or the ticket's existing
106
+ `implementer` for single-ticket work. Do not ask for confirmation or
107
+ create a second implementer when the gate proves the repair is already
108
+ authorized.
109
+ 8. After each repair batch, rerun affected proof and invoke `$code-review` on
110
+ the new revision. Run only the affected lens: a Spec-only repair returns to
111
+ Spec, a Standards-only repair returns to Standards, and a mixed or
112
+ unisolatable repair returns to both lenses. Limit targeted Review to the
113
+ repair delta, direct impact cone, and affected proof; untouched approval
114
+ remains valid. Repeat until approval. Reviewer count is not a stop condition;
115
+ use a complete two-lens Review only when the repair cannot be isolated.
116
+ 9. After proof and every applicable review approve, create one scoped local
117
+ commit for direct and single-ticket Implement when the scope is isolatable.
118
+ Stage only owned paths and recheck that unrelated dirty paths remain
119
+ unchanged. If user or repository policy explicitly forbids or reserves Git
120
+ to another actor, return the proven uncommitted diff.
121
+
122
+ Missing proof, terminally failed or interrupted review, dirty overlap, or
123
+ unisolatable scope produces no affected staging or commit. Never push or open a
124
+ PR without separate authority.
125
+
126
+ ## Handoff
127
+
128
+ Return:
129
+
130
+ - changed and deleted files;
131
+ - observable scenarios proved;
132
+ - exact commands and results;
133
+ - skipped checks and why;
134
+ - residual risks, decision deltas, overlap, and blockers;
135
+ - Git actions performed, or explicit confirmation that none occurred.
@@ -0,0 +1,6 @@
1
+ interface:
2
+ display_name: "Implement"
3
+ short_description: "Single owner for authorized coding execution"
4
+ default_prompt: "Use $implement to deliver the authorized outcome through observable proof and applicable Review."
5
+ policy:
6
+ allow_implicit_invocation: true