codex-orchestrator 2.0.2 → 2.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +44 -426
  2. package/README.md +135 -34
  3. package/dist/src/index.d.ts +11 -1
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/index.js +5 -0
  6. package/dist/src/index.js.map +1 -1
  7. package/dist/src/v2/acceptance-proof.d.ts +3 -0
  8. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  9. package/dist/src/v2/acceptance-proof.js +2 -8
  10. package/dist/src/v2/acceptance-proof.js.map +1 -1
  11. package/dist/src/v2/adapters/gh-issue-adapter.d.ts +5 -3
  12. package/dist/src/v2/adapters/gh-issue-adapter.d.ts.map +1 -1
  13. package/dist/src/v2/adapters/gh-issue-adapter.js +67 -12
  14. package/dist/src/v2/adapters/gh-issue-adapter.js.map +1 -1
  15. package/dist/src/v2/adapters/issues.d.ts +16 -2
  16. package/dist/src/v2/adapters/issues.d.ts.map +1 -1
  17. package/dist/src/v2/adapters/issues.js +15 -5
  18. package/dist/src/v2/adapters/issues.js.map +1 -1
  19. package/dist/src/v2/adapters/mission-coordinator-lock.d.ts +1 -0
  20. package/dist/src/v2/adapters/mission-coordinator-lock.d.ts.map +1 -1
  21. package/dist/src/v2/adapters/mission-coordinator-lock.js +5 -1
  22. package/dist/src/v2/adapters/mission-coordinator-lock.js.map +1 -1
  23. package/dist/src/v2/cli-contract.d.ts +3 -3
  24. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  25. package/dist/src/v2/cli-contract.js +9 -1
  26. package/dist/src/v2/cli-contract.js.map +1 -1
  27. package/dist/src/v2/cli.d.ts +24 -0
  28. package/dist/src/v2/cli.d.ts.map +1 -0
  29. package/dist/src/v2/{candidate-cli.js → cli.js} +30 -26
  30. package/dist/src/v2/cli.js.map +1 -0
  31. package/dist/src/v2/code-review-report.d.ts +66 -0
  32. package/dist/src/v2/code-review-report.d.ts.map +1 -0
  33. package/dist/src/v2/code-review-report.js +259 -0
  34. package/dist/src/v2/code-review-report.js.map +1 -0
  35. package/dist/src/v2/codex-process.d.ts +8 -1
  36. package/dist/src/v2/codex-process.d.ts.map +1 -1
  37. package/dist/src/v2/codex-process.js +22 -0
  38. package/dist/src/v2/codex-process.js.map +1 -1
  39. package/dist/src/v2/config.d.ts +4 -3
  40. package/dist/src/v2/config.d.ts.map +1 -1
  41. package/dist/src/v2/config.js +8 -3
  42. package/dist/src/v2/config.js.map +1 -1
  43. package/dist/src/v2/contained-report-operation.d.ts +100 -0
  44. package/dist/src/v2/contained-report-operation.d.ts.map +1 -0
  45. package/dist/src/v2/contained-report-operation.js +200 -0
  46. package/dist/src/v2/contained-report-operation.js.map +1 -0
  47. package/dist/src/v2/containment.d.ts +10 -0
  48. package/dist/src/v2/containment.d.ts.map +1 -1
  49. package/dist/src/v2/containment.js +49 -1
  50. package/dist/src/v2/containment.js.map +1 -1
  51. package/dist/src/v2/direct-delivery.d.ts +96 -0
  52. package/dist/src/v2/direct-delivery.d.ts.map +1 -0
  53. package/dist/src/v2/direct-delivery.js +482 -0
  54. package/dist/src/v2/direct-delivery.js.map +1 -0
  55. package/dist/src/v2/immutable-workflow-publisher.d.ts +40 -0
  56. package/dist/src/v2/immutable-workflow-publisher.d.ts.map +1 -0
  57. package/dist/src/v2/immutable-workflow-publisher.js +218 -0
  58. package/dist/src/v2/immutable-workflow-publisher.js.map +1 -0
  59. package/dist/src/v2/implementation-reviewer.d.ts +81 -0
  60. package/dist/src/v2/implementation-reviewer.d.ts.map +1 -0
  61. package/dist/src/v2/implementation-reviewer.js +157 -0
  62. package/dist/src/v2/implementation-reviewer.js.map +1 -0
  63. package/dist/src/v2/owner-control-lock.d.ts +41 -0
  64. package/dist/src/v2/owner-control-lock.d.ts.map +1 -0
  65. package/dist/src/v2/owner-control-lock.js +174 -0
  66. package/dist/src/v2/owner-control-lock.js.map +1 -0
  67. package/dist/src/v2/proof-report.d.ts.map +1 -1
  68. package/dist/src/v2/proof-report.js +55 -29
  69. package/dist/src/v2/proof-report.js.map +1 -1
  70. package/dist/src/v2/route-continuations.d.ts +32 -0
  71. package/dist/src/v2/route-continuations.d.ts.map +1 -0
  72. package/dist/src/v2/route-continuations.js +2 -0
  73. package/dist/src/v2/route-continuations.js.map +1 -0
  74. package/dist/src/v2/route-coordinator.d.ts +77 -0
  75. package/dist/src/v2/route-coordinator.d.ts.map +1 -0
  76. package/dist/src/v2/route-coordinator.js +370 -0
  77. package/dist/src/v2/route-coordinator.js.map +1 -0
  78. package/dist/src/v2/route-decision.d.ts +129 -0
  79. package/dist/src/v2/route-decision.d.ts.map +1 -0
  80. package/dist/src/v2/route-decision.js +400 -0
  81. package/dist/src/v2/route-decision.js.map +1 -0
  82. package/dist/src/v2/run-issue.d.ts +64 -6
  83. package/dist/src/v2/run-issue.d.ts.map +1 -1
  84. package/dist/src/v2/run-issue.js +962 -92
  85. package/dist/src/v2/run-issue.js.map +1 -1
  86. package/dist/src/v2/run-store.d.ts +25 -1
  87. package/dist/src/v2/run-store.d.ts.map +1 -1
  88. package/dist/src/v2/run-store.js +129 -4
  89. package/dist/src/v2/run-store.js.map +1 -1
  90. package/dist/src/v2/runtime-assets.d.ts +15 -13
  91. package/dist/src/v2/runtime-assets.d.ts.map +1 -1
  92. package/dist/src/v2/runtime-assets.js +263 -416
  93. package/dist/src/v2/runtime-assets.js.map +1 -1
  94. package/dist/src/v2/runtime.d.ts +17 -9
  95. package/dist/src/v2/runtime.d.ts.map +1 -1
  96. package/dist/src/v2/runtime.js +564 -64
  97. package/dist/src/v2/runtime.js.map +1 -1
  98. package/dist/src/v2/setup-cli.d.ts.map +1 -1
  99. package/dist/src/v2/setup-cli.js +4 -10
  100. package/dist/src/v2/setup-cli.js.map +1 -1
  101. package/dist/src/v2/setup-runtime.d.ts.map +1 -1
  102. package/dist/src/v2/setup-runtime.js +19 -131
  103. package/dist/src/v2/setup-runtime.js.map +1 -1
  104. package/dist/src/v2/setup-store.d.ts +0 -5
  105. package/dist/src/v2/setup-store.d.ts.map +1 -1
  106. package/dist/src/v2/setup-store.js +3 -106
  107. package/dist/src/v2/setup-store.js.map +1 -1
  108. package/dist/src/v2/setup.d.ts +6 -43
  109. package/dist/src/v2/setup.d.ts.map +1 -1
  110. package/dist/src/v2/setup.js +13 -192
  111. package/dist/src/v2/setup.js.map +1 -1
  112. package/dist/src/v2/spec-coordinator.d.ts +85 -0
  113. package/dist/src/v2/spec-coordinator.d.ts.map +1 -0
  114. package/dist/src/v2/spec-coordinator.js +88 -0
  115. package/dist/src/v2/spec-coordinator.js.map +1 -0
  116. package/dist/src/v2/spec-delivery.d.ts +143 -0
  117. package/dist/src/v2/spec-delivery.d.ts.map +1 -0
  118. package/dist/src/v2/spec-delivery.js +401 -0
  119. package/dist/src/v2/spec-delivery.js.map +1 -0
  120. package/dist/src/v2/triage-route.d.ts +68 -0
  121. package/dist/src/v2/triage-route.d.ts.map +1 -0
  122. package/dist/src/v2/triage-route.js +223 -0
  123. package/dist/src/v2/triage-route.js.map +1 -0
  124. package/dist/src/v2/waiting-human-coordinator.d.ts +49 -0
  125. package/dist/src/v2/waiting-human-coordinator.d.ts.map +1 -0
  126. package/dist/src/v2/waiting-human-coordinator.js +509 -0
  127. package/dist/src/v2/waiting-human-coordinator.js.map +1 -0
  128. package/dist/src/v2/waiting-human.d.ts +143 -0
  129. package/dist/src/v2/waiting-human.d.ts.map +1 -0
  130. package/dist/src/v2/waiting-human.js +408 -0
  131. package/dist/src/v2/waiting-human.js.map +1 -0
  132. package/dist/src/v2/workflow-assets.d.ts +98 -0
  133. package/dist/src/v2/workflow-assets.d.ts.map +1 -0
  134. package/dist/src/v2/workflow-assets.js +646 -0
  135. package/dist/src/v2/workflow-assets.js.map +1 -0
  136. package/docs/deep-dive.md +275 -52
  137. package/internal-workflow/docs/agents/bug-workflow-routing.md +24 -0
  138. package/internal-workflow/docs/agents/bugfix-quality-gate.md +11 -0
  139. package/internal-workflow/docs/agents/coding-skill-routing.md +123 -0
  140. package/internal-workflow/docs/agents/confidence-rubric.md +65 -0
  141. package/internal-workflow/docs/agents/contract-test-ledger.md +60 -0
  142. package/internal-workflow/docs/agents/review-gates.md +42 -0
  143. package/internal-workflow/docs/agents/review-protocol.md +98 -0
  144. package/internal-workflow/docs/agents/tool-usage.md +88 -0
  145. package/internal-workflow/evals/coding-skill-evals.json +66 -0
  146. package/internal-workflow/manifest.json +1 -0
  147. package/internal-workflow/operations/acceptance-proof/SKILL.md +9 -0
  148. package/internal-workflow/operations/ambiguity-review/SKILL.md +5 -0
  149. package/internal-workflow/operations/code-review/SKILL.md +23 -0
  150. package/internal-workflow/operations/implementation/SKILL.md +24 -0
  151. package/internal-workflow/operations/spec-author/SKILL.md +12 -0
  152. package/internal-workflow/operations/spec-review/SKILL.md +12 -0
  153. package/internal-workflow/operations/triage/SKILL.md +12 -0
  154. package/internal-workflow/profiles/analyst_deep.toml +9 -0
  155. package/internal-workflow/profiles/implementer_standard.toml +9 -0
  156. package/internal-workflow/profiles/proof_agent.toml +8 -0
  157. package/internal-workflow/profiles/reviewer_deep.toml +9 -0
  158. package/internal-workflow/profiles/reviewer_standard.toml +9 -0
  159. package/internal-workflow/schemas/ambiguity-review-v1.json +1 -0
  160. package/internal-workflow/schemas/code-review-v1.json +1 -0
  161. package/internal-workflow/schemas/implementation-report-v1.json +1 -0
  162. package/internal-workflow/schemas/proof-report-v1.json +1 -0
  163. package/internal-workflow/schemas/spec-author-v1.json +1 -0
  164. package/internal-workflow/schemas/spec-review-v1.json +30 -0
  165. package/internal-workflow/schemas/triage-route-v1.json +1 -0
  166. package/internal-workflow/skills/acceptance-proof/agents/openai.yaml +6 -0
  167. package/{internal-skills → internal-workflow/skills}/agent-auto/SKILL.md +6 -1
  168. package/internal-workflow/skills/agent-auto/agents/openai.yaml +6 -0
  169. package/internal-workflow/skills/code-debugger/SKILL.md +122 -0
  170. package/internal-workflow/skills/code-debugger/agents/openai.yaml +7 -0
  171. package/internal-workflow/skills/code-review/SKILL.md +279 -0
  172. package/internal-workflow/skills/code-review/agents/openai.yaml +4 -0
  173. package/internal-workflow/skills/code-review/references/bug-classes.md +56 -0
  174. package/internal-workflow/skills/code-review/references/cleanup-lens.md +52 -0
  175. package/internal-workflow/skills/code-review/references/framework-lenses.md +34 -0
  176. package/internal-workflow/skills/code-review/references/targeted-recipes.md +49 -0
  177. package/internal-workflow/skills/diagnosing-bugs/SKILL.md +138 -0
  178. package/internal-workflow/skills/diagnosing-bugs/agents/openai.yaml +6 -0
  179. package/internal-workflow/skills/diagnosing-bugs/scripts/hitl-loop.template.sh +41 -0
  180. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +102 -0
  181. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +6 -0
  182. package/internal-workflow/skills/implementation-spec-maker/references/source-modes.md +31 -0
  183. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +146 -0
  184. package/internal-workflow/skills/implementation-spec-review/SKILL.md +115 -0
  185. package/internal-workflow/skills/implementation-spec-review/agents/openai.yaml +6 -0
  186. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +24 -0
  187. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +93 -0
  188. package/internal-workflow/skills/small-task-implementer/SKILL.md +104 -0
  189. package/internal-workflow/skills/small-task-implementer/agents/openai.yaml +6 -0
  190. package/internal-workflow/skills/spec-implementer/SKILL.md +126 -0
  191. package/internal-workflow/skills/spec-implementer/agents/openai.yaml +6 -0
  192. package/internal-workflow/skills/spec-implementer/evals/evals.json +30 -0
  193. package/internal-workflow/skills/spec-implementer/references/review-loop.md +94 -0
  194. package/internal-workflow/skills/tdd/SKILL.md +72 -0
  195. package/internal-workflow/skills/tdd/agents/openai.yaml +6 -0
  196. package/internal-workflow/skills/tdd/interface-design.md +31 -0
  197. package/internal-workflow/skills/tdd/mocking.md +59 -0
  198. package/internal-workflow/skills/tdd/refactoring.md +10 -0
  199. package/internal-workflow/skills/tdd/tests.md +77 -0
  200. package/internal-workflow/skills/triage/AGENT-BRIEF.md +192 -0
  201. package/internal-workflow/skills/triage/OUT-OF-SCOPE.md +101 -0
  202. package/internal-workflow/skills/triage/SKILL.md +134 -0
  203. package/internal-workflow/skills/triage/agents/openai.yaml +6 -0
  204. package/package.json +14 -8
  205. package/dist/src/v2/adapters/target-activity-fence.d.ts +0 -23
  206. package/dist/src/v2/adapters/target-activity-fence.d.ts.map +0 -1
  207. package/dist/src/v2/adapters/target-activity-fence.js +0 -249
  208. package/dist/src/v2/adapters/target-activity-fence.js.map +0 -1
  209. package/dist/src/v2/candidate-cli.d.ts +0 -22
  210. package/dist/src/v2/candidate-cli.d.ts.map +0 -1
  211. package/dist/src/v2/candidate-cli.js.map +0 -1
  212. package/dist/src/v2/legacy-cutover.d.ts +0 -52
  213. package/dist/src/v2/legacy-cutover.d.ts.map +0 -1
  214. package/dist/src/v2/legacy-cutover.js +0 -87
  215. package/dist/src/v2/legacy-cutover.js.map +0 -1
  216. /package/{internal-skills → internal-workflow/skills}/acceptance-proof/SKILL.md +0 -0
  217. /package/{internal-skills → internal-workflow/skills}/acceptance-proof/references/android.md +0 -0
  218. /package/{internal-skills → internal-workflow/skills}/acceptance-proof/references/browser.md +0 -0
  219. /package/{internal-skills → internal-workflow/skills}/acceptance-proof/references/ios.md +0 -0
  220. /package/{internal-skills → internal-workflow/skills}/acceptance-proof/tools/android-lease.mjs +0 -0
  221. /package/{internal-skills → internal-workflow/skills}/acceptance-proof/tools/ios-lease.mjs +0 -0
@@ -0,0 +1,104 @@
1
+ ---
2
+ name: small-task-implementer
3
+ description: Implement small low-risk coding tasks with narrow edits and targeted validation. Use for tiny fixes, UI/copy changes, config/build corrections, simple tests, or one-module changes that do not need plans, specs, orchestration, or heavy review.
4
+ ---
5
+
6
+ # Small Task Implementer
7
+
8
+ Use this skill for fast, bounded implementation when creating a PRD and approved ticket-delivery flow would be disproportionate.
9
+
10
+ ## Fit Gate
11
+
12
+ Proceed only when all are true:
13
+
14
+ - The requested behavior is clear or can be inferred from local code/tests without a product decision.
15
+ - The change is expected to touch one small area or a few tightly related files.
16
+ - There is a narrow validation path: targeted test, lint/typecheck, build check, UI proof, or direct command.
17
+ - The task does not require a new plan, PRD, issue breakdown, implementation spec, migration, rollout, or multi-agent orchestration.
18
+
19
+ Escalate out of the tiny-task route when the work has more than one coherent
20
+ behavior, a broad ownership boundary, material rollback/recovery risk, unclear
21
+ product intent, no credible affected validation, or genuine multi-agent/live
22
+ coordination. Statefulness alone does not require escalation into planning or
23
+ orchestration; a clear API, persistence, cache, queue, or DTO change normally
24
+ becomes direct medium implementation.
25
+
26
+ Escalation rule:
27
+
28
+ - For clear authority and one coherent outcome, escalate to direct medium root
29
+ implementation under `$tdd`, affected validation, and one final review when
30
+ `review-gates.md` applies.
31
+ - Use optional `$grilling`, then `$spec-to-tickets` and `$tickets-orchestrator`,
32
+ only for unresolved product decisions or a real approved ticket graph,
33
+ delivery dependency, or explicit orchestration request.
34
+ - For one risky behavior or technical contract, prefer one approved ticket and mark `compact spec` or `standard spec` only when the ticket plus repository evidence cannot remove execution ambiguity.
35
+ - For several tickets sharing one unresolved contract or validation path, make the contract-defining ticket block its consumers; merge tickets that cannot be specified or verified independently instead of creating a wave-level implementation spec.
36
+ - Escalate if the bug requires Bugfix Quality Gate analysis across multiple paths, states, async events, persistence, auth, cache, retries, workers, or contracts.
37
+
38
+ ## Workflow
39
+
40
+ 1. Inspect local context just enough to confirm fit.
41
+ - Read repo instructions and the smallest relevant code/test files.
42
+ - Check `git status --short` before editing.
43
+ - Preserve unrelated dirty work.
44
+
45
+ 2. Write a compact contract in the working update or internal task notes:
46
+
47
+ ```text
48
+ Behavior:
49
+ Scope boundary:
50
+ Validation:
51
+ ```
52
+
53
+ 3. Implement the smallest complete change.
54
+ - Prefer existing patterns and owner modules.
55
+ - Avoid unrelated refactors, abstractions, cleanup, and compatibility paths.
56
+ - Before adding a helper, module, layer, or seam, apply the deletion test; keep it only if it improves current locality or leverage.
57
+ - Do not add pass-through modules, one-adapter seams, or tests coupled to Implementation details; escalate if no natural public test seam exists.
58
+ - Add or update a focused test only when behavior risk justifies it and the repo has a natural seam.
59
+ - For pure copy/docs/config changes, do not invent tests; run the cheapest relevant syntax/lint/check instead.
60
+
61
+ 4. Run targeted validation.
62
+ - Use the narrowest meaningful command first.
63
+ - If validation is unavailable or too expensive, state the concrete reason and residual risk.
64
+ - Do not run full CI unless local policy or the changed surface makes it necessary.
65
+
66
+ 5. Stop and escalate if implementation reveals hidden risk.
67
+ - Examples: shared contract drift, duplicate source of truth, missing test
68
+ seam, material concurrency/recovery uncertainty, or product ambiguity.
69
+ - Leave a short explanation of what was discovered and which heavier flow should take over.
70
+
71
+ ## Output
72
+
73
+ Final response must stay compact:
74
+
75
+ ```text
76
+ Small Task Result
77
+
78
+ Changed:
79
+ - ...
80
+
81
+ Proof:
82
+ - ...
83
+
84
+ Skipped:
85
+ - none / ...
86
+
87
+ Risk:
88
+ - low / reason
89
+ ```
90
+
91
+ If escalated, use:
92
+
93
+ ```text
94
+ Escalated
95
+
96
+ Reason:
97
+ - ...
98
+
99
+ Recommended flow:
100
+ - Direct medium implementation / canonical ticket delivery
101
+
102
+ Evidence:
103
+ - ...
104
+ ```
@@ -0,0 +1,6 @@
1
+ interface:
2
+ display_name: "Small Task Implementer"
3
+ short_description: "Fast bounded low-risk code changes"
4
+ default_prompt: "Use $small-task-implementer to make this small low-risk change with targeted validation."
5
+ policy:
6
+ allow_implicit_invocation: true
@@ -0,0 +1,126 @@
1
+ ---
2
+ name: "spec-implementer"
3
+ description: "Executes approved specs continuously with honest checklist updates, proportional validation, opt-in Git checkpoints, and required review/signoff."
4
+ ---
5
+
6
+ # Spec Implementer
7
+
8
+ Execute one approved implementation spec continuously. Keep its checklist
9
+ truthful and stop only at a real authority, evidence, safety, or explicit pause
10
+ boundary. Do not redesign approved scope.
11
+
12
+ Read before editing:
13
+
14
+ - the complete approved spec and applicable repository instructions;
15
+ - `references/review-loop.md` for approved-spec review ownership;
16
+ - `../../docs/agents/contract-test-ledger.md` only when the spec contains
17
+ material contract invariants.
18
+
19
+ ## Modes
20
+
21
+ Compact and full specs use the same direct phase flow. `compact` describes
22
+ document density, not implementation size or risk. Full mode adds only the
23
+ concrete `Risk Controls`, stop conditions, validation, or coordination contract
24
+ already present in the approved spec; it does not add review or reporting
25
+ ceremony by itself.
26
+
27
+ Use multiple agents only when the spec defines perfectly disjoint write scopes
28
+ and one integrator. Otherwise execute single-agent.
29
+
30
+ ## Preflight
31
+
32
+ 1. Confirm `status`, `spec_mode`, `implementation_size`, `review_profile`,
33
+ repository count, authority, scope, and exclusions.
34
+ 2. Confirm first-phase targets, required services/env/data/fixtures, commands,
35
+ observable proof, protected paths, and rejected approaches.
36
+ 3. Stop if execution needs invented paths, symbols, contracts, commands, or
37
+ product decisions; if reality differs only in a bounded technical detail,
38
+ resolve it from repository evidence and record the adjustment.
39
+ 4. Keep an intermediate review checkpoint only when the spec explicitly names
40
+ a stable high-risk slice that later work will not invalidate. Move every
41
+ ordinary or unstable checkpoint to final review.
42
+ 5. Do not create `## Implementation Review State` yet.
43
+
44
+ ## Implementation
45
+
46
+ For each phase:
47
+
48
+ 1. Re-read its scope and preconditions.
49
+ 2. Implement narrow vertical behavior slices through `$tdd` when applicable.
50
+ 3. Update reached checklist and Contract Test Ledger items at natural
51
+ checkpoints; never save all updates for the end.
52
+ 4. Run the phase's targeted exit proof.
53
+ 5. Record a short `Blocked:` note for any item that cannot complete.
54
+ 6. Continue immediately when the exit proof passes and no stop condition
55
+ applies.
56
+
57
+ Do not add cleanup, comments, helpers, abstractions, retries, flags, fallbacks,
58
+ or compatibility paths outside the spec. A small implementation adjustment is
59
+ allowed only when it preserves approved behavior and is supported by current
60
+ repository evidence.
61
+
62
+ ## Git Checkpoints
63
+
64
+ Default to no commits. `$spec-implementer` does not authorize Git writes.
65
+
66
+ Use per-slice commits only when explicitly authorized and they materially help
67
+ a disjoint multi-agent handoff, planned cross-session pause, or approved
68
+ rollback boundary. Require a passed slice proof, stage only owned paths, follow
69
+ `$commit`, and never push without separate authority. File/slice count and risk
70
+ profile alone do not justify checkpoints.
71
+
72
+ ## Review And Validation
73
+
74
+ Run targeted behavior tests and the smallest affected integration checks first.
75
+ Use a full repository suite only when the spec or repository policy requires
76
+ it, broad contract fan-out cannot be isolated, or the task is genuinely high.
77
+
78
+ Run final `$code-review` only when the spec, repository policy, or
79
+ `../../docs/agents/review-gates.md` applies. Ordinary medium work gets one
80
+ `reviewer_standard` Full review on the settled diff. High gets two disjoint
81
+ `reviewer_deep` lenses. Cleanup stays inside spec/standards review.
82
+
83
+ Immediately before the first reviewer launch, create only the minimal
84
+ `## Implementation Review State` required by `references/review-loop.md`.
85
+ Reconcile a recorded live session before replacement after interruption.
86
+
87
+ Repair compatible findings once and rerun affected validation. Coordinator
88
+ verification closes ordinary medium/low behavior-preserving repairs. Use
89
+ Closure only for critical/high, protected trust/data/concurrency/shared-contract
90
+ impact, or invalidated mandatory coverage. Do not restart broad Full review
91
+ unless the repair actually invalidated its coverage.
92
+
93
+ ## Stop Conditions
94
+
95
+ Stop and report the exact blocker when:
96
+
97
+ - a precondition, source contract, required proof, or protected path differs
98
+ materially from the approved spec;
99
+ - exact execution requires a new product, scope, ownership, or risky trade-off
100
+ decision;
101
+ - required validation or reviewer is unavailable and no approved substitute
102
+ exists;
103
+ - multi-agent scopes overlap or integration ownership is missing;
104
+ - an explicit user pause or halt condition is reached.
105
+
106
+ Do not stop merely because the work is broad, review took time, or a medium/low
107
+ finding required one repair.
108
+
109
+ ## Completion
110
+
111
+ Complete only when reached checklist items are reconciled, affected proof and
112
+ required review pass, protected paths and rejected approaches remain intact,
113
+ and every unfinished item has a concrete status.
114
+
115
+ For ordinary medium work report only:
116
+
117
+ - behavior/contract implemented;
118
+ - review result and repaired/open findings;
119
+ - affected validation;
120
+ - skipped checks and residual risk;
121
+ - changed files and any authorized commits.
122
+
123
+ For high, actual Closure, accepted risk, interrupted recovery, or multi-agent
124
+ delivery, add the relevant invariants, reviewer coverage, defect IDs, session
125
+ recovery, and handoff ownership. Do not create a separate report file unless
126
+ the spec or the complexity of that exceptional handoff requires it.
@@ -0,0 +1,6 @@
1
+ interface:
2
+ display_name: "Spec Implementer"
3
+ short_description: "Execute approved specs with lean delivery"
4
+ default_prompt: "Use $spec-implementer to execute the approved spec continuously, default to no Git checkpoints, validate proportionately, and launch only stable required review gates."
5
+ policy:
6
+ allow_implicit_invocation: true
@@ -0,0 +1,30 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "spec-implementer",
4
+ "cases": [
5
+ {
6
+ "id": "state-created-at-launch",
7
+ "prompt": "Execute an approved medium spec whose final review has not started yet.",
8
+ "expected": ["do not create review state during implementation", "persist minimal state immediately before reviewer launch"],
9
+ "forbidden": ["create per-slice review bookkeeping"]
10
+ },
11
+ {
12
+ "id": "medium-one-final-review",
13
+ "prompt": "Execute a normal medium spec with several vertical slices and no stable high-risk checkpoint.",
14
+ "expected": ["implement continuously", "run one final reviewer_standard on the settled diff"],
15
+ "forbidden": ["review after every slice", "run a full repository suite from file count"]
16
+ },
17
+ {
18
+ "id": "closure-stays-affected",
19
+ "prompt": "Final review finds one high defect in a shared schema and several ordinary low findings.",
20
+ "expected": ["repair once", "Closure verifies only the affected high-risk contract"],
21
+ "forbidden": ["restart all Full reviewers"]
22
+ },
23
+ {
24
+ "id": "resume-live-reviewer",
25
+ "prompt": "Resume after a reviewer poll timed out while the recorded session is still live.",
26
+ "expected": ["reconcile the existing session"],
27
+ "forbidden": ["mark it failed from timeout alone", "launch a duplicate reviewer"]
28
+ }
29
+ ]
30
+ }
@@ -0,0 +1,94 @@
1
+ # Approved Spec Implementation Review Loop
2
+
3
+ This reference owns review orchestration for `$spec-implementer`. Read it only
4
+ when executing an approved implementation spec. Shared Full/Closure and defect
5
+ mechanics live in `../../../docs/agents/review-protocol.md`.
6
+
7
+ Direct work and deterministic issue delivery use normal TDD and review gates;
8
+ they must not create Implementation Review State.
9
+
10
+ ## Authority
11
+
12
+ Only an approved implementation spec may own `## Implementation Review State`.
13
+ PRDs, tickets, architecture notes, and chat summaries are not review-state
14
+ owners. A substantive spec change returns through artifact review before
15
+ implementation continues.
16
+
17
+ Use the spec's `review_profile`; if absent, infer it from current evidence:
18
+
19
+ - `simple`: narrow change with direct proof;
20
+ - `medium`: default for ordinary implementation;
21
+ - `high`: material failure consequence plus an uncertainty amplifier.
22
+
23
+ Implementation evidence may raise but never lower the approved profile.
24
+
25
+ ## Default Review Shape
26
+
27
+ Implement continuously through vertical slices. Validate each affected behavior
28
+ and run one final review on the settled diff when the gate applies.
29
+
30
+ - `simple`: one `reviewer_fast` when review is required.
31
+ - `medium`: one `reviewer_standard`, one bounded final Full, no intermediate
32
+ checkpoint by default.
33
+ - `high`: two parallel `reviewer_deep` Full reviews with disjoint correctness
34
+ and spec/standards lenses.
35
+
36
+ Add an intermediate checkpoint only when the approved spec explicitly names a
37
+ stable high-risk slice whose review will remain valid after later work. Do not
38
+ review unstable intermediate diffs or create per-slice review cycles.
39
+
40
+ Cleanup stays inside the spec/standards lens. A concrete simplification risk may
41
+ amplify that lens; size and profile labels alone do not create another gate.
42
+
43
+ ## Minimal Durable State
44
+
45
+ Do not create review state during preflight or implementation. Immediately
46
+ before the first actual reviewer launch, persist:
47
+
48
+ - profile, authority path, settled target revision, and assigned lenses;
49
+ - launch ID, reviewer/session handle, lineage, and `pending | completed | failed`;
50
+ - returned findings, repair revision, affected validation, and Closure need.
51
+
52
+ Write `pending` before launch and reconcile that session before replacing it
53
+ after interruption or resume. A usable result becomes `completed`; an explicit
54
+ failure becomes `failed`. A poll timeout while the session remains live is not
55
+ a failure and does not authorize duplicate review.
56
+
57
+ Record extended lineage/session history only for `high`, a real intermediate
58
+ checkpoint, actual Closure, accepted risk, or interrupted recovery. Normal
59
+ medium execution does not keep epochs, pass thresholds, activation counters, or
60
+ per-slice handoff bookkeeping.
61
+
62
+ ## Findings And Closure
63
+
64
+ Root aggregates findings, repairs compatible defects once, and reruns only
65
+ affected validation. Coordinator verification closes ordinary medium/low
66
+ behavior-preserving findings after confirming the repair matches the failure.
67
+
68
+ Use shared-protocol Closure only for critical/high defects, protected
69
+ trust/data/concurrency/shared API impact, or invalidated mandatory coverage.
70
+ Closure stays with the affected reviewer lineage and repaired targets. Start a
71
+ new Full only when the repair invalidated mandatory-lens coverage.
72
+
73
+ Do not repeat review without a material change in target, evidence, repair, or
74
+ source decision. Stop and surface the actual decision or evidence blocker when
75
+ no progress is possible.
76
+
77
+ ## Completion
78
+
79
+ Run gates in this order:
80
+
81
+ 1. affected behavior and integration validation;
82
+ 2. applicable final code review;
83
+ 3. Closure only when triggered;
84
+ 4. repository architecture/build/smoke gates required by policy or the spec;
85
+ 5. delivery actions explicitly authorized by the user or workflow.
86
+
87
+ Return `Approved` only for the final settled revision when mandatory lenses and
88
+ validation are complete and shared protocol state is clear. `Waived` records
89
+ skipped coverage but is not approval. `Blocked` requires a concrete authority,
90
+ evidence, reviewer, or convergence blocker—not elapsed time or review count.
91
+
92
+ For normal medium work report only profile, review result, repaired/open
93
+ findings, affected validation, skipped checks, and residual risk. Add extended
94
+ session/defect accounting only when the exceptional state above exists.
@@ -0,0 +1,72 @@
1
+ ---
2
+ name: tdd
3
+ description: Test-driven development for changes that alter observable behavior, have a natural public test seam, and can produce a meaningful failing test before implementation. Use after the global TDD Fit Gate passes, or when the user explicitly requests red-green-refactor, test-first development, or TDD.
4
+ ---
5
+
6
+ # Test-Driven Development
7
+
8
+ Use short vertical RED -> GREEN cycles. Make each test prove observable behavior through the same public seam real callers use.
9
+
10
+ ## Fit
11
+
12
+ Use this skill only when the change alters observable behavior, a natural public
13
+ seam exists, and the pre-change test will fail for the intended behavioral
14
+ reason. If an implicit activation fails this gate, stop the TDD route and use
15
+ existing regression tests plus affected validation. For mixed tasks, apply TDD
16
+ only to the behavioral slice.
17
+
18
+ Behavior-preserving cleanup, dead-code deletion, documentation, copy,
19
+ formatting, generated assets, package maintenance, simple config, builds, and
20
+ read-only work do not need TDD. Absence and architecture guards added after a
21
+ cleanup are validation, not RED proofs.
22
+
23
+ ## Core Contract
24
+
25
+ - Lock expected behavior from the request, specification, design, bug report, or existing product behavior before changing implementation.
26
+ - Derive expected values from an independent source, never from the production algorithm.
27
+ - Prove RED on the old behavior for the same observable reason the user reported or requested.
28
+ - Add only enough implementation to make the current test pass; do not anticipate later tests.
29
+ - Keep tests stable across behavior-preserving refactors and refactor only while GREEN.
30
+
31
+ Read [tests.md](tests.md) when choosing or reviewing test shape. Read [mocking.md](mocking.md) before introducing test doubles.
32
+
33
+ ## Before the First RED
34
+
35
+ 1. Read local instructions, domain language, existing tests, and relevant ADRs.
36
+ 2. List the prioritized observable behaviors, not implementation steps.
37
+ 3. Select the public seam where callers observe each behavior.
38
+ 4. Ask the user only when the seam changes the public contract, product intent is unclear, or behavior priorities materially conflict.
39
+ 5. For contract-risk changes, create or update the shared [Contract Test Ledger](../../docs/agents/contract-test-ledger.md) and map each invariant to its first failing test or observable proof.
40
+ 6. If no natural public seam exists, stop the TDD route. Consult [interface-design.md](interface-design.md) only when changing the interface is itself required by the task.
41
+
42
+ For UI behavior, define proof at the rendered seam: visible content and order, interaction result, semantics, or screenshot when layout direction or scrolling matters.
43
+
44
+ ## RED -> GREEN Cycle
45
+
46
+ For each behavior:
47
+
48
+ 1. **RED:** Write one test through the selected seam.
49
+ 2. Confirm it fails on current behavior for the expected reason. A passing test or an internal-only failure is not valid RED.
50
+ 3. **GREEN:** Add the minimal implementation required for that test.
51
+ 4. Run the proof and update the ledger status to `red`, `green`, or `blocked` with the missing seam or evidence.
52
+
53
+ Keep each cycle to one seam, one behavior, one test, and one minimal implementation. Do not rewrite the test to fit the code. For state, async, lifecycle, retry, cache, or auth defects, include the competing condition when feasible.
54
+
55
+ Handle reviewer repairs inside the same activation only under [bug workflow routing](../../docs/agents/bug-workflow-routing.md); group related cases by protected invariant.
56
+
57
+ ## After GREEN
58
+
59
+ Refactor as a separate review-stage activity, never while RED. Use [refactoring.md](refactoring.md) for candidates and rerun affected tests after each step.
60
+
61
+ ## Cycle Checklist
62
+
63
+ ```text
64
+ [ ] Behavior is proved through the caller's public seam
65
+ [ ] Expected behavior is locked and the expected value is independent
66
+ [ ] RED fails on old behavior for the correct observable reason
67
+ [ ] Test was not fitted to implementation details
68
+ [ ] GREEN uses only the code needed for the current behavior
69
+ [ ] Final outcome and relevant competing condition are proved
70
+ [ ] Contract Test Ledger is current when applicable
71
+ [ ] Refactoring starts only after GREEN
72
+ ```
@@ -0,0 +1,6 @@
1
+ interface:
2
+ display_name: "Test-Driven Development"
3
+ short_description: "Use TDD only when its behavioral fit gate passes"
4
+ default_prompt: "Use $tdd after confirming an observable behavior change, a public test seam, and a meaningful pre-change failure."
5
+ policy:
6
+ allow_implicit_invocation: true
@@ -0,0 +1,31 @@
1
+ # Interface Design for Testability
2
+
3
+ Good interfaces make testing natural:
4
+
5
+ 1. **Accept dependencies, don't create them**
6
+
7
+ ```typescript
8
+ // Testable
9
+ function processOrder(order, paymentGateway) {}
10
+
11
+ // Hard to test
12
+ function processOrder(order) {
13
+ const gateway = new StripeGateway();
14
+ }
15
+ ```
16
+
17
+ 2. **Return results, don't produce side effects**
18
+
19
+ ```typescript
20
+ // Testable
21
+ function calculateDiscount(cart): Discount {}
22
+
23
+ // Hard to test
24
+ function applyDiscount(cart): void {
25
+ cart.total -= discount;
26
+ }
27
+ ```
28
+
29
+ 3. **Small surface area**
30
+ - Fewer methods = fewer tests needed
31
+ - Fewer params = simpler test setup
@@ -0,0 +1,59 @@
1
+ # When to Mock
2
+
3
+ Mock at **system boundaries** only:
4
+
5
+ - External APIs (payment, email, etc.)
6
+ - Databases (sometimes - prefer test DB)
7
+ - Time/randomness
8
+ - File system (sometimes)
9
+
10
+ Don't mock:
11
+
12
+ - Your own classes/modules
13
+ - Internal collaborators
14
+ - Anything you control
15
+
16
+ ## Designing for Mockability
17
+
18
+ At system boundaries, design interfaces that are easy to mock:
19
+
20
+ **1. Use dependency injection**
21
+
22
+ Pass external dependencies in rather than creating them internally:
23
+
24
+ ```typescript
25
+ // Easy to mock
26
+ function processPayment(order, paymentClient) {
27
+ return paymentClient.charge(order.total);
28
+ }
29
+
30
+ // Hard to mock
31
+ function processPayment(order) {
32
+ const client = new StripeClient(process.env.STRIPE_KEY);
33
+ return client.charge(order.total);
34
+ }
35
+ ```
36
+
37
+ **2. Prefer SDK-style interfaces over generic fetchers**
38
+
39
+ Create specific functions for each external operation instead of one generic function with conditional logic:
40
+
41
+ ```typescript
42
+ // GOOD: Each function is independently mockable
43
+ const api = {
44
+ getUser: (id) => fetch(`/users/${id}`),
45
+ getOrders: (userId) => fetch(`/users/${userId}/orders`),
46
+ createOrder: (data) => fetch('/orders', { method: 'POST', body: data }),
47
+ };
48
+
49
+ // BAD: Mocking requires conditional logic inside the mock
50
+ const api = {
51
+ fetch: (endpoint, options) => fetch(endpoint, options),
52
+ };
53
+ ```
54
+
55
+ The SDK approach means:
56
+ - Each mock returns one specific shape
57
+ - No conditional logic in test setup
58
+ - Easier to see which endpoints a test exercises
59
+ - Type safety per endpoint
@@ -0,0 +1,10 @@
1
+ # Refactor Candidates
2
+
3
+ After TDD cycle, look for:
4
+
5
+ - **Duplication** → Extract function/class
6
+ - **Long methods** → Break into private helpers (keep tests on public interface)
7
+ - **Shallow modules** → Combine or deepen
8
+ - **Feature envy** → Move logic to where data lives
9
+ - **Primitive obsession** → Introduce value objects
10
+ - **Existing code** the new code reveals as problematic
@@ -0,0 +1,77 @@
1
+ # Good and Bad Tests
2
+
3
+ ## Good Tests
4
+
5
+ **Integration-style**: Test through real interfaces, not mocks of internal parts.
6
+
7
+ ```typescript
8
+ // GOOD: Tests observable behavior
9
+ test("user can checkout with valid cart", async () => {
10
+ const cart = createCart();
11
+ cart.add(product);
12
+ const result = await checkout(cart, paymentMethod);
13
+ expect(result.status).toBe("confirmed");
14
+ });
15
+ ```
16
+
17
+ Characteristics:
18
+
19
+ - Tests behavior users/callers care about
20
+ - Uses public API only
21
+ - Survives internal refactors
22
+ - Describes WHAT, not HOW
23
+ - One logical assertion per test
24
+
25
+ ## Bad Tests
26
+
27
+ **Implementation-detail tests**: Coupled to internal structure.
28
+
29
+ ```typescript
30
+ // BAD: Tests implementation details
31
+ test("checkout calls paymentService.process", async () => {
32
+ const mockPayment = jest.mock(paymentService);
33
+ await checkout(cart, payment);
34
+ expect(mockPayment.process).toHaveBeenCalledWith(cart.total);
35
+ });
36
+ ```
37
+
38
+ Red flags:
39
+
40
+ - Mocking internal collaborators
41
+ - Testing private methods
42
+ - Asserting on call counts/order
43
+ - Test breaks when refactoring without behavior change
44
+ - Test name describes HOW not WHAT
45
+ - Verifying through external means instead of interface
46
+
47
+ ```typescript
48
+ // BAD: Bypasses interface to verify
49
+ test("createUser saves to database", async () => {
50
+ await createUser({ name: "Alice" });
51
+ const row = await db.query("SELECT * FROM users WHERE name = ?", ["Alice"]);
52
+ expect(row).toBeDefined();
53
+ });
54
+
55
+ // GOOD: Verifies through interface
56
+ test("createUser makes user retrievable", async () => {
57
+ const user = await createUser({ name: "Alice" });
58
+ const retrieved = await getUser(user.id);
59
+ expect(retrieved.name).toBe("Alice");
60
+ });
61
+ ```
62
+
63
+ **Tautological tests**: Expected value restates the implementation, so the test passes by construction.
64
+
65
+ ```typescript
66
+ // BAD: Expected value is recomputed the way the code computes it
67
+ test("calculateTotal sums line items", () => {
68
+ const items = [{ price: 10 }, { price: 5 }];
69
+ const expected = items.reduce((sum, item) => sum + item.price, 0);
70
+ expect(calculateTotal(items)).toBe(expected);
71
+ });
72
+
73
+ // GOOD: Expected value comes from an independent, known literal
74
+ test("calculateTotal sums line items", () => {
75
+ expect(calculateTotal([{ price: 10 }, { price: 5 }])).toBe(15);
76
+ });
77
+ ```