codex-orchestrator 2.0.3 → 2.0.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (161) hide show
  1. package/CHANGELOG.md +51 -427
  2. package/README.md +161 -37
  3. package/dist/src/index.d.ts +1 -1
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/v2/acceptance-proof.d.ts +5 -0
  6. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  7. package/dist/src/v2/acceptance-proof.js +10 -2
  8. package/dist/src/v2/acceptance-proof.js.map +1 -1
  9. package/dist/src/v2/adapters/gh-issue-adapter.d.ts.map +1 -1
  10. package/dist/src/v2/adapters/gh-issue-adapter.js +6 -7
  11. package/dist/src/v2/adapters/gh-issue-adapter.js.map +1 -1
  12. package/dist/src/v2/adapters/gh-pull-request-adapter.d.ts +7 -1
  13. package/dist/src/v2/adapters/gh-pull-request-adapter.d.ts.map +1 -1
  14. package/dist/src/v2/adapters/gh-pull-request-adapter.js +288 -0
  15. package/dist/src/v2/adapters/gh-pull-request-adapter.js.map +1 -1
  16. package/dist/src/v2/adapters/pull-requests.d.ts +69 -0
  17. package/dist/src/v2/adapters/pull-requests.d.ts.map +1 -1
  18. package/dist/src/v2/adapters/pull-requests.js +48 -0
  19. package/dist/src/v2/adapters/pull-requests.js.map +1 -1
  20. package/dist/src/v2/adapters/worktree.d.ts +1 -0
  21. package/dist/src/v2/adapters/worktree.d.ts.map +1 -1
  22. package/dist/src/v2/adapters/worktree.js +10 -1
  23. package/dist/src/v2/adapters/worktree.js.map +1 -1
  24. package/dist/src/v2/cli-contract.d.ts +3 -3
  25. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  26. package/dist/src/v2/cli-contract.js +1 -3
  27. package/dist/src/v2/cli-contract.js.map +1 -1
  28. package/dist/src/v2/cli.d.ts +33 -0
  29. package/dist/src/v2/cli.d.ts.map +1 -0
  30. package/dist/src/v2/{candidate-cli.js → cli.js} +55 -39
  31. package/dist/src/v2/cli.js.map +1 -0
  32. package/dist/src/v2/code-review-report.d.ts +1 -1
  33. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  34. package/dist/src/v2/code-review-report.js +2 -2
  35. package/dist/src/v2/code-review-report.js.map +1 -1
  36. package/dist/src/v2/codex-process.d.ts.map +1 -1
  37. package/dist/src/v2/codex-process.js +12 -1
  38. package/dist/src/v2/codex-process.js.map +1 -1
  39. package/dist/src/v2/config.d.ts +2 -3
  40. package/dist/src/v2/config.d.ts.map +1 -1
  41. package/dist/src/v2/config.js +0 -3
  42. package/dist/src/v2/config.js.map +1 -1
  43. package/dist/src/v2/contained-report-operation.d.ts +2 -2
  44. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  45. package/dist/src/v2/contained-report-operation.js +1 -1
  46. package/dist/src/v2/contained-report-operation.js.map +1 -1
  47. package/dist/src/v2/containment.d.ts +15 -2
  48. package/dist/src/v2/containment.d.ts.map +1 -1
  49. package/dist/src/v2/containment.js +43 -6
  50. package/dist/src/v2/containment.js.map +1 -1
  51. package/dist/src/v2/direct-delivery.d.ts +5 -10
  52. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  53. package/dist/src/v2/direct-delivery.js +32 -91
  54. package/dist/src/v2/direct-delivery.js.map +1 -1
  55. package/dist/src/v2/proof-report.d.ts.map +1 -1
  56. package/dist/src/v2/proof-report.js +55 -29
  57. package/dist/src/v2/proof-report.js.map +1 -1
  58. package/dist/src/v2/review-feedback-coordinator.d.ts +54 -0
  59. package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -0
  60. package/dist/src/v2/review-feedback-coordinator.js +245 -0
  61. package/dist/src/v2/review-feedback-coordinator.js.map +1 -0
  62. package/dist/src/v2/review-feedback.d.ts +127 -0
  63. package/dist/src/v2/review-feedback.d.ts.map +1 -0
  64. package/dist/src/v2/review-feedback.js +436 -0
  65. package/dist/src/v2/review-feedback.js.map +1 -0
  66. package/dist/src/v2/run-issue.d.ts +63 -9
  67. package/dist/src/v2/run-issue.d.ts.map +1 -1
  68. package/dist/src/v2/run-issue.js +798 -78
  69. package/dist/src/v2/run-issue.js.map +1 -1
  70. package/dist/src/v2/run-store.d.ts +49 -5
  71. package/dist/src/v2/run-store.d.ts.map +1 -1
  72. package/dist/src/v2/run-store.js +138 -44
  73. package/dist/src/v2/run-store.js.map +1 -1
  74. package/dist/src/v2/runtime.d.ts +40 -3
  75. package/dist/src/v2/runtime.d.ts.map +1 -1
  76. package/dist/src/v2/runtime.js +245 -52
  77. package/dist/src/v2/runtime.js.map +1 -1
  78. package/dist/src/v2/setup-cli.d.ts.map +1 -1
  79. package/dist/src/v2/setup-cli.js +4 -11
  80. package/dist/src/v2/setup-cli.js.map +1 -1
  81. package/dist/src/v2/setup-runtime.d.ts.map +1 -1
  82. package/dist/src/v2/setup-runtime.js +1 -61
  83. package/dist/src/v2/setup-runtime.js.map +1 -1
  84. package/dist/src/v2/setup-store.d.ts +0 -5
  85. package/dist/src/v2/setup-store.d.ts.map +1 -1
  86. package/dist/src/v2/setup-store.js +3 -106
  87. package/dist/src/v2/setup-store.js.map +1 -1
  88. package/dist/src/v2/setup.d.ts +6 -46
  89. package/dist/src/v2/setup.d.ts.map +1 -1
  90. package/dist/src/v2/setup.js +12 -294
  91. package/dist/src/v2/setup.js.map +1 -1
  92. package/dist/src/v2/workflow-assets.d.ts +19 -11
  93. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  94. package/dist/src/v2/workflow-assets.js +132 -40
  95. package/dist/src/v2/workflow-assets.js.map +1 -1
  96. package/docs/deep-dive.md +328 -56
  97. package/internal-workflow/docs/agents/bugfix-quality-gate.md +11 -0
  98. package/internal-workflow/docs/agents/coding-skill-routing.md +116 -196
  99. package/internal-workflow/docs/agents/contract-test-ledger.md +11 -1
  100. package/internal-workflow/docs/agents/review-gates.md +32 -39
  101. package/internal-workflow/docs/agents/review-protocol.md +75 -147
  102. package/internal-workflow/evals/coding-skill-evals.json +84 -0
  103. package/internal-workflow/manifest.json +1 -1
  104. package/internal-workflow/operations/acceptance-proof/SKILL.md +7 -1
  105. package/internal-workflow/operations/ambiguity-review/SKILL.md +2 -0
  106. package/internal-workflow/operations/code-review/SKILL.md +21 -1
  107. package/internal-workflow/operations/implementation/SKILL.md +22 -1
  108. package/internal-workflow/operations/spec-author/SKILL.md +10 -1
  109. package/internal-workflow/operations/spec-review/SKILL.md +10 -1
  110. package/internal-workflow/operations/triage/SKILL.md +10 -1
  111. package/internal-workflow/schemas/code-review-v1.json +1 -1
  112. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  113. package/internal-workflow/skills/agent-auto/SKILL.md +6 -1
  114. package/internal-workflow/skills/code-debugger/SKILL.md +122 -0
  115. package/internal-workflow/skills/code-debugger/agents/openai.yaml +7 -0
  116. package/internal-workflow/skills/code-review/SKILL.md +51 -17
  117. package/internal-workflow/skills/code-review/references/cleanup-lens.md +52 -0
  118. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +15 -6
  119. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +1 -1
  120. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +2 -2
  121. package/internal-workflow/skills/implementation-spec-review/SKILL.md +108 -204
  122. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +24 -0
  123. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +93 -0
  124. package/internal-workflow/skills/small-task-implementer/SKILL.md +15 -8
  125. package/internal-workflow/skills/spec-implementer/SKILL.md +101 -172
  126. package/internal-workflow/skills/spec-implementer/evals/evals.json +30 -0
  127. package/internal-workflow/skills/spec-implementer/references/review-loop.md +100 -0
  128. package/internal-workflow/skills/tdd/SKILL.md +20 -6
  129. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  130. package/internal-workflow/skills/tdd/evals/evals.json +18 -0
  131. package/internal-workflow/skills/tdd/mocking.md +3 -42
  132. package/internal-workflow/skills/tdd/refactoring.md +6 -8
  133. package/package.json +9 -6
  134. package/dist/src/v2/adapters/target-activity-fence.d.ts +0 -23
  135. package/dist/src/v2/adapters/target-activity-fence.d.ts.map +0 -1
  136. package/dist/src/v2/adapters/target-activity-fence.js +0 -249
  137. package/dist/src/v2/adapters/target-activity-fence.js.map +0 -1
  138. package/dist/src/v2/candidate-cli.d.ts +0 -26
  139. package/dist/src/v2/candidate-cli.d.ts.map +0 -1
  140. package/dist/src/v2/candidate-cli.js.map +0 -1
  141. package/dist/src/v2/legacy-cutover.d.ts +0 -52
  142. package/dist/src/v2/legacy-cutover.d.ts.map +0 -1
  143. package/dist/src/v2/legacy-cutover.js +0 -87
  144. package/dist/src/v2/legacy-cutover.js.map +0 -1
  145. package/internal-workflow/docs/agents/artifact-review-loop.md +0 -267
  146. package/internal-workflow/docs/agents/implementation-review-loop.md +0 -302
  147. package/internal-workflow/operations/cleanup-review/SKILL.md +0 -3
  148. package/internal-workflow/operations/spec-implementation/SKILL.md +0 -3
  149. package/internal-workflow/profiles/implementer_deep.toml +0 -9
  150. package/internal-workflow/profiles/researcher_standard.toml +0 -9
  151. package/internal-workflow/profiles/reviewer_fast.toml +0 -9
  152. package/internal-workflow/skills/cleanup-review/SKILL.md +0 -84
  153. package/internal-workflow/skills/cleanup-review/agents/openai.yaml +0 -6
  154. package/internal-workflow/skills/codebase-design/DEEPENING.md +0 -35
  155. package/internal-workflow/skills/codebase-design/DESIGN-IT-TWICE.md +0 -50
  156. package/internal-workflow/skills/codebase-design/SKILL.md +0 -82
  157. package/internal-workflow/skills/codebase-design/agents/openai.yaml +0 -6
  158. package/internal-workflow/skills/research/SKILL.md +0 -107
  159. package/internal-workflow/skills/research/agents/openai.yaml +0 -6
  160. package/internal-workflow/skills/ui-evidence-proof/SKILL.md +0 -123
  161. package/internal-workflow/skills/ui-evidence-proof/agents/openai.yaml +0 -6
@@ -5,193 +5,122 @@ description: "Executes approved specs continuously with honest checklist updates
5
5
 
6
6
  # Spec Implementer
7
7
 
8
- Execute an approved implementation spec. Your job is to carry out the chosen spec, keep its checklist honest, and stop at the right boundaries. Do not redesign the work unless the spec or repo reality proves a blocker.
9
-
10
- This skill is standalone by default: it executes only an approved spec that the
11
- user has chosen to run. `$tickets-orchestrator` may invoke it inline at root for
12
- an accepted compact or standard ticket spec inside user-authorized orchestration;
13
- that does not authorize unrelated specs or broader delivery scope.
14
-
15
- When a spec contains a Contract Test Ledger, treat it as part of the execution contract. The shared reference is `../../docs/agents/contract-test-ledger.md`.
16
-
17
- All implementation review checkpoints and final review gates use
18
- `../../docs/agents/implementation-review-loop.md` as their caller-facing owner.
19
- This skill must not create a per-slice review flow or retry loop.
20
-
21
- ## Spec Modes
22
-
23
- - **Compact specs:** execute directly as one continuous implementation flow with lightweight phase checkpoints. `compact` describes document density, not implementation size or risk. Use review/signoff gates only when the spec, repo policy, or change risk requires them.
24
- - **Full specs:** execute the same phase flow, but treat `Risk Controls`, task-specific `Halt Conditions`, and validation proof as hard constraints. Do not invent extra process just because the spec is full.
25
- - **Multi-agent specs:** follow the integrator contract exactly. If write scopes are not perfectly disjoint, stop before spawning workers.
26
-
27
- ## Core Rules
28
-
29
- 1. Follow the spec literally. Do not broaden scope, add cleanup, or re-plan unless a blocker is proven.
30
- 2. Treat the spec checklist as the execution ledger.
31
- 3. Update checklist items during implementation. For compact specs, update at natural checkpoints and phase exits. For full specs, or when the spec says so, update completed leaf items immediately.
32
- 4. Do not save all checklist updates only for the final response.
33
- 5. Re-read the current phase before moving on and reconcile already-completed unchecked items.
34
- 6. If a step is blocked, leave it unchecked and record one short `Blocked:` note with the concrete reason.
35
- 7. Treat Preconditions as hard blockers. Do not start a phase until they are satisfied, explicitly not applicable, or blocked.
36
- 8. If the saved spec contains unresolved template text, placeholders, alternative commands, or pseudo-paths, stop and escalate.
37
- 9. Honor Protected Paths and Rejected Approaches exactly as written.
38
- 10. Apply the `$codebase-design` lens only when ownership or a public Module Interface or Seam changes. For any new private helper, run the deletion test and keep it only if it improves locality or leverage, without activating architecture workflow.
39
- 11. Do not add pass-through modules, one-adapter seams, or test-only helpers unless the spec explicitly approves them.
40
- 12. Add comments/docblocks only where the spec explicitly requires them.
41
-
42
- ## Before Editing
43
-
44
- - Read the complete frontmatter before announcing execution strategy. Confirm the spec path, status, `spec_mode`, `implementation_size`, `review_profile`, and `expected_repositories`; infer a missing classification from evidence without rewriting the approved design.
45
- - Identify whether it is compact, full, or multi-agent. Do not call a compact spec full or equate compact with small.
46
- - Check required services, env vars, fixtures, repo state, and prerequisite issues.
47
- - Confirm the first phase targets exist as described.
48
- - Confirm validation commands are executable or explicitly not applicable.
49
- - If the spec has a Contract Test Ledger, confirm each reached invariant has a first test/proof or a concrete blocked reason before implementation.
50
- - If the spec has `Review Checkpoints` or `Review Focus`, keep only checkpoints whose target becomes stable before later slices. If later work will touch the same files, owners, or contracts, fold that coverage into the final parallel review instead of reviewing an unstable slice.
51
- - Resolve the implementation review profile and plan mandatory final coverage before launching any reviewer. Do not manufacture an early checkpoint merely because the profile is high.
52
- - Do not write `## Implementation Review State` during ordinary preflight or implementation. Immediately before the first actual reviewer launch, persist the short Review Plan and pending launch required by the Module; then update it after every usable reviewer result, repair batch, closure, waiver, or terminal outcome.
53
- - If the spec has `Final Handoff Requirements`, treat them as the final response contract. For medium/high-risk specs without explicit requirements, prepare the standard Final Risk Handoff anyway.
54
- - For full specs, read `Risk Controls` before editing and translate each applicable control into a concrete execution constraint.
55
- - Use `Write Scope Summary` when present as an audit aid. If it is absent, rely on phase targets unless the write set is ambiguous.
56
- - Stop if exact execution would require guessing.
8
+ Execute one approved implementation spec continuously. Keep its checklist
9
+ truthful and stop only at a real authority, evidence, safety, or explicit pause
10
+ boundary. Do not redesign approved scope.
11
+
12
+ Read before editing:
13
+
14
+ - the complete approved spec and applicable repository instructions;
15
+ - `references/review-loop.md` for approved-spec review ownership;
16
+ - `../../docs/agents/contract-test-ledger.md` only when the spec contains
17
+ material contract invariants.
18
+
19
+ ## Modes
20
+
21
+ Compact and full specs use the same direct phase flow. `compact` describes
22
+ document density, not implementation size or risk. Full mode adds only the
23
+ concrete `Risk Controls`, stop conditions, validation, or coordination contract
24
+ already present in the approved spec; it does not add review or reporting
25
+ ceremony by itself.
26
+
27
+ Use multiple agents only when the spec defines perfectly disjoint write scopes
28
+ and one integrator. Otherwise execute single-agent.
29
+
30
+ ## Preflight
31
+
32
+ 1. Confirm `status`, `spec_mode`, `implementation_size`, `review_profile`,
33
+ repository count, authority, scope, and exclusions.
34
+ 2. Confirm first-phase targets, required services/env/data/fixtures, commands,
35
+ observable proof, protected paths, and rejected approaches.
36
+ 3. Stop if execution needs invented paths, symbols, contracts, commands, or
37
+ product decisions; if reality differs only in a bounded technical detail,
38
+ resolve it from repository evidence and record the adjustment.
39
+ 4. Keep an intermediate review checkpoint only when the spec explicitly names
40
+ a stable high-risk slice that later work will not invalidate. Move every
41
+ ordinary or unstable checkpoint to final review.
42
+ 5. Do not create `## Implementation Review State` yet.
43
+
44
+ ## Implementation
45
+
46
+ For each phase:
47
+
48
+ 1. Re-read its scope and preconditions.
49
+ 2. Implement narrow vertical behavior slices through `$tdd` when applicable.
50
+ 3. Update reached checklist and Contract Test Ledger items at natural
51
+ checkpoints; never save all updates for the end.
52
+ 4. Run the phase's targeted exit proof.
53
+ 5. Record a short `Blocked:` note for any item that cannot complete.
54
+ 6. Continue immediately when the exit proof passes and no stop condition
55
+ applies.
56
+
57
+ Do not add cleanup, comments, helpers, abstractions, retries, flags, fallbacks,
58
+ or compatibility paths outside the spec. A small implementation adjustment is
59
+ allowed only when it preserves approved behavior and is supported by current
60
+ repository evidence.
57
61
 
58
62
  ## Git Checkpoints
59
63
 
60
- Default to `none` and begin implementation without checkpoint ceremony. Mention the strategy once only when it materially affects delivery. Invoking `$spec-implementer` does not by itself authorize commits.
64
+ Default to no commits. `$spec-implementer` does not authorize Git writes.
61
65
 
62
- Choose `per-slice` only when commits are explicitly authorized by the user or approved spec, slice diffs are truly isolated, and checkpoints materially improve recovery or handoff safety. Valid reasons are:
66
+ Use per-slice commits only when explicitly authorized and they materially help
67
+ a disjoint multi-agent handoff, planned cross-session pause, or approved
68
+ rollback boundary. Require a passed slice proof, stage only owned paths, follow
69
+ `$commit`, and never push without separate authority. File/slice count and risk
70
+ profile alone do not justify checkpoints.
63
71
 
64
- - a multi-agent merge/handoff boundary
65
- - an explicitly planned pause or continuation in another session
66
- - a destructive or rollback boundary whose isolated commit is part of the approved safety plan
72
+ ## Review And Validation
67
73
 
68
- Use `none` for ordinary single-agent execution, including compact/high specs, overlapping slices, and continuous work in one session. Slice count, file count, or review profile alone never justifies commits.
74
+ Run targeted behavior tests and the smallest affected integration checks first.
75
+ Use a full repository suite only when the spec or repository policy requires
76
+ it, broad contract fan-out cannot be isolated, or the task is genuinely high.
69
77
 
70
- At each checkpoint:
78
+ Run final `$code-review` only when the spec, repository policy, or
79
+ `../../docs/agents/review-gates.md` applies. Ordinary medium work gets one
80
+ `reviewer_standard` Full review on the settled diff. High gets two disjoint
81
+ `reviewer_deep` lenses. Cleanup stays inside spec/standards review.
71
82
 
72
- - require a passed exit gate and applicable slice review, then reconcile and include tracked checklist/ledger updates
73
- - inspect the full diff, stage only slice-owned paths or hunks, follow `$commit` safety rules, and verify the hash
74
- - treat the applicable slice review as the pre-commit gate; final review still runs at the end
83
+ Immediately before the first reviewer launch, create only the minimal
84
+ `## Implementation Review State` required by `references/review-loop.md`.
85
+ Reconcile a recorded live session before replacement after interruption.
75
86
 
76
- If isolation later becomes unsafe, skip and record the reason. Never commit RED state, failed validation, unresolved findings, discovery-only work, or a partial slice. Never amend a checkpoint commit; put later fixes in a new commit. Never push unless explicitly requested. Report the strategy and slice-to-commit mapping, or the reason no checkpoints were created.
87
+ Repair compatible findings once and rerun affected validation. Coordinator
88
+ verification closes ordinary medium/low behavior-preserving repairs. Use
89
+ Closure only for critical/high, protected trust/data/concurrency/shared-contract
90
+ impact, or invalidated mandatory coverage. Do not restart broad Full review
91
+ unless the repair actually invalidated its coverage.
77
92
 
78
- ## Phase Workflow
79
-
80
- For every phase:
81
-
82
- - Confirm phase dependencies and preconditions.
83
- - Execute only the steps assigned to that phase.
84
- - Update the spec checklist according to its mode.
85
- - Update any reached Contract Test Ledger rows as planned -> red -> green, or blocked with the missing seam/proof.
86
- - Run the phase exit gate.
87
- - If a `Review Checkpoint` applies after this phase, execute it through the
88
- Module only when the target is settled and later slices do not invalidate its
89
- files/contracts. Otherwise record that its lenses moved to final coverage and
90
- continue without launching an unstable review.
91
- - Reconcile unchecked items for the current phase.
92
- - Re-check applicable `Risk Controls` before leaving the phase.
93
- - Check whether the phase introduced shallow modules, duplicated source-of-truth logic, or tests coupled to implementation details; fix only when inside approved scope, otherwise report it.
94
- - Run the repo architecture check when available, applicable, and required by the spec or repo policy.
95
- - If `per-slice` applies and this phase completes an implementation slice, create and verify its checkpoint commit before continuing.
96
- - Continue to the next phase only when the exit gate passes and no stop condition applies.
97
- - If the phase exit gate says `User Pause: Required`, stop and wait for the user's explicit command.
98
-
99
- ## Review And Signoff
100
-
101
- Do not run a dedicated review subagent after every phase by default. Do run one at explicit `Review Checkpoints`; these are risk gates, not optional status updates.
102
-
103
- Before every reviewer launch, apply the Module's launch and reconciliation
104
- rules to the persisted `## Implementation Review State`; do not restate or
105
- replace those rules in this skill.
106
-
107
- Require final `$code-review` coverage when any of these apply:
108
-
109
- - the spec explicitly requires it
110
- - the repo policy requires it
111
- - the change is medium or large
112
- - the change touches multiple runtime files or shared behavior
113
- - the change touches API contracts, DTOs, schemas, persistence, auth, permissions, payments, caching, concurrency, background jobs, or shared state
114
-
115
- When `$code-review` is required for a checkpoint or final gate, keep orchestration at root so `$code-review` can launch the profile-selected reviewer topology. Invoking `$spec-implementer` authorizes that review; if the required role is unavailable, report the gate as unavailable/blocked instead of self-certifying it.
93
+ ## Stop Conditions
116
94
 
117
- Use one final `$code-review` wave after the implementation and validation settle.
118
- For `simple` and `medium`, one reviewer covers both lenses. For `high`, launch
119
- the correctness and spec/standards reviewers in parallel; the spec/standards
120
- lens includes bounded cleanup. Run separate `$cleanup-review` only when the
121
- user, approved source, or repo policy names a concrete evidenced reason that
122
- cannot fit that lens; size or risk labels alone are insufficient. Integrate safe
123
- fixes and rerun relevant validation before continuing.
124
- Before launching a fresh final reviewer, reconcile the settled revision against
125
- the Review Plan. Stop when an approved Full or Closure already covers every
126
- mandatory final lens; otherwise run `$code-review` only for the uncovered
127
- lenses. A `cleanup-only` result never substitutes for correctness or
128
- spec/standards coverage.
95
+ Stop and report the exact blocker when:
129
96
 
130
- For compact low-risk specs, final validation plus checklist reconciliation is enough unless the spec says otherwise.
97
+ - a precondition, source contract, required proof, or protected path differs
98
+ materially from the approved spec;
99
+ - exact execution requires a new product, scope, ownership, or risky trade-off
100
+ decision;
101
+ - required validation or reviewer is unavailable and no approved substitute
102
+ exists;
103
+ - multi-agent scopes overlap or integration ownership is missing;
104
+ - an explicit user pause or halt condition is reached.
131
105
 
132
- Treat review feedback as mandatory remediation when it is grounded in code or the spec. If review reveals ambiguity that cannot be resolved from the spec, code, or docs, pause and ask the user.
106
+ Do not stop merely because the work is broad, review took time, or a medium/low
107
+ finding required one repair.
133
108
 
134
- Apply the Module's convergence and stop rules after every usable result and
135
- before every new launch.
109
+ ## Completion
136
110
 
137
- ## Multi-Agent Execution
111
+ Complete only when reached checklist items are reconciled, affected proof and
112
+ required review pass, protected paths and rejected approaches remain intact,
113
+ and every unfinished item has a concrete status.
138
114
 
139
- - Use multiple agents only when the spec has explicit disjoint write scopes.
140
- - Keep one integrator responsible for merge sequencing, handoff checks, final validation, and checklist reconciliation.
141
- - Respect exclusive write scopes, handoff artifacts, forbidden overlap, and merge order exactly as written.
142
- - Never let two agents edit the same file, generated artifact, schema, source-of-truth rule, or shared contract at the same time.
143
- - If the spec lacks a clear integrator contract, execute single-agent or stop and ask for clarification.
115
+ For ordinary medium work report only:
144
116
 
145
- ## Stop Conditions
117
+ - behavior/contract implemented;
118
+ - review result and repaired/open findings;
119
+ - affected validation;
120
+ - skipped checks and residual risk;
121
+ - changed files and any authorized commits.
146
122
 
147
- Stop immediately and escalate if:
148
-
149
- - a required precondition cannot be satisfied exactly
150
- - a required file, symbol, command, dependency, or interface differs from the spec
151
- - the saved spec contains unresolved template text or alternative commands
152
- - completing the task would require touching a Protected Path or using a Rejected Approach
153
- - validation cannot prove the intended behavior with available repo context
154
- - implementation would require unapproved scope, abstraction, migration, compatibility logic, or cleanup
155
- - a `Risk Controls` rule would be violated or is contradicted by repo reality
156
- - review exposes a real ambiguity that would require guessing
157
- - a user pause is required and the user has not explicitly said to proceed
158
-
159
- ## Completion Standard
160
-
161
- Do not mark the task complete until:
162
-
163
- - completed checklist items are checked off according to the spec mode
164
- - every remaining unchecked item is blocked, intentionally unfinished, not applicable, or halted by an explicit stop condition
165
- - every reached phase exit gate has passed
166
- - applicable `Risk Controls` remained satisfied
167
- - reached Contract Test Ledger rows are green or explicitly blocked with evidence
168
- - validation commands and behavior proof have run, or skipped checks have a concrete reason
169
- - required review/signoff gates have run and grounded findings are fixed or blocked with evidence
170
- - the whole-spec Review Plan, review-pass history, and stable defect lifecycle remain
171
- consistent with `implementation-review-loop.md`
172
- - protected paths remained untouched and rejected approaches were not used
173
- - required comments/docblocks were added only where the spec demanded them
174
- - the chosen Git checkpoint strategy was followed, and every created or skipped checkpoint was recorded
175
- - final user handoff is allowed by the spec
176
-
177
- ## Final Risk Handoff
178
-
179
- For medium/high-risk specs, the final chat response must include a compact `Final Risk Handoff` block. Do not make the user ask for this separately, and do not replace it with a generic summary.
180
-
181
- Include:
182
-
183
- - **Contract implemented:** the one behavior/contract delivered, in user-facing terms.
184
- - **High-risk checkpoints:** each required checkpoint, review result, fixed findings, and any stop/continue decision.
185
- - **Main invariants proved:** the key Contract Test Ledger rows or equivalent proofs and their status.
186
- - **Code-review findings:** high/critical findings fixed, remaining medium/low findings, or `none`.
187
- - **Fixes after review:** concrete fixes made because of cleanup/code review, or `none`.
188
- - **Validation:** exact commands/proofs that passed.
189
- - **Skipped checks:** skipped or blocked checks with concrete reasons.
190
- - **Residual risks:** accepted remaining risks or `none`.
191
- - **Checkpoint commits:** slice-to-commit mapping or `none`; do not narrate hypothetical checkpoints that were never authorized or attempted.
192
- - **Implementation reviews:** profile, total review passes, Full/Closure count,
193
- mandatory coverage, verified defect IDs, accepted-risk IDs with authority and
194
- reason, and open defect IDs.
195
- - **Files by role:** state owner, orchestration, side effects, UI/projection, tests, docs/copy, as applicable.
196
-
197
- Only create a separate report file when the spec requires it or the work is broad enough that chat would lose important evidence, such as multi-agent execution, multiple review passes with findings, skipped live checks, production validation, or handoff to another person. Otherwise keep the spec checklist/ledger as the durable artifact and the final response as the concise decision packet.
123
+ For high, actual Closure, accepted risk, interrupted recovery, or multi-agent
124
+ delivery, add the relevant invariants, reviewer coverage, defect IDs, session
125
+ recovery, and handoff ownership. Do not create a separate report file unless
126
+ the spec or the complexity of that exceptional handoff requires it.
@@ -0,0 +1,30 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "spec-implementer",
4
+ "cases": [
5
+ {
6
+ "id": "state-created-at-launch",
7
+ "prompt": "Execute an approved medium spec whose final review has not started yet.",
8
+ "expected": ["do not create review state during implementation", "persist minimal state immediately before reviewer launch"],
9
+ "forbidden": ["create per-slice review bookkeeping"]
10
+ },
11
+ {
12
+ "id": "medium-one-final-review",
13
+ "prompt": "Execute a normal medium spec with several vertical slices and no stable high-risk checkpoint.",
14
+ "expected": ["implement continuously", "run one final reviewer_standard on the settled diff"],
15
+ "forbidden": ["review after every slice", "run a full repository suite from file count"]
16
+ },
17
+ {
18
+ "id": "closure-stays-affected",
19
+ "prompt": "Final review finds one high defect in a shared schema and several ordinary low findings.",
20
+ "expected": ["repair once", "Closure verifies only the affected high-risk contract"],
21
+ "forbidden": ["restart all Full reviewers"]
22
+ },
23
+ {
24
+ "id": "resume-live-reviewer",
25
+ "prompt": "Resume after a reviewer poll timed out while the recorded session is still live.",
26
+ "expected": ["reconcile the existing session"],
27
+ "forbidden": ["mark it failed from timeout alone", "launch a duplicate reviewer"]
28
+ }
29
+ ]
30
+ }
@@ -0,0 +1,100 @@
1
+ # Approved Spec Implementation Review Loop
2
+
3
+ This reference owns review orchestration for `$spec-implementer`. Read it only
4
+ when executing an approved implementation spec. Shared Full/Closure and defect
5
+ mechanics live in `../../../docs/agents/review-protocol.md`.
6
+
7
+ Direct work and deterministic issue delivery use normal TDD and review gates;
8
+ they must not create Implementation Review State.
9
+
10
+ ## Authority
11
+
12
+ Only an approved implementation spec may own `## Implementation Review State`.
13
+ PRDs, tickets, architecture notes, and chat summaries are not review-state
14
+ owners. A substantive spec change returns through artifact review before
15
+ implementation continues.
16
+
17
+ Use the spec's `review_profile`; if absent, infer it from current evidence:
18
+
19
+ - `simple`: narrow change with direct proof;
20
+ - `medium`: default for ordinary implementation;
21
+ - `high`: material failure consequence (financial side effect,
22
+ unauthorized/cross-owner behavior, durable corruption, or materially false
23
+ production result) plus an uncertainty amplifier (concurrency or event
24
+ ordering, delayed/background callbacks, retry/idempotency/recovery,
25
+ ownership transitions, or shared state across consumers).
26
+
27
+ Implementation evidence may raise but never lower the approved profile.
28
+ Recheck the settled diff immediately before the first reviewer launch and
29
+ persist any required raise before launching reviewers.
30
+
31
+ ## Default Review Shape
32
+
33
+ Implement continuously through vertical slices. Validate each affected behavior
34
+ and run one final review on the settled diff when the gate applies.
35
+
36
+ - `simple`: one `reviewer_fast` when review is required.
37
+ - `medium`: one `reviewer_standard`, one bounded final Full, no intermediate
38
+ checkpoint by default.
39
+ - `high`: two parallel `reviewer_deep` Full reviews with disjoint correctness
40
+ and spec/standards lenses.
41
+
42
+ Add an intermediate checkpoint only when the approved spec explicitly names a
43
+ stable high-risk slice whose review will remain valid after later work. Do not
44
+ review unstable intermediate diffs or create per-slice review cycles.
45
+
46
+ Cleanup stays inside the spec/standards lens. A concrete simplification risk may
47
+ amplify that lens; size and profile labels alone do not create another gate.
48
+
49
+ ## Minimal Durable State
50
+
51
+ Do not create review state during preflight or implementation. Immediately
52
+ before the first actual reviewer launch, persist:
53
+
54
+ - profile, authority path, settled target revision, and assigned lenses;
55
+ - launch ID, reviewer/session handle, lineage, and `pending | completed | failed`;
56
+ - returned findings, repair revision, affected validation, and Closure need.
57
+
58
+ Write `pending` before launch and reconcile that session before replacing it
59
+ after interruption or resume. A usable result becomes `completed`; an explicit
60
+ failure becomes `failed`. A poll timeout while the session remains live is not
61
+ a failure and does not authorize duplicate review.
62
+
63
+ Record extended lineage/session history only for `high`, a real intermediate
64
+ checkpoint, actual Closure, accepted risk, or interrupted recovery. Normal
65
+ medium execution does not keep epochs, pass thresholds, activation counters, or
66
+ per-slice handoff bookkeeping.
67
+
68
+ ## Findings And Closure
69
+
70
+ Root aggregates findings, repairs compatible defects once, and reruns only
71
+ affected validation. Coordinator verification closes ordinary medium/low
72
+ behavior-preserving findings after confirming the repair matches the failure.
73
+
74
+ Use shared-protocol Closure only for critical/high defects, protected
75
+ trust/data/concurrency/shared API impact, or invalidated mandatory coverage.
76
+ Closure stays with the affected reviewer lineage and repaired targets. Start a
77
+ new Full only when the repair invalidated mandatory-lens coverage.
78
+
79
+ Do not repeat review without a material change in target, evidence, repair, or
80
+ source decision. Stop and surface the actual decision or evidence blocker when
81
+ no progress is possible.
82
+
83
+ ## Completion
84
+
85
+ Run gates in this order:
86
+
87
+ 1. affected behavior and integration validation;
88
+ 2. applicable final code review;
89
+ 3. Closure only when triggered;
90
+ 4. repository architecture/build/smoke gates required by policy or the spec;
91
+ 5. delivery actions explicitly authorized by the user or workflow.
92
+
93
+ Return `Approved` only for the final settled revision when mandatory lenses and
94
+ validation are complete and shared protocol state is clear. `Waived` records
95
+ skipped coverage but is not approval. `Blocked` requires a concrete authority,
96
+ evidence, reviewer, or convergence blocker—not elapsed time or review count.
97
+
98
+ For normal medium work report only profile, review result, repaired/open
99
+ findings, affected validation, skipped checks, and residual risk. Add extended
100
+ session/defect accounting only when the exceptional state above exists.
@@ -1,19 +1,33 @@
1
1
  ---
2
2
  name: tdd
3
- description: Test-driven development policy gate for implementation, bugfix, and new feature work unless the user explicitly opts out. Use before planning or editing code to shape the first behavior proof, and when the user mentions red-green-refactor, integration tests, or test-first development.
3
+ description: Test-driven development for changes that alter observable behavior, have a natural public test seam, and can produce a meaningful failing test before implementation. Use after the global TDD Fit Gate passes, or when the user explicitly requests red-green-refactor, test-first development, or TDD.
4
4
  ---
5
5
 
6
6
  # Test-Driven Development
7
7
 
8
8
  Use short vertical RED -> GREEN cycles. Make each test prove observable behavior through the same public seam real callers use.
9
9
 
10
+ ## Fit
11
+
12
+ Use this skill only when the change alters observable behavior, a natural public
13
+ seam exists, and the pre-change test will fail for the intended behavioral
14
+ reason. If an implicit activation fails this gate, stop the TDD route and use
15
+ existing regression tests plus affected validation. For mixed tasks, apply TDD
16
+ only to the behavioral slice.
17
+
18
+ Behavior-preserving cleanup, dead-code deletion, documentation, copy,
19
+ formatting, generated assets, package maintenance, simple config, builds, and
20
+ read-only work do not need TDD. Absence and architecture guards added after a
21
+ cleanup are validation, not RED proofs.
22
+
10
23
  ## Core Contract
11
24
 
12
25
  - Lock expected behavior from the request, specification, design, bug report, or existing product behavior before changing implementation.
13
26
  - Derive expected values from an independent source, never from the production algorithm.
14
27
  - Prove RED on the old behavior for the same observable reason the user reported or requested.
15
28
  - Add only enough implementation to make the current test pass; do not anticipate later tests.
16
- - Keep tests stable across behavior-preserving refactors and refactor only while GREEN.
29
+ - Keep tests stable across behavior-preserving refactors.
30
+ - After sufficient GREEN, stop by default. Refactor only to reduce concrete complexity introduced by the change.
17
31
 
18
32
  Read [tests.md](tests.md) when choosing or reviewing test shape. Read [mocking.md](mocking.md) before introducing test doubles.
19
33
 
@@ -23,8 +37,8 @@ Read [tests.md](tests.md) when choosing or reviewing test shape. Read [mocking.m
23
37
  2. List the prioritized observable behaviors, not implementation steps.
24
38
  3. Select the public seam where callers observe each behavior.
25
39
  4. Ask the user only when the seam changes the public contract, product intent is unclear, or behavior priorities materially conflict.
26
- 5. For contract-risk changes, create or update the shared [Contract Test Ledger](../../docs/agents/contract-test-ledger.md) and map each invariant to its first failing test or observable proof.
27
- 6. If no natural public seam exists, consult [interface-design.md](interface-design.md) instead of testing internals.
40
+ 5. Use the shared [Contract Test Ledger](../../docs/agents/contract-test-ledger.md) only when its material-delta and missed-failure gate passes.
41
+ 6. If no natural public seam exists, stop the TDD route. Consult [interface-design.md](interface-design.md) only when changing the interface is itself required by the task.
28
42
 
29
43
  For UI behavior, define proof at the rendered seam: visible content and order, interaction result, semantics, or screenshot when layout direction or scrolling matters.
30
44
 
@@ -43,7 +57,7 @@ Handle reviewer repairs inside the same activation only under [bug workflow rout
43
57
 
44
58
  ## After GREEN
45
59
 
46
- Refactor as a separate review-stage activity, never while RED. Use [refactoring.md](refactoring.md) for candidates and rerun affected tests after each step.
60
+ GREEN is a valid stopping point. If the current change created concrete local complexity, use [refactoring.md](refactoring.md) and rerun affected tests.
47
61
 
48
62
  ## Cycle Checklist
49
63
 
@@ -55,5 +69,5 @@ Refactor as a separate review-stage activity, never while RED. Use [refactoring.
55
69
  [ ] GREEN uses only the code needed for the current behavior
56
70
  [ ] Final outcome and relevant competing condition are proved
57
71
  [ ] Contract Test Ledger is current when applicable
58
- [ ] Refactoring starts only after GREEN
72
+ [ ] Any refactor is local and reduces current-change complexity
59
73
  ```
@@ -1,6 +1,6 @@
1
1
  interface:
2
2
  display_name: "Test-Driven Development"
3
- short_description: "Apply behavior-first red-green development"
4
- default_prompt: "Use $tdd to define the public test seam and implement this behavior through red-green cycles."
3
+ short_description: "Use TDD only when its behavioral fit gate passes"
4
+ default_prompt: "Use $tdd after confirming an observable behavior change, a public test seam, and a meaningful pre-change failure."
5
5
  policy:
6
6
  allow_implicit_invocation: true
@@ -0,0 +1,18 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "tdd",
4
+ "cases": [
5
+ {
6
+ "id": "green-can-stop",
7
+ "prompt": "The requested behavior is green and the changed code is already clear and local.",
8
+ "expected": ["stop after green", "keep the current structure"],
9
+ "forbidden": ["add helpers, classes, or value objects", "refactor unrelated code"]
10
+ },
11
+ {
12
+ "id": "no-test-only-seam",
13
+ "prompt": "A behavior test can use the existing public seam, but dependency injection would make mocking easier.",
14
+ "expected": ["use the existing public seam"],
15
+ "forbidden": ["add production dependency injection only for tests", "wrap an SDK only for mockability"]
16
+ }
17
+ ]
18
+ }
@@ -15,45 +15,6 @@ Don't mock:
15
15
 
16
16
  ## Designing for Mockability
17
17
 
18
- At system boundaries, design interfaces that are easy to mock:
19
-
20
- **1. Use dependency injection**
21
-
22
- Pass external dependencies in rather than creating them internally:
23
-
24
- ```typescript
25
- // Easy to mock
26
- function processPayment(order, paymentClient) {
27
- return paymentClient.charge(order.total);
28
- }
29
-
30
- // Hard to mock
31
- function processPayment(order) {
32
- const client = new StripeClient(process.env.STRIPE_KEY);
33
- return client.charge(order.total);
34
- }
35
- ```
36
-
37
- **2. Prefer SDK-style interfaces over generic fetchers**
38
-
39
- Create specific functions for each external operation instead of one generic function with conditional logic:
40
-
41
- ```typescript
42
- // GOOD: Each function is independently mockable
43
- const api = {
44
- getUser: (id) => fetch(`/users/${id}`),
45
- getOrders: (userId) => fetch(`/users/${userId}/orders`),
46
- createOrder: (data) => fetch('/orders', { method: 'POST', body: data }),
47
- };
48
-
49
- // BAD: Mocking requires conditional logic inside the mock
50
- const api = {
51
- fetch: (endpoint, options) => fetch(endpoint, options),
52
- };
53
- ```
54
-
55
- The SDK approach means:
56
- - Each mock returns one specific shape
57
- - No conditional logic in test setup
58
- - Easier to see which endpoints a test exercises
59
- - Type safety per endpoint
18
+ Use the existing public or system-boundary seam first. Add dependency injection,
19
+ an adapter, or an SDK wrapper only when production ownership or the requested
20
+ contract requires it—not only to make a test easier to mock.
@@ -1,10 +1,8 @@
1
- # Refactor Candidates
1
+ # Refactoring After GREEN
2
2
 
3
- After TDD cycle, look for:
3
+ Stop when GREEN code is clear and local. Refactor only when the current change
4
+ introduced concrete duplication, confusion, or misplaced ownership and the edit
5
+ reduces total complexity.
4
6
 
5
- - **Duplication** → Extract function/class
6
- - **Long methods** → Break into private helpers (keep tests on public interface)
7
- - **Shallow modules** → Combine or deepen
8
- - **Feature envy** → Move logic to where data lives
9
- - **Primitive obsession** → Introduce value objects
10
- - **Existing code** the new code reveals as problematic
7
+ Keep it local. Do not add helpers, classes, value objects, deeper modules, or
8
+ unrelated cleanup from pattern preference alone. Rerun affected tests.