codex-orchestrator 2.0.3 → 2.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/CHANGELOG.md +28 -427
  2. package/README.md +135 -37
  3. package/dist/src/index.d.ts +1 -1
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/v2/adapters/gh-issue-adapter.d.ts.map +1 -1
  6. package/dist/src/v2/adapters/gh-issue-adapter.js +6 -7
  7. package/dist/src/v2/adapters/gh-issue-adapter.js.map +1 -1
  8. package/dist/src/v2/cli-contract.d.ts +3 -3
  9. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  10. package/dist/src/v2/cli-contract.js +1 -3
  11. package/dist/src/v2/cli-contract.js.map +1 -1
  12. package/dist/src/v2/cli.d.ts +24 -0
  13. package/dist/src/v2/cli.d.ts.map +1 -0
  14. package/dist/src/v2/{candidate-cli.js → cli.js} +18 -29
  15. package/dist/src/v2/cli.js.map +1 -0
  16. package/dist/src/v2/code-review-report.d.ts +1 -1
  17. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  18. package/dist/src/v2/code-review-report.js +2 -2
  19. package/dist/src/v2/code-review-report.js.map +1 -1
  20. package/dist/src/v2/codex-process.d.ts.map +1 -1
  21. package/dist/src/v2/codex-process.js +12 -1
  22. package/dist/src/v2/codex-process.js.map +1 -1
  23. package/dist/src/v2/config.d.ts +2 -2
  24. package/dist/src/v2/config.d.ts.map +1 -1
  25. package/dist/src/v2/config.js.map +1 -1
  26. package/dist/src/v2/contained-report-operation.d.ts +2 -2
  27. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  28. package/dist/src/v2/contained-report-operation.js +1 -1
  29. package/dist/src/v2/contained-report-operation.js.map +1 -1
  30. package/dist/src/v2/containment.d.ts +4 -0
  31. package/dist/src/v2/containment.d.ts.map +1 -1
  32. package/dist/src/v2/containment.js +9 -0
  33. package/dist/src/v2/containment.js.map +1 -1
  34. package/dist/src/v2/direct-delivery.d.ts +5 -10
  35. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  36. package/dist/src/v2/direct-delivery.js +25 -90
  37. package/dist/src/v2/direct-delivery.js.map +1 -1
  38. package/dist/src/v2/proof-report.d.ts.map +1 -1
  39. package/dist/src/v2/proof-report.js +55 -29
  40. package/dist/src/v2/proof-report.js.map +1 -1
  41. package/dist/src/v2/run-issue.d.ts +6 -9
  42. package/dist/src/v2/run-issue.d.ts.map +1 -1
  43. package/dist/src/v2/run-issue.js +98 -43
  44. package/dist/src/v2/run-issue.js.map +1 -1
  45. package/dist/src/v2/run-store.d.ts +4 -4
  46. package/dist/src/v2/run-store.d.ts.map +1 -1
  47. package/dist/src/v2/run-store.js +25 -40
  48. package/dist/src/v2/run-store.js.map +1 -1
  49. package/dist/src/v2/runtime.d.ts +3 -3
  50. package/dist/src/v2/runtime.d.ts.map +1 -1
  51. package/dist/src/v2/runtime.js +125 -47
  52. package/dist/src/v2/runtime.js.map +1 -1
  53. package/dist/src/v2/setup-cli.d.ts.map +1 -1
  54. package/dist/src/v2/setup-cli.js +4 -11
  55. package/dist/src/v2/setup-cli.js.map +1 -1
  56. package/dist/src/v2/setup-runtime.d.ts.map +1 -1
  57. package/dist/src/v2/setup-runtime.js +1 -61
  58. package/dist/src/v2/setup-runtime.js.map +1 -1
  59. package/dist/src/v2/setup-store.d.ts +0 -5
  60. package/dist/src/v2/setup-store.d.ts.map +1 -1
  61. package/dist/src/v2/setup-store.js +3 -106
  62. package/dist/src/v2/setup-store.js.map +1 -1
  63. package/dist/src/v2/setup.d.ts +6 -46
  64. package/dist/src/v2/setup.d.ts.map +1 -1
  65. package/dist/src/v2/setup.js +11 -293
  66. package/dist/src/v2/setup.js.map +1 -1
  67. package/dist/src/v2/workflow-assets.d.ts +19 -11
  68. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  69. package/dist/src/v2/workflow-assets.js +132 -40
  70. package/dist/src/v2/workflow-assets.js.map +1 -1
  71. package/docs/deep-dive.md +272 -56
  72. package/internal-workflow/docs/agents/bugfix-quality-gate.md +11 -0
  73. package/internal-workflow/docs/agents/coding-skill-routing.md +116 -196
  74. package/internal-workflow/docs/agents/review-gates.md +32 -39
  75. package/internal-workflow/docs/agents/review-protocol.md +75 -147
  76. package/internal-workflow/evals/coding-skill-evals.json +66 -0
  77. package/internal-workflow/manifest.json +1 -1
  78. package/internal-workflow/operations/acceptance-proof/SKILL.md +7 -1
  79. package/internal-workflow/operations/ambiguity-review/SKILL.md +2 -0
  80. package/internal-workflow/operations/code-review/SKILL.md +21 -1
  81. package/internal-workflow/operations/implementation/SKILL.md +22 -1
  82. package/internal-workflow/operations/spec-author/SKILL.md +10 -1
  83. package/internal-workflow/operations/spec-review/SKILL.md +10 -1
  84. package/internal-workflow/operations/triage/SKILL.md +10 -1
  85. package/internal-workflow/schemas/code-review-v1.json +1 -1
  86. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  87. package/internal-workflow/skills/agent-auto/SKILL.md +6 -1
  88. package/internal-workflow/skills/code-debugger/SKILL.md +122 -0
  89. package/internal-workflow/skills/code-debugger/agents/openai.yaml +7 -0
  90. package/internal-workflow/skills/code-review/SKILL.md +33 -11
  91. package/internal-workflow/skills/code-review/references/cleanup-lens.md +52 -0
  92. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +15 -6
  93. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +2 -2
  94. package/internal-workflow/skills/implementation-spec-review/SKILL.md +108 -204
  95. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +24 -0
  96. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +93 -0
  97. package/internal-workflow/skills/small-task-implementer/SKILL.md +15 -8
  98. package/internal-workflow/skills/spec-implementer/SKILL.md +101 -172
  99. package/internal-workflow/skills/spec-implementer/evals/evals.json +30 -0
  100. package/internal-workflow/skills/spec-implementer/references/review-loop.md +94 -0
  101. package/internal-workflow/skills/tdd/SKILL.md +15 -2
  102. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  103. package/package.json +9 -6
  104. package/dist/src/v2/adapters/target-activity-fence.d.ts +0 -23
  105. package/dist/src/v2/adapters/target-activity-fence.d.ts.map +0 -1
  106. package/dist/src/v2/adapters/target-activity-fence.js +0 -249
  107. package/dist/src/v2/adapters/target-activity-fence.js.map +0 -1
  108. package/dist/src/v2/candidate-cli.d.ts +0 -26
  109. package/dist/src/v2/candidate-cli.d.ts.map +0 -1
  110. package/dist/src/v2/candidate-cli.js.map +0 -1
  111. package/dist/src/v2/legacy-cutover.d.ts +0 -52
  112. package/dist/src/v2/legacy-cutover.d.ts.map +0 -1
  113. package/dist/src/v2/legacy-cutover.js +0 -87
  114. package/dist/src/v2/legacy-cutover.js.map +0 -1
  115. package/internal-workflow/docs/agents/artifact-review-loop.md +0 -267
  116. package/internal-workflow/docs/agents/implementation-review-loop.md +0 -302
  117. package/internal-workflow/operations/cleanup-review/SKILL.md +0 -3
  118. package/internal-workflow/operations/spec-implementation/SKILL.md +0 -3
  119. package/internal-workflow/profiles/implementer_deep.toml +0 -9
  120. package/internal-workflow/profiles/researcher_standard.toml +0 -9
  121. package/internal-workflow/profiles/reviewer_fast.toml +0 -9
  122. package/internal-workflow/skills/cleanup-review/SKILL.md +0 -84
  123. package/internal-workflow/skills/cleanup-review/agents/openai.yaml +0 -6
  124. package/internal-workflow/skills/codebase-design/DEEPENING.md +0 -35
  125. package/internal-workflow/skills/codebase-design/DESIGN-IT-TWICE.md +0 -50
  126. package/internal-workflow/skills/codebase-design/SKILL.md +0 -82
  127. package/internal-workflow/skills/codebase-design/agents/openai.yaml +0 -6
  128. package/internal-workflow/skills/research/SKILL.md +0 -107
  129. package/internal-workflow/skills/research/agents/openai.yaml +0 -6
  130. package/internal-workflow/skills/ui-evidence-proof/SKILL.md +0 -123
  131. package/internal-workflow/skills/ui-evidence-proof/agents/openai.yaml +0 -6
@@ -5,193 +5,122 @@ description: "Executes approved specs continuously with honest checklist updates
5
5
 
6
6
  # Spec Implementer
7
7
 
8
- Execute an approved implementation spec. Your job is to carry out the chosen spec, keep its checklist honest, and stop at the right boundaries. Do not redesign the work unless the spec or repo reality proves a blocker.
9
-
10
- This skill is standalone by default: it executes only an approved spec that the
11
- user has chosen to run. `$tickets-orchestrator` may invoke it inline at root for
12
- an accepted compact or standard ticket spec inside user-authorized orchestration;
13
- that does not authorize unrelated specs or broader delivery scope.
14
-
15
- When a spec contains a Contract Test Ledger, treat it as part of the execution contract. The shared reference is `../../docs/agents/contract-test-ledger.md`.
16
-
17
- All implementation review checkpoints and final review gates use
18
- `../../docs/agents/implementation-review-loop.md` as their caller-facing owner.
19
- This skill must not create a per-slice review flow or retry loop.
20
-
21
- ## Spec Modes
22
-
23
- - **Compact specs:** execute directly as one continuous implementation flow with lightweight phase checkpoints. `compact` describes document density, not implementation size or risk. Use review/signoff gates only when the spec, repo policy, or change risk requires them.
24
- - **Full specs:** execute the same phase flow, but treat `Risk Controls`, task-specific `Halt Conditions`, and validation proof as hard constraints. Do not invent extra process just because the spec is full.
25
- - **Multi-agent specs:** follow the integrator contract exactly. If write scopes are not perfectly disjoint, stop before spawning workers.
26
-
27
- ## Core Rules
28
-
29
- 1. Follow the spec literally. Do not broaden scope, add cleanup, or re-plan unless a blocker is proven.
30
- 2. Treat the spec checklist as the execution ledger.
31
- 3. Update checklist items during implementation. For compact specs, update at natural checkpoints and phase exits. For full specs, or when the spec says so, update completed leaf items immediately.
32
- 4. Do not save all checklist updates only for the final response.
33
- 5. Re-read the current phase before moving on and reconcile already-completed unchecked items.
34
- 6. If a step is blocked, leave it unchecked and record one short `Blocked:` note with the concrete reason.
35
- 7. Treat Preconditions as hard blockers. Do not start a phase until they are satisfied, explicitly not applicable, or blocked.
36
- 8. If the saved spec contains unresolved template text, placeholders, alternative commands, or pseudo-paths, stop and escalate.
37
- 9. Honor Protected Paths and Rejected Approaches exactly as written.
38
- 10. Apply the `$codebase-design` lens only when ownership or a public Module Interface or Seam changes. For any new private helper, run the deletion test and keep it only if it improves locality or leverage, without activating architecture workflow.
39
- 11. Do not add pass-through modules, one-adapter seams, or test-only helpers unless the spec explicitly approves them.
40
- 12. Add comments/docblocks only where the spec explicitly requires them.
41
-
42
- ## Before Editing
43
-
44
- - Read the complete frontmatter before announcing execution strategy. Confirm the spec path, status, `spec_mode`, `implementation_size`, `review_profile`, and `expected_repositories`; infer a missing classification from evidence without rewriting the approved design.
45
- - Identify whether it is compact, full, or multi-agent. Do not call a compact spec full or equate compact with small.
46
- - Check required services, env vars, fixtures, repo state, and prerequisite issues.
47
- - Confirm the first phase targets exist as described.
48
- - Confirm validation commands are executable or explicitly not applicable.
49
- - If the spec has a Contract Test Ledger, confirm each reached invariant has a first test/proof or a concrete blocked reason before implementation.
50
- - If the spec has `Review Checkpoints` or `Review Focus`, keep only checkpoints whose target becomes stable before later slices. If later work will touch the same files, owners, or contracts, fold that coverage into the final parallel review instead of reviewing an unstable slice.
51
- - Resolve the implementation review profile and plan mandatory final coverage before launching any reviewer. Do not manufacture an early checkpoint merely because the profile is high.
52
- - Do not write `## Implementation Review State` during ordinary preflight or implementation. Immediately before the first actual reviewer launch, persist the short Review Plan and pending launch required by the Module; then update it after every usable reviewer result, repair batch, closure, waiver, or terminal outcome.
53
- - If the spec has `Final Handoff Requirements`, treat them as the final response contract. For medium/high-risk specs without explicit requirements, prepare the standard Final Risk Handoff anyway.
54
- - For full specs, read `Risk Controls` before editing and translate each applicable control into a concrete execution constraint.
55
- - Use `Write Scope Summary` when present as an audit aid. If it is absent, rely on phase targets unless the write set is ambiguous.
56
- - Stop if exact execution would require guessing.
8
+ Execute one approved implementation spec continuously. Keep its checklist
9
+ truthful and stop only at a real authority, evidence, safety, or explicit pause
10
+ boundary. Do not redesign approved scope.
11
+
12
+ Read before editing:
13
+
14
+ - the complete approved spec and applicable repository instructions;
15
+ - `references/review-loop.md` for approved-spec review ownership;
16
+ - `../../docs/agents/contract-test-ledger.md` only when the spec contains
17
+ material contract invariants.
18
+
19
+ ## Modes
20
+
21
+ Compact and full specs use the same direct phase flow. `compact` describes
22
+ document density, not implementation size or risk. Full mode adds only the
23
+ concrete `Risk Controls`, stop conditions, validation, or coordination contract
24
+ already present in the approved spec; it does not add review or reporting
25
+ ceremony by itself.
26
+
27
+ Use multiple agents only when the spec defines perfectly disjoint write scopes
28
+ and one integrator. Otherwise execute single-agent.
29
+
30
+ ## Preflight
31
+
32
+ 1. Confirm `status`, `spec_mode`, `implementation_size`, `review_profile`,
33
+ repository count, authority, scope, and exclusions.
34
+ 2. Confirm first-phase targets, required services/env/data/fixtures, commands,
35
+ observable proof, protected paths, and rejected approaches.
36
+ 3. Stop if execution needs invented paths, symbols, contracts, commands, or
37
+ product decisions; if reality differs only in a bounded technical detail,
38
+ resolve it from repository evidence and record the adjustment.
39
+ 4. Keep an intermediate review checkpoint only when the spec explicitly names
40
+ a stable high-risk slice that later work will not invalidate. Move every
41
+ ordinary or unstable checkpoint to final review.
42
+ 5. Do not create `## Implementation Review State` yet.
43
+
44
+ ## Implementation
45
+
46
+ For each phase:
47
+
48
+ 1. Re-read its scope and preconditions.
49
+ 2. Implement narrow vertical behavior slices through `$tdd` when applicable.
50
+ 3. Update reached checklist and Contract Test Ledger items at natural
51
+ checkpoints; never save all updates for the end.
52
+ 4. Run the phase's targeted exit proof.
53
+ 5. Record a short `Blocked:` note for any item that cannot complete.
54
+ 6. Continue immediately when the exit proof passes and no stop condition
55
+ applies.
56
+
57
+ Do not add cleanup, comments, helpers, abstractions, retries, flags, fallbacks,
58
+ or compatibility paths outside the spec. A small implementation adjustment is
59
+ allowed only when it preserves approved behavior and is supported by current
60
+ repository evidence.
57
61
 
58
62
  ## Git Checkpoints
59
63
 
60
- Default to `none` and begin implementation without checkpoint ceremony. Mention the strategy once only when it materially affects delivery. Invoking `$spec-implementer` does not by itself authorize commits.
64
+ Default to no commits. `$spec-implementer` does not authorize Git writes.
61
65
 
62
- Choose `per-slice` only when commits are explicitly authorized by the user or approved spec, slice diffs are truly isolated, and checkpoints materially improve recovery or handoff safety. Valid reasons are:
66
+ Use per-slice commits only when explicitly authorized and they materially help
67
+ a disjoint multi-agent handoff, planned cross-session pause, or approved
68
+ rollback boundary. Require a passed slice proof, stage only owned paths, follow
69
+ `$commit`, and never push without separate authority. File/slice count and risk
70
+ profile alone do not justify checkpoints.
63
71
 
64
- - a multi-agent merge/handoff boundary
65
- - an explicitly planned pause or continuation in another session
66
- - a destructive or rollback boundary whose isolated commit is part of the approved safety plan
72
+ ## Review And Validation
67
73
 
68
- Use `none` for ordinary single-agent execution, including compact/high specs, overlapping slices, and continuous work in one session. Slice count, file count, or review profile alone never justifies commits.
74
+ Run targeted behavior tests and the smallest affected integration checks first.
75
+ Use a full repository suite only when the spec or repository policy requires
76
+ it, broad contract fan-out cannot be isolated, or the task is genuinely high.
69
77
 
70
- At each checkpoint:
78
+ Run final `$code-review` only when the spec, repository policy, or
79
+ `../../docs/agents/review-gates.md` applies. Ordinary medium work gets one
80
+ `reviewer_standard` Full review on the settled diff. High gets two disjoint
81
+ `reviewer_deep` lenses. Cleanup stays inside spec/standards review.
71
82
 
72
- - require a passed exit gate and applicable slice review, then reconcile and include tracked checklist/ledger updates
73
- - inspect the full diff, stage only slice-owned paths or hunks, follow `$commit` safety rules, and verify the hash
74
- - treat the applicable slice review as the pre-commit gate; final review still runs at the end
83
+ Immediately before the first reviewer launch, create only the minimal
84
+ `## Implementation Review State` required by `references/review-loop.md`.
85
+ Reconcile a recorded live session before replacement after interruption.
75
86
 
76
- If isolation later becomes unsafe, skip and record the reason. Never commit RED state, failed validation, unresolved findings, discovery-only work, or a partial slice. Never amend a checkpoint commit; put later fixes in a new commit. Never push unless explicitly requested. Report the strategy and slice-to-commit mapping, or the reason no checkpoints were created.
87
+ Repair compatible findings once and rerun affected validation. Coordinator
88
+ verification closes ordinary medium/low behavior-preserving repairs. Use
89
+ Closure only for critical/high, protected trust/data/concurrency/shared-contract
90
+ impact, or invalidated mandatory coverage. Do not restart broad Full review
91
+ unless the repair actually invalidated its coverage.
77
92
 
78
- ## Phase Workflow
79
-
80
- For every phase:
81
-
82
- - Confirm phase dependencies and preconditions.
83
- - Execute only the steps assigned to that phase.
84
- - Update the spec checklist according to its mode.
85
- - Update any reached Contract Test Ledger rows as planned -> red -> green, or blocked with the missing seam/proof.
86
- - Run the phase exit gate.
87
- - If a `Review Checkpoint` applies after this phase, execute it through the
88
- Module only when the target is settled and later slices do not invalidate its
89
- files/contracts. Otherwise record that its lenses moved to final coverage and
90
- continue without launching an unstable review.
91
- - Reconcile unchecked items for the current phase.
92
- - Re-check applicable `Risk Controls` before leaving the phase.
93
- - Check whether the phase introduced shallow modules, duplicated source-of-truth logic, or tests coupled to implementation details; fix only when inside approved scope, otherwise report it.
94
- - Run the repo architecture check when available, applicable, and required by the spec or repo policy.
95
- - If `per-slice` applies and this phase completes an implementation slice, create and verify its checkpoint commit before continuing.
96
- - Continue to the next phase only when the exit gate passes and no stop condition applies.
97
- - If the phase exit gate says `User Pause: Required`, stop and wait for the user's explicit command.
98
-
99
- ## Review And Signoff
100
-
101
- Do not run a dedicated review subagent after every phase by default. Do run one at explicit `Review Checkpoints`; these are risk gates, not optional status updates.
102
-
103
- Before every reviewer launch, apply the Module's launch and reconciliation
104
- rules to the persisted `## Implementation Review State`; do not restate or
105
- replace those rules in this skill.
106
-
107
- Require final `$code-review` coverage when any of these apply:
108
-
109
- - the spec explicitly requires it
110
- - the repo policy requires it
111
- - the change is medium or large
112
- - the change touches multiple runtime files or shared behavior
113
- - the change touches API contracts, DTOs, schemas, persistence, auth, permissions, payments, caching, concurrency, background jobs, or shared state
114
-
115
- When `$code-review` is required for a checkpoint or final gate, keep orchestration at root so `$code-review` can launch the profile-selected reviewer topology. Invoking `$spec-implementer` authorizes that review; if the required role is unavailable, report the gate as unavailable/blocked instead of self-certifying it.
93
+ ## Stop Conditions
116
94
 
117
- Use one final `$code-review` wave after the implementation and validation settle.
118
- For `simple` and `medium`, one reviewer covers both lenses. For `high`, launch
119
- the correctness and spec/standards reviewers in parallel; the spec/standards
120
- lens includes bounded cleanup. Run separate `$cleanup-review` only when the
121
- user, approved source, or repo policy names a concrete evidenced reason that
122
- cannot fit that lens; size or risk labels alone are insufficient. Integrate safe
123
- fixes and rerun relevant validation before continuing.
124
- Before launching a fresh final reviewer, reconcile the settled revision against
125
- the Review Plan. Stop when an approved Full or Closure already covers every
126
- mandatory final lens; otherwise run `$code-review` only for the uncovered
127
- lenses. A `cleanup-only` result never substitutes for correctness or
128
- spec/standards coverage.
95
+ Stop and report the exact blocker when:
129
96
 
130
- For compact low-risk specs, final validation plus checklist reconciliation is enough unless the spec says otherwise.
97
+ - a precondition, source contract, required proof, or protected path differs
98
+ materially from the approved spec;
99
+ - exact execution requires a new product, scope, ownership, or risky trade-off
100
+ decision;
101
+ - required validation or reviewer is unavailable and no approved substitute
102
+ exists;
103
+ - multi-agent scopes overlap or integration ownership is missing;
104
+ - an explicit user pause or halt condition is reached.
131
105
 
132
- Treat review feedback as mandatory remediation when it is grounded in code or the spec. If review reveals ambiguity that cannot be resolved from the spec, code, or docs, pause and ask the user.
106
+ Do not stop merely because the work is broad, review took time, or a medium/low
107
+ finding required one repair.
133
108
 
134
- Apply the Module's convergence and stop rules after every usable result and
135
- before every new launch.
109
+ ## Completion
136
110
 
137
- ## Multi-Agent Execution
111
+ Complete only when reached checklist items are reconciled, affected proof and
112
+ required review pass, protected paths and rejected approaches remain intact,
113
+ and every unfinished item has a concrete status.
138
114
 
139
- - Use multiple agents only when the spec has explicit disjoint write scopes.
140
- - Keep one integrator responsible for merge sequencing, handoff checks, final validation, and checklist reconciliation.
141
- - Respect exclusive write scopes, handoff artifacts, forbidden overlap, and merge order exactly as written.
142
- - Never let two agents edit the same file, generated artifact, schema, source-of-truth rule, or shared contract at the same time.
143
- - If the spec lacks a clear integrator contract, execute single-agent or stop and ask for clarification.
115
+ For ordinary medium work report only:
144
116
 
145
- ## Stop Conditions
117
+ - behavior/contract implemented;
118
+ - review result and repaired/open findings;
119
+ - affected validation;
120
+ - skipped checks and residual risk;
121
+ - changed files and any authorized commits.
146
122
 
147
- Stop immediately and escalate if:
148
-
149
- - a required precondition cannot be satisfied exactly
150
- - a required file, symbol, command, dependency, or interface differs from the spec
151
- - the saved spec contains unresolved template text or alternative commands
152
- - completing the task would require touching a Protected Path or using a Rejected Approach
153
- - validation cannot prove the intended behavior with available repo context
154
- - implementation would require unapproved scope, abstraction, migration, compatibility logic, or cleanup
155
- - a `Risk Controls` rule would be violated or is contradicted by repo reality
156
- - review exposes a real ambiguity that would require guessing
157
- - a user pause is required and the user has not explicitly said to proceed
158
-
159
- ## Completion Standard
160
-
161
- Do not mark the task complete until:
162
-
163
- - completed checklist items are checked off according to the spec mode
164
- - every remaining unchecked item is blocked, intentionally unfinished, not applicable, or halted by an explicit stop condition
165
- - every reached phase exit gate has passed
166
- - applicable `Risk Controls` remained satisfied
167
- - reached Contract Test Ledger rows are green or explicitly blocked with evidence
168
- - validation commands and behavior proof have run, or skipped checks have a concrete reason
169
- - required review/signoff gates have run and grounded findings are fixed or blocked with evidence
170
- - the whole-spec Review Plan, review-pass history, and stable defect lifecycle remain
171
- consistent with `implementation-review-loop.md`
172
- - protected paths remained untouched and rejected approaches were not used
173
- - required comments/docblocks were added only where the spec demanded them
174
- - the chosen Git checkpoint strategy was followed, and every created or skipped checkpoint was recorded
175
- - final user handoff is allowed by the spec
176
-
177
- ## Final Risk Handoff
178
-
179
- For medium/high-risk specs, the final chat response must include a compact `Final Risk Handoff` block. Do not make the user ask for this separately, and do not replace it with a generic summary.
180
-
181
- Include:
182
-
183
- - **Contract implemented:** the one behavior/contract delivered, in user-facing terms.
184
- - **High-risk checkpoints:** each required checkpoint, review result, fixed findings, and any stop/continue decision.
185
- - **Main invariants proved:** the key Contract Test Ledger rows or equivalent proofs and their status.
186
- - **Code-review findings:** high/critical findings fixed, remaining medium/low findings, or `none`.
187
- - **Fixes after review:** concrete fixes made because of cleanup/code review, or `none`.
188
- - **Validation:** exact commands/proofs that passed.
189
- - **Skipped checks:** skipped or blocked checks with concrete reasons.
190
- - **Residual risks:** accepted remaining risks or `none`.
191
- - **Checkpoint commits:** slice-to-commit mapping or `none`; do not narrate hypothetical checkpoints that were never authorized or attempted.
192
- - **Implementation reviews:** profile, total review passes, Full/Closure count,
193
- mandatory coverage, verified defect IDs, accepted-risk IDs with authority and
194
- reason, and open defect IDs.
195
- - **Files by role:** state owner, orchestration, side effects, UI/projection, tests, docs/copy, as applicable.
196
-
197
- Only create a separate report file when the spec requires it or the work is broad enough that chat would lose important evidence, such as multi-agent execution, multiple review passes with findings, skipped live checks, production validation, or handoff to another person. Otherwise keep the spec checklist/ledger as the durable artifact and the final response as the concise decision packet.
123
+ For high, actual Closure, accepted risk, interrupted recovery, or multi-agent
124
+ delivery, add the relevant invariants, reviewer coverage, defect IDs, session
125
+ recovery, and handoff ownership. Do not create a separate report file unless
126
+ the spec or the complexity of that exceptional handoff requires it.
@@ -0,0 +1,30 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "spec-implementer",
4
+ "cases": [
5
+ {
6
+ "id": "state-created-at-launch",
7
+ "prompt": "Execute an approved medium spec whose final review has not started yet.",
8
+ "expected": ["do not create review state during implementation", "persist minimal state immediately before reviewer launch"],
9
+ "forbidden": ["create per-slice review bookkeeping"]
10
+ },
11
+ {
12
+ "id": "medium-one-final-review",
13
+ "prompt": "Execute a normal medium spec with several vertical slices and no stable high-risk checkpoint.",
14
+ "expected": ["implement continuously", "run one final reviewer_standard on the settled diff"],
15
+ "forbidden": ["review after every slice", "run a full repository suite from file count"]
16
+ },
17
+ {
18
+ "id": "closure-stays-affected",
19
+ "prompt": "Final review finds one high defect in a shared schema and several ordinary low findings.",
20
+ "expected": ["repair once", "Closure verifies only the affected high-risk contract"],
21
+ "forbidden": ["restart all Full reviewers"]
22
+ },
23
+ {
24
+ "id": "resume-live-reviewer",
25
+ "prompt": "Resume after a reviewer poll timed out while the recorded session is still live.",
26
+ "expected": ["reconcile the existing session"],
27
+ "forbidden": ["mark it failed from timeout alone", "launch a duplicate reviewer"]
28
+ }
29
+ ]
30
+ }
@@ -0,0 +1,94 @@
1
+ # Approved Spec Implementation Review Loop
2
+
3
+ This reference owns review orchestration for `$spec-implementer`. Read it only
4
+ when executing an approved implementation spec. Shared Full/Closure and defect
5
+ mechanics live in `../../../docs/agents/review-protocol.md`.
6
+
7
+ Direct work and deterministic issue delivery use normal TDD and review gates;
8
+ they must not create Implementation Review State.
9
+
10
+ ## Authority
11
+
12
+ Only an approved implementation spec may own `## Implementation Review State`.
13
+ PRDs, tickets, architecture notes, and chat summaries are not review-state
14
+ owners. A substantive spec change returns through artifact review before
15
+ implementation continues.
16
+
17
+ Use the spec's `review_profile`; if absent, infer it from current evidence:
18
+
19
+ - `simple`: narrow change with direct proof;
20
+ - `medium`: default for ordinary implementation;
21
+ - `high`: material failure consequence plus an uncertainty amplifier.
22
+
23
+ Implementation evidence may raise but never lower the approved profile.
24
+
25
+ ## Default Review Shape
26
+
27
+ Implement continuously through vertical slices. Validate each affected behavior
28
+ and run one final review on the settled diff when the gate applies.
29
+
30
+ - `simple`: one `reviewer_fast` when review is required.
31
+ - `medium`: one `reviewer_standard`, one bounded final Full, no intermediate
32
+ checkpoint by default.
33
+ - `high`: two parallel `reviewer_deep` Full reviews with disjoint correctness
34
+ and spec/standards lenses.
35
+
36
+ Add an intermediate checkpoint only when the approved spec explicitly names a
37
+ stable high-risk slice whose review will remain valid after later work. Do not
38
+ review unstable intermediate diffs or create per-slice review cycles.
39
+
40
+ Cleanup stays inside the spec/standards lens. A concrete simplification risk may
41
+ amplify that lens; size and profile labels alone do not create another gate.
42
+
43
+ ## Minimal Durable State
44
+
45
+ Do not create review state during preflight or implementation. Immediately
46
+ before the first actual reviewer launch, persist:
47
+
48
+ - profile, authority path, settled target revision, and assigned lenses;
49
+ - launch ID, reviewer/session handle, lineage, and `pending | completed | failed`;
50
+ - returned findings, repair revision, affected validation, and Closure need.
51
+
52
+ Write `pending` before launch and reconcile that session before replacing it
53
+ after interruption or resume. A usable result becomes `completed`; an explicit
54
+ failure becomes `failed`. A poll timeout while the session remains live is not
55
+ a failure and does not authorize duplicate review.
56
+
57
+ Record extended lineage/session history only for `high`, a real intermediate
58
+ checkpoint, actual Closure, accepted risk, or interrupted recovery. Normal
59
+ medium execution does not keep epochs, pass thresholds, activation counters, or
60
+ per-slice handoff bookkeeping.
61
+
62
+ ## Findings And Closure
63
+
64
+ Root aggregates findings, repairs compatible defects once, and reruns only
65
+ affected validation. Coordinator verification closes ordinary medium/low
66
+ behavior-preserving findings after confirming the repair matches the failure.
67
+
68
+ Use shared-protocol Closure only for critical/high defects, protected
69
+ trust/data/concurrency/shared API impact, or invalidated mandatory coverage.
70
+ Closure stays with the affected reviewer lineage and repaired targets. Start a
71
+ new Full only when the repair invalidated mandatory-lens coverage.
72
+
73
+ Do not repeat review without a material change in target, evidence, repair, or
74
+ source decision. Stop and surface the actual decision or evidence blocker when
75
+ no progress is possible.
76
+
77
+ ## Completion
78
+
79
+ Run gates in this order:
80
+
81
+ 1. affected behavior and integration validation;
82
+ 2. applicable final code review;
83
+ 3. Closure only when triggered;
84
+ 4. repository architecture/build/smoke gates required by policy or the spec;
85
+ 5. delivery actions explicitly authorized by the user or workflow.
86
+
87
+ Return `Approved` only for the final settled revision when mandatory lenses and
88
+ validation are complete and shared protocol state is clear. `Waived` records
89
+ skipped coverage but is not approval. `Blocked` requires a concrete authority,
90
+ evidence, reviewer, or convergence blocker—not elapsed time or review count.
91
+
92
+ For normal medium work report only profile, review result, repaired/open
93
+ findings, affected validation, skipped checks, and residual risk. Add extended
94
+ session/defect accounting only when the exceptional state above exists.
@@ -1,12 +1,25 @@
1
1
  ---
2
2
  name: tdd
3
- description: Test-driven development policy gate for implementation, bugfix, and new feature work unless the user explicitly opts out. Use before planning or editing code to shape the first behavior proof, and when the user mentions red-green-refactor, integration tests, or test-first development.
3
+ description: Test-driven development for changes that alter observable behavior, have a natural public test seam, and can produce a meaningful failing test before implementation. Use after the global TDD Fit Gate passes, or when the user explicitly requests red-green-refactor, test-first development, or TDD.
4
4
  ---
5
5
 
6
6
  # Test-Driven Development
7
7
 
8
8
  Use short vertical RED -> GREEN cycles. Make each test prove observable behavior through the same public seam real callers use.
9
9
 
10
+ ## Fit
11
+
12
+ Use this skill only when the change alters observable behavior, a natural public
13
+ seam exists, and the pre-change test will fail for the intended behavioral
14
+ reason. If an implicit activation fails this gate, stop the TDD route and use
15
+ existing regression tests plus affected validation. For mixed tasks, apply TDD
16
+ only to the behavioral slice.
17
+
18
+ Behavior-preserving cleanup, dead-code deletion, documentation, copy,
19
+ formatting, generated assets, package maintenance, simple config, builds, and
20
+ read-only work do not need TDD. Absence and architecture guards added after a
21
+ cleanup are validation, not RED proofs.
22
+
10
23
  ## Core Contract
11
24
 
12
25
  - Lock expected behavior from the request, specification, design, bug report, or existing product behavior before changing implementation.
@@ -24,7 +37,7 @@ Read [tests.md](tests.md) when choosing or reviewing test shape. Read [mocking.m
24
37
  3. Select the public seam where callers observe each behavior.
25
38
  4. Ask the user only when the seam changes the public contract, product intent is unclear, or behavior priorities materially conflict.
26
39
  5. For contract-risk changes, create or update the shared [Contract Test Ledger](../../docs/agents/contract-test-ledger.md) and map each invariant to its first failing test or observable proof.
27
- 6. If no natural public seam exists, consult [interface-design.md](interface-design.md) instead of testing internals.
40
+ 6. If no natural public seam exists, stop the TDD route. Consult [interface-design.md](interface-design.md) only when changing the interface is itself required by the task.
28
41
 
29
42
  For UI behavior, define proof at the rendered seam: visible content and order, interaction result, semantics, or screenshot when layout direction or scrolling matters.
30
43
 
@@ -1,6 +1,6 @@
1
1
  interface:
2
2
  display_name: "Test-Driven Development"
3
- short_description: "Apply behavior-first red-green development"
4
- default_prompt: "Use $tdd to define the public test seam and implement this behavior through red-green cycles."
3
+ short_description: "Use TDD only when its behavioral fit gate passes"
4
+ default_prompt: "Use $tdd after confirming an observable behavior change, a public test seam, and a meaningful pre-change failure."
5
5
  policy:
6
6
  allow_implicit_invocation: true
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "codex-orchestrator",
3
- "version": "2.0.3",
3
+ "version": "2.0.4",
4
4
  "description": "Reusable GitHub Issues runner for Codex.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -13,7 +13,7 @@
13
13
  }
14
14
  },
15
15
  "bin": {
16
- "codex-orchestrator": "dist/src/v2/candidate-cli.js"
16
+ "codex-orchestrator": "dist/src/v2/cli.js"
17
17
  },
18
18
  "files": [
19
19
  "dist/src",
@@ -31,10 +31,10 @@
31
31
  "url": "git+https://github.com/SergiiMytakii/codex-orchestrator.git"
32
32
  },
33
33
  "scripts": {
34
- "orchestrator:doctor": "npm run build --silent && node dist/src/v2/candidate-cli.js doctor --target \"$PWD\"",
35
- "orchestrator:status": "npm run build --silent && node dist/src/v2/candidate-cli.js status --target \"$PWD\"",
36
- "orchestrator:daemon": "npm run build --silent && node dist/src/v2/candidate-cli.js daemon --target \"$PWD\"",
37
- "orchestrator:daemon:once": "npm run build --silent && node dist/src/v2/candidate-cli.js daemon --target \"$PWD\" --once",
34
+ "orchestrator:doctor": "npm run build --silent && node dist/src/v2/cli.js doctor --target \"$PWD\"",
35
+ "orchestrator:status": "npm run build --silent && node dist/src/v2/cli.js status --target \"$PWD\"",
36
+ "orchestrator:daemon": "npm run build --silent && node dist/src/v2/cli.js daemon --target \"$PWD\"",
37
+ "orchestrator:daemon:once": "npm run build --silent && node dist/src/v2/cli.js daemon --target \"$PWD\" --once",
38
38
  "clean": "node --input-type=module -e \"import { rmSync } from 'node:fs'; rmSync('dist', { recursive: true, force: true })\"",
39
39
  "build": "npm run clean --silent && tsc -p tsconfig.json",
40
40
  "test": "npm run build && node --test dist/test/*.test.js",
@@ -43,6 +43,9 @@
43
43
  "sync:workflow": "node scripts/sync-agent-auto-workflow.mjs sync --codex-home \"${CODEX_HOME:-$HOME/.codex}\" --repo-root \"$PWD\" --config scripts/agent-auto-workflow-source.json --output-root internal-workflow",
44
44
  "check:workflow": "node scripts/sync-agent-auto-workflow.mjs check --codex-home \"${CODEX_HOME:-$HOME/.codex}\" --repo-root \"$PWD\" --config scripts/agent-auto-workflow-source.json --output-root internal-workflow",
45
45
  "verify:workflow": "node scripts/sync-agent-auto-workflow.mjs verify --output-root internal-workflow",
46
+ "test:workflow:contracts": "npm run build --silent && node --test dist/test/v2-workflow-import.test.js dist/test/v2-workflow-assets.test.js dist/test/v2-runtime-assets.test.js",
47
+ "test:workflow": "npm run test:workflow:contracts --silent && node --test dist/test/v2-package-consumer.test.js",
48
+ "refresh:workflow": "npm run sync:workflow --silent && npm run check:workflow --silent && npm run verify:workflow --silent && npm run test:workflow --silent",
46
49
  "smoke:live": "node scripts/live-smoke.mjs",
47
50
  "prepack": "npm run verify:workflow --silent && npm run build --silent"
48
51
  },
@@ -1,23 +0,0 @@
1
- export type TargetActivityFenceMode = 'shared' | 'exclusive';
2
- export type TargetActivityPurpose = 'daemon' | 'claim' | 'setup' | 'preparation' | 'migration';
3
- export interface TargetActivityFenceInput {
4
- targetRoot: string;
5
- stateDir: string;
6
- mode: TargetActivityFenceMode;
7
- purpose: TargetActivityPurpose;
8
- hostId?: string;
9
- bootNonce?: string;
10
- pid?: number;
11
- now?: Date;
12
- isProcessAlive?: (pid: number) => boolean;
13
- }
14
- export interface TargetActivityFenceLease {
15
- canonicalTargetRoot: string;
16
- generation: number;
17
- metadataPath: string;
18
- release(): Promise<void>;
19
- }
20
- export declare function acquireTargetActivityFence(input: TargetActivityFenceInput): Promise<TargetActivityFenceLease>;
21
- export declare function readTargetActivityFenceGeneration(targetRoot: string, stateDir: string): Promise<number>;
22
- export declare function readCurrentBootNonce(platform?: NodeJS.Platform): Promise<string>;
23
- //# sourceMappingURL=target-activity-fence.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"target-activity-fence.d.ts","sourceRoot":"","sources":["../../../../src/v2/adapters/target-activity-fence.ts"],"names":[],"mappings":"AAmBA,MAAM,MAAM,uBAAuB,GAAG,QAAQ,GAAG,WAAW,CAAC;AAC7D,MAAM,MAAM,qBAAqB,GAAG,QAAQ,GAAG,OAAO,GAAG,OAAO,GAAG,aAAa,GAAG,WAAW,CAAC;AAE/F,MAAM,WAAW,wBAAwB;IACvC,UAAU,EAAE,MAAM,CAAC;IACnB,QAAQ,EAAE,MAAM,CAAC;IACjB,IAAI,EAAE,uBAAuB,CAAC;IAC9B,OAAO,EAAE,qBAAqB,CAAC;IAC/B,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,GAAG,CAAC,EAAE,IAAI,CAAC;IACX,cAAc,CAAC,EAAE,CAAC,GAAG,EAAE,MAAM,KAAK,OAAO,CAAC;CAC3C;AAED,MAAM,WAAW,wBAAwB;IACvC,mBAAmB,EAAE,MAAM,CAAC;IAC5B,UAAU,EAAE,MAAM,CAAC;IACnB,YAAY,EAAE,MAAM,CAAC;IACrB,OAAO,IAAI,OAAO,CAAC,IAAI,CAAC,CAAC;CAC1B;AAsBD,wBAAsB,0BAA0B,CAC9C,KAAK,EAAE,wBAAwB,GAC9B,OAAO,CAAC,wBAAwB,CAAC,CAoEnC;AAED,wBAAsB,iCAAiC,CAAC,UAAU,EAAE,MAAM,EAAE,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,CAAC,CAG7G;AAED,wBAAsB,oBAAoB,CAAC,QAAQ,GAAE,MAAM,CAAC,QAA2B,GAAG,OAAO,CAAC,MAAM,CAAC,CAiBxG"}