codex-orchestrator 2.0.3 → 2.0.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (161) hide show
  1. package/CHANGELOG.md +51 -427
  2. package/README.md +161 -37
  3. package/dist/src/index.d.ts +1 -1
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/v2/acceptance-proof.d.ts +5 -0
  6. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  7. package/dist/src/v2/acceptance-proof.js +10 -2
  8. package/dist/src/v2/acceptance-proof.js.map +1 -1
  9. package/dist/src/v2/adapters/gh-issue-adapter.d.ts.map +1 -1
  10. package/dist/src/v2/adapters/gh-issue-adapter.js +6 -7
  11. package/dist/src/v2/adapters/gh-issue-adapter.js.map +1 -1
  12. package/dist/src/v2/adapters/gh-pull-request-adapter.d.ts +7 -1
  13. package/dist/src/v2/adapters/gh-pull-request-adapter.d.ts.map +1 -1
  14. package/dist/src/v2/adapters/gh-pull-request-adapter.js +288 -0
  15. package/dist/src/v2/adapters/gh-pull-request-adapter.js.map +1 -1
  16. package/dist/src/v2/adapters/pull-requests.d.ts +69 -0
  17. package/dist/src/v2/adapters/pull-requests.d.ts.map +1 -1
  18. package/dist/src/v2/adapters/pull-requests.js +48 -0
  19. package/dist/src/v2/adapters/pull-requests.js.map +1 -1
  20. package/dist/src/v2/adapters/worktree.d.ts +1 -0
  21. package/dist/src/v2/adapters/worktree.d.ts.map +1 -1
  22. package/dist/src/v2/adapters/worktree.js +10 -1
  23. package/dist/src/v2/adapters/worktree.js.map +1 -1
  24. package/dist/src/v2/cli-contract.d.ts +3 -3
  25. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  26. package/dist/src/v2/cli-contract.js +1 -3
  27. package/dist/src/v2/cli-contract.js.map +1 -1
  28. package/dist/src/v2/cli.d.ts +33 -0
  29. package/dist/src/v2/cli.d.ts.map +1 -0
  30. package/dist/src/v2/{candidate-cli.js → cli.js} +55 -39
  31. package/dist/src/v2/cli.js.map +1 -0
  32. package/dist/src/v2/code-review-report.d.ts +1 -1
  33. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  34. package/dist/src/v2/code-review-report.js +2 -2
  35. package/dist/src/v2/code-review-report.js.map +1 -1
  36. package/dist/src/v2/codex-process.d.ts.map +1 -1
  37. package/dist/src/v2/codex-process.js +12 -1
  38. package/dist/src/v2/codex-process.js.map +1 -1
  39. package/dist/src/v2/config.d.ts +2 -3
  40. package/dist/src/v2/config.d.ts.map +1 -1
  41. package/dist/src/v2/config.js +0 -3
  42. package/dist/src/v2/config.js.map +1 -1
  43. package/dist/src/v2/contained-report-operation.d.ts +2 -2
  44. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  45. package/dist/src/v2/contained-report-operation.js +1 -1
  46. package/dist/src/v2/contained-report-operation.js.map +1 -1
  47. package/dist/src/v2/containment.d.ts +15 -2
  48. package/dist/src/v2/containment.d.ts.map +1 -1
  49. package/dist/src/v2/containment.js +43 -6
  50. package/dist/src/v2/containment.js.map +1 -1
  51. package/dist/src/v2/direct-delivery.d.ts +5 -10
  52. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  53. package/dist/src/v2/direct-delivery.js +32 -91
  54. package/dist/src/v2/direct-delivery.js.map +1 -1
  55. package/dist/src/v2/proof-report.d.ts.map +1 -1
  56. package/dist/src/v2/proof-report.js +55 -29
  57. package/dist/src/v2/proof-report.js.map +1 -1
  58. package/dist/src/v2/review-feedback-coordinator.d.ts +54 -0
  59. package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -0
  60. package/dist/src/v2/review-feedback-coordinator.js +245 -0
  61. package/dist/src/v2/review-feedback-coordinator.js.map +1 -0
  62. package/dist/src/v2/review-feedback.d.ts +127 -0
  63. package/dist/src/v2/review-feedback.d.ts.map +1 -0
  64. package/dist/src/v2/review-feedback.js +436 -0
  65. package/dist/src/v2/review-feedback.js.map +1 -0
  66. package/dist/src/v2/run-issue.d.ts +63 -9
  67. package/dist/src/v2/run-issue.d.ts.map +1 -1
  68. package/dist/src/v2/run-issue.js +798 -78
  69. package/dist/src/v2/run-issue.js.map +1 -1
  70. package/dist/src/v2/run-store.d.ts +49 -5
  71. package/dist/src/v2/run-store.d.ts.map +1 -1
  72. package/dist/src/v2/run-store.js +138 -44
  73. package/dist/src/v2/run-store.js.map +1 -1
  74. package/dist/src/v2/runtime.d.ts +40 -3
  75. package/dist/src/v2/runtime.d.ts.map +1 -1
  76. package/dist/src/v2/runtime.js +245 -52
  77. package/dist/src/v2/runtime.js.map +1 -1
  78. package/dist/src/v2/setup-cli.d.ts.map +1 -1
  79. package/dist/src/v2/setup-cli.js +4 -11
  80. package/dist/src/v2/setup-cli.js.map +1 -1
  81. package/dist/src/v2/setup-runtime.d.ts.map +1 -1
  82. package/dist/src/v2/setup-runtime.js +1 -61
  83. package/dist/src/v2/setup-runtime.js.map +1 -1
  84. package/dist/src/v2/setup-store.d.ts +0 -5
  85. package/dist/src/v2/setup-store.d.ts.map +1 -1
  86. package/dist/src/v2/setup-store.js +3 -106
  87. package/dist/src/v2/setup-store.js.map +1 -1
  88. package/dist/src/v2/setup.d.ts +6 -46
  89. package/dist/src/v2/setup.d.ts.map +1 -1
  90. package/dist/src/v2/setup.js +12 -294
  91. package/dist/src/v2/setup.js.map +1 -1
  92. package/dist/src/v2/workflow-assets.d.ts +19 -11
  93. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  94. package/dist/src/v2/workflow-assets.js +132 -40
  95. package/dist/src/v2/workflow-assets.js.map +1 -1
  96. package/docs/deep-dive.md +328 -56
  97. package/internal-workflow/docs/agents/bugfix-quality-gate.md +11 -0
  98. package/internal-workflow/docs/agents/coding-skill-routing.md +116 -196
  99. package/internal-workflow/docs/agents/contract-test-ledger.md +11 -1
  100. package/internal-workflow/docs/agents/review-gates.md +32 -39
  101. package/internal-workflow/docs/agents/review-protocol.md +75 -147
  102. package/internal-workflow/evals/coding-skill-evals.json +84 -0
  103. package/internal-workflow/manifest.json +1 -1
  104. package/internal-workflow/operations/acceptance-proof/SKILL.md +7 -1
  105. package/internal-workflow/operations/ambiguity-review/SKILL.md +2 -0
  106. package/internal-workflow/operations/code-review/SKILL.md +21 -1
  107. package/internal-workflow/operations/implementation/SKILL.md +22 -1
  108. package/internal-workflow/operations/spec-author/SKILL.md +10 -1
  109. package/internal-workflow/operations/spec-review/SKILL.md +10 -1
  110. package/internal-workflow/operations/triage/SKILL.md +10 -1
  111. package/internal-workflow/schemas/code-review-v1.json +1 -1
  112. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  113. package/internal-workflow/skills/agent-auto/SKILL.md +6 -1
  114. package/internal-workflow/skills/code-debugger/SKILL.md +122 -0
  115. package/internal-workflow/skills/code-debugger/agents/openai.yaml +7 -0
  116. package/internal-workflow/skills/code-review/SKILL.md +51 -17
  117. package/internal-workflow/skills/code-review/references/cleanup-lens.md +52 -0
  118. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +15 -6
  119. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +1 -1
  120. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +2 -2
  121. package/internal-workflow/skills/implementation-spec-review/SKILL.md +108 -204
  122. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +24 -0
  123. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +93 -0
  124. package/internal-workflow/skills/small-task-implementer/SKILL.md +15 -8
  125. package/internal-workflow/skills/spec-implementer/SKILL.md +101 -172
  126. package/internal-workflow/skills/spec-implementer/evals/evals.json +30 -0
  127. package/internal-workflow/skills/spec-implementer/references/review-loop.md +100 -0
  128. package/internal-workflow/skills/tdd/SKILL.md +20 -6
  129. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  130. package/internal-workflow/skills/tdd/evals/evals.json +18 -0
  131. package/internal-workflow/skills/tdd/mocking.md +3 -42
  132. package/internal-workflow/skills/tdd/refactoring.md +6 -8
  133. package/package.json +9 -6
  134. package/dist/src/v2/adapters/target-activity-fence.d.ts +0 -23
  135. package/dist/src/v2/adapters/target-activity-fence.d.ts.map +0 -1
  136. package/dist/src/v2/adapters/target-activity-fence.js +0 -249
  137. package/dist/src/v2/adapters/target-activity-fence.js.map +0 -1
  138. package/dist/src/v2/candidate-cli.d.ts +0 -26
  139. package/dist/src/v2/candidate-cli.d.ts.map +0 -1
  140. package/dist/src/v2/candidate-cli.js.map +0 -1
  141. package/dist/src/v2/legacy-cutover.d.ts +0 -52
  142. package/dist/src/v2/legacy-cutover.d.ts.map +0 -1
  143. package/dist/src/v2/legacy-cutover.js +0 -87
  144. package/dist/src/v2/legacy-cutover.js.map +0 -1
  145. package/internal-workflow/docs/agents/artifact-review-loop.md +0 -267
  146. package/internal-workflow/docs/agents/implementation-review-loop.md +0 -302
  147. package/internal-workflow/operations/cleanup-review/SKILL.md +0 -3
  148. package/internal-workflow/operations/spec-implementation/SKILL.md +0 -3
  149. package/internal-workflow/profiles/implementer_deep.toml +0 -9
  150. package/internal-workflow/profiles/researcher_standard.toml +0 -9
  151. package/internal-workflow/profiles/reviewer_fast.toml +0 -9
  152. package/internal-workflow/skills/cleanup-review/SKILL.md +0 -84
  153. package/internal-workflow/skills/cleanup-review/agents/openai.yaml +0 -6
  154. package/internal-workflow/skills/codebase-design/DEEPENING.md +0 -35
  155. package/internal-workflow/skills/codebase-design/DESIGN-IT-TWICE.md +0 -50
  156. package/internal-workflow/skills/codebase-design/SKILL.md +0 -82
  157. package/internal-workflow/skills/codebase-design/agents/openai.yaml +0 -6
  158. package/internal-workflow/skills/research/SKILL.md +0 -107
  159. package/internal-workflow/skills/research/agents/openai.yaml +0 -6
  160. package/internal-workflow/skills/ui-evidence-proof/SKILL.md +0 -123
  161. package/internal-workflow/skills/ui-evidence-proof/agents/openai.yaml +0 -6
@@ -1,267 +0,0 @@
1
- # Artifact Review Loop
2
-
3
- Use this policy whenever `plans-maker` or `implementation-spec-maker` reviews an
4
- artifact. This target-specific Module owns artifact preflight, review-risk
5
- classification, scope conservation, reviewer topology, and artifact outcome
6
- mapping. It applies [`review-protocol.md`](review-protocol.md) for context
7
- transfer, Full/Closure mechanics, defect lifecycle, no-progress, and the common
8
- result envelope.
9
-
10
- The Module sits at the Seam between artifact-authoring skills and the existing
11
- `plan-review` and `implementation-spec-review` Adapters. Callers provide an
12
- artifact and source authority; they do not reproduce either Module.
13
-
14
- ## Interface
15
-
16
- Conceptually, callers use one operation:
17
-
18
- ```text
19
- review_artifact(
20
- artifact_kind,
21
- artifact_path,
22
- source_references,
23
- approved_decisions,
24
- risk_profile = auto
25
- ) -> review_outcome
26
- ```
27
-
28
- `review_outcome` contains:
29
-
30
- - `outcome`: `Approved | Blocked | Waived`
31
- - `adapter_verdict`: `Approved | Needs Work | Rejected | Not run`
32
- - `risk_profile`: `simple | medium | high`
33
- - `risk_reasons`: evidence-backed classification signals
34
- - `review_passes`, `full_reviews`, `closure_reviews`, and `fresh_sessions`
35
- - mandatory-lens coverage
36
- - the Defect Ledger, including every unresolved blocker or execution risk
37
-
38
- The caller may explicitly raise the risk profile. It must not lower an
39
- evidence-backed profile or override a hard escalator.
40
-
41
- Reviewer Adapters themselves return only `Approved`, `Needs Work`, or
42
- `Rejected`. `Blocked` is a Module-level terminal outcome produced by preflight
43
- or a convergence stop rule; it is never invented by an individual reviewer.
44
- `Waived` is also Module-level and requires an explicit user instruction to skip
45
- remaining artifact review. Preserve the latest Adapter verdict when reviews
46
- already ran; use `Not run` only when no reviewer ran.
47
-
48
- ## Preflight And Risk Classification
49
-
50
- Run preflight before the first reviewer. Confirm source authority, approved
51
- scope, user decisions, external contracts, and the current artifact revision.
52
- Unknown product decisions or missing mandatory evidence block before review and
53
- do not start a reviewer.
54
-
55
- A useful preflight-blocked artifact may be saved with zero reviews. Record the
56
- Module outcome as `Blocked`, `Review Passes: 0`, and the exact blocking
57
- unknown; do not invent an Adapter verdict for a reviewer that never ran.
58
-
59
- Classify risk by the consequence and uncertainty of being wrong, not by file
60
- count or estimated implementation effort.
61
-
62
- ### Hard Escalators
63
-
64
- `high` requires a sensitive mechanism, an evidence-backed material consequence,
65
- and at least one uncertainty amplifier: competing or unclear ownership, a
66
- cross-trust-boundary effect, non-local rollback/recovery, an unproven external
67
- contract, or proof that cannot isolate the material failure. A sensitive
68
- mechanism with one proven owner, a local pattern, bounded rollback, and direct
69
- proof remains a standard signal rather than a hard escalator.
70
-
71
- | Sensitive mechanism | Material consequence required for `high` |
72
- | --- | --- |
73
- | Durable data, transaction, migration, schema | Corruption/loss, irreversible reinterpretation, or non-trivial rollback/backfill |
74
- | Concurrency, ordering, queue, retry, idempotency, shared state | Duplicate external effect, cross-worker invariant violation, corrupt/false durable state |
75
- | Auth, secrets, permissions, payments, destructive writes | Access/secret leak, incorrect money movement, or user/production data destruction |
76
- | Background processing or multiple writers | Ambiguous recovery ownership or competing source-of-truth ownership |
77
- | Unknown external contract, multi-agent integration, rollout | Safety-critical/irreversible failure or any material consequence above |
78
-
79
- ### Standard Signals
80
-
81
- Without a hard escalator, count: user-visible behavior; multiple Modules/runtime
82
- surfaces; shared Interface/DTO/event/serialization changes; non-trivial
83
- error/fallback/timeout/recovery; a known external contract; or a narrow sensitive
84
- mechanism retained inside one proven owner and local pattern.
85
-
86
- Use this deterministic mapping:
87
-
88
- | Profile | Classification |
89
- | --- | --- |
90
- | `simple` | No hard escalator and 0-2 standard signals |
91
- | `medium` | No hard escalator and 3-5 standard signals |
92
- | `high` | Any hard escalator or 6+ standard signals |
93
-
94
- File count is never decisive: one trust-boundary line can be `high`, while a
95
- broad mechanical rename can remain `simple` or `medium`.
96
-
97
- Record the decision in the artifact:
98
-
99
- ```yaml
100
- review_profile: simple | medium | high
101
- review_reasons:
102
- - "<signal>: <source or repo evidence>"
103
- ```
104
-
105
- ## Scope Conservation
106
-
107
- Review profile controls review depth, not artifact size or solution breadth.
108
- Risk may require stronger proof or more independent review, but it does not
109
- authorize extra product behavior, runtime layers, or operational machinery.
110
-
111
- Each review repair must preserve the approved scope. When it resolves the
112
- failure, delete or narrow the proposal before adding a mechanism. Do not add
113
- feature flags, telemetry systems, dashboards, rollout machinery, compatibility
114
- paths, generic fallbacks, or other operational features unless the source
115
- explicitly requires them or a concrete evidenced failure path makes them
116
- necessary. Optional improvements remain optional and outside the artifact
117
- unless the source authority or user explicitly approves them.
118
-
119
- If review discovers a higher-risk signal, escalate immediately. Keep completed
120
- review passes in the audit trail; do not discard valid coverage or automatically
121
- restart the loop. A user decision within approved scope also does not reset the
122
- defect history. A genuinely new product scope starts a new artifact revision
123
- only when the root explicitly says so and reports the previous outcome.
124
-
125
- When escalation to `high` occurs after reviews have started, do not pretend the
126
- initial full reviews were parallel. Preserve completed coverage, assign it to
127
- the matching high-risk lens set, and run only the missing full lens review. Then
128
- use affected-lens Closure for repaired defects.
129
-
130
- ## Review Capsule
131
-
132
- Use the protocol capsule with these artifact fields:
133
-
134
- - the unanswered artifact question and any prior coverage it invalidates
135
- - current saved artifact content and artifact kind
136
- - approved scope, out of scope, and Decision Snapshot
137
- - source-authority paths or URLs
138
- - Evidence Index entries of `claim -> file:symbol`, artifact section, or trusted
139
- URL
140
-
141
- For Closure, map artifact sections into the protocol Revision Map.
142
-
143
- ## Review Modes
144
-
145
- Use protocol Full and Closure without redefining them. Artifact Closure maps
146
- `affected_targets` to changed sections and source contracts. A substantive
147
- artifact rewrite requires a new Full pass only when existing mandatory-lens
148
- coverage is no longer valid.
149
-
150
- ## Risk-Aware Topology
151
-
152
- `simple` uses one `reviewer_fast` session, `medium` uses one
153
- `reviewer_standard` session, and `high` uses two independent `reviewer_deep`
154
- sessions with disjoint primary lenses. Root always launches these children and
155
- never executes the Adapter as self-review. The Adapter runs inline only inside
156
- the already assigned reviewer child because children cannot spawn grandchildren.
157
- Both high-profile full reviews are required for approval. Review counts are
158
- audit and performance metrics, never terminal limits.
159
-
160
- ### Simple And Medium
161
-
162
- Use one profile-selected reviewer lineage and one Full pass. After each
163
- consolidated repair batch, use protocol Closure until its lenses are clear or
164
- protocol no-progress/stop rules apply.
165
-
166
- Default initial sessions: 1. Default full reviews: 1.
167
-
168
- ### High
169
-
170
- When `high` is known at preflight, split primary lenses exactly as follows:
171
-
172
- - **Architecture/Execution:** source authority, scope, determinism, evidence,
173
- preconditions, architecture and ownership, sequencing/slices, reuse and
174
- simplicity, validation, and handoff.
175
- - **Failure/Contracts:** state transitions, concurrency and ordering,
176
- persistence, retry/idempotency, external/runtime contracts, auth/security,
177
- destructive behavior, partial failure/recovery, and Contract Test Ledger.
178
-
179
- Each reviewer owns its assigned primary lenses. It may report a concrete
180
- cross-lens defect, but it does not repeat the other reviewer's broad discovery.
181
- Root aggregates both results and verifies that their combined coverage includes
182
- all mandatory lenses and cross-lens defects before approval.
183
-
184
- 1. Reviewer A performs a full Architecture/Execution review.
185
- 2. Reviewer B performs a full Failure/Contracts review in parallel with review
186
- 1. The two briefs have disjoint primary lenses and both reviewers remain
187
- independent.
188
- 3. After one consolidated repair batch, apply protocol Closure only to affected
189
- A/B lineages, in parallel when both are affected.
190
- 4. Stop when both lens sets are clear on the same settled revision or protocol
191
- stop/no-progress rules apply.
192
-
193
- Default initial sessions: 2. Default full reviews: 2. Closure sessions rotate by
194
- protocol; another Full requires invalidated mandatory-lens coverage.
195
-
196
- Root owns launches, aggregation, repairs, and final decisions. Parallel launch
197
- must preserve and close every fulfilled handle even after partial failure.
198
-
199
- ## Defect Ledger
200
-
201
- Use the canonical protocol ledger. Artifact locations are stored in
202
- `affected_targets` as sections or source contracts. Artifact proof-contract gaps
203
- always use `repair-now` and reopen artifact review; they cannot use
204
- `planned-final-verification`.
205
-
206
- ## Convergence And Stop Rules
207
-
208
- Apply protocol repair, no-progress, stop, and waiver semantics first. Return
209
- `Approved` only when all of these artifact conditions also hold:
210
-
211
- - the verdict applies to the current saved artifact content
212
- - source authority and approved scope still match the artifact
213
-
214
- Persist mandatory-lens coverage with the outcome. Any substantive edit after
215
- approval invalidates `Approved` until the saved artifact passes the applicable
216
- Module path again. Updating only lifecycle and review-result metadata does not
217
- invalidate approval.
218
-
219
- Map protocol `stopped` to `Blocked`. Artifact-specific blockers also include:
220
-
221
- - a defect repeatedly reopens and exposes a source-of-truth or repair-design
222
- conflict that root cannot resolve from current evidence
223
- - the next repair needs a product, scope, ownership, or risky trade-off decision
224
- - the artifact did not change after `Needs Work` or `Rejected`
225
-
226
- Map protocol `waived` to `Waived` only when preflight is complete and no known
227
- blocker or unaccepted execution risk is open, blocked, or fixed-but-unverified.
228
- Otherwise record the waiver and return `Blocked`. Preserve `Adapter Verdict: Not
229
- run` or the last real Adapter verdict.
230
-
231
- Map terminal outcomes to artifact status without guesswork:
232
-
233
- - `Approved`: plan may be `ready-for-approval` or user-approved; spec may be
234
- `ready`.
235
- - `Blocked`: plan/spec status is `blocked`.
236
- - `Waived`: plan/spec may be ready under the eligible mapped waiver; keep it
237
- visible in metadata and the final response.
238
-
239
- `Needs Work` and `Rejected` remain Adapter verdicts, never durable Module
240
- outcomes.
241
-
242
- Skipping further artifact review never implicitly skips implementation
243
- checkpoints, cleanup review, final code review, tests, or runtime validation.
244
- Those are separate contracts and require their own explicit waiver when policy
245
- allows one.
246
-
247
- ## Required Outcome Summary
248
-
249
- Use the protocol result envelope and add:
250
-
251
- ```text
252
- Review Profile: <simple | medium | high>
253
- Review Outcome: <Approved | Blocked | Waived>
254
- Adapter Verdict: <Approved | Needs Work | Rejected | Not run>
255
- ```
256
-
257
- For performance evaluation, also retain per-review mode, start/end timestamps,
258
- and tool-call count when the runtime exposes them. Compare wall-clock time and
259
- token/tool usage separately.
260
-
261
- ## Contract Test Ledger
262
-
263
- | Invariant | Risk It Prevents | First Test / Proof | Status |
264
- | --- | --- | --- | --- |
265
- | Risk classification requires a material consequence for hard escalation; narrow use of a sensitive mechanism remains a standard signal. | Simple stateful work is over-reviewed or genuinely dangerous work is under-reviewed. | Manual eval scenario 10 | planned |
266
- | High-risk work uses two parallel Full lineages while protocol owns affected-lens Closure and session rotation. | Follow-up restarts Full review, retains stale context, or loses mandatory-lens coverage. | Manual eval scenarios 10-11 | green |
267
- | Substantive edits invalidate approval while lifecycle-only edits do not. | A changed plan/spec is executed under stale approval. | Review-state inspection | green |
@@ -1,302 +0,0 @@
1
- # Implementation Review Loop
2
-
3
- Use this policy for every approved implementation-spec execution. This
4
- target-specific Module owns implementation authority, durable Review State,
5
- checkpoint/final topology, validation reuse, gate ordering, audit epochs, and
6
- implementation outcome mapping. It applies
7
- [`review-protocol.md`](review-protocol.md) for context transfer, Full/Closure
8
- mechanics, defect lifecycle, no-progress, and the common result envelope.
9
- `spec-implementer`, `cleanup-review`, and `code-review` are callers or Adapters;
10
- they must not reproduce either Module. Deterministic tickets-orchestrator work
11
- using its issue as authority remains outside this Module and uses direct TDD
12
- plus repo review gates.
13
-
14
- This policy does not replace tests, architecture checks, smoke tests, or Git
15
- checkpoints. Those proofs remain independent evidence.
16
-
17
- ## Interface
18
-
19
- Conceptually, the executor uses:
20
-
21
- ```text
22
- review_implementation(
23
- authority_artifact_kind,
24
- authority_artifact_path,
25
- review_profile,
26
- current_revision,
27
- checkpoint,
28
- review_focus,
29
- defect_ledger
30
- ) -> review_outcome
31
- ```
32
-
33
- The outcome records:
34
-
35
- - `outcome`: `Approved | Blocked | Waived`
36
- - `authority_artifact_kind`: `approved-spec`
37
- - `authority_artifact_path`: the sole artifact that stores review state
38
- - `review_profile`: `simple | medium | high`
39
- - completed review passes and pending launches
40
- - review mode, reviewer/session identity, target revision, and assigned lenses
41
- - stable defect IDs and their current status
42
- - logical skill activations and their open/closed state
43
- - mandatory final coverage still required
44
-
45
- ## Authority
46
-
47
- An approved implementation spec stores the sole `## Implementation Review
48
- State`. Architecture RFCs, product PRDs, tickets, `ready-for-approval`
49
- artifacts, inferred approval, and ordinary direct-ticket waves are ineligible.
50
- Persist `authority_artifact_kind` and `authority_artifact_path` and never create
51
- a second ledger in an upstream artifact, caller, ticket, or Adapter.
52
-
53
- Lifecycle and proof updates to the selected artifact do not change its approved
54
- status or substantive design. A substantive authority change requires the
55
- normal artifact revision/review path before implementation continues.
56
-
57
- ## Profile And Review Shape
58
-
59
- Prefer the selected authority artifact's `review_profile`. If it is absent, use the same
60
- evidence-based classification and hard escalators as
61
- [`artifact-review-loop.md`](artifact-review-loop.md). Actual implementation
62
- evidence may raise the profile but must not lower it.
63
-
64
- Review profile selects mandatory lenses and independence. Protocol pass-count
65
- semantics apply; parallel reviewer results remain separate passes even when they
66
- reduce wall-clock time.
67
-
68
- Select the reviewer role from the profile: `simple` uses `reviewer_fast`,
69
- `medium` uses `reviewer_standard`, and `high` uses `reviewer_deep`. Root always
70
- launches reviewer children; it never performs an implementation review inline.
71
- An Adapter runs inline only inside its already assigned reviewer child.
72
-
73
- A reviewer that fails before returning a usable result is recorded as failed
74
- and closed, not as completed coverage. Escalation preserves completed coverage
75
- and the stable Defect Ledger. A user pause, context compaction, slice commit,
76
- worker replacement, or new turn also preserves them; none restarts the review
77
- topology automatically.
78
-
79
- Stop early when mandatory coverage and protocol clear-state requirements hold.
80
-
81
- ## Review Planning
82
-
83
- Before the first implementation reviewer, root creates a short Review Plan:
84
-
85
- - current profile and required independent lenses
86
- - only stable intermediate checkpoints and their required lenses; move an unstable checkpoint to final coverage when later slices touch the same files, owners, or contracts
87
- - any separate cleanup requirement, which must name a concrete evidenced reason that cannot fit the final spec/standards lens
88
- - final code-review lenses and minimum independent coverage
89
- - reviewer lineages that own affected-lens Closure
90
-
91
- Create one durable activation record for each logical skill invocation. Record
92
- `activation_id`, skill, owner, opened/closed state, and resume rule. Review Full
93
- and lineage-preserving Closure passes stay inside that review skill's activation;
94
- TDD repair cycles stay inside the active TDD activation. Cleanup, code review,
95
- TDD, and debugger activations never share an ID, and a continuation resumes an
96
- ID only for the same skill and authorized flow.
97
-
98
- ## Durable Review State
99
-
100
- Do not create durable review state during implementation preflight. Immediately
101
- before the first actual reviewer launch, persist the short Review Plan and
102
- pending launch in the selected authority artifact under
103
- `## Implementation Review State`. From that point onward this is the execution
104
- ledger for review state; do not keep the authoritative history only in chat
105
- context or a subagent summary.
106
-
107
- Record at least:
108
-
109
- - profile, completed pass count, and required coverage still outstanding
110
- - current checkpoint/gate, review timing baseline, gate-local consecutive
111
- Closure-wave count, and latest Closure wave ID
112
- - planned mandatory final reviews and their lenses
113
- - authority artifact kind/path and logical skill activation records
114
- - each lineage ID, origin Full session, active session generation, Closure count,
115
- rotation reason, live/timeout state, and `conclude_requested_at`
116
- - any convergence audit epoch: trigger, triggering pass/wave/revision,
117
- completion, dispositions, selected sessions, resume reason, and pass/wave/time
118
- baselines used for its next rearm
119
- - pending reviewer launches with launch ID, mode, lineage/session identity,
120
- activation ID, target revision, checkpoint/gate, Closure wave ID when
121
- applicable, assigned lenses, and start timestamp
122
- - every completed review's mode, lineage/session identity, target revision,
123
- checkpoint/gate, Closure wave ID, assigned lenses, start/end timestamps, and
124
- outcome
125
- - the stable Defect Ledger with transition history, reopen count, fixed revision,
126
- verifying review, and any explicit risk acceptance
127
-
128
- Write a pending launch before starting its reviewer. After the launch returns,
129
- replace the pending record with either its usable completed result or a failed,
130
- closed session record; only a usable result increments `review_passes`. A context
131
- compaction, new turn, resumed task, or different root agent must reconcile every
132
- pending launch with its recorded session before starting a replacement.
133
-
134
- Update this section after each launch, usable reviewer result, repair batch,
135
- closure, waiver, acceptance, reopen, or terminal outcome. A resumed executor
136
- reconstructs accounting and lifecycle history from this persisted state. If the
137
- state is missing or internally inconsistent after reviews began, return
138
- `Blocked` until it is reconciled from available thread/session evidence; never
139
- assume zero completed passes or silently replace an in-flight reviewer.
140
-
141
- Plan mandatory final coverage before launching an intermediate review. Launch a
142
- checkpoint only when its target is settled and later slices will not invalidate
143
- the reviewed files, owners, or contracts. Otherwise move its lenses to final
144
- coverage. Do not replace a required final lens with another fresh checkpoint
145
- reviewer or a repeat broad audit.
146
-
147
- The default shapes are:
148
-
149
- - `simple`: validation only when policy does not require review; otherwise one
150
- final Full review and affected-lens Closure only after repairs.
151
- - `medium`: an explicit intermediate checkpoint may provide one required lens;
152
- the final integrator covers every remaining lens, includes bounded cleanup in
153
- spec/standards, and verifies its defects without a separate cleanup pass.
154
- - `high`: use parallel independent tracks only for disjoint mandatory lenses;
155
- the spec/standards track includes bounded cleanup, and affected-lens Closure
156
- follows only after consolidated repairs. There is no separate cleanup pass by
157
- default.
158
-
159
- `code-review` uses one final reviewer covering both correctness and
160
- spec/standards for `simple` and `medium`. For `high`, it uses two disjoint final
161
- tracks unless earlier independent coverage already covered both axes and one
162
- fresh final integrator receives their compact handoffs.
163
-
164
- Before launching a fresh final reviewer, reconcile coverage on the settled
165
- revision. If the latest usable Full or Closure covered every mandatory final
166
- lens and left no open defect, mark final review complete and stop. Count cleanup
167
- Closure only for the lenses explicitly assigned in the Review Plan; a
168
- `cleanup-only` pass does not satisfy correctness or spec/standards coverage.
169
-
170
- ## Review Capsule
171
-
172
- Use the protocol capsule with these implementation fields:
173
-
174
- - the unanswered implementation question and any prior coverage it invalidates
175
- - authority artifact kind/path, profile, current revision, checkpoint, and exact diff command
176
- - changed paths and assigned `Review Focus` lenses
177
- - source-of-truth docs and relevant Contract Test Ledger rows
178
- - compact validation results and known verification gaps
179
-
180
- For Closure, map changed paths and tests into the protocol Revision Map.
181
-
182
- ## Review Modes
183
-
184
- Use protocol Full and Closure without redefining them. Implementation Closure
185
- maps `affected_targets` to paths, tests, runtime contracts, and Review Focus
186
- lenses. An already planned Full reviewer may verify a repair when its assigned
187
- lenses cover it.
188
-
189
- ## Defect Lifecycle
190
-
191
- Use the canonical protocol ledger and lifecycle without local aliases.
192
-
193
- Implementation proof-only gaps may use `planned-final-verification` only when an
194
- already scheduled code-review lens owns the proof; they remain open until
195
- independently verified and never re-enter cleanup. Artifact proof-contract gaps
196
- reopen artifact review. A pre-existing adjacent issue is non-blocking only as an
197
- `improvement` with `follow-up-improvement`.
198
-
199
- ## Validation Evidence Reuse
200
-
201
- Persist command/config identity, failure signature, target revision and changed
202
- path/contract impact basis, secret-safe environment fingerprint, transitive
203
- ownership/contract impact, and result. Reuse a known unrelated suite failure
204
- only when every field matches; unknown environment or transitive impact fails
205
- closed. Focused tests and every required check for a repair always rerun.
206
-
207
- ## Gate Ordering
208
-
209
- 1. Implement the slice and pass its tests/exit gate.
210
- 2. At an explicit intermediate checkpoint, run the required targeted
211
- `code-review` directly under the Review Plan.
212
- 3. Repair one consolidated finding batch and use protocol Closure for the
213
- affected lineages. An already planned Full reviewer may verify the repair when
214
- its assigned lenses cover it.
215
- At one gate, collect the usable results from all already-launched reviewers
216
- before repairing, unless an immediate blocker invalidates the remaining
217
- work. Do not turn individual findings into serial repair, validation, and
218
- Closure micro-cycles. Repair compatible findings once, rerun each affected
219
- validation once on the resulting revision, then launch one affected-lens
220
- Closure wave.
221
- 4. Continue implementation only when checkpoint blockers are verified or the
222
- Review Plan explicitly assigns their verification to an already planned reviewer
223
- without violating the checkpoint's safety purpose.
224
- 5. After all implementation slices and validations settle, run one final code
225
- review wave. `simple` and `medium` use one reviewer; `high` launches two
226
- disjoint reviewer tracks in parallel. The spec/standards lens owns bounded
227
- cleanup.
228
-
229
- Intermediate code-review checkpoints do not run cleanup-review. A separate
230
- cleanup pass is exceptional: run it only when the user, approved source, or repo
231
- policy names a concrete evidenced simplification risk that cannot fit the final
232
- spec/standards lens. `large` or `high` alone is not a reason. If an approved spec
233
- names an intermediate cleanup checkpoint, return `Blocked` for spec revision.
234
-
235
- Cleanup review runs at most once as a Full review for the whole spec. After its
236
- findings are repaired, either use protocol Closure or give the final code
237
- reviewer those stable defect IDs for verification. Never launch another Full
238
- cleanup review over the repaired whole diff.
239
-
240
- ## Convergence And Stop Rules
241
-
242
- Apply protocol repair, no-progress, stop, and waiver semantics. The following
243
- audit is implementation-specific.
244
-
245
- Before launching more reviewers, run one non-terminal convergence audit after
246
- two consecutive Closure waves in one gate, ten total implementation review
247
- passes, or 90 minutes when timing is available. One coordinated launch over all
248
- affected lineages is one wave regardless of parallel pass count.
249
-
250
- Persist one audit epoch with trigger, triggering pass/wave/revision,
251
- completion, dispositions, selected sessions, and resume reason. It survives
252
- resume. After a material repair/evidence change proves progress, rearm by
253
- recording the current total pass count, current gate-local Closure-wave count,
254
- and current timestamp as new baselines. The next audit opens only after a
255
- post-rearm delta reaches two Closure waves in that gate, ten implementation
256
- review passes, or 90 minutes; already-consumed counts or time cannot reopen it
257
- immediately. Without progress do not rearm and use the existing stop rules.
258
- Thresholds never approve, waive, downgrade, or block by themselves, and
259
- distinct failure mechanics remain distinct even when they protect one
260
- invariant.
261
-
262
- Return `Approved` for the final settled revision only when protocol state is
263
- `clear` and:
264
-
265
- - every mandatory lens has independent coverage
266
- - final validation and required cleanup/code-review gates ran
267
-
268
- Map protocol `stopped` to `Blocked`. Implementation-specific blockers also
269
- include:
270
-
271
- - a defect reopens repeatedly and exposes a source-of-truth or repair-design
272
- contradiction that root cannot resolve from current evidence
273
- - an execution risk remains open without an explicit user decision to accept it
274
-
275
- Do not mark implementation `Blocked` merely because review has run several
276
- times. Resolve the repair, evidence, or decision problem first.
277
-
278
- Map protocol `waived` to `Waived` and preserve skipped coverage and open risks.
279
- It remains non-approval; target authority or downstream policy may still block
280
- delivery.
281
-
282
- ## Required Handoff
283
-
284
- Use the protocol result envelope and add:
285
-
286
- ```text
287
- Implementation Review Profile: <simple | medium | high>
288
- Review Outcome: <Approved | Blocked | Waived>
289
- Authority Artifact: <approved spec path>
290
- Implementation Checkpoint: <checkpoint or final>
291
- ```
292
-
293
- ## Contract Test Ledger
294
-
295
- | Invariant | Risk It Prevents | First Test / Proof | Status |
296
- | --- | --- | --- | --- |
297
- | Review pass counts are audit metrics, while one durable Review Plan and Defect Ledger span the entire spec. | Each slice silently recreates a new loop or a repairable spec blocks on an arbitrary count. | Manual eval scenario 15 | planned |
298
- | Intermediate checkpoints never trigger cleanup; final review is one settled profile-selected wave, and separate cleanup is exceptional. | Per-slice hygiene or size-driven cleanup adds latency and restarts review over unstable work. | Manual eval scenario 17 | planned |
299
- | Audit epochs persist across resume and rearm only after material progress. | Thresholds repeatedly trigger audits or become terminal limits. | Manual eval scenario 12 | planned |
300
-
301
- Keep these rows `planned` until the corresponding operator eval is run and its
302
- result is saved.
@@ -1,3 +0,0 @@
1
- # Cleanup Review Operation
2
-
3
- Follow `skills/cleanup-review/SKILL.md`. Review only; do not edit files or external state. Return only `schemas/code-review-v1.json`.
@@ -1,3 +0,0 @@
1
- # Spec Implementation Operation
2
-
3
- Follow `skills/spec-implementer/SKILL.md` and the exact frozen spec. Never commit, publish, or mutate GitHub. Return only `schemas/implementation-report-v1.json`.
@@ -1,9 +0,0 @@
1
- name = "implementer_deep"
2
- description = "Write-capable implementation worker for an isolated approved slice with material technical uncertainty."
3
- nickname_candidates = ["Foundry", "Helix", "Vector"]
4
- model = "gpt-5.6-sol"
5
- model_reasoning_effort = "high"
6
- sandbox_mode = "workspace-write"
7
- developer_instructions = """
8
- Implement only the assigned isolated high-complexity ticket slice. Resolve technical uncertainty from repository evidence without changing approved product behavior, ownership, or slice boundaries. Respect exclusive write scope and stop on overlap or decision drift. You are not alone in the repository: preserve unrelated and concurrent changes and never revert work you do not own. Use behavior-first proof and return changed files, acceptance proof, skipped checks, risks, decision deltas, and blockers to the root integrator.
9
- """
@@ -1,9 +0,0 @@
1
- name = "researcher_standard"
2
- description = "Read-only external research against primary sources with claim-level citations."
3
- nickname_candidates = ["Atlas", "Index", "Scribe"]
4
- model = "gpt-5.6-sol"
5
- model_reasoning_effort = "medium"
6
- sandbox_mode = "read-only"
7
- developer_instructions = """
8
- Research one bounded external question from the supplied Research Capsule. Prefer official documentation, specifications, first-party source code, changelogs, release notes, issue trackers, APIs, and schemas. Map every material claim to the exact primary source and include source version/date when available. Separate sourced facts from repository inference; expose conflicts, stale evidence, uncertainty, and missing proof. Never edit files, create artifacts, change code, or broaden into unrelated reading. Return a concise short answer, claim-to-source ledger, decision implications, and unresolved questions to the root integrator.
9
- """
@@ -1,9 +0,0 @@
1
- name = "reviewer_fast"
2
- description = "Fast independent read-only reviewer for simple plans, specs, tickets, cleanup, and code changes."
3
- nickname_candidates = ["Dash", "Jet", "Swift"]
4
- model = "gpt-5.6-terra"
5
- model_reasoning_effort = "medium"
6
- sandbox_mode = "read-only"
7
- developer_instructions = """
8
- Review independently and proportionately. Prioritize concrete correctness, scope, contract, and verification gaps; avoid speculative edge cases and cosmetic comments. Never edit files. Return concise findings with evidence, severity, confidence, and proposed fixes to the parent.
9
- """