codex-orchestrator 2.0.3 → 2.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -427
- package/README.md +135 -37
- package/dist/src/index.d.ts +1 -1
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/v2/adapters/gh-issue-adapter.d.ts.map +1 -1
- package/dist/src/v2/adapters/gh-issue-adapter.js +6 -7
- package/dist/src/v2/adapters/gh-issue-adapter.js.map +1 -1
- package/dist/src/v2/cli-contract.d.ts +3 -3
- package/dist/src/v2/cli-contract.d.ts.map +1 -1
- package/dist/src/v2/cli-contract.js +1 -3
- package/dist/src/v2/cli-contract.js.map +1 -1
- package/dist/src/v2/cli.d.ts +24 -0
- package/dist/src/v2/cli.d.ts.map +1 -0
- package/dist/src/v2/{candidate-cli.js → cli.js} +18 -29
- package/dist/src/v2/cli.js.map +1 -0
- package/dist/src/v2/code-review-report.d.ts +1 -1
- package/dist/src/v2/code-review-report.d.ts.map +1 -1
- package/dist/src/v2/code-review-report.js +2 -2
- package/dist/src/v2/code-review-report.js.map +1 -1
- package/dist/src/v2/codex-process.d.ts.map +1 -1
- package/dist/src/v2/codex-process.js +12 -1
- package/dist/src/v2/codex-process.js.map +1 -1
- package/dist/src/v2/config.d.ts +2 -2
- package/dist/src/v2/config.d.ts.map +1 -1
- package/dist/src/v2/config.js.map +1 -1
- package/dist/src/v2/contained-report-operation.d.ts +2 -2
- package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
- package/dist/src/v2/contained-report-operation.js +1 -1
- package/dist/src/v2/contained-report-operation.js.map +1 -1
- package/dist/src/v2/containment.d.ts +4 -0
- package/dist/src/v2/containment.d.ts.map +1 -1
- package/dist/src/v2/containment.js +9 -0
- package/dist/src/v2/containment.js.map +1 -1
- package/dist/src/v2/direct-delivery.d.ts +5 -10
- package/dist/src/v2/direct-delivery.d.ts.map +1 -1
- package/dist/src/v2/direct-delivery.js +25 -90
- package/dist/src/v2/direct-delivery.js.map +1 -1
- package/dist/src/v2/proof-report.d.ts.map +1 -1
- package/dist/src/v2/proof-report.js +55 -29
- package/dist/src/v2/proof-report.js.map +1 -1
- package/dist/src/v2/run-issue.d.ts +6 -9
- package/dist/src/v2/run-issue.d.ts.map +1 -1
- package/dist/src/v2/run-issue.js +98 -43
- package/dist/src/v2/run-issue.js.map +1 -1
- package/dist/src/v2/run-store.d.ts +4 -4
- package/dist/src/v2/run-store.d.ts.map +1 -1
- package/dist/src/v2/run-store.js +25 -40
- package/dist/src/v2/run-store.js.map +1 -1
- package/dist/src/v2/runtime.d.ts +3 -3
- package/dist/src/v2/runtime.d.ts.map +1 -1
- package/dist/src/v2/runtime.js +125 -47
- package/dist/src/v2/runtime.js.map +1 -1
- package/dist/src/v2/setup-cli.d.ts.map +1 -1
- package/dist/src/v2/setup-cli.js +4 -11
- package/dist/src/v2/setup-cli.js.map +1 -1
- package/dist/src/v2/setup-runtime.d.ts.map +1 -1
- package/dist/src/v2/setup-runtime.js +1 -61
- package/dist/src/v2/setup-runtime.js.map +1 -1
- package/dist/src/v2/setup-store.d.ts +0 -5
- package/dist/src/v2/setup-store.d.ts.map +1 -1
- package/dist/src/v2/setup-store.js +3 -106
- package/dist/src/v2/setup-store.js.map +1 -1
- package/dist/src/v2/setup.d.ts +6 -46
- package/dist/src/v2/setup.d.ts.map +1 -1
- package/dist/src/v2/setup.js +11 -293
- package/dist/src/v2/setup.js.map +1 -1
- package/dist/src/v2/workflow-assets.d.ts +19 -11
- package/dist/src/v2/workflow-assets.d.ts.map +1 -1
- package/dist/src/v2/workflow-assets.js +132 -40
- package/dist/src/v2/workflow-assets.js.map +1 -1
- package/docs/deep-dive.md +272 -56
- package/internal-workflow/docs/agents/bugfix-quality-gate.md +11 -0
- package/internal-workflow/docs/agents/coding-skill-routing.md +116 -196
- package/internal-workflow/docs/agents/review-gates.md +32 -39
- package/internal-workflow/docs/agents/review-protocol.md +75 -147
- package/internal-workflow/evals/coding-skill-evals.json +66 -0
- package/internal-workflow/manifest.json +1 -1
- package/internal-workflow/operations/acceptance-proof/SKILL.md +7 -1
- package/internal-workflow/operations/ambiguity-review/SKILL.md +2 -0
- package/internal-workflow/operations/code-review/SKILL.md +21 -1
- package/internal-workflow/operations/implementation/SKILL.md +22 -1
- package/internal-workflow/operations/spec-author/SKILL.md +10 -1
- package/internal-workflow/operations/spec-review/SKILL.md +10 -1
- package/internal-workflow/operations/triage/SKILL.md +10 -1
- package/internal-workflow/schemas/code-review-v1.json +1 -1
- package/internal-workflow/schemas/proof-report-v1.json +1 -1
- package/internal-workflow/skills/agent-auto/SKILL.md +6 -1
- package/internal-workflow/skills/code-debugger/SKILL.md +122 -0
- package/internal-workflow/skills/code-debugger/agents/openai.yaml +7 -0
- package/internal-workflow/skills/code-review/SKILL.md +33 -11
- package/internal-workflow/skills/code-review/references/cleanup-lens.md +52 -0
- package/internal-workflow/skills/implementation-spec-maker/SKILL.md +15 -6
- package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +2 -2
- package/internal-workflow/skills/implementation-spec-review/SKILL.md +108 -204
- package/internal-workflow/skills/implementation-spec-review/evals/evals.json +24 -0
- package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +93 -0
- package/internal-workflow/skills/small-task-implementer/SKILL.md +15 -8
- package/internal-workflow/skills/spec-implementer/SKILL.md +101 -172
- package/internal-workflow/skills/spec-implementer/evals/evals.json +30 -0
- package/internal-workflow/skills/spec-implementer/references/review-loop.md +94 -0
- package/internal-workflow/skills/tdd/SKILL.md +15 -2
- package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
- package/package.json +9 -6
- package/dist/src/v2/adapters/target-activity-fence.d.ts +0 -23
- package/dist/src/v2/adapters/target-activity-fence.d.ts.map +0 -1
- package/dist/src/v2/adapters/target-activity-fence.js +0 -249
- package/dist/src/v2/adapters/target-activity-fence.js.map +0 -1
- package/dist/src/v2/candidate-cli.d.ts +0 -26
- package/dist/src/v2/candidate-cli.d.ts.map +0 -1
- package/dist/src/v2/candidate-cli.js.map +0 -1
- package/dist/src/v2/legacy-cutover.d.ts +0 -52
- package/dist/src/v2/legacy-cutover.d.ts.map +0 -1
- package/dist/src/v2/legacy-cutover.js +0 -87
- package/dist/src/v2/legacy-cutover.js.map +0 -1
- package/internal-workflow/docs/agents/artifact-review-loop.md +0 -267
- package/internal-workflow/docs/agents/implementation-review-loop.md +0 -302
- package/internal-workflow/operations/cleanup-review/SKILL.md +0 -3
- package/internal-workflow/operations/spec-implementation/SKILL.md +0 -3
- package/internal-workflow/profiles/implementer_deep.toml +0 -9
- package/internal-workflow/profiles/researcher_standard.toml +0 -9
- package/internal-workflow/profiles/reviewer_fast.toml +0 -9
- package/internal-workflow/skills/cleanup-review/SKILL.md +0 -84
- package/internal-workflow/skills/cleanup-review/agents/openai.yaml +0 -6
- package/internal-workflow/skills/codebase-design/DEEPENING.md +0 -35
- package/internal-workflow/skills/codebase-design/DESIGN-IT-TWICE.md +0 -50
- package/internal-workflow/skills/codebase-design/SKILL.md +0 -82
- package/internal-workflow/skills/codebase-design/agents/openai.yaml +0 -6
- package/internal-workflow/skills/research/SKILL.md +0 -107
- package/internal-workflow/skills/research/agents/openai.yaml +0 -6
- package/internal-workflow/skills/ui-evidence-proof/SKILL.md +0 -123
- package/internal-workflow/skills/ui-evidence-proof/agents/openai.yaml +0 -6
|
@@ -1,267 +0,0 @@
|
|
|
1
|
-
# Artifact Review Loop
|
|
2
|
-
|
|
3
|
-
Use this policy whenever `plans-maker` or `implementation-spec-maker` reviews an
|
|
4
|
-
artifact. This target-specific Module owns artifact preflight, review-risk
|
|
5
|
-
classification, scope conservation, reviewer topology, and artifact outcome
|
|
6
|
-
mapping. It applies [`review-protocol.md`](review-protocol.md) for context
|
|
7
|
-
transfer, Full/Closure mechanics, defect lifecycle, no-progress, and the common
|
|
8
|
-
result envelope.
|
|
9
|
-
|
|
10
|
-
The Module sits at the Seam between artifact-authoring skills and the existing
|
|
11
|
-
`plan-review` and `implementation-spec-review` Adapters. Callers provide an
|
|
12
|
-
artifact and source authority; they do not reproduce either Module.
|
|
13
|
-
|
|
14
|
-
## Interface
|
|
15
|
-
|
|
16
|
-
Conceptually, callers use one operation:
|
|
17
|
-
|
|
18
|
-
```text
|
|
19
|
-
review_artifact(
|
|
20
|
-
artifact_kind,
|
|
21
|
-
artifact_path,
|
|
22
|
-
source_references,
|
|
23
|
-
approved_decisions,
|
|
24
|
-
risk_profile = auto
|
|
25
|
-
) -> review_outcome
|
|
26
|
-
```
|
|
27
|
-
|
|
28
|
-
`review_outcome` contains:
|
|
29
|
-
|
|
30
|
-
- `outcome`: `Approved | Blocked | Waived`
|
|
31
|
-
- `adapter_verdict`: `Approved | Needs Work | Rejected | Not run`
|
|
32
|
-
- `risk_profile`: `simple | medium | high`
|
|
33
|
-
- `risk_reasons`: evidence-backed classification signals
|
|
34
|
-
- `review_passes`, `full_reviews`, `closure_reviews`, and `fresh_sessions`
|
|
35
|
-
- mandatory-lens coverage
|
|
36
|
-
- the Defect Ledger, including every unresolved blocker or execution risk
|
|
37
|
-
|
|
38
|
-
The caller may explicitly raise the risk profile. It must not lower an
|
|
39
|
-
evidence-backed profile or override a hard escalator.
|
|
40
|
-
|
|
41
|
-
Reviewer Adapters themselves return only `Approved`, `Needs Work`, or
|
|
42
|
-
`Rejected`. `Blocked` is a Module-level terminal outcome produced by preflight
|
|
43
|
-
or a convergence stop rule; it is never invented by an individual reviewer.
|
|
44
|
-
`Waived` is also Module-level and requires an explicit user instruction to skip
|
|
45
|
-
remaining artifact review. Preserve the latest Adapter verdict when reviews
|
|
46
|
-
already ran; use `Not run` only when no reviewer ran.
|
|
47
|
-
|
|
48
|
-
## Preflight And Risk Classification
|
|
49
|
-
|
|
50
|
-
Run preflight before the first reviewer. Confirm source authority, approved
|
|
51
|
-
scope, user decisions, external contracts, and the current artifact revision.
|
|
52
|
-
Unknown product decisions or missing mandatory evidence block before review and
|
|
53
|
-
do not start a reviewer.
|
|
54
|
-
|
|
55
|
-
A useful preflight-blocked artifact may be saved with zero reviews. Record the
|
|
56
|
-
Module outcome as `Blocked`, `Review Passes: 0`, and the exact blocking
|
|
57
|
-
unknown; do not invent an Adapter verdict for a reviewer that never ran.
|
|
58
|
-
|
|
59
|
-
Classify risk by the consequence and uncertainty of being wrong, not by file
|
|
60
|
-
count or estimated implementation effort.
|
|
61
|
-
|
|
62
|
-
### Hard Escalators
|
|
63
|
-
|
|
64
|
-
`high` requires a sensitive mechanism, an evidence-backed material consequence,
|
|
65
|
-
and at least one uncertainty amplifier: competing or unclear ownership, a
|
|
66
|
-
cross-trust-boundary effect, non-local rollback/recovery, an unproven external
|
|
67
|
-
contract, or proof that cannot isolate the material failure. A sensitive
|
|
68
|
-
mechanism with one proven owner, a local pattern, bounded rollback, and direct
|
|
69
|
-
proof remains a standard signal rather than a hard escalator.
|
|
70
|
-
|
|
71
|
-
| Sensitive mechanism | Material consequence required for `high` |
|
|
72
|
-
| --- | --- |
|
|
73
|
-
| Durable data, transaction, migration, schema | Corruption/loss, irreversible reinterpretation, or non-trivial rollback/backfill |
|
|
74
|
-
| Concurrency, ordering, queue, retry, idempotency, shared state | Duplicate external effect, cross-worker invariant violation, corrupt/false durable state |
|
|
75
|
-
| Auth, secrets, permissions, payments, destructive writes | Access/secret leak, incorrect money movement, or user/production data destruction |
|
|
76
|
-
| Background processing or multiple writers | Ambiguous recovery ownership or competing source-of-truth ownership |
|
|
77
|
-
| Unknown external contract, multi-agent integration, rollout | Safety-critical/irreversible failure or any material consequence above |
|
|
78
|
-
|
|
79
|
-
### Standard Signals
|
|
80
|
-
|
|
81
|
-
Without a hard escalator, count: user-visible behavior; multiple Modules/runtime
|
|
82
|
-
surfaces; shared Interface/DTO/event/serialization changes; non-trivial
|
|
83
|
-
error/fallback/timeout/recovery; a known external contract; or a narrow sensitive
|
|
84
|
-
mechanism retained inside one proven owner and local pattern.
|
|
85
|
-
|
|
86
|
-
Use this deterministic mapping:
|
|
87
|
-
|
|
88
|
-
| Profile | Classification |
|
|
89
|
-
| --- | --- |
|
|
90
|
-
| `simple` | No hard escalator and 0-2 standard signals |
|
|
91
|
-
| `medium` | No hard escalator and 3-5 standard signals |
|
|
92
|
-
| `high` | Any hard escalator or 6+ standard signals |
|
|
93
|
-
|
|
94
|
-
File count is never decisive: one trust-boundary line can be `high`, while a
|
|
95
|
-
broad mechanical rename can remain `simple` or `medium`.
|
|
96
|
-
|
|
97
|
-
Record the decision in the artifact:
|
|
98
|
-
|
|
99
|
-
```yaml
|
|
100
|
-
review_profile: simple | medium | high
|
|
101
|
-
review_reasons:
|
|
102
|
-
- "<signal>: <source or repo evidence>"
|
|
103
|
-
```
|
|
104
|
-
|
|
105
|
-
## Scope Conservation
|
|
106
|
-
|
|
107
|
-
Review profile controls review depth, not artifact size or solution breadth.
|
|
108
|
-
Risk may require stronger proof or more independent review, but it does not
|
|
109
|
-
authorize extra product behavior, runtime layers, or operational machinery.
|
|
110
|
-
|
|
111
|
-
Each review repair must preserve the approved scope. When it resolves the
|
|
112
|
-
failure, delete or narrow the proposal before adding a mechanism. Do not add
|
|
113
|
-
feature flags, telemetry systems, dashboards, rollout machinery, compatibility
|
|
114
|
-
paths, generic fallbacks, or other operational features unless the source
|
|
115
|
-
explicitly requires them or a concrete evidenced failure path makes them
|
|
116
|
-
necessary. Optional improvements remain optional and outside the artifact
|
|
117
|
-
unless the source authority or user explicitly approves them.
|
|
118
|
-
|
|
119
|
-
If review discovers a higher-risk signal, escalate immediately. Keep completed
|
|
120
|
-
review passes in the audit trail; do not discard valid coverage or automatically
|
|
121
|
-
restart the loop. A user decision within approved scope also does not reset the
|
|
122
|
-
defect history. A genuinely new product scope starts a new artifact revision
|
|
123
|
-
only when the root explicitly says so and reports the previous outcome.
|
|
124
|
-
|
|
125
|
-
When escalation to `high` occurs after reviews have started, do not pretend the
|
|
126
|
-
initial full reviews were parallel. Preserve completed coverage, assign it to
|
|
127
|
-
the matching high-risk lens set, and run only the missing full lens review. Then
|
|
128
|
-
use affected-lens Closure for repaired defects.
|
|
129
|
-
|
|
130
|
-
## Review Capsule
|
|
131
|
-
|
|
132
|
-
Use the protocol capsule with these artifact fields:
|
|
133
|
-
|
|
134
|
-
- the unanswered artifact question and any prior coverage it invalidates
|
|
135
|
-
- current saved artifact content and artifact kind
|
|
136
|
-
- approved scope, out of scope, and Decision Snapshot
|
|
137
|
-
- source-authority paths or URLs
|
|
138
|
-
- Evidence Index entries of `claim -> file:symbol`, artifact section, or trusted
|
|
139
|
-
URL
|
|
140
|
-
|
|
141
|
-
For Closure, map artifact sections into the protocol Revision Map.
|
|
142
|
-
|
|
143
|
-
## Review Modes
|
|
144
|
-
|
|
145
|
-
Use protocol Full and Closure without redefining them. Artifact Closure maps
|
|
146
|
-
`affected_targets` to changed sections and source contracts. A substantive
|
|
147
|
-
artifact rewrite requires a new Full pass only when existing mandatory-lens
|
|
148
|
-
coverage is no longer valid.
|
|
149
|
-
|
|
150
|
-
## Risk-Aware Topology
|
|
151
|
-
|
|
152
|
-
`simple` uses one `reviewer_fast` session, `medium` uses one
|
|
153
|
-
`reviewer_standard` session, and `high` uses two independent `reviewer_deep`
|
|
154
|
-
sessions with disjoint primary lenses. Root always launches these children and
|
|
155
|
-
never executes the Adapter as self-review. The Adapter runs inline only inside
|
|
156
|
-
the already assigned reviewer child because children cannot spawn grandchildren.
|
|
157
|
-
Both high-profile full reviews are required for approval. Review counts are
|
|
158
|
-
audit and performance metrics, never terminal limits.
|
|
159
|
-
|
|
160
|
-
### Simple And Medium
|
|
161
|
-
|
|
162
|
-
Use one profile-selected reviewer lineage and one Full pass. After each
|
|
163
|
-
consolidated repair batch, use protocol Closure until its lenses are clear or
|
|
164
|
-
protocol no-progress/stop rules apply.
|
|
165
|
-
|
|
166
|
-
Default initial sessions: 1. Default full reviews: 1.
|
|
167
|
-
|
|
168
|
-
### High
|
|
169
|
-
|
|
170
|
-
When `high` is known at preflight, split primary lenses exactly as follows:
|
|
171
|
-
|
|
172
|
-
- **Architecture/Execution:** source authority, scope, determinism, evidence,
|
|
173
|
-
preconditions, architecture and ownership, sequencing/slices, reuse and
|
|
174
|
-
simplicity, validation, and handoff.
|
|
175
|
-
- **Failure/Contracts:** state transitions, concurrency and ordering,
|
|
176
|
-
persistence, retry/idempotency, external/runtime contracts, auth/security,
|
|
177
|
-
destructive behavior, partial failure/recovery, and Contract Test Ledger.
|
|
178
|
-
|
|
179
|
-
Each reviewer owns its assigned primary lenses. It may report a concrete
|
|
180
|
-
cross-lens defect, but it does not repeat the other reviewer's broad discovery.
|
|
181
|
-
Root aggregates both results and verifies that their combined coverage includes
|
|
182
|
-
all mandatory lenses and cross-lens defects before approval.
|
|
183
|
-
|
|
184
|
-
1. Reviewer A performs a full Architecture/Execution review.
|
|
185
|
-
2. Reviewer B performs a full Failure/Contracts review in parallel with review
|
|
186
|
-
1. The two briefs have disjoint primary lenses and both reviewers remain
|
|
187
|
-
independent.
|
|
188
|
-
3. After one consolidated repair batch, apply protocol Closure only to affected
|
|
189
|
-
A/B lineages, in parallel when both are affected.
|
|
190
|
-
4. Stop when both lens sets are clear on the same settled revision or protocol
|
|
191
|
-
stop/no-progress rules apply.
|
|
192
|
-
|
|
193
|
-
Default initial sessions: 2. Default full reviews: 2. Closure sessions rotate by
|
|
194
|
-
protocol; another Full requires invalidated mandatory-lens coverage.
|
|
195
|
-
|
|
196
|
-
Root owns launches, aggregation, repairs, and final decisions. Parallel launch
|
|
197
|
-
must preserve and close every fulfilled handle even after partial failure.
|
|
198
|
-
|
|
199
|
-
## Defect Ledger
|
|
200
|
-
|
|
201
|
-
Use the canonical protocol ledger. Artifact locations are stored in
|
|
202
|
-
`affected_targets` as sections or source contracts. Artifact proof-contract gaps
|
|
203
|
-
always use `repair-now` and reopen artifact review; they cannot use
|
|
204
|
-
`planned-final-verification`.
|
|
205
|
-
|
|
206
|
-
## Convergence And Stop Rules
|
|
207
|
-
|
|
208
|
-
Apply protocol repair, no-progress, stop, and waiver semantics first. Return
|
|
209
|
-
`Approved` only when all of these artifact conditions also hold:
|
|
210
|
-
|
|
211
|
-
- the verdict applies to the current saved artifact content
|
|
212
|
-
- source authority and approved scope still match the artifact
|
|
213
|
-
|
|
214
|
-
Persist mandatory-lens coverage with the outcome. Any substantive edit after
|
|
215
|
-
approval invalidates `Approved` until the saved artifact passes the applicable
|
|
216
|
-
Module path again. Updating only lifecycle and review-result metadata does not
|
|
217
|
-
invalidate approval.
|
|
218
|
-
|
|
219
|
-
Map protocol `stopped` to `Blocked`. Artifact-specific blockers also include:
|
|
220
|
-
|
|
221
|
-
- a defect repeatedly reopens and exposes a source-of-truth or repair-design
|
|
222
|
-
conflict that root cannot resolve from current evidence
|
|
223
|
-
- the next repair needs a product, scope, ownership, or risky trade-off decision
|
|
224
|
-
- the artifact did not change after `Needs Work` or `Rejected`
|
|
225
|
-
|
|
226
|
-
Map protocol `waived` to `Waived` only when preflight is complete and no known
|
|
227
|
-
blocker or unaccepted execution risk is open, blocked, or fixed-but-unverified.
|
|
228
|
-
Otherwise record the waiver and return `Blocked`. Preserve `Adapter Verdict: Not
|
|
229
|
-
run` or the last real Adapter verdict.
|
|
230
|
-
|
|
231
|
-
Map terminal outcomes to artifact status without guesswork:
|
|
232
|
-
|
|
233
|
-
- `Approved`: plan may be `ready-for-approval` or user-approved; spec may be
|
|
234
|
-
`ready`.
|
|
235
|
-
- `Blocked`: plan/spec status is `blocked`.
|
|
236
|
-
- `Waived`: plan/spec may be ready under the eligible mapped waiver; keep it
|
|
237
|
-
visible in metadata and the final response.
|
|
238
|
-
|
|
239
|
-
`Needs Work` and `Rejected` remain Adapter verdicts, never durable Module
|
|
240
|
-
outcomes.
|
|
241
|
-
|
|
242
|
-
Skipping further artifact review never implicitly skips implementation
|
|
243
|
-
checkpoints, cleanup review, final code review, tests, or runtime validation.
|
|
244
|
-
Those are separate contracts and require their own explicit waiver when policy
|
|
245
|
-
allows one.
|
|
246
|
-
|
|
247
|
-
## Required Outcome Summary
|
|
248
|
-
|
|
249
|
-
Use the protocol result envelope and add:
|
|
250
|
-
|
|
251
|
-
```text
|
|
252
|
-
Review Profile: <simple | medium | high>
|
|
253
|
-
Review Outcome: <Approved | Blocked | Waived>
|
|
254
|
-
Adapter Verdict: <Approved | Needs Work | Rejected | Not run>
|
|
255
|
-
```
|
|
256
|
-
|
|
257
|
-
For performance evaluation, also retain per-review mode, start/end timestamps,
|
|
258
|
-
and tool-call count when the runtime exposes them. Compare wall-clock time and
|
|
259
|
-
token/tool usage separately.
|
|
260
|
-
|
|
261
|
-
## Contract Test Ledger
|
|
262
|
-
|
|
263
|
-
| Invariant | Risk It Prevents | First Test / Proof | Status |
|
|
264
|
-
| --- | --- | --- | --- |
|
|
265
|
-
| Risk classification requires a material consequence for hard escalation; narrow use of a sensitive mechanism remains a standard signal. | Simple stateful work is over-reviewed or genuinely dangerous work is under-reviewed. | Manual eval scenario 10 | planned |
|
|
266
|
-
| High-risk work uses two parallel Full lineages while protocol owns affected-lens Closure and session rotation. | Follow-up restarts Full review, retains stale context, or loses mandatory-lens coverage. | Manual eval scenarios 10-11 | green |
|
|
267
|
-
| Substantive edits invalidate approval while lifecycle-only edits do not. | A changed plan/spec is executed under stale approval. | Review-state inspection | green |
|
|
@@ -1,302 +0,0 @@
|
|
|
1
|
-
# Implementation Review Loop
|
|
2
|
-
|
|
3
|
-
Use this policy for every approved implementation-spec execution. This
|
|
4
|
-
target-specific Module owns implementation authority, durable Review State,
|
|
5
|
-
checkpoint/final topology, validation reuse, gate ordering, audit epochs, and
|
|
6
|
-
implementation outcome mapping. It applies
|
|
7
|
-
[`review-protocol.md`](review-protocol.md) for context transfer, Full/Closure
|
|
8
|
-
mechanics, defect lifecycle, no-progress, and the common result envelope.
|
|
9
|
-
`spec-implementer`, `cleanup-review`, and `code-review` are callers or Adapters;
|
|
10
|
-
they must not reproduce either Module. Deterministic tickets-orchestrator work
|
|
11
|
-
using its issue as authority remains outside this Module and uses direct TDD
|
|
12
|
-
plus repo review gates.
|
|
13
|
-
|
|
14
|
-
This policy does not replace tests, architecture checks, smoke tests, or Git
|
|
15
|
-
checkpoints. Those proofs remain independent evidence.
|
|
16
|
-
|
|
17
|
-
## Interface
|
|
18
|
-
|
|
19
|
-
Conceptually, the executor uses:
|
|
20
|
-
|
|
21
|
-
```text
|
|
22
|
-
review_implementation(
|
|
23
|
-
authority_artifact_kind,
|
|
24
|
-
authority_artifact_path,
|
|
25
|
-
review_profile,
|
|
26
|
-
current_revision,
|
|
27
|
-
checkpoint,
|
|
28
|
-
review_focus,
|
|
29
|
-
defect_ledger
|
|
30
|
-
) -> review_outcome
|
|
31
|
-
```
|
|
32
|
-
|
|
33
|
-
The outcome records:
|
|
34
|
-
|
|
35
|
-
- `outcome`: `Approved | Blocked | Waived`
|
|
36
|
-
- `authority_artifact_kind`: `approved-spec`
|
|
37
|
-
- `authority_artifact_path`: the sole artifact that stores review state
|
|
38
|
-
- `review_profile`: `simple | medium | high`
|
|
39
|
-
- completed review passes and pending launches
|
|
40
|
-
- review mode, reviewer/session identity, target revision, and assigned lenses
|
|
41
|
-
- stable defect IDs and their current status
|
|
42
|
-
- logical skill activations and their open/closed state
|
|
43
|
-
- mandatory final coverage still required
|
|
44
|
-
|
|
45
|
-
## Authority
|
|
46
|
-
|
|
47
|
-
An approved implementation spec stores the sole `## Implementation Review
|
|
48
|
-
State`. Architecture RFCs, product PRDs, tickets, `ready-for-approval`
|
|
49
|
-
artifacts, inferred approval, and ordinary direct-ticket waves are ineligible.
|
|
50
|
-
Persist `authority_artifact_kind` and `authority_artifact_path` and never create
|
|
51
|
-
a second ledger in an upstream artifact, caller, ticket, or Adapter.
|
|
52
|
-
|
|
53
|
-
Lifecycle and proof updates to the selected artifact do not change its approved
|
|
54
|
-
status or substantive design. A substantive authority change requires the
|
|
55
|
-
normal artifact revision/review path before implementation continues.
|
|
56
|
-
|
|
57
|
-
## Profile And Review Shape
|
|
58
|
-
|
|
59
|
-
Prefer the selected authority artifact's `review_profile`. If it is absent, use the same
|
|
60
|
-
evidence-based classification and hard escalators as
|
|
61
|
-
[`artifact-review-loop.md`](artifact-review-loop.md). Actual implementation
|
|
62
|
-
evidence may raise the profile but must not lower it.
|
|
63
|
-
|
|
64
|
-
Review profile selects mandatory lenses and independence. Protocol pass-count
|
|
65
|
-
semantics apply; parallel reviewer results remain separate passes even when they
|
|
66
|
-
reduce wall-clock time.
|
|
67
|
-
|
|
68
|
-
Select the reviewer role from the profile: `simple` uses `reviewer_fast`,
|
|
69
|
-
`medium` uses `reviewer_standard`, and `high` uses `reviewer_deep`. Root always
|
|
70
|
-
launches reviewer children; it never performs an implementation review inline.
|
|
71
|
-
An Adapter runs inline only inside its already assigned reviewer child.
|
|
72
|
-
|
|
73
|
-
A reviewer that fails before returning a usable result is recorded as failed
|
|
74
|
-
and closed, not as completed coverage. Escalation preserves completed coverage
|
|
75
|
-
and the stable Defect Ledger. A user pause, context compaction, slice commit,
|
|
76
|
-
worker replacement, or new turn also preserves them; none restarts the review
|
|
77
|
-
topology automatically.
|
|
78
|
-
|
|
79
|
-
Stop early when mandatory coverage and protocol clear-state requirements hold.
|
|
80
|
-
|
|
81
|
-
## Review Planning
|
|
82
|
-
|
|
83
|
-
Before the first implementation reviewer, root creates a short Review Plan:
|
|
84
|
-
|
|
85
|
-
- current profile and required independent lenses
|
|
86
|
-
- only stable intermediate checkpoints and their required lenses; move an unstable checkpoint to final coverage when later slices touch the same files, owners, or contracts
|
|
87
|
-
- any separate cleanup requirement, which must name a concrete evidenced reason that cannot fit the final spec/standards lens
|
|
88
|
-
- final code-review lenses and minimum independent coverage
|
|
89
|
-
- reviewer lineages that own affected-lens Closure
|
|
90
|
-
|
|
91
|
-
Create one durable activation record for each logical skill invocation. Record
|
|
92
|
-
`activation_id`, skill, owner, opened/closed state, and resume rule. Review Full
|
|
93
|
-
and lineage-preserving Closure passes stay inside that review skill's activation;
|
|
94
|
-
TDD repair cycles stay inside the active TDD activation. Cleanup, code review,
|
|
95
|
-
TDD, and debugger activations never share an ID, and a continuation resumes an
|
|
96
|
-
ID only for the same skill and authorized flow.
|
|
97
|
-
|
|
98
|
-
## Durable Review State
|
|
99
|
-
|
|
100
|
-
Do not create durable review state during implementation preflight. Immediately
|
|
101
|
-
before the first actual reviewer launch, persist the short Review Plan and
|
|
102
|
-
pending launch in the selected authority artifact under
|
|
103
|
-
`## Implementation Review State`. From that point onward this is the execution
|
|
104
|
-
ledger for review state; do not keep the authoritative history only in chat
|
|
105
|
-
context or a subagent summary.
|
|
106
|
-
|
|
107
|
-
Record at least:
|
|
108
|
-
|
|
109
|
-
- profile, completed pass count, and required coverage still outstanding
|
|
110
|
-
- current checkpoint/gate, review timing baseline, gate-local consecutive
|
|
111
|
-
Closure-wave count, and latest Closure wave ID
|
|
112
|
-
- planned mandatory final reviews and their lenses
|
|
113
|
-
- authority artifact kind/path and logical skill activation records
|
|
114
|
-
- each lineage ID, origin Full session, active session generation, Closure count,
|
|
115
|
-
rotation reason, live/timeout state, and `conclude_requested_at`
|
|
116
|
-
- any convergence audit epoch: trigger, triggering pass/wave/revision,
|
|
117
|
-
completion, dispositions, selected sessions, resume reason, and pass/wave/time
|
|
118
|
-
baselines used for its next rearm
|
|
119
|
-
- pending reviewer launches with launch ID, mode, lineage/session identity,
|
|
120
|
-
activation ID, target revision, checkpoint/gate, Closure wave ID when
|
|
121
|
-
applicable, assigned lenses, and start timestamp
|
|
122
|
-
- every completed review's mode, lineage/session identity, target revision,
|
|
123
|
-
checkpoint/gate, Closure wave ID, assigned lenses, start/end timestamps, and
|
|
124
|
-
outcome
|
|
125
|
-
- the stable Defect Ledger with transition history, reopen count, fixed revision,
|
|
126
|
-
verifying review, and any explicit risk acceptance
|
|
127
|
-
|
|
128
|
-
Write a pending launch before starting its reviewer. After the launch returns,
|
|
129
|
-
replace the pending record with either its usable completed result or a failed,
|
|
130
|
-
closed session record; only a usable result increments `review_passes`. A context
|
|
131
|
-
compaction, new turn, resumed task, or different root agent must reconcile every
|
|
132
|
-
pending launch with its recorded session before starting a replacement.
|
|
133
|
-
|
|
134
|
-
Update this section after each launch, usable reviewer result, repair batch,
|
|
135
|
-
closure, waiver, acceptance, reopen, or terminal outcome. A resumed executor
|
|
136
|
-
reconstructs accounting and lifecycle history from this persisted state. If the
|
|
137
|
-
state is missing or internally inconsistent after reviews began, return
|
|
138
|
-
`Blocked` until it is reconciled from available thread/session evidence; never
|
|
139
|
-
assume zero completed passes or silently replace an in-flight reviewer.
|
|
140
|
-
|
|
141
|
-
Plan mandatory final coverage before launching an intermediate review. Launch a
|
|
142
|
-
checkpoint only when its target is settled and later slices will not invalidate
|
|
143
|
-
the reviewed files, owners, or contracts. Otherwise move its lenses to final
|
|
144
|
-
coverage. Do not replace a required final lens with another fresh checkpoint
|
|
145
|
-
reviewer or a repeat broad audit.
|
|
146
|
-
|
|
147
|
-
The default shapes are:
|
|
148
|
-
|
|
149
|
-
- `simple`: validation only when policy does not require review; otherwise one
|
|
150
|
-
final Full review and affected-lens Closure only after repairs.
|
|
151
|
-
- `medium`: an explicit intermediate checkpoint may provide one required lens;
|
|
152
|
-
the final integrator covers every remaining lens, includes bounded cleanup in
|
|
153
|
-
spec/standards, and verifies its defects without a separate cleanup pass.
|
|
154
|
-
- `high`: use parallel independent tracks only for disjoint mandatory lenses;
|
|
155
|
-
the spec/standards track includes bounded cleanup, and affected-lens Closure
|
|
156
|
-
follows only after consolidated repairs. There is no separate cleanup pass by
|
|
157
|
-
default.
|
|
158
|
-
|
|
159
|
-
`code-review` uses one final reviewer covering both correctness and
|
|
160
|
-
spec/standards for `simple` and `medium`. For `high`, it uses two disjoint final
|
|
161
|
-
tracks unless earlier independent coverage already covered both axes and one
|
|
162
|
-
fresh final integrator receives their compact handoffs.
|
|
163
|
-
|
|
164
|
-
Before launching a fresh final reviewer, reconcile coverage on the settled
|
|
165
|
-
revision. If the latest usable Full or Closure covered every mandatory final
|
|
166
|
-
lens and left no open defect, mark final review complete and stop. Count cleanup
|
|
167
|
-
Closure only for the lenses explicitly assigned in the Review Plan; a
|
|
168
|
-
`cleanup-only` pass does not satisfy correctness or spec/standards coverage.
|
|
169
|
-
|
|
170
|
-
## Review Capsule
|
|
171
|
-
|
|
172
|
-
Use the protocol capsule with these implementation fields:
|
|
173
|
-
|
|
174
|
-
- the unanswered implementation question and any prior coverage it invalidates
|
|
175
|
-
- authority artifact kind/path, profile, current revision, checkpoint, and exact diff command
|
|
176
|
-
- changed paths and assigned `Review Focus` lenses
|
|
177
|
-
- source-of-truth docs and relevant Contract Test Ledger rows
|
|
178
|
-
- compact validation results and known verification gaps
|
|
179
|
-
|
|
180
|
-
For Closure, map changed paths and tests into the protocol Revision Map.
|
|
181
|
-
|
|
182
|
-
## Review Modes
|
|
183
|
-
|
|
184
|
-
Use protocol Full and Closure without redefining them. Implementation Closure
|
|
185
|
-
maps `affected_targets` to paths, tests, runtime contracts, and Review Focus
|
|
186
|
-
lenses. An already planned Full reviewer may verify a repair when its assigned
|
|
187
|
-
lenses cover it.
|
|
188
|
-
|
|
189
|
-
## Defect Lifecycle
|
|
190
|
-
|
|
191
|
-
Use the canonical protocol ledger and lifecycle without local aliases.
|
|
192
|
-
|
|
193
|
-
Implementation proof-only gaps may use `planned-final-verification` only when an
|
|
194
|
-
already scheduled code-review lens owns the proof; they remain open until
|
|
195
|
-
independently verified and never re-enter cleanup. Artifact proof-contract gaps
|
|
196
|
-
reopen artifact review. A pre-existing adjacent issue is non-blocking only as an
|
|
197
|
-
`improvement` with `follow-up-improvement`.
|
|
198
|
-
|
|
199
|
-
## Validation Evidence Reuse
|
|
200
|
-
|
|
201
|
-
Persist command/config identity, failure signature, target revision and changed
|
|
202
|
-
path/contract impact basis, secret-safe environment fingerprint, transitive
|
|
203
|
-
ownership/contract impact, and result. Reuse a known unrelated suite failure
|
|
204
|
-
only when every field matches; unknown environment or transitive impact fails
|
|
205
|
-
closed. Focused tests and every required check for a repair always rerun.
|
|
206
|
-
|
|
207
|
-
## Gate Ordering
|
|
208
|
-
|
|
209
|
-
1. Implement the slice and pass its tests/exit gate.
|
|
210
|
-
2. At an explicit intermediate checkpoint, run the required targeted
|
|
211
|
-
`code-review` directly under the Review Plan.
|
|
212
|
-
3. Repair one consolidated finding batch and use protocol Closure for the
|
|
213
|
-
affected lineages. An already planned Full reviewer may verify the repair when
|
|
214
|
-
its assigned lenses cover it.
|
|
215
|
-
At one gate, collect the usable results from all already-launched reviewers
|
|
216
|
-
before repairing, unless an immediate blocker invalidates the remaining
|
|
217
|
-
work. Do not turn individual findings into serial repair, validation, and
|
|
218
|
-
Closure micro-cycles. Repair compatible findings once, rerun each affected
|
|
219
|
-
validation once on the resulting revision, then launch one affected-lens
|
|
220
|
-
Closure wave.
|
|
221
|
-
4. Continue implementation only when checkpoint blockers are verified or the
|
|
222
|
-
Review Plan explicitly assigns their verification to an already planned reviewer
|
|
223
|
-
without violating the checkpoint's safety purpose.
|
|
224
|
-
5. After all implementation slices and validations settle, run one final code
|
|
225
|
-
review wave. `simple` and `medium` use one reviewer; `high` launches two
|
|
226
|
-
disjoint reviewer tracks in parallel. The spec/standards lens owns bounded
|
|
227
|
-
cleanup.
|
|
228
|
-
|
|
229
|
-
Intermediate code-review checkpoints do not run cleanup-review. A separate
|
|
230
|
-
cleanup pass is exceptional: run it only when the user, approved source, or repo
|
|
231
|
-
policy names a concrete evidenced simplification risk that cannot fit the final
|
|
232
|
-
spec/standards lens. `large` or `high` alone is not a reason. If an approved spec
|
|
233
|
-
names an intermediate cleanup checkpoint, return `Blocked` for spec revision.
|
|
234
|
-
|
|
235
|
-
Cleanup review runs at most once as a Full review for the whole spec. After its
|
|
236
|
-
findings are repaired, either use protocol Closure or give the final code
|
|
237
|
-
reviewer those stable defect IDs for verification. Never launch another Full
|
|
238
|
-
cleanup review over the repaired whole diff.
|
|
239
|
-
|
|
240
|
-
## Convergence And Stop Rules
|
|
241
|
-
|
|
242
|
-
Apply protocol repair, no-progress, stop, and waiver semantics. The following
|
|
243
|
-
audit is implementation-specific.
|
|
244
|
-
|
|
245
|
-
Before launching more reviewers, run one non-terminal convergence audit after
|
|
246
|
-
two consecutive Closure waves in one gate, ten total implementation review
|
|
247
|
-
passes, or 90 minutes when timing is available. One coordinated launch over all
|
|
248
|
-
affected lineages is one wave regardless of parallel pass count.
|
|
249
|
-
|
|
250
|
-
Persist one audit epoch with trigger, triggering pass/wave/revision,
|
|
251
|
-
completion, dispositions, selected sessions, and resume reason. It survives
|
|
252
|
-
resume. After a material repair/evidence change proves progress, rearm by
|
|
253
|
-
recording the current total pass count, current gate-local Closure-wave count,
|
|
254
|
-
and current timestamp as new baselines. The next audit opens only after a
|
|
255
|
-
post-rearm delta reaches two Closure waves in that gate, ten implementation
|
|
256
|
-
review passes, or 90 minutes; already-consumed counts or time cannot reopen it
|
|
257
|
-
immediately. Without progress do not rearm and use the existing stop rules.
|
|
258
|
-
Thresholds never approve, waive, downgrade, or block by themselves, and
|
|
259
|
-
distinct failure mechanics remain distinct even when they protect one
|
|
260
|
-
invariant.
|
|
261
|
-
|
|
262
|
-
Return `Approved` for the final settled revision only when protocol state is
|
|
263
|
-
`clear` and:
|
|
264
|
-
|
|
265
|
-
- every mandatory lens has independent coverage
|
|
266
|
-
- final validation and required cleanup/code-review gates ran
|
|
267
|
-
|
|
268
|
-
Map protocol `stopped` to `Blocked`. Implementation-specific blockers also
|
|
269
|
-
include:
|
|
270
|
-
|
|
271
|
-
- a defect reopens repeatedly and exposes a source-of-truth or repair-design
|
|
272
|
-
contradiction that root cannot resolve from current evidence
|
|
273
|
-
- an execution risk remains open without an explicit user decision to accept it
|
|
274
|
-
|
|
275
|
-
Do not mark implementation `Blocked` merely because review has run several
|
|
276
|
-
times. Resolve the repair, evidence, or decision problem first.
|
|
277
|
-
|
|
278
|
-
Map protocol `waived` to `Waived` and preserve skipped coverage and open risks.
|
|
279
|
-
It remains non-approval; target authority or downstream policy may still block
|
|
280
|
-
delivery.
|
|
281
|
-
|
|
282
|
-
## Required Handoff
|
|
283
|
-
|
|
284
|
-
Use the protocol result envelope and add:
|
|
285
|
-
|
|
286
|
-
```text
|
|
287
|
-
Implementation Review Profile: <simple | medium | high>
|
|
288
|
-
Review Outcome: <Approved | Blocked | Waived>
|
|
289
|
-
Authority Artifact: <approved spec path>
|
|
290
|
-
Implementation Checkpoint: <checkpoint or final>
|
|
291
|
-
```
|
|
292
|
-
|
|
293
|
-
## Contract Test Ledger
|
|
294
|
-
|
|
295
|
-
| Invariant | Risk It Prevents | First Test / Proof | Status |
|
|
296
|
-
| --- | --- | --- | --- |
|
|
297
|
-
| Review pass counts are audit metrics, while one durable Review Plan and Defect Ledger span the entire spec. | Each slice silently recreates a new loop or a repairable spec blocks on an arbitrary count. | Manual eval scenario 15 | planned |
|
|
298
|
-
| Intermediate checkpoints never trigger cleanup; final review is one settled profile-selected wave, and separate cleanup is exceptional. | Per-slice hygiene or size-driven cleanup adds latency and restarts review over unstable work. | Manual eval scenario 17 | planned |
|
|
299
|
-
| Audit epochs persist across resume and rearm only after material progress. | Thresholds repeatedly trigger audits or become terminal limits. | Manual eval scenario 12 | planned |
|
|
300
|
-
|
|
301
|
-
Keep these rows `planned` until the corresponding operator eval is run and its
|
|
302
|
-
result is saved.
|
|
@@ -1,9 +0,0 @@
|
|
|
1
|
-
name = "implementer_deep"
|
|
2
|
-
description = "Write-capable implementation worker for an isolated approved slice with material technical uncertainty."
|
|
3
|
-
nickname_candidates = ["Foundry", "Helix", "Vector"]
|
|
4
|
-
model = "gpt-5.6-sol"
|
|
5
|
-
model_reasoning_effort = "high"
|
|
6
|
-
sandbox_mode = "workspace-write"
|
|
7
|
-
developer_instructions = """
|
|
8
|
-
Implement only the assigned isolated high-complexity ticket slice. Resolve technical uncertainty from repository evidence without changing approved product behavior, ownership, or slice boundaries. Respect exclusive write scope and stop on overlap or decision drift. You are not alone in the repository: preserve unrelated and concurrent changes and never revert work you do not own. Use behavior-first proof and return changed files, acceptance proof, skipped checks, risks, decision deltas, and blockers to the root integrator.
|
|
9
|
-
"""
|
|
@@ -1,9 +0,0 @@
|
|
|
1
|
-
name = "researcher_standard"
|
|
2
|
-
description = "Read-only external research against primary sources with claim-level citations."
|
|
3
|
-
nickname_candidates = ["Atlas", "Index", "Scribe"]
|
|
4
|
-
model = "gpt-5.6-sol"
|
|
5
|
-
model_reasoning_effort = "medium"
|
|
6
|
-
sandbox_mode = "read-only"
|
|
7
|
-
developer_instructions = """
|
|
8
|
-
Research one bounded external question from the supplied Research Capsule. Prefer official documentation, specifications, first-party source code, changelogs, release notes, issue trackers, APIs, and schemas. Map every material claim to the exact primary source and include source version/date when available. Separate sourced facts from repository inference; expose conflicts, stale evidence, uncertainty, and missing proof. Never edit files, create artifacts, change code, or broaden into unrelated reading. Return a concise short answer, claim-to-source ledger, decision implications, and unresolved questions to the root integrator.
|
|
9
|
-
"""
|
|
@@ -1,9 +0,0 @@
|
|
|
1
|
-
name = "reviewer_fast"
|
|
2
|
-
description = "Fast independent read-only reviewer for simple plans, specs, tickets, cleanup, and code changes."
|
|
3
|
-
nickname_candidates = ["Dash", "Jet", "Swift"]
|
|
4
|
-
model = "gpt-5.6-terra"
|
|
5
|
-
model_reasoning_effort = "medium"
|
|
6
|
-
sandbox_mode = "read-only"
|
|
7
|
-
developer_instructions = """
|
|
8
|
-
Review independently and proportionately. Prioritize concrete correctness, scope, contract, and verification gaps; avoid speculative edge cases and cosmetic comments. Never edit files. Return concise findings with evidence, severity, confidence, and proposed fixes to the parent.
|
|
9
|
-
"""
|