codex-orchestrator 2.0.3 → 2.0.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -427
- package/README.md +161 -37
- package/dist/src/index.d.ts +1 -1
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/v2/acceptance-proof.d.ts +5 -0
- package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
- package/dist/src/v2/acceptance-proof.js +10 -2
- package/dist/src/v2/acceptance-proof.js.map +1 -1
- package/dist/src/v2/adapters/gh-issue-adapter.d.ts.map +1 -1
- package/dist/src/v2/adapters/gh-issue-adapter.js +6 -7
- package/dist/src/v2/adapters/gh-issue-adapter.js.map +1 -1
- package/dist/src/v2/adapters/gh-pull-request-adapter.d.ts +7 -1
- package/dist/src/v2/adapters/gh-pull-request-adapter.d.ts.map +1 -1
- package/dist/src/v2/adapters/gh-pull-request-adapter.js +288 -0
- package/dist/src/v2/adapters/gh-pull-request-adapter.js.map +1 -1
- package/dist/src/v2/adapters/pull-requests.d.ts +69 -0
- package/dist/src/v2/adapters/pull-requests.d.ts.map +1 -1
- package/dist/src/v2/adapters/pull-requests.js +48 -0
- package/dist/src/v2/adapters/pull-requests.js.map +1 -1
- package/dist/src/v2/adapters/worktree.d.ts +1 -0
- package/dist/src/v2/adapters/worktree.d.ts.map +1 -1
- package/dist/src/v2/adapters/worktree.js +10 -1
- package/dist/src/v2/adapters/worktree.js.map +1 -1
- package/dist/src/v2/cli-contract.d.ts +3 -3
- package/dist/src/v2/cli-contract.d.ts.map +1 -1
- package/dist/src/v2/cli-contract.js +1 -3
- package/dist/src/v2/cli-contract.js.map +1 -1
- package/dist/src/v2/cli.d.ts +33 -0
- package/dist/src/v2/cli.d.ts.map +1 -0
- package/dist/src/v2/{candidate-cli.js → cli.js} +55 -39
- package/dist/src/v2/cli.js.map +1 -0
- package/dist/src/v2/code-review-report.d.ts +1 -1
- package/dist/src/v2/code-review-report.d.ts.map +1 -1
- package/dist/src/v2/code-review-report.js +2 -2
- package/dist/src/v2/code-review-report.js.map +1 -1
- package/dist/src/v2/codex-process.d.ts.map +1 -1
- package/dist/src/v2/codex-process.js +12 -1
- package/dist/src/v2/codex-process.js.map +1 -1
- package/dist/src/v2/config.d.ts +2 -3
- package/dist/src/v2/config.d.ts.map +1 -1
- package/dist/src/v2/config.js +0 -3
- package/dist/src/v2/config.js.map +1 -1
- package/dist/src/v2/contained-report-operation.d.ts +2 -2
- package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
- package/dist/src/v2/contained-report-operation.js +1 -1
- package/dist/src/v2/contained-report-operation.js.map +1 -1
- package/dist/src/v2/containment.d.ts +15 -2
- package/dist/src/v2/containment.d.ts.map +1 -1
- package/dist/src/v2/containment.js +43 -6
- package/dist/src/v2/containment.js.map +1 -1
- package/dist/src/v2/direct-delivery.d.ts +5 -10
- package/dist/src/v2/direct-delivery.d.ts.map +1 -1
- package/dist/src/v2/direct-delivery.js +32 -91
- package/dist/src/v2/direct-delivery.js.map +1 -1
- package/dist/src/v2/proof-report.d.ts.map +1 -1
- package/dist/src/v2/proof-report.js +55 -29
- package/dist/src/v2/proof-report.js.map +1 -1
- package/dist/src/v2/review-feedback-coordinator.d.ts +54 -0
- package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -0
- package/dist/src/v2/review-feedback-coordinator.js +245 -0
- package/dist/src/v2/review-feedback-coordinator.js.map +1 -0
- package/dist/src/v2/review-feedback.d.ts +127 -0
- package/dist/src/v2/review-feedback.d.ts.map +1 -0
- package/dist/src/v2/review-feedback.js +436 -0
- package/dist/src/v2/review-feedback.js.map +1 -0
- package/dist/src/v2/run-issue.d.ts +63 -9
- package/dist/src/v2/run-issue.d.ts.map +1 -1
- package/dist/src/v2/run-issue.js +798 -78
- package/dist/src/v2/run-issue.js.map +1 -1
- package/dist/src/v2/run-store.d.ts +49 -5
- package/dist/src/v2/run-store.d.ts.map +1 -1
- package/dist/src/v2/run-store.js +138 -44
- package/dist/src/v2/run-store.js.map +1 -1
- package/dist/src/v2/runtime.d.ts +40 -3
- package/dist/src/v2/runtime.d.ts.map +1 -1
- package/dist/src/v2/runtime.js +245 -52
- package/dist/src/v2/runtime.js.map +1 -1
- package/dist/src/v2/setup-cli.d.ts.map +1 -1
- package/dist/src/v2/setup-cli.js +4 -11
- package/dist/src/v2/setup-cli.js.map +1 -1
- package/dist/src/v2/setup-runtime.d.ts.map +1 -1
- package/dist/src/v2/setup-runtime.js +1 -61
- package/dist/src/v2/setup-runtime.js.map +1 -1
- package/dist/src/v2/setup-store.d.ts +0 -5
- package/dist/src/v2/setup-store.d.ts.map +1 -1
- package/dist/src/v2/setup-store.js +3 -106
- package/dist/src/v2/setup-store.js.map +1 -1
- package/dist/src/v2/setup.d.ts +6 -46
- package/dist/src/v2/setup.d.ts.map +1 -1
- package/dist/src/v2/setup.js +12 -294
- package/dist/src/v2/setup.js.map +1 -1
- package/dist/src/v2/workflow-assets.d.ts +19 -11
- package/dist/src/v2/workflow-assets.d.ts.map +1 -1
- package/dist/src/v2/workflow-assets.js +132 -40
- package/dist/src/v2/workflow-assets.js.map +1 -1
- package/docs/deep-dive.md +328 -56
- package/internal-workflow/docs/agents/bugfix-quality-gate.md +11 -0
- package/internal-workflow/docs/agents/coding-skill-routing.md +116 -196
- package/internal-workflow/docs/agents/contract-test-ledger.md +11 -1
- package/internal-workflow/docs/agents/review-gates.md +32 -39
- package/internal-workflow/docs/agents/review-protocol.md +75 -147
- package/internal-workflow/evals/coding-skill-evals.json +84 -0
- package/internal-workflow/manifest.json +1 -1
- package/internal-workflow/operations/acceptance-proof/SKILL.md +7 -1
- package/internal-workflow/operations/ambiguity-review/SKILL.md +2 -0
- package/internal-workflow/operations/code-review/SKILL.md +21 -1
- package/internal-workflow/operations/implementation/SKILL.md +22 -1
- package/internal-workflow/operations/spec-author/SKILL.md +10 -1
- package/internal-workflow/operations/spec-review/SKILL.md +10 -1
- package/internal-workflow/operations/triage/SKILL.md +10 -1
- package/internal-workflow/schemas/code-review-v1.json +1 -1
- package/internal-workflow/schemas/proof-report-v1.json +1 -1
- package/internal-workflow/skills/agent-auto/SKILL.md +6 -1
- package/internal-workflow/skills/code-debugger/SKILL.md +122 -0
- package/internal-workflow/skills/code-debugger/agents/openai.yaml +7 -0
- package/internal-workflow/skills/code-review/SKILL.md +51 -17
- package/internal-workflow/skills/code-review/references/cleanup-lens.md +52 -0
- package/internal-workflow/skills/implementation-spec-maker/SKILL.md +15 -6
- package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +1 -1
- package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +2 -2
- package/internal-workflow/skills/implementation-spec-review/SKILL.md +108 -204
- package/internal-workflow/skills/implementation-spec-review/evals/evals.json +24 -0
- package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +93 -0
- package/internal-workflow/skills/small-task-implementer/SKILL.md +15 -8
- package/internal-workflow/skills/spec-implementer/SKILL.md +101 -172
- package/internal-workflow/skills/spec-implementer/evals/evals.json +30 -0
- package/internal-workflow/skills/spec-implementer/references/review-loop.md +100 -0
- package/internal-workflow/skills/tdd/SKILL.md +20 -6
- package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
- package/internal-workflow/skills/tdd/evals/evals.json +18 -0
- package/internal-workflow/skills/tdd/mocking.md +3 -42
- package/internal-workflow/skills/tdd/refactoring.md +6 -8
- package/package.json +9 -6
- package/dist/src/v2/adapters/target-activity-fence.d.ts +0 -23
- package/dist/src/v2/adapters/target-activity-fence.d.ts.map +0 -1
- package/dist/src/v2/adapters/target-activity-fence.js +0 -249
- package/dist/src/v2/adapters/target-activity-fence.js.map +0 -1
- package/dist/src/v2/candidate-cli.d.ts +0 -26
- package/dist/src/v2/candidate-cli.d.ts.map +0 -1
- package/dist/src/v2/candidate-cli.js.map +0 -1
- package/dist/src/v2/legacy-cutover.d.ts +0 -52
- package/dist/src/v2/legacy-cutover.d.ts.map +0 -1
- package/dist/src/v2/legacy-cutover.js +0 -87
- package/dist/src/v2/legacy-cutover.js.map +0 -1
- package/internal-workflow/docs/agents/artifact-review-loop.md +0 -267
- package/internal-workflow/docs/agents/implementation-review-loop.md +0 -302
- package/internal-workflow/operations/cleanup-review/SKILL.md +0 -3
- package/internal-workflow/operations/spec-implementation/SKILL.md +0 -3
- package/internal-workflow/profiles/implementer_deep.toml +0 -9
- package/internal-workflow/profiles/researcher_standard.toml +0 -9
- package/internal-workflow/profiles/reviewer_fast.toml +0 -9
- package/internal-workflow/skills/cleanup-review/SKILL.md +0 -84
- package/internal-workflow/skills/cleanup-review/agents/openai.yaml +0 -6
- package/internal-workflow/skills/codebase-design/DEEPENING.md +0 -35
- package/internal-workflow/skills/codebase-design/DESIGN-IT-TWICE.md +0 -50
- package/internal-workflow/skills/codebase-design/SKILL.md +0 -82
- package/internal-workflow/skills/codebase-design/agents/openai.yaml +0 -6
- package/internal-workflow/skills/research/SKILL.md +0 -107
- package/internal-workflow/skills/research/agents/openai.yaml +0 -6
- package/internal-workflow/skills/ui-evidence-proof/SKILL.md +0 -123
- package/internal-workflow/skills/ui-evidence-proof/agents/openai.yaml +0 -6
|
@@ -5,193 +5,122 @@ description: "Executes approved specs continuously with honest checklist updates
|
|
|
5
5
|
|
|
6
6
|
# Spec Implementer
|
|
7
7
|
|
|
8
|
-
Execute
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
8
|
+
Execute one approved implementation spec continuously. Keep its checklist
|
|
9
|
+
truthful and stop only at a real authority, evidence, safety, or explicit pause
|
|
10
|
+
boundary. Do not redesign approved scope.
|
|
11
|
+
|
|
12
|
+
Read before editing:
|
|
13
|
+
|
|
14
|
+
- the complete approved spec and applicable repository instructions;
|
|
15
|
+
- `references/review-loop.md` for approved-spec review ownership;
|
|
16
|
+
- `../../docs/agents/contract-test-ledger.md` only when the spec contains
|
|
17
|
+
material contract invariants.
|
|
18
|
+
|
|
19
|
+
## Modes
|
|
20
|
+
|
|
21
|
+
Compact and full specs use the same direct phase flow. `compact` describes
|
|
22
|
+
document density, not implementation size or risk. Full mode adds only the
|
|
23
|
+
concrete `Risk Controls`, stop conditions, validation, or coordination contract
|
|
24
|
+
already present in the approved spec; it does not add review or reporting
|
|
25
|
+
ceremony by itself.
|
|
26
|
+
|
|
27
|
+
Use multiple agents only when the spec defines perfectly disjoint write scopes
|
|
28
|
+
and one integrator. Otherwise execute single-agent.
|
|
29
|
+
|
|
30
|
+
## Preflight
|
|
31
|
+
|
|
32
|
+
1. Confirm `status`, `spec_mode`, `implementation_size`, `review_profile`,
|
|
33
|
+
repository count, authority, scope, and exclusions.
|
|
34
|
+
2. Confirm first-phase targets, required services/env/data/fixtures, commands,
|
|
35
|
+
observable proof, protected paths, and rejected approaches.
|
|
36
|
+
3. Stop if execution needs invented paths, symbols, contracts, commands, or
|
|
37
|
+
product decisions; if reality differs only in a bounded technical detail,
|
|
38
|
+
resolve it from repository evidence and record the adjustment.
|
|
39
|
+
4. Keep an intermediate review checkpoint only when the spec explicitly names
|
|
40
|
+
a stable high-risk slice that later work will not invalidate. Move every
|
|
41
|
+
ordinary or unstable checkpoint to final review.
|
|
42
|
+
5. Do not create `## Implementation Review State` yet.
|
|
43
|
+
|
|
44
|
+
## Implementation
|
|
45
|
+
|
|
46
|
+
For each phase:
|
|
47
|
+
|
|
48
|
+
1. Re-read its scope and preconditions.
|
|
49
|
+
2. Implement narrow vertical behavior slices through `$tdd` when applicable.
|
|
50
|
+
3. Update reached checklist and Contract Test Ledger items at natural
|
|
51
|
+
checkpoints; never save all updates for the end.
|
|
52
|
+
4. Run the phase's targeted exit proof.
|
|
53
|
+
5. Record a short `Blocked:` note for any item that cannot complete.
|
|
54
|
+
6. Continue immediately when the exit proof passes and no stop condition
|
|
55
|
+
applies.
|
|
56
|
+
|
|
57
|
+
Do not add cleanup, comments, helpers, abstractions, retries, flags, fallbacks,
|
|
58
|
+
or compatibility paths outside the spec. A small implementation adjustment is
|
|
59
|
+
allowed only when it preserves approved behavior and is supported by current
|
|
60
|
+
repository evidence.
|
|
57
61
|
|
|
58
62
|
## Git Checkpoints
|
|
59
63
|
|
|
60
|
-
Default to
|
|
64
|
+
Default to no commits. `$spec-implementer` does not authorize Git writes.
|
|
61
65
|
|
|
62
|
-
|
|
66
|
+
Use per-slice commits only when explicitly authorized and they materially help
|
|
67
|
+
a disjoint multi-agent handoff, planned cross-session pause, or approved
|
|
68
|
+
rollback boundary. Require a passed slice proof, stage only owned paths, follow
|
|
69
|
+
`$commit`, and never push without separate authority. File/slice count and risk
|
|
70
|
+
profile alone do not justify checkpoints.
|
|
63
71
|
|
|
64
|
-
|
|
65
|
-
- an explicitly planned pause or continuation in another session
|
|
66
|
-
- a destructive or rollback boundary whose isolated commit is part of the approved safety plan
|
|
72
|
+
## Review And Validation
|
|
67
73
|
|
|
68
|
-
|
|
74
|
+
Run targeted behavior tests and the smallest affected integration checks first.
|
|
75
|
+
Use a full repository suite only when the spec or repository policy requires
|
|
76
|
+
it, broad contract fan-out cannot be isolated, or the task is genuinely high.
|
|
69
77
|
|
|
70
|
-
|
|
78
|
+
Run final `$code-review` only when the spec, repository policy, or
|
|
79
|
+
`../../docs/agents/review-gates.md` applies. Ordinary medium work gets one
|
|
80
|
+
`reviewer_standard` Full review on the settled diff. High gets two disjoint
|
|
81
|
+
`reviewer_deep` lenses. Cleanup stays inside spec/standards review.
|
|
71
82
|
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
83
|
+
Immediately before the first reviewer launch, create only the minimal
|
|
84
|
+
`## Implementation Review State` required by `references/review-loop.md`.
|
|
85
|
+
Reconcile a recorded live session before replacement after interruption.
|
|
75
86
|
|
|
76
|
-
|
|
87
|
+
Repair compatible findings once and rerun affected validation. Coordinator
|
|
88
|
+
verification closes ordinary medium/low behavior-preserving repairs. Use
|
|
89
|
+
Closure only for critical/high, protected trust/data/concurrency/shared-contract
|
|
90
|
+
impact, or invalidated mandatory coverage. Do not restart broad Full review
|
|
91
|
+
unless the repair actually invalidated its coverage.
|
|
77
92
|
|
|
78
|
-
##
|
|
79
|
-
|
|
80
|
-
For every phase:
|
|
81
|
-
|
|
82
|
-
- Confirm phase dependencies and preconditions.
|
|
83
|
-
- Execute only the steps assigned to that phase.
|
|
84
|
-
- Update the spec checklist according to its mode.
|
|
85
|
-
- Update any reached Contract Test Ledger rows as planned -> red -> green, or blocked with the missing seam/proof.
|
|
86
|
-
- Run the phase exit gate.
|
|
87
|
-
- If a `Review Checkpoint` applies after this phase, execute it through the
|
|
88
|
-
Module only when the target is settled and later slices do not invalidate its
|
|
89
|
-
files/contracts. Otherwise record that its lenses moved to final coverage and
|
|
90
|
-
continue without launching an unstable review.
|
|
91
|
-
- Reconcile unchecked items for the current phase.
|
|
92
|
-
- Re-check applicable `Risk Controls` before leaving the phase.
|
|
93
|
-
- Check whether the phase introduced shallow modules, duplicated source-of-truth logic, or tests coupled to implementation details; fix only when inside approved scope, otherwise report it.
|
|
94
|
-
- Run the repo architecture check when available, applicable, and required by the spec or repo policy.
|
|
95
|
-
- If `per-slice` applies and this phase completes an implementation slice, create and verify its checkpoint commit before continuing.
|
|
96
|
-
- Continue to the next phase only when the exit gate passes and no stop condition applies.
|
|
97
|
-
- If the phase exit gate says `User Pause: Required`, stop and wait for the user's explicit command.
|
|
98
|
-
|
|
99
|
-
## Review And Signoff
|
|
100
|
-
|
|
101
|
-
Do not run a dedicated review subagent after every phase by default. Do run one at explicit `Review Checkpoints`; these are risk gates, not optional status updates.
|
|
102
|
-
|
|
103
|
-
Before every reviewer launch, apply the Module's launch and reconciliation
|
|
104
|
-
rules to the persisted `## Implementation Review State`; do not restate or
|
|
105
|
-
replace those rules in this skill.
|
|
106
|
-
|
|
107
|
-
Require final `$code-review` coverage when any of these apply:
|
|
108
|
-
|
|
109
|
-
- the spec explicitly requires it
|
|
110
|
-
- the repo policy requires it
|
|
111
|
-
- the change is medium or large
|
|
112
|
-
- the change touches multiple runtime files or shared behavior
|
|
113
|
-
- the change touches API contracts, DTOs, schemas, persistence, auth, permissions, payments, caching, concurrency, background jobs, or shared state
|
|
114
|
-
|
|
115
|
-
When `$code-review` is required for a checkpoint or final gate, keep orchestration at root so `$code-review` can launch the profile-selected reviewer topology. Invoking `$spec-implementer` authorizes that review; if the required role is unavailable, report the gate as unavailable/blocked instead of self-certifying it.
|
|
93
|
+
## Stop Conditions
|
|
116
94
|
|
|
117
|
-
|
|
118
|
-
For `simple` and `medium`, one reviewer covers both lenses. For `high`, launch
|
|
119
|
-
the correctness and spec/standards reviewers in parallel; the spec/standards
|
|
120
|
-
lens includes bounded cleanup. Run separate `$cleanup-review` only when the
|
|
121
|
-
user, approved source, or repo policy names a concrete evidenced reason that
|
|
122
|
-
cannot fit that lens; size or risk labels alone are insufficient. Integrate safe
|
|
123
|
-
fixes and rerun relevant validation before continuing.
|
|
124
|
-
Before launching a fresh final reviewer, reconcile the settled revision against
|
|
125
|
-
the Review Plan. Stop when an approved Full or Closure already covers every
|
|
126
|
-
mandatory final lens; otherwise run `$code-review` only for the uncovered
|
|
127
|
-
lenses. A `cleanup-only` result never substitutes for correctness or
|
|
128
|
-
spec/standards coverage.
|
|
95
|
+
Stop and report the exact blocker when:
|
|
129
96
|
|
|
130
|
-
|
|
97
|
+
- a precondition, source contract, required proof, or protected path differs
|
|
98
|
+
materially from the approved spec;
|
|
99
|
+
- exact execution requires a new product, scope, ownership, or risky trade-off
|
|
100
|
+
decision;
|
|
101
|
+
- required validation or reviewer is unavailable and no approved substitute
|
|
102
|
+
exists;
|
|
103
|
+
- multi-agent scopes overlap or integration ownership is missing;
|
|
104
|
+
- an explicit user pause or halt condition is reached.
|
|
131
105
|
|
|
132
|
-
|
|
106
|
+
Do not stop merely because the work is broad, review took time, or a medium/low
|
|
107
|
+
finding required one repair.
|
|
133
108
|
|
|
134
|
-
|
|
135
|
-
before every new launch.
|
|
109
|
+
## Completion
|
|
136
110
|
|
|
137
|
-
|
|
111
|
+
Complete only when reached checklist items are reconciled, affected proof and
|
|
112
|
+
required review pass, protected paths and rejected approaches remain intact,
|
|
113
|
+
and every unfinished item has a concrete status.
|
|
138
114
|
|
|
139
|
-
|
|
140
|
-
- Keep one integrator responsible for merge sequencing, handoff checks, final validation, and checklist reconciliation.
|
|
141
|
-
- Respect exclusive write scopes, handoff artifacts, forbidden overlap, and merge order exactly as written.
|
|
142
|
-
- Never let two agents edit the same file, generated artifact, schema, source-of-truth rule, or shared contract at the same time.
|
|
143
|
-
- If the spec lacks a clear integrator contract, execute single-agent or stop and ask for clarification.
|
|
115
|
+
For ordinary medium work report only:
|
|
144
116
|
|
|
145
|
-
|
|
117
|
+
- behavior/contract implemented;
|
|
118
|
+
- review result and repaired/open findings;
|
|
119
|
+
- affected validation;
|
|
120
|
+
- skipped checks and residual risk;
|
|
121
|
+
- changed files and any authorized commits.
|
|
146
122
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
- the saved spec contains unresolved template text or alternative commands
|
|
152
|
-
- completing the task would require touching a Protected Path or using a Rejected Approach
|
|
153
|
-
- validation cannot prove the intended behavior with available repo context
|
|
154
|
-
- implementation would require unapproved scope, abstraction, migration, compatibility logic, or cleanup
|
|
155
|
-
- a `Risk Controls` rule would be violated or is contradicted by repo reality
|
|
156
|
-
- review exposes a real ambiguity that would require guessing
|
|
157
|
-
- a user pause is required and the user has not explicitly said to proceed
|
|
158
|
-
|
|
159
|
-
## Completion Standard
|
|
160
|
-
|
|
161
|
-
Do not mark the task complete until:
|
|
162
|
-
|
|
163
|
-
- completed checklist items are checked off according to the spec mode
|
|
164
|
-
- every remaining unchecked item is blocked, intentionally unfinished, not applicable, or halted by an explicit stop condition
|
|
165
|
-
- every reached phase exit gate has passed
|
|
166
|
-
- applicable `Risk Controls` remained satisfied
|
|
167
|
-
- reached Contract Test Ledger rows are green or explicitly blocked with evidence
|
|
168
|
-
- validation commands and behavior proof have run, or skipped checks have a concrete reason
|
|
169
|
-
- required review/signoff gates have run and grounded findings are fixed or blocked with evidence
|
|
170
|
-
- the whole-spec Review Plan, review-pass history, and stable defect lifecycle remain
|
|
171
|
-
consistent with `implementation-review-loop.md`
|
|
172
|
-
- protected paths remained untouched and rejected approaches were not used
|
|
173
|
-
- required comments/docblocks were added only where the spec demanded them
|
|
174
|
-
- the chosen Git checkpoint strategy was followed, and every created or skipped checkpoint was recorded
|
|
175
|
-
- final user handoff is allowed by the spec
|
|
176
|
-
|
|
177
|
-
## Final Risk Handoff
|
|
178
|
-
|
|
179
|
-
For medium/high-risk specs, the final chat response must include a compact `Final Risk Handoff` block. Do not make the user ask for this separately, and do not replace it with a generic summary.
|
|
180
|
-
|
|
181
|
-
Include:
|
|
182
|
-
|
|
183
|
-
- **Contract implemented:** the one behavior/contract delivered, in user-facing terms.
|
|
184
|
-
- **High-risk checkpoints:** each required checkpoint, review result, fixed findings, and any stop/continue decision.
|
|
185
|
-
- **Main invariants proved:** the key Contract Test Ledger rows or equivalent proofs and their status.
|
|
186
|
-
- **Code-review findings:** high/critical findings fixed, remaining medium/low findings, or `none`.
|
|
187
|
-
- **Fixes after review:** concrete fixes made because of cleanup/code review, or `none`.
|
|
188
|
-
- **Validation:** exact commands/proofs that passed.
|
|
189
|
-
- **Skipped checks:** skipped or blocked checks with concrete reasons.
|
|
190
|
-
- **Residual risks:** accepted remaining risks or `none`.
|
|
191
|
-
- **Checkpoint commits:** slice-to-commit mapping or `none`; do not narrate hypothetical checkpoints that were never authorized or attempted.
|
|
192
|
-
- **Implementation reviews:** profile, total review passes, Full/Closure count,
|
|
193
|
-
mandatory coverage, verified defect IDs, accepted-risk IDs with authority and
|
|
194
|
-
reason, and open defect IDs.
|
|
195
|
-
- **Files by role:** state owner, orchestration, side effects, UI/projection, tests, docs/copy, as applicable.
|
|
196
|
-
|
|
197
|
-
Only create a separate report file when the spec requires it or the work is broad enough that chat would lose important evidence, such as multi-agent execution, multiple review passes with findings, skipped live checks, production validation, or handoff to another person. Otherwise keep the spec checklist/ledger as the durable artifact and the final response as the concise decision packet.
|
|
123
|
+
For high, actual Closure, accepted risk, interrupted recovery, or multi-agent
|
|
124
|
+
delivery, add the relevant invariants, reviewer coverage, defect IDs, session
|
|
125
|
+
recovery, and handoff ownership. Do not create a separate report file unless
|
|
126
|
+
the spec or the complexity of that exceptional handoff requires it.
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schema_version": 1,
|
|
3
|
+
"skill": "spec-implementer",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "state-created-at-launch",
|
|
7
|
+
"prompt": "Execute an approved medium spec whose final review has not started yet.",
|
|
8
|
+
"expected": ["do not create review state during implementation", "persist minimal state immediately before reviewer launch"],
|
|
9
|
+
"forbidden": ["create per-slice review bookkeeping"]
|
|
10
|
+
},
|
|
11
|
+
{
|
|
12
|
+
"id": "medium-one-final-review",
|
|
13
|
+
"prompt": "Execute a normal medium spec with several vertical slices and no stable high-risk checkpoint.",
|
|
14
|
+
"expected": ["implement continuously", "run one final reviewer_standard on the settled diff"],
|
|
15
|
+
"forbidden": ["review after every slice", "run a full repository suite from file count"]
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"id": "closure-stays-affected",
|
|
19
|
+
"prompt": "Final review finds one high defect in a shared schema and several ordinary low findings.",
|
|
20
|
+
"expected": ["repair once", "Closure verifies only the affected high-risk contract"],
|
|
21
|
+
"forbidden": ["restart all Full reviewers"]
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"id": "resume-live-reviewer",
|
|
25
|
+
"prompt": "Resume after a reviewer poll timed out while the recorded session is still live.",
|
|
26
|
+
"expected": ["reconcile the existing session"],
|
|
27
|
+
"forbidden": ["mark it failed from timeout alone", "launch a duplicate reviewer"]
|
|
28
|
+
}
|
|
29
|
+
]
|
|
30
|
+
}
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
# Approved Spec Implementation Review Loop
|
|
2
|
+
|
|
3
|
+
This reference owns review orchestration for `$spec-implementer`. Read it only
|
|
4
|
+
when executing an approved implementation spec. Shared Full/Closure and defect
|
|
5
|
+
mechanics live in `../../../docs/agents/review-protocol.md`.
|
|
6
|
+
|
|
7
|
+
Direct work and deterministic issue delivery use normal TDD and review gates;
|
|
8
|
+
they must not create Implementation Review State.
|
|
9
|
+
|
|
10
|
+
## Authority
|
|
11
|
+
|
|
12
|
+
Only an approved implementation spec may own `## Implementation Review State`.
|
|
13
|
+
PRDs, tickets, architecture notes, and chat summaries are not review-state
|
|
14
|
+
owners. A substantive spec change returns through artifact review before
|
|
15
|
+
implementation continues.
|
|
16
|
+
|
|
17
|
+
Use the spec's `review_profile`; if absent, infer it from current evidence:
|
|
18
|
+
|
|
19
|
+
- `simple`: narrow change with direct proof;
|
|
20
|
+
- `medium`: default for ordinary implementation;
|
|
21
|
+
- `high`: material failure consequence (financial side effect,
|
|
22
|
+
unauthorized/cross-owner behavior, durable corruption, or materially false
|
|
23
|
+
production result) plus an uncertainty amplifier (concurrency or event
|
|
24
|
+
ordering, delayed/background callbacks, retry/idempotency/recovery,
|
|
25
|
+
ownership transitions, or shared state across consumers).
|
|
26
|
+
|
|
27
|
+
Implementation evidence may raise but never lower the approved profile.
|
|
28
|
+
Recheck the settled diff immediately before the first reviewer launch and
|
|
29
|
+
persist any required raise before launching reviewers.
|
|
30
|
+
|
|
31
|
+
## Default Review Shape
|
|
32
|
+
|
|
33
|
+
Implement continuously through vertical slices. Validate each affected behavior
|
|
34
|
+
and run one final review on the settled diff when the gate applies.
|
|
35
|
+
|
|
36
|
+
- `simple`: one `reviewer_fast` when review is required.
|
|
37
|
+
- `medium`: one `reviewer_standard`, one bounded final Full, no intermediate
|
|
38
|
+
checkpoint by default.
|
|
39
|
+
- `high`: two parallel `reviewer_deep` Full reviews with disjoint correctness
|
|
40
|
+
and spec/standards lenses.
|
|
41
|
+
|
|
42
|
+
Add an intermediate checkpoint only when the approved spec explicitly names a
|
|
43
|
+
stable high-risk slice whose review will remain valid after later work. Do not
|
|
44
|
+
review unstable intermediate diffs or create per-slice review cycles.
|
|
45
|
+
|
|
46
|
+
Cleanup stays inside the spec/standards lens. A concrete simplification risk may
|
|
47
|
+
amplify that lens; size and profile labels alone do not create another gate.
|
|
48
|
+
|
|
49
|
+
## Minimal Durable State
|
|
50
|
+
|
|
51
|
+
Do not create review state during preflight or implementation. Immediately
|
|
52
|
+
before the first actual reviewer launch, persist:
|
|
53
|
+
|
|
54
|
+
- profile, authority path, settled target revision, and assigned lenses;
|
|
55
|
+
- launch ID, reviewer/session handle, lineage, and `pending | completed | failed`;
|
|
56
|
+
- returned findings, repair revision, affected validation, and Closure need.
|
|
57
|
+
|
|
58
|
+
Write `pending` before launch and reconcile that session before replacing it
|
|
59
|
+
after interruption or resume. A usable result becomes `completed`; an explicit
|
|
60
|
+
failure becomes `failed`. A poll timeout while the session remains live is not
|
|
61
|
+
a failure and does not authorize duplicate review.
|
|
62
|
+
|
|
63
|
+
Record extended lineage/session history only for `high`, a real intermediate
|
|
64
|
+
checkpoint, actual Closure, accepted risk, or interrupted recovery. Normal
|
|
65
|
+
medium execution does not keep epochs, pass thresholds, activation counters, or
|
|
66
|
+
per-slice handoff bookkeeping.
|
|
67
|
+
|
|
68
|
+
## Findings And Closure
|
|
69
|
+
|
|
70
|
+
Root aggregates findings, repairs compatible defects once, and reruns only
|
|
71
|
+
affected validation. Coordinator verification closes ordinary medium/low
|
|
72
|
+
behavior-preserving findings after confirming the repair matches the failure.
|
|
73
|
+
|
|
74
|
+
Use shared-protocol Closure only for critical/high defects, protected
|
|
75
|
+
trust/data/concurrency/shared API impact, or invalidated mandatory coverage.
|
|
76
|
+
Closure stays with the affected reviewer lineage and repaired targets. Start a
|
|
77
|
+
new Full only when the repair invalidated mandatory-lens coverage.
|
|
78
|
+
|
|
79
|
+
Do not repeat review without a material change in target, evidence, repair, or
|
|
80
|
+
source decision. Stop and surface the actual decision or evidence blocker when
|
|
81
|
+
no progress is possible.
|
|
82
|
+
|
|
83
|
+
## Completion
|
|
84
|
+
|
|
85
|
+
Run gates in this order:
|
|
86
|
+
|
|
87
|
+
1. affected behavior and integration validation;
|
|
88
|
+
2. applicable final code review;
|
|
89
|
+
3. Closure only when triggered;
|
|
90
|
+
4. repository architecture/build/smoke gates required by policy or the spec;
|
|
91
|
+
5. delivery actions explicitly authorized by the user or workflow.
|
|
92
|
+
|
|
93
|
+
Return `Approved` only for the final settled revision when mandatory lenses and
|
|
94
|
+
validation are complete and shared protocol state is clear. `Waived` records
|
|
95
|
+
skipped coverage but is not approval. `Blocked` requires a concrete authority,
|
|
96
|
+
evidence, reviewer, or convergence blocker—not elapsed time or review count.
|
|
97
|
+
|
|
98
|
+
For normal medium work report only profile, review result, repaired/open
|
|
99
|
+
findings, affected validation, skipped checks, and residual risk. Add extended
|
|
100
|
+
session/defect accounting only when the exceptional state above exists.
|
|
@@ -1,19 +1,33 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: tdd
|
|
3
|
-
description: Test-driven development
|
|
3
|
+
description: Test-driven development for changes that alter observable behavior, have a natural public test seam, and can produce a meaningful failing test before implementation. Use after the global TDD Fit Gate passes, or when the user explicitly requests red-green-refactor, test-first development, or TDD.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Test-Driven Development
|
|
7
7
|
|
|
8
8
|
Use short vertical RED -> GREEN cycles. Make each test prove observable behavior through the same public seam real callers use.
|
|
9
9
|
|
|
10
|
+
## Fit
|
|
11
|
+
|
|
12
|
+
Use this skill only when the change alters observable behavior, a natural public
|
|
13
|
+
seam exists, and the pre-change test will fail for the intended behavioral
|
|
14
|
+
reason. If an implicit activation fails this gate, stop the TDD route and use
|
|
15
|
+
existing regression tests plus affected validation. For mixed tasks, apply TDD
|
|
16
|
+
only to the behavioral slice.
|
|
17
|
+
|
|
18
|
+
Behavior-preserving cleanup, dead-code deletion, documentation, copy,
|
|
19
|
+
formatting, generated assets, package maintenance, simple config, builds, and
|
|
20
|
+
read-only work do not need TDD. Absence and architecture guards added after a
|
|
21
|
+
cleanup are validation, not RED proofs.
|
|
22
|
+
|
|
10
23
|
## Core Contract
|
|
11
24
|
|
|
12
25
|
- Lock expected behavior from the request, specification, design, bug report, or existing product behavior before changing implementation.
|
|
13
26
|
- Derive expected values from an independent source, never from the production algorithm.
|
|
14
27
|
- Prove RED on the old behavior for the same observable reason the user reported or requested.
|
|
15
28
|
- Add only enough implementation to make the current test pass; do not anticipate later tests.
|
|
16
|
-
- Keep tests stable across behavior-preserving refactors
|
|
29
|
+
- Keep tests stable across behavior-preserving refactors.
|
|
30
|
+
- After sufficient GREEN, stop by default. Refactor only to reduce concrete complexity introduced by the change.
|
|
17
31
|
|
|
18
32
|
Read [tests.md](tests.md) when choosing or reviewing test shape. Read [mocking.md](mocking.md) before introducing test doubles.
|
|
19
33
|
|
|
@@ -23,8 +37,8 @@ Read [tests.md](tests.md) when choosing or reviewing test shape. Read [mocking.m
|
|
|
23
37
|
2. List the prioritized observable behaviors, not implementation steps.
|
|
24
38
|
3. Select the public seam where callers observe each behavior.
|
|
25
39
|
4. Ask the user only when the seam changes the public contract, product intent is unclear, or behavior priorities materially conflict.
|
|
26
|
-
5.
|
|
27
|
-
6. If no natural public seam exists,
|
|
40
|
+
5. Use the shared [Contract Test Ledger](../../docs/agents/contract-test-ledger.md) only when its material-delta and missed-failure gate passes.
|
|
41
|
+
6. If no natural public seam exists, stop the TDD route. Consult [interface-design.md](interface-design.md) only when changing the interface is itself required by the task.
|
|
28
42
|
|
|
29
43
|
For UI behavior, define proof at the rendered seam: visible content and order, interaction result, semantics, or screenshot when layout direction or scrolling matters.
|
|
30
44
|
|
|
@@ -43,7 +57,7 @@ Handle reviewer repairs inside the same activation only under [bug workflow rout
|
|
|
43
57
|
|
|
44
58
|
## After GREEN
|
|
45
59
|
|
|
46
|
-
|
|
60
|
+
GREEN is a valid stopping point. If the current change created concrete local complexity, use [refactoring.md](refactoring.md) and rerun affected tests.
|
|
47
61
|
|
|
48
62
|
## Cycle Checklist
|
|
49
63
|
|
|
@@ -55,5 +69,5 @@ Refactor as a separate review-stage activity, never while RED. Use [refactoring.
|
|
|
55
69
|
[ ] GREEN uses only the code needed for the current behavior
|
|
56
70
|
[ ] Final outcome and relevant competing condition are proved
|
|
57
71
|
[ ] Contract Test Ledger is current when applicable
|
|
58
|
-
[ ]
|
|
72
|
+
[ ] Any refactor is local and reduces current-change complexity
|
|
59
73
|
```
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
interface:
|
|
2
2
|
display_name: "Test-Driven Development"
|
|
3
|
-
short_description: "
|
|
4
|
-
default_prompt: "Use $tdd
|
|
3
|
+
short_description: "Use TDD only when its behavioral fit gate passes"
|
|
4
|
+
default_prompt: "Use $tdd after confirming an observable behavior change, a public test seam, and a meaningful pre-change failure."
|
|
5
5
|
policy:
|
|
6
6
|
allow_implicit_invocation: true
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schema_version": 1,
|
|
3
|
+
"skill": "tdd",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "green-can-stop",
|
|
7
|
+
"prompt": "The requested behavior is green and the changed code is already clear and local.",
|
|
8
|
+
"expected": ["stop after green", "keep the current structure"],
|
|
9
|
+
"forbidden": ["add helpers, classes, or value objects", "refactor unrelated code"]
|
|
10
|
+
},
|
|
11
|
+
{
|
|
12
|
+
"id": "no-test-only-seam",
|
|
13
|
+
"prompt": "A behavior test can use the existing public seam, but dependency injection would make mocking easier.",
|
|
14
|
+
"expected": ["use the existing public seam"],
|
|
15
|
+
"forbidden": ["add production dependency injection only for tests", "wrap an SDK only for mockability"]
|
|
16
|
+
}
|
|
17
|
+
]
|
|
18
|
+
}
|
|
@@ -15,45 +15,6 @@ Don't mock:
|
|
|
15
15
|
|
|
16
16
|
## Designing for Mockability
|
|
17
17
|
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
Pass external dependencies in rather than creating them internally:
|
|
23
|
-
|
|
24
|
-
```typescript
|
|
25
|
-
// Easy to mock
|
|
26
|
-
function processPayment(order, paymentClient) {
|
|
27
|
-
return paymentClient.charge(order.total);
|
|
28
|
-
}
|
|
29
|
-
|
|
30
|
-
// Hard to mock
|
|
31
|
-
function processPayment(order) {
|
|
32
|
-
const client = new StripeClient(process.env.STRIPE_KEY);
|
|
33
|
-
return client.charge(order.total);
|
|
34
|
-
}
|
|
35
|
-
```
|
|
36
|
-
|
|
37
|
-
**2. Prefer SDK-style interfaces over generic fetchers**
|
|
38
|
-
|
|
39
|
-
Create specific functions for each external operation instead of one generic function with conditional logic:
|
|
40
|
-
|
|
41
|
-
```typescript
|
|
42
|
-
// GOOD: Each function is independently mockable
|
|
43
|
-
const api = {
|
|
44
|
-
getUser: (id) => fetch(`/users/${id}`),
|
|
45
|
-
getOrders: (userId) => fetch(`/users/${userId}/orders`),
|
|
46
|
-
createOrder: (data) => fetch('/orders', { method: 'POST', body: data }),
|
|
47
|
-
};
|
|
48
|
-
|
|
49
|
-
// BAD: Mocking requires conditional logic inside the mock
|
|
50
|
-
const api = {
|
|
51
|
-
fetch: (endpoint, options) => fetch(endpoint, options),
|
|
52
|
-
};
|
|
53
|
-
```
|
|
54
|
-
|
|
55
|
-
The SDK approach means:
|
|
56
|
-
- Each mock returns one specific shape
|
|
57
|
-
- No conditional logic in test setup
|
|
58
|
-
- Easier to see which endpoints a test exercises
|
|
59
|
-
- Type safety per endpoint
|
|
18
|
+
Use the existing public or system-boundary seam first. Add dependency injection,
|
|
19
|
+
an adapter, or an SDK wrapper only when production ownership or the requested
|
|
20
|
+
contract requires it—not only to make a test easier to mock.
|
|
@@ -1,10 +1,8 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Refactoring After GREEN
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Stop when GREEN code is clear and local. Refactor only when the current change
|
|
4
|
+
introduced concrete duplication, confusion, or misplaced ownership and the edit
|
|
5
|
+
reduces total complexity.
|
|
4
6
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
- **Shallow modules** → Combine or deepen
|
|
8
|
-
- **Feature envy** → Move logic to where data lives
|
|
9
|
-
- **Primitive obsession** → Introduce value objects
|
|
10
|
-
- **Existing code** the new code reveals as problematic
|
|
7
|
+
Keep it local. Do not add helpers, classes, value objects, deeper modules, or
|
|
8
|
+
unrelated cleanup from pattern preference alone. Rerun affected tests.
|