codex-orchestrator 2.0.2 → 2.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -426
- package/README.md +135 -34
- package/dist/src/index.d.ts +11 -1
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +5 -0
- package/dist/src/index.js.map +1 -1
- package/dist/src/v2/acceptance-proof.d.ts +3 -0
- package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
- package/dist/src/v2/acceptance-proof.js +2 -8
- package/dist/src/v2/acceptance-proof.js.map +1 -1
- package/dist/src/v2/adapters/gh-issue-adapter.d.ts +5 -3
- package/dist/src/v2/adapters/gh-issue-adapter.d.ts.map +1 -1
- package/dist/src/v2/adapters/gh-issue-adapter.js +67 -12
- package/dist/src/v2/adapters/gh-issue-adapter.js.map +1 -1
- package/dist/src/v2/adapters/issues.d.ts +16 -2
- package/dist/src/v2/adapters/issues.d.ts.map +1 -1
- package/dist/src/v2/adapters/issues.js +15 -5
- package/dist/src/v2/adapters/issues.js.map +1 -1
- package/dist/src/v2/adapters/mission-coordinator-lock.d.ts +1 -0
- package/dist/src/v2/adapters/mission-coordinator-lock.d.ts.map +1 -1
- package/dist/src/v2/adapters/mission-coordinator-lock.js +5 -1
- package/dist/src/v2/adapters/mission-coordinator-lock.js.map +1 -1
- package/dist/src/v2/cli-contract.d.ts +3 -3
- package/dist/src/v2/cli-contract.d.ts.map +1 -1
- package/dist/src/v2/cli-contract.js +9 -1
- package/dist/src/v2/cli-contract.js.map +1 -1
- package/dist/src/v2/cli.d.ts +24 -0
- package/dist/src/v2/cli.d.ts.map +1 -0
- package/dist/src/v2/{candidate-cli.js → cli.js} +30 -26
- package/dist/src/v2/cli.js.map +1 -0
- package/dist/src/v2/code-review-report.d.ts +66 -0
- package/dist/src/v2/code-review-report.d.ts.map +1 -0
- package/dist/src/v2/code-review-report.js +259 -0
- package/dist/src/v2/code-review-report.js.map +1 -0
- package/dist/src/v2/codex-process.d.ts +8 -1
- package/dist/src/v2/codex-process.d.ts.map +1 -1
- package/dist/src/v2/codex-process.js +22 -0
- package/dist/src/v2/codex-process.js.map +1 -1
- package/dist/src/v2/config.d.ts +4 -3
- package/dist/src/v2/config.d.ts.map +1 -1
- package/dist/src/v2/config.js +8 -3
- package/dist/src/v2/config.js.map +1 -1
- package/dist/src/v2/contained-report-operation.d.ts +100 -0
- package/dist/src/v2/contained-report-operation.d.ts.map +1 -0
- package/dist/src/v2/contained-report-operation.js +200 -0
- package/dist/src/v2/contained-report-operation.js.map +1 -0
- package/dist/src/v2/containment.d.ts +10 -0
- package/dist/src/v2/containment.d.ts.map +1 -1
- package/dist/src/v2/containment.js +49 -1
- package/dist/src/v2/containment.js.map +1 -1
- package/dist/src/v2/direct-delivery.d.ts +96 -0
- package/dist/src/v2/direct-delivery.d.ts.map +1 -0
- package/dist/src/v2/direct-delivery.js +482 -0
- package/dist/src/v2/direct-delivery.js.map +1 -0
- package/dist/src/v2/immutable-workflow-publisher.d.ts +40 -0
- package/dist/src/v2/immutable-workflow-publisher.d.ts.map +1 -0
- package/dist/src/v2/immutable-workflow-publisher.js +218 -0
- package/dist/src/v2/immutable-workflow-publisher.js.map +1 -0
- package/dist/src/v2/implementation-reviewer.d.ts +81 -0
- package/dist/src/v2/implementation-reviewer.d.ts.map +1 -0
- package/dist/src/v2/implementation-reviewer.js +157 -0
- package/dist/src/v2/implementation-reviewer.js.map +1 -0
- package/dist/src/v2/owner-control-lock.d.ts +41 -0
- package/dist/src/v2/owner-control-lock.d.ts.map +1 -0
- package/dist/src/v2/owner-control-lock.js +174 -0
- package/dist/src/v2/owner-control-lock.js.map +1 -0
- package/dist/src/v2/proof-report.d.ts.map +1 -1
- package/dist/src/v2/proof-report.js +55 -29
- package/dist/src/v2/proof-report.js.map +1 -1
- package/dist/src/v2/route-continuations.d.ts +32 -0
- package/dist/src/v2/route-continuations.d.ts.map +1 -0
- package/dist/src/v2/route-continuations.js +2 -0
- package/dist/src/v2/route-continuations.js.map +1 -0
- package/dist/src/v2/route-coordinator.d.ts +77 -0
- package/dist/src/v2/route-coordinator.d.ts.map +1 -0
- package/dist/src/v2/route-coordinator.js +370 -0
- package/dist/src/v2/route-coordinator.js.map +1 -0
- package/dist/src/v2/route-decision.d.ts +129 -0
- package/dist/src/v2/route-decision.d.ts.map +1 -0
- package/dist/src/v2/route-decision.js +400 -0
- package/dist/src/v2/route-decision.js.map +1 -0
- package/dist/src/v2/run-issue.d.ts +64 -6
- package/dist/src/v2/run-issue.d.ts.map +1 -1
- package/dist/src/v2/run-issue.js +962 -92
- package/dist/src/v2/run-issue.js.map +1 -1
- package/dist/src/v2/run-store.d.ts +25 -1
- package/dist/src/v2/run-store.d.ts.map +1 -1
- package/dist/src/v2/run-store.js +129 -4
- package/dist/src/v2/run-store.js.map +1 -1
- package/dist/src/v2/runtime-assets.d.ts +15 -13
- package/dist/src/v2/runtime-assets.d.ts.map +1 -1
- package/dist/src/v2/runtime-assets.js +263 -416
- package/dist/src/v2/runtime-assets.js.map +1 -1
- package/dist/src/v2/runtime.d.ts +17 -9
- package/dist/src/v2/runtime.d.ts.map +1 -1
- package/dist/src/v2/runtime.js +564 -64
- package/dist/src/v2/runtime.js.map +1 -1
- package/dist/src/v2/setup-cli.d.ts.map +1 -1
- package/dist/src/v2/setup-cli.js +4 -10
- package/dist/src/v2/setup-cli.js.map +1 -1
- package/dist/src/v2/setup-runtime.d.ts.map +1 -1
- package/dist/src/v2/setup-runtime.js +19 -131
- package/dist/src/v2/setup-runtime.js.map +1 -1
- package/dist/src/v2/setup-store.d.ts +0 -5
- package/dist/src/v2/setup-store.d.ts.map +1 -1
- package/dist/src/v2/setup-store.js +3 -106
- package/dist/src/v2/setup-store.js.map +1 -1
- package/dist/src/v2/setup.d.ts +6 -43
- package/dist/src/v2/setup.d.ts.map +1 -1
- package/dist/src/v2/setup.js +13 -192
- package/dist/src/v2/setup.js.map +1 -1
- package/dist/src/v2/spec-coordinator.d.ts +85 -0
- package/dist/src/v2/spec-coordinator.d.ts.map +1 -0
- package/dist/src/v2/spec-coordinator.js +88 -0
- package/dist/src/v2/spec-coordinator.js.map +1 -0
- package/dist/src/v2/spec-delivery.d.ts +143 -0
- package/dist/src/v2/spec-delivery.d.ts.map +1 -0
- package/dist/src/v2/spec-delivery.js +401 -0
- package/dist/src/v2/spec-delivery.js.map +1 -0
- package/dist/src/v2/triage-route.d.ts +68 -0
- package/dist/src/v2/triage-route.d.ts.map +1 -0
- package/dist/src/v2/triage-route.js +223 -0
- package/dist/src/v2/triage-route.js.map +1 -0
- package/dist/src/v2/waiting-human-coordinator.d.ts +49 -0
- package/dist/src/v2/waiting-human-coordinator.d.ts.map +1 -0
- package/dist/src/v2/waiting-human-coordinator.js +509 -0
- package/dist/src/v2/waiting-human-coordinator.js.map +1 -0
- package/dist/src/v2/waiting-human.d.ts +143 -0
- package/dist/src/v2/waiting-human.d.ts.map +1 -0
- package/dist/src/v2/waiting-human.js +408 -0
- package/dist/src/v2/waiting-human.js.map +1 -0
- package/dist/src/v2/workflow-assets.d.ts +98 -0
- package/dist/src/v2/workflow-assets.d.ts.map +1 -0
- package/dist/src/v2/workflow-assets.js +646 -0
- package/dist/src/v2/workflow-assets.js.map +1 -0
- package/docs/deep-dive.md +275 -52
- package/internal-workflow/docs/agents/bug-workflow-routing.md +24 -0
- package/internal-workflow/docs/agents/bugfix-quality-gate.md +11 -0
- package/internal-workflow/docs/agents/coding-skill-routing.md +123 -0
- package/internal-workflow/docs/agents/confidence-rubric.md +65 -0
- package/internal-workflow/docs/agents/contract-test-ledger.md +60 -0
- package/internal-workflow/docs/agents/review-gates.md +42 -0
- package/internal-workflow/docs/agents/review-protocol.md +98 -0
- package/internal-workflow/docs/agents/tool-usage.md +88 -0
- package/internal-workflow/evals/coding-skill-evals.json +66 -0
- package/internal-workflow/manifest.json +1 -0
- package/internal-workflow/operations/acceptance-proof/SKILL.md +9 -0
- package/internal-workflow/operations/ambiguity-review/SKILL.md +5 -0
- package/internal-workflow/operations/code-review/SKILL.md +23 -0
- package/internal-workflow/operations/implementation/SKILL.md +24 -0
- package/internal-workflow/operations/spec-author/SKILL.md +12 -0
- package/internal-workflow/operations/spec-review/SKILL.md +12 -0
- package/internal-workflow/operations/triage/SKILL.md +12 -0
- package/internal-workflow/profiles/analyst_deep.toml +9 -0
- package/internal-workflow/profiles/implementer_standard.toml +9 -0
- package/internal-workflow/profiles/proof_agent.toml +8 -0
- package/internal-workflow/profiles/reviewer_deep.toml +9 -0
- package/internal-workflow/profiles/reviewer_standard.toml +9 -0
- package/internal-workflow/schemas/ambiguity-review-v1.json +1 -0
- package/internal-workflow/schemas/code-review-v1.json +1 -0
- package/internal-workflow/schemas/implementation-report-v1.json +1 -0
- package/internal-workflow/schemas/proof-report-v1.json +1 -0
- package/internal-workflow/schemas/spec-author-v1.json +1 -0
- package/internal-workflow/schemas/spec-review-v1.json +30 -0
- package/internal-workflow/schemas/triage-route-v1.json +1 -0
- package/internal-workflow/skills/acceptance-proof/agents/openai.yaml +6 -0
- package/{internal-skills → internal-workflow/skills}/agent-auto/SKILL.md +6 -1
- package/internal-workflow/skills/agent-auto/agents/openai.yaml +6 -0
- package/internal-workflow/skills/code-debugger/SKILL.md +122 -0
- package/internal-workflow/skills/code-debugger/agents/openai.yaml +7 -0
- package/internal-workflow/skills/code-review/SKILL.md +279 -0
- package/internal-workflow/skills/code-review/agents/openai.yaml +4 -0
- package/internal-workflow/skills/code-review/references/bug-classes.md +56 -0
- package/internal-workflow/skills/code-review/references/cleanup-lens.md +52 -0
- package/internal-workflow/skills/code-review/references/framework-lenses.md +34 -0
- package/internal-workflow/skills/code-review/references/targeted-recipes.md +49 -0
- package/internal-workflow/skills/diagnosing-bugs/SKILL.md +138 -0
- package/internal-workflow/skills/diagnosing-bugs/agents/openai.yaml +6 -0
- package/internal-workflow/skills/diagnosing-bugs/scripts/hitl-loop.template.sh +41 -0
- package/internal-workflow/skills/implementation-spec-maker/SKILL.md +102 -0
- package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +6 -0
- package/internal-workflow/skills/implementation-spec-maker/references/source-modes.md +31 -0
- package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +146 -0
- package/internal-workflow/skills/implementation-spec-review/SKILL.md +115 -0
- package/internal-workflow/skills/implementation-spec-review/agents/openai.yaml +6 -0
- package/internal-workflow/skills/implementation-spec-review/evals/evals.json +24 -0
- package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +93 -0
- package/internal-workflow/skills/small-task-implementer/SKILL.md +104 -0
- package/internal-workflow/skills/small-task-implementer/agents/openai.yaml +6 -0
- package/internal-workflow/skills/spec-implementer/SKILL.md +126 -0
- package/internal-workflow/skills/spec-implementer/agents/openai.yaml +6 -0
- package/internal-workflow/skills/spec-implementer/evals/evals.json +30 -0
- package/internal-workflow/skills/spec-implementer/references/review-loop.md +94 -0
- package/internal-workflow/skills/tdd/SKILL.md +72 -0
- package/internal-workflow/skills/tdd/agents/openai.yaml +6 -0
- package/internal-workflow/skills/tdd/interface-design.md +31 -0
- package/internal-workflow/skills/tdd/mocking.md +59 -0
- package/internal-workflow/skills/tdd/refactoring.md +10 -0
- package/internal-workflow/skills/tdd/tests.md +77 -0
- package/internal-workflow/skills/triage/AGENT-BRIEF.md +192 -0
- package/internal-workflow/skills/triage/OUT-OF-SCOPE.md +101 -0
- package/internal-workflow/skills/triage/SKILL.md +134 -0
- package/internal-workflow/skills/triage/agents/openai.yaml +6 -0
- package/package.json +14 -8
- package/dist/src/v2/adapters/target-activity-fence.d.ts +0 -23
- package/dist/src/v2/adapters/target-activity-fence.d.ts.map +0 -1
- package/dist/src/v2/adapters/target-activity-fence.js +0 -249
- package/dist/src/v2/adapters/target-activity-fence.js.map +0 -1
- package/dist/src/v2/candidate-cli.d.ts +0 -22
- package/dist/src/v2/candidate-cli.d.ts.map +0 -1
- package/dist/src/v2/candidate-cli.js.map +0 -1
- package/dist/src/v2/legacy-cutover.d.ts +0 -52
- package/dist/src/v2/legacy-cutover.d.ts.map +0 -1
- package/dist/src/v2/legacy-cutover.js +0 -87
- package/dist/src/v2/legacy-cutover.js.map +0 -1
- /package/{internal-skills → internal-workflow/skills}/acceptance-proof/SKILL.md +0 -0
- /package/{internal-skills → internal-workflow/skills}/acceptance-proof/references/android.md +0 -0
- /package/{internal-skills → internal-workflow/skills}/acceptance-proof/references/browser.md +0 -0
- /package/{internal-skills → internal-workflow/skills}/acceptance-proof/references/ios.md +0 -0
- /package/{internal-skills → internal-workflow/skills}/acceptance-proof/tools/android-lease.mjs +0 -0
- /package/{internal-skills → internal-workflow/skills}/acceptance-proof/tools/ios-lease.mjs +0 -0
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: small-task-implementer
|
|
3
|
+
description: Implement small low-risk coding tasks with narrow edits and targeted validation. Use for tiny fixes, UI/copy changes, config/build corrections, simple tests, or one-module changes that do not need plans, specs, orchestration, or heavy review.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Small Task Implementer
|
|
7
|
+
|
|
8
|
+
Use this skill for fast, bounded implementation when creating a PRD and approved ticket-delivery flow would be disproportionate.
|
|
9
|
+
|
|
10
|
+
## Fit Gate
|
|
11
|
+
|
|
12
|
+
Proceed only when all are true:
|
|
13
|
+
|
|
14
|
+
- The requested behavior is clear or can be inferred from local code/tests without a product decision.
|
|
15
|
+
- The change is expected to touch one small area or a few tightly related files.
|
|
16
|
+
- There is a narrow validation path: targeted test, lint/typecheck, build check, UI proof, or direct command.
|
|
17
|
+
- The task does not require a new plan, PRD, issue breakdown, implementation spec, migration, rollout, or multi-agent orchestration.
|
|
18
|
+
|
|
19
|
+
Escalate out of the tiny-task route when the work has more than one coherent
|
|
20
|
+
behavior, a broad ownership boundary, material rollback/recovery risk, unclear
|
|
21
|
+
product intent, no credible affected validation, or genuine multi-agent/live
|
|
22
|
+
coordination. Statefulness alone does not require escalation into planning or
|
|
23
|
+
orchestration; a clear API, persistence, cache, queue, or DTO change normally
|
|
24
|
+
becomes direct medium implementation.
|
|
25
|
+
|
|
26
|
+
Escalation rule:
|
|
27
|
+
|
|
28
|
+
- For clear authority and one coherent outcome, escalate to direct medium root
|
|
29
|
+
implementation under `$tdd`, affected validation, and one final review when
|
|
30
|
+
`review-gates.md` applies.
|
|
31
|
+
- Use optional `$grilling`, then `$spec-to-tickets` and `$tickets-orchestrator`,
|
|
32
|
+
only for unresolved product decisions or a real approved ticket graph,
|
|
33
|
+
delivery dependency, or explicit orchestration request.
|
|
34
|
+
- For one risky behavior or technical contract, prefer one approved ticket and mark `compact spec` or `standard spec` only when the ticket plus repository evidence cannot remove execution ambiguity.
|
|
35
|
+
- For several tickets sharing one unresolved contract or validation path, make the contract-defining ticket block its consumers; merge tickets that cannot be specified or verified independently instead of creating a wave-level implementation spec.
|
|
36
|
+
- Escalate if the bug requires Bugfix Quality Gate analysis across multiple paths, states, async events, persistence, auth, cache, retries, workers, or contracts.
|
|
37
|
+
|
|
38
|
+
## Workflow
|
|
39
|
+
|
|
40
|
+
1. Inspect local context just enough to confirm fit.
|
|
41
|
+
- Read repo instructions and the smallest relevant code/test files.
|
|
42
|
+
- Check `git status --short` before editing.
|
|
43
|
+
- Preserve unrelated dirty work.
|
|
44
|
+
|
|
45
|
+
2. Write a compact contract in the working update or internal task notes:
|
|
46
|
+
|
|
47
|
+
```text
|
|
48
|
+
Behavior:
|
|
49
|
+
Scope boundary:
|
|
50
|
+
Validation:
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
3. Implement the smallest complete change.
|
|
54
|
+
- Prefer existing patterns and owner modules.
|
|
55
|
+
- Avoid unrelated refactors, abstractions, cleanup, and compatibility paths.
|
|
56
|
+
- Before adding a helper, module, layer, or seam, apply the deletion test; keep it only if it improves current locality or leverage.
|
|
57
|
+
- Do not add pass-through modules, one-adapter seams, or tests coupled to Implementation details; escalate if no natural public test seam exists.
|
|
58
|
+
- Add or update a focused test only when behavior risk justifies it and the repo has a natural seam.
|
|
59
|
+
- For pure copy/docs/config changes, do not invent tests; run the cheapest relevant syntax/lint/check instead.
|
|
60
|
+
|
|
61
|
+
4. Run targeted validation.
|
|
62
|
+
- Use the narrowest meaningful command first.
|
|
63
|
+
- If validation is unavailable or too expensive, state the concrete reason and residual risk.
|
|
64
|
+
- Do not run full CI unless local policy or the changed surface makes it necessary.
|
|
65
|
+
|
|
66
|
+
5. Stop and escalate if implementation reveals hidden risk.
|
|
67
|
+
- Examples: shared contract drift, duplicate source of truth, missing test
|
|
68
|
+
seam, material concurrency/recovery uncertainty, or product ambiguity.
|
|
69
|
+
- Leave a short explanation of what was discovered and which heavier flow should take over.
|
|
70
|
+
|
|
71
|
+
## Output
|
|
72
|
+
|
|
73
|
+
Final response must stay compact:
|
|
74
|
+
|
|
75
|
+
```text
|
|
76
|
+
Small Task Result
|
|
77
|
+
|
|
78
|
+
Changed:
|
|
79
|
+
- ...
|
|
80
|
+
|
|
81
|
+
Proof:
|
|
82
|
+
- ...
|
|
83
|
+
|
|
84
|
+
Skipped:
|
|
85
|
+
- none / ...
|
|
86
|
+
|
|
87
|
+
Risk:
|
|
88
|
+
- low / reason
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
If escalated, use:
|
|
92
|
+
|
|
93
|
+
```text
|
|
94
|
+
Escalated
|
|
95
|
+
|
|
96
|
+
Reason:
|
|
97
|
+
- ...
|
|
98
|
+
|
|
99
|
+
Recommended flow:
|
|
100
|
+
- Direct medium implementation / canonical ticket delivery
|
|
101
|
+
|
|
102
|
+
Evidence:
|
|
103
|
+
- ...
|
|
104
|
+
```
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: "spec-implementer"
|
|
3
|
+
description: "Executes approved specs continuously with honest checklist updates, proportional validation, opt-in Git checkpoints, and required review/signoff."
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Spec Implementer
|
|
7
|
+
|
|
8
|
+
Execute one approved implementation spec continuously. Keep its checklist
|
|
9
|
+
truthful and stop only at a real authority, evidence, safety, or explicit pause
|
|
10
|
+
boundary. Do not redesign approved scope.
|
|
11
|
+
|
|
12
|
+
Read before editing:
|
|
13
|
+
|
|
14
|
+
- the complete approved spec and applicable repository instructions;
|
|
15
|
+
- `references/review-loop.md` for approved-spec review ownership;
|
|
16
|
+
- `../../docs/agents/contract-test-ledger.md` only when the spec contains
|
|
17
|
+
material contract invariants.
|
|
18
|
+
|
|
19
|
+
## Modes
|
|
20
|
+
|
|
21
|
+
Compact and full specs use the same direct phase flow. `compact` describes
|
|
22
|
+
document density, not implementation size or risk. Full mode adds only the
|
|
23
|
+
concrete `Risk Controls`, stop conditions, validation, or coordination contract
|
|
24
|
+
already present in the approved spec; it does not add review or reporting
|
|
25
|
+
ceremony by itself.
|
|
26
|
+
|
|
27
|
+
Use multiple agents only when the spec defines perfectly disjoint write scopes
|
|
28
|
+
and one integrator. Otherwise execute single-agent.
|
|
29
|
+
|
|
30
|
+
## Preflight
|
|
31
|
+
|
|
32
|
+
1. Confirm `status`, `spec_mode`, `implementation_size`, `review_profile`,
|
|
33
|
+
repository count, authority, scope, and exclusions.
|
|
34
|
+
2. Confirm first-phase targets, required services/env/data/fixtures, commands,
|
|
35
|
+
observable proof, protected paths, and rejected approaches.
|
|
36
|
+
3. Stop if execution needs invented paths, symbols, contracts, commands, or
|
|
37
|
+
product decisions; if reality differs only in a bounded technical detail,
|
|
38
|
+
resolve it from repository evidence and record the adjustment.
|
|
39
|
+
4. Keep an intermediate review checkpoint only when the spec explicitly names
|
|
40
|
+
a stable high-risk slice that later work will not invalidate. Move every
|
|
41
|
+
ordinary or unstable checkpoint to final review.
|
|
42
|
+
5. Do not create `## Implementation Review State` yet.
|
|
43
|
+
|
|
44
|
+
## Implementation
|
|
45
|
+
|
|
46
|
+
For each phase:
|
|
47
|
+
|
|
48
|
+
1. Re-read its scope and preconditions.
|
|
49
|
+
2. Implement narrow vertical behavior slices through `$tdd` when applicable.
|
|
50
|
+
3. Update reached checklist and Contract Test Ledger items at natural
|
|
51
|
+
checkpoints; never save all updates for the end.
|
|
52
|
+
4. Run the phase's targeted exit proof.
|
|
53
|
+
5. Record a short `Blocked:` note for any item that cannot complete.
|
|
54
|
+
6. Continue immediately when the exit proof passes and no stop condition
|
|
55
|
+
applies.
|
|
56
|
+
|
|
57
|
+
Do not add cleanup, comments, helpers, abstractions, retries, flags, fallbacks,
|
|
58
|
+
or compatibility paths outside the spec. A small implementation adjustment is
|
|
59
|
+
allowed only when it preserves approved behavior and is supported by current
|
|
60
|
+
repository evidence.
|
|
61
|
+
|
|
62
|
+
## Git Checkpoints
|
|
63
|
+
|
|
64
|
+
Default to no commits. `$spec-implementer` does not authorize Git writes.
|
|
65
|
+
|
|
66
|
+
Use per-slice commits only when explicitly authorized and they materially help
|
|
67
|
+
a disjoint multi-agent handoff, planned cross-session pause, or approved
|
|
68
|
+
rollback boundary. Require a passed slice proof, stage only owned paths, follow
|
|
69
|
+
`$commit`, and never push without separate authority. File/slice count and risk
|
|
70
|
+
profile alone do not justify checkpoints.
|
|
71
|
+
|
|
72
|
+
## Review And Validation
|
|
73
|
+
|
|
74
|
+
Run targeted behavior tests and the smallest affected integration checks first.
|
|
75
|
+
Use a full repository suite only when the spec or repository policy requires
|
|
76
|
+
it, broad contract fan-out cannot be isolated, or the task is genuinely high.
|
|
77
|
+
|
|
78
|
+
Run final `$code-review` only when the spec, repository policy, or
|
|
79
|
+
`../../docs/agents/review-gates.md` applies. Ordinary medium work gets one
|
|
80
|
+
`reviewer_standard` Full review on the settled diff. High gets two disjoint
|
|
81
|
+
`reviewer_deep` lenses. Cleanup stays inside spec/standards review.
|
|
82
|
+
|
|
83
|
+
Immediately before the first reviewer launch, create only the minimal
|
|
84
|
+
`## Implementation Review State` required by `references/review-loop.md`.
|
|
85
|
+
Reconcile a recorded live session before replacement after interruption.
|
|
86
|
+
|
|
87
|
+
Repair compatible findings once and rerun affected validation. Coordinator
|
|
88
|
+
verification closes ordinary medium/low behavior-preserving repairs. Use
|
|
89
|
+
Closure only for critical/high, protected trust/data/concurrency/shared-contract
|
|
90
|
+
impact, or invalidated mandatory coverage. Do not restart broad Full review
|
|
91
|
+
unless the repair actually invalidated its coverage.
|
|
92
|
+
|
|
93
|
+
## Stop Conditions
|
|
94
|
+
|
|
95
|
+
Stop and report the exact blocker when:
|
|
96
|
+
|
|
97
|
+
- a precondition, source contract, required proof, or protected path differs
|
|
98
|
+
materially from the approved spec;
|
|
99
|
+
- exact execution requires a new product, scope, ownership, or risky trade-off
|
|
100
|
+
decision;
|
|
101
|
+
- required validation or reviewer is unavailable and no approved substitute
|
|
102
|
+
exists;
|
|
103
|
+
- multi-agent scopes overlap or integration ownership is missing;
|
|
104
|
+
- an explicit user pause or halt condition is reached.
|
|
105
|
+
|
|
106
|
+
Do not stop merely because the work is broad, review took time, or a medium/low
|
|
107
|
+
finding required one repair.
|
|
108
|
+
|
|
109
|
+
## Completion
|
|
110
|
+
|
|
111
|
+
Complete only when reached checklist items are reconciled, affected proof and
|
|
112
|
+
required review pass, protected paths and rejected approaches remain intact,
|
|
113
|
+
and every unfinished item has a concrete status.
|
|
114
|
+
|
|
115
|
+
For ordinary medium work report only:
|
|
116
|
+
|
|
117
|
+
- behavior/contract implemented;
|
|
118
|
+
- review result and repaired/open findings;
|
|
119
|
+
- affected validation;
|
|
120
|
+
- skipped checks and residual risk;
|
|
121
|
+
- changed files and any authorized commits.
|
|
122
|
+
|
|
123
|
+
For high, actual Closure, accepted risk, interrupted recovery, or multi-agent
|
|
124
|
+
delivery, add the relevant invariants, reviewer coverage, defect IDs, session
|
|
125
|
+
recovery, and handoff ownership. Do not create a separate report file unless
|
|
126
|
+
the spec or the complexity of that exceptional handoff requires it.
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
interface:
|
|
2
|
+
display_name: "Spec Implementer"
|
|
3
|
+
short_description: "Execute approved specs with lean delivery"
|
|
4
|
+
default_prompt: "Use $spec-implementer to execute the approved spec continuously, default to no Git checkpoints, validate proportionately, and launch only stable required review gates."
|
|
5
|
+
policy:
|
|
6
|
+
allow_implicit_invocation: true
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schema_version": 1,
|
|
3
|
+
"skill": "spec-implementer",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "state-created-at-launch",
|
|
7
|
+
"prompt": "Execute an approved medium spec whose final review has not started yet.",
|
|
8
|
+
"expected": ["do not create review state during implementation", "persist minimal state immediately before reviewer launch"],
|
|
9
|
+
"forbidden": ["create per-slice review bookkeeping"]
|
|
10
|
+
},
|
|
11
|
+
{
|
|
12
|
+
"id": "medium-one-final-review",
|
|
13
|
+
"prompt": "Execute a normal medium spec with several vertical slices and no stable high-risk checkpoint.",
|
|
14
|
+
"expected": ["implement continuously", "run one final reviewer_standard on the settled diff"],
|
|
15
|
+
"forbidden": ["review after every slice", "run a full repository suite from file count"]
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"id": "closure-stays-affected",
|
|
19
|
+
"prompt": "Final review finds one high defect in a shared schema and several ordinary low findings.",
|
|
20
|
+
"expected": ["repair once", "Closure verifies only the affected high-risk contract"],
|
|
21
|
+
"forbidden": ["restart all Full reviewers"]
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"id": "resume-live-reviewer",
|
|
25
|
+
"prompt": "Resume after a reviewer poll timed out while the recorded session is still live.",
|
|
26
|
+
"expected": ["reconcile the existing session"],
|
|
27
|
+
"forbidden": ["mark it failed from timeout alone", "launch a duplicate reviewer"]
|
|
28
|
+
}
|
|
29
|
+
]
|
|
30
|
+
}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# Approved Spec Implementation Review Loop
|
|
2
|
+
|
|
3
|
+
This reference owns review orchestration for `$spec-implementer`. Read it only
|
|
4
|
+
when executing an approved implementation spec. Shared Full/Closure and defect
|
|
5
|
+
mechanics live in `../../../docs/agents/review-protocol.md`.
|
|
6
|
+
|
|
7
|
+
Direct work and deterministic issue delivery use normal TDD and review gates;
|
|
8
|
+
they must not create Implementation Review State.
|
|
9
|
+
|
|
10
|
+
## Authority
|
|
11
|
+
|
|
12
|
+
Only an approved implementation spec may own `## Implementation Review State`.
|
|
13
|
+
PRDs, tickets, architecture notes, and chat summaries are not review-state
|
|
14
|
+
owners. A substantive spec change returns through artifact review before
|
|
15
|
+
implementation continues.
|
|
16
|
+
|
|
17
|
+
Use the spec's `review_profile`; if absent, infer it from current evidence:
|
|
18
|
+
|
|
19
|
+
- `simple`: narrow change with direct proof;
|
|
20
|
+
- `medium`: default for ordinary implementation;
|
|
21
|
+
- `high`: material failure consequence plus an uncertainty amplifier.
|
|
22
|
+
|
|
23
|
+
Implementation evidence may raise but never lower the approved profile.
|
|
24
|
+
|
|
25
|
+
## Default Review Shape
|
|
26
|
+
|
|
27
|
+
Implement continuously through vertical slices. Validate each affected behavior
|
|
28
|
+
and run one final review on the settled diff when the gate applies.
|
|
29
|
+
|
|
30
|
+
- `simple`: one `reviewer_fast` when review is required.
|
|
31
|
+
- `medium`: one `reviewer_standard`, one bounded final Full, no intermediate
|
|
32
|
+
checkpoint by default.
|
|
33
|
+
- `high`: two parallel `reviewer_deep` Full reviews with disjoint correctness
|
|
34
|
+
and spec/standards lenses.
|
|
35
|
+
|
|
36
|
+
Add an intermediate checkpoint only when the approved spec explicitly names a
|
|
37
|
+
stable high-risk slice whose review will remain valid after later work. Do not
|
|
38
|
+
review unstable intermediate diffs or create per-slice review cycles.
|
|
39
|
+
|
|
40
|
+
Cleanup stays inside the spec/standards lens. A concrete simplification risk may
|
|
41
|
+
amplify that lens; size and profile labels alone do not create another gate.
|
|
42
|
+
|
|
43
|
+
## Minimal Durable State
|
|
44
|
+
|
|
45
|
+
Do not create review state during preflight or implementation. Immediately
|
|
46
|
+
before the first actual reviewer launch, persist:
|
|
47
|
+
|
|
48
|
+
- profile, authority path, settled target revision, and assigned lenses;
|
|
49
|
+
- launch ID, reviewer/session handle, lineage, and `pending | completed | failed`;
|
|
50
|
+
- returned findings, repair revision, affected validation, and Closure need.
|
|
51
|
+
|
|
52
|
+
Write `pending` before launch and reconcile that session before replacing it
|
|
53
|
+
after interruption or resume. A usable result becomes `completed`; an explicit
|
|
54
|
+
failure becomes `failed`. A poll timeout while the session remains live is not
|
|
55
|
+
a failure and does not authorize duplicate review.
|
|
56
|
+
|
|
57
|
+
Record extended lineage/session history only for `high`, a real intermediate
|
|
58
|
+
checkpoint, actual Closure, accepted risk, or interrupted recovery. Normal
|
|
59
|
+
medium execution does not keep epochs, pass thresholds, activation counters, or
|
|
60
|
+
per-slice handoff bookkeeping.
|
|
61
|
+
|
|
62
|
+
## Findings And Closure
|
|
63
|
+
|
|
64
|
+
Root aggregates findings, repairs compatible defects once, and reruns only
|
|
65
|
+
affected validation. Coordinator verification closes ordinary medium/low
|
|
66
|
+
behavior-preserving findings after confirming the repair matches the failure.
|
|
67
|
+
|
|
68
|
+
Use shared-protocol Closure only for critical/high defects, protected
|
|
69
|
+
trust/data/concurrency/shared API impact, or invalidated mandatory coverage.
|
|
70
|
+
Closure stays with the affected reviewer lineage and repaired targets. Start a
|
|
71
|
+
new Full only when the repair invalidated mandatory-lens coverage.
|
|
72
|
+
|
|
73
|
+
Do not repeat review without a material change in target, evidence, repair, or
|
|
74
|
+
source decision. Stop and surface the actual decision or evidence blocker when
|
|
75
|
+
no progress is possible.
|
|
76
|
+
|
|
77
|
+
## Completion
|
|
78
|
+
|
|
79
|
+
Run gates in this order:
|
|
80
|
+
|
|
81
|
+
1. affected behavior and integration validation;
|
|
82
|
+
2. applicable final code review;
|
|
83
|
+
3. Closure only when triggered;
|
|
84
|
+
4. repository architecture/build/smoke gates required by policy or the spec;
|
|
85
|
+
5. delivery actions explicitly authorized by the user or workflow.
|
|
86
|
+
|
|
87
|
+
Return `Approved` only for the final settled revision when mandatory lenses and
|
|
88
|
+
validation are complete and shared protocol state is clear. `Waived` records
|
|
89
|
+
skipped coverage but is not approval. `Blocked` requires a concrete authority,
|
|
90
|
+
evidence, reviewer, or convergence blocker—not elapsed time or review count.
|
|
91
|
+
|
|
92
|
+
For normal medium work report only profile, review result, repaired/open
|
|
93
|
+
findings, affected validation, skipped checks, and residual risk. Add extended
|
|
94
|
+
session/defect accounting only when the exceptional state above exists.
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: tdd
|
|
3
|
+
description: Test-driven development for changes that alter observable behavior, have a natural public test seam, and can produce a meaningful failing test before implementation. Use after the global TDD Fit Gate passes, or when the user explicitly requests red-green-refactor, test-first development, or TDD.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Test-Driven Development
|
|
7
|
+
|
|
8
|
+
Use short vertical RED -> GREEN cycles. Make each test prove observable behavior through the same public seam real callers use.
|
|
9
|
+
|
|
10
|
+
## Fit
|
|
11
|
+
|
|
12
|
+
Use this skill only when the change alters observable behavior, a natural public
|
|
13
|
+
seam exists, and the pre-change test will fail for the intended behavioral
|
|
14
|
+
reason. If an implicit activation fails this gate, stop the TDD route and use
|
|
15
|
+
existing regression tests plus affected validation. For mixed tasks, apply TDD
|
|
16
|
+
only to the behavioral slice.
|
|
17
|
+
|
|
18
|
+
Behavior-preserving cleanup, dead-code deletion, documentation, copy,
|
|
19
|
+
formatting, generated assets, package maintenance, simple config, builds, and
|
|
20
|
+
read-only work do not need TDD. Absence and architecture guards added after a
|
|
21
|
+
cleanup are validation, not RED proofs.
|
|
22
|
+
|
|
23
|
+
## Core Contract
|
|
24
|
+
|
|
25
|
+
- Lock expected behavior from the request, specification, design, bug report, or existing product behavior before changing implementation.
|
|
26
|
+
- Derive expected values from an independent source, never from the production algorithm.
|
|
27
|
+
- Prove RED on the old behavior for the same observable reason the user reported or requested.
|
|
28
|
+
- Add only enough implementation to make the current test pass; do not anticipate later tests.
|
|
29
|
+
- Keep tests stable across behavior-preserving refactors and refactor only while GREEN.
|
|
30
|
+
|
|
31
|
+
Read [tests.md](tests.md) when choosing or reviewing test shape. Read [mocking.md](mocking.md) before introducing test doubles.
|
|
32
|
+
|
|
33
|
+
## Before the First RED
|
|
34
|
+
|
|
35
|
+
1. Read local instructions, domain language, existing tests, and relevant ADRs.
|
|
36
|
+
2. List the prioritized observable behaviors, not implementation steps.
|
|
37
|
+
3. Select the public seam where callers observe each behavior.
|
|
38
|
+
4. Ask the user only when the seam changes the public contract, product intent is unclear, or behavior priorities materially conflict.
|
|
39
|
+
5. For contract-risk changes, create or update the shared [Contract Test Ledger](../../docs/agents/contract-test-ledger.md) and map each invariant to its first failing test or observable proof.
|
|
40
|
+
6. If no natural public seam exists, stop the TDD route. Consult [interface-design.md](interface-design.md) only when changing the interface is itself required by the task.
|
|
41
|
+
|
|
42
|
+
For UI behavior, define proof at the rendered seam: visible content and order, interaction result, semantics, or screenshot when layout direction or scrolling matters.
|
|
43
|
+
|
|
44
|
+
## RED -> GREEN Cycle
|
|
45
|
+
|
|
46
|
+
For each behavior:
|
|
47
|
+
|
|
48
|
+
1. **RED:** Write one test through the selected seam.
|
|
49
|
+
2. Confirm it fails on current behavior for the expected reason. A passing test or an internal-only failure is not valid RED.
|
|
50
|
+
3. **GREEN:** Add the minimal implementation required for that test.
|
|
51
|
+
4. Run the proof and update the ledger status to `red`, `green`, or `blocked` with the missing seam or evidence.
|
|
52
|
+
|
|
53
|
+
Keep each cycle to one seam, one behavior, one test, and one minimal implementation. Do not rewrite the test to fit the code. For state, async, lifecycle, retry, cache, or auth defects, include the competing condition when feasible.
|
|
54
|
+
|
|
55
|
+
Handle reviewer repairs inside the same activation only under [bug workflow routing](../../docs/agents/bug-workflow-routing.md); group related cases by protected invariant.
|
|
56
|
+
|
|
57
|
+
## After GREEN
|
|
58
|
+
|
|
59
|
+
Refactor as a separate review-stage activity, never while RED. Use [refactoring.md](refactoring.md) for candidates and rerun affected tests after each step.
|
|
60
|
+
|
|
61
|
+
## Cycle Checklist
|
|
62
|
+
|
|
63
|
+
```text
|
|
64
|
+
[ ] Behavior is proved through the caller's public seam
|
|
65
|
+
[ ] Expected behavior is locked and the expected value is independent
|
|
66
|
+
[ ] RED fails on old behavior for the correct observable reason
|
|
67
|
+
[ ] Test was not fitted to implementation details
|
|
68
|
+
[ ] GREEN uses only the code needed for the current behavior
|
|
69
|
+
[ ] Final outcome and relevant competing condition are proved
|
|
70
|
+
[ ] Contract Test Ledger is current when applicable
|
|
71
|
+
[ ] Refactoring starts only after GREEN
|
|
72
|
+
```
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
interface:
|
|
2
|
+
display_name: "Test-Driven Development"
|
|
3
|
+
short_description: "Use TDD only when its behavioral fit gate passes"
|
|
4
|
+
default_prompt: "Use $tdd after confirming an observable behavior change, a public test seam, and a meaningful pre-change failure."
|
|
5
|
+
policy:
|
|
6
|
+
allow_implicit_invocation: true
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Interface Design for Testability
|
|
2
|
+
|
|
3
|
+
Good interfaces make testing natural:
|
|
4
|
+
|
|
5
|
+
1. **Accept dependencies, don't create them**
|
|
6
|
+
|
|
7
|
+
```typescript
|
|
8
|
+
// Testable
|
|
9
|
+
function processOrder(order, paymentGateway) {}
|
|
10
|
+
|
|
11
|
+
// Hard to test
|
|
12
|
+
function processOrder(order) {
|
|
13
|
+
const gateway = new StripeGateway();
|
|
14
|
+
}
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
2. **Return results, don't produce side effects**
|
|
18
|
+
|
|
19
|
+
```typescript
|
|
20
|
+
// Testable
|
|
21
|
+
function calculateDiscount(cart): Discount {}
|
|
22
|
+
|
|
23
|
+
// Hard to test
|
|
24
|
+
function applyDiscount(cart): void {
|
|
25
|
+
cart.total -= discount;
|
|
26
|
+
}
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
3. **Small surface area**
|
|
30
|
+
- Fewer methods = fewer tests needed
|
|
31
|
+
- Fewer params = simpler test setup
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# When to Mock
|
|
2
|
+
|
|
3
|
+
Mock at **system boundaries** only:
|
|
4
|
+
|
|
5
|
+
- External APIs (payment, email, etc.)
|
|
6
|
+
- Databases (sometimes - prefer test DB)
|
|
7
|
+
- Time/randomness
|
|
8
|
+
- File system (sometimes)
|
|
9
|
+
|
|
10
|
+
Don't mock:
|
|
11
|
+
|
|
12
|
+
- Your own classes/modules
|
|
13
|
+
- Internal collaborators
|
|
14
|
+
- Anything you control
|
|
15
|
+
|
|
16
|
+
## Designing for Mockability
|
|
17
|
+
|
|
18
|
+
At system boundaries, design interfaces that are easy to mock:
|
|
19
|
+
|
|
20
|
+
**1. Use dependency injection**
|
|
21
|
+
|
|
22
|
+
Pass external dependencies in rather than creating them internally:
|
|
23
|
+
|
|
24
|
+
```typescript
|
|
25
|
+
// Easy to mock
|
|
26
|
+
function processPayment(order, paymentClient) {
|
|
27
|
+
return paymentClient.charge(order.total);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
// Hard to mock
|
|
31
|
+
function processPayment(order) {
|
|
32
|
+
const client = new StripeClient(process.env.STRIPE_KEY);
|
|
33
|
+
return client.charge(order.total);
|
|
34
|
+
}
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
**2. Prefer SDK-style interfaces over generic fetchers**
|
|
38
|
+
|
|
39
|
+
Create specific functions for each external operation instead of one generic function with conditional logic:
|
|
40
|
+
|
|
41
|
+
```typescript
|
|
42
|
+
// GOOD: Each function is independently mockable
|
|
43
|
+
const api = {
|
|
44
|
+
getUser: (id) => fetch(`/users/${id}`),
|
|
45
|
+
getOrders: (userId) => fetch(`/users/${userId}/orders`),
|
|
46
|
+
createOrder: (data) => fetch('/orders', { method: 'POST', body: data }),
|
|
47
|
+
};
|
|
48
|
+
|
|
49
|
+
// BAD: Mocking requires conditional logic inside the mock
|
|
50
|
+
const api = {
|
|
51
|
+
fetch: (endpoint, options) => fetch(endpoint, options),
|
|
52
|
+
};
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
The SDK approach means:
|
|
56
|
+
- Each mock returns one specific shape
|
|
57
|
+
- No conditional logic in test setup
|
|
58
|
+
- Easier to see which endpoints a test exercises
|
|
59
|
+
- Type safety per endpoint
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Refactor Candidates
|
|
2
|
+
|
|
3
|
+
After TDD cycle, look for:
|
|
4
|
+
|
|
5
|
+
- **Duplication** → Extract function/class
|
|
6
|
+
- **Long methods** → Break into private helpers (keep tests on public interface)
|
|
7
|
+
- **Shallow modules** → Combine or deepen
|
|
8
|
+
- **Feature envy** → Move logic to where data lives
|
|
9
|
+
- **Primitive obsession** → Introduce value objects
|
|
10
|
+
- **Existing code** the new code reveals as problematic
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
# Good and Bad Tests
|
|
2
|
+
|
|
3
|
+
## Good Tests
|
|
4
|
+
|
|
5
|
+
**Integration-style**: Test through real interfaces, not mocks of internal parts.
|
|
6
|
+
|
|
7
|
+
```typescript
|
|
8
|
+
// GOOD: Tests observable behavior
|
|
9
|
+
test("user can checkout with valid cart", async () => {
|
|
10
|
+
const cart = createCart();
|
|
11
|
+
cart.add(product);
|
|
12
|
+
const result = await checkout(cart, paymentMethod);
|
|
13
|
+
expect(result.status).toBe("confirmed");
|
|
14
|
+
});
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Characteristics:
|
|
18
|
+
|
|
19
|
+
- Tests behavior users/callers care about
|
|
20
|
+
- Uses public API only
|
|
21
|
+
- Survives internal refactors
|
|
22
|
+
- Describes WHAT, not HOW
|
|
23
|
+
- One logical assertion per test
|
|
24
|
+
|
|
25
|
+
## Bad Tests
|
|
26
|
+
|
|
27
|
+
**Implementation-detail tests**: Coupled to internal structure.
|
|
28
|
+
|
|
29
|
+
```typescript
|
|
30
|
+
// BAD: Tests implementation details
|
|
31
|
+
test("checkout calls paymentService.process", async () => {
|
|
32
|
+
const mockPayment = jest.mock(paymentService);
|
|
33
|
+
await checkout(cart, payment);
|
|
34
|
+
expect(mockPayment.process).toHaveBeenCalledWith(cart.total);
|
|
35
|
+
});
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Red flags:
|
|
39
|
+
|
|
40
|
+
- Mocking internal collaborators
|
|
41
|
+
- Testing private methods
|
|
42
|
+
- Asserting on call counts/order
|
|
43
|
+
- Test breaks when refactoring without behavior change
|
|
44
|
+
- Test name describes HOW not WHAT
|
|
45
|
+
- Verifying through external means instead of interface
|
|
46
|
+
|
|
47
|
+
```typescript
|
|
48
|
+
// BAD: Bypasses interface to verify
|
|
49
|
+
test("createUser saves to database", async () => {
|
|
50
|
+
await createUser({ name: "Alice" });
|
|
51
|
+
const row = await db.query("SELECT * FROM users WHERE name = ?", ["Alice"]);
|
|
52
|
+
expect(row).toBeDefined();
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
// GOOD: Verifies through interface
|
|
56
|
+
test("createUser makes user retrievable", async () => {
|
|
57
|
+
const user = await createUser({ name: "Alice" });
|
|
58
|
+
const retrieved = await getUser(user.id);
|
|
59
|
+
expect(retrieved.name).toBe("Alice");
|
|
60
|
+
});
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
**Tautological tests**: Expected value restates the implementation, so the test passes by construction.
|
|
64
|
+
|
|
65
|
+
```typescript
|
|
66
|
+
// BAD: Expected value is recomputed the way the code computes it
|
|
67
|
+
test("calculateTotal sums line items", () => {
|
|
68
|
+
const items = [{ price: 10 }, { price: 5 }];
|
|
69
|
+
const expected = items.reduce((sum, item) => sum + item.price, 0);
|
|
70
|
+
expect(calculateTotal(items)).toBe(expected);
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
// GOOD: Expected value comes from an independent, known literal
|
|
74
|
+
test("calculateTotal sums line items", () => {
|
|
75
|
+
expect(calculateTotal([{ price: 10 }, { price: 5 }])).toBe(15);
|
|
76
|
+
});
|
|
77
|
+
```
|