forge-workflow 0.0.3 → 0.0.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (209) hide show
  1. package/.claude/commands/dev.md +340 -314
  2. package/.claude/commands/plan.md +521 -478
  3. package/.claude/commands/premerge.md +176 -179
  4. package/.claude/commands/research.md +42 -42
  5. package/.claude/commands/review.md +442 -442
  6. package/.claude/commands/rollback.md +721 -721
  7. package/.claude/commands/ship.md +164 -134
  8. package/.claude/commands/sonarcloud.md +152 -152
  9. package/.claude/commands/status.md +48 -77
  10. package/.claude/commands/validate.md +282 -237
  11. package/.claude/commands/verify.md +221 -221
  12. package/.claude/rules/greptile-review-process.md +285 -285
  13. package/.claude/rules/workflow.md +105 -105
  14. package/.claude/scripts/greptile-resolve.sh +526 -526
  15. package/.claude/scripts/load-env.sh +32 -32
  16. package/.cline/workflows/dev.md +337 -311
  17. package/.cline/workflows/plan.md +518 -475
  18. package/.cline/workflows/premerge.md +173 -176
  19. package/.cline/workflows/research.md +39 -39
  20. package/.cline/workflows/review.md +439 -439
  21. package/.cline/workflows/rollback.md +718 -718
  22. package/.cline/workflows/ship.md +161 -131
  23. package/.cline/workflows/sonarcloud.md +146 -146
  24. package/.cline/workflows/status.md +45 -74
  25. package/.cline/workflows/validate.md +279 -234
  26. package/.cline/workflows/verify.md +218 -218
  27. package/.codex/config.toml +11 -11
  28. package/.codex/skills/dev/SKILL.md +340 -314
  29. package/.codex/skills/plan/SKILL.md +521 -478
  30. package/.codex/skills/premerge/SKILL.md +176 -179
  31. package/.codex/skills/research/SKILL.md +42 -42
  32. package/.codex/skills/review/SKILL.md +442 -442
  33. package/.codex/skills/rollback/SKILL.md +721 -721
  34. package/.codex/skills/ship/SKILL.md +164 -134
  35. package/.codex/skills/sonarcloud/SKILL.md +149 -149
  36. package/.codex/skills/status/SKILL.md +48 -77
  37. package/.codex/skills/validate/SKILL.md +282 -237
  38. package/.codex/skills/verify/SKILL.md +221 -221
  39. package/.cursor/commands/dev.md +337 -311
  40. package/.cursor/commands/plan.md +518 -475
  41. package/.cursor/commands/premerge.md +173 -176
  42. package/.cursor/commands/research.md +39 -39
  43. package/.cursor/commands/review.md +439 -439
  44. package/.cursor/commands/rollback.md +718 -718
  45. package/.cursor/commands/ship.md +161 -131
  46. package/.cursor/commands/sonarcloud.md +146 -146
  47. package/.cursor/commands/status.md +45 -74
  48. package/.cursor/commands/validate.md +279 -234
  49. package/.cursor/commands/verify.md +218 -218
  50. package/.cursor/rules/permissions-guidance.mdc +37 -37
  51. package/.forge/hooks/check-tdd.js +240 -240
  52. package/.github/PLUGIN_TEMPLATE.json +32 -32
  53. package/.github/prompts/dev.prompt.md +342 -316
  54. package/.github/prompts/plan.prompt.md +523 -480
  55. package/.github/prompts/premerge.prompt.md +178 -181
  56. package/.github/prompts/research.prompt.md +44 -44
  57. package/.github/prompts/review.prompt.md +444 -444
  58. package/.github/prompts/rollback.prompt.md +723 -723
  59. package/.github/prompts/ship.prompt.md +166 -136
  60. package/.github/prompts/sonarcloud.prompt.md +151 -151
  61. package/.github/prompts/status.prompt.md +50 -79
  62. package/.github/prompts/validate.prompt.md +284 -239
  63. package/.github/prompts/verify.prompt.md +223 -223
  64. package/.github/workflows/beads-to-github.yml +56 -0
  65. package/.github/workflows/github-to-beads.yml +97 -0
  66. package/.kilocode/workflows/dev.md +341 -315
  67. package/.kilocode/workflows/plan.md +522 -479
  68. package/.kilocode/workflows/premerge.md +177 -180
  69. package/.kilocode/workflows/research.md +43 -43
  70. package/.kilocode/workflows/review.md +443 -443
  71. package/.kilocode/workflows/rollback.md +722 -722
  72. package/.kilocode/workflows/ship.md +165 -135
  73. package/.kilocode/workflows/sonarcloud.md +150 -150
  74. package/.kilocode/workflows/status.md +49 -78
  75. package/.kilocode/workflows/validate.md +283 -238
  76. package/.kilocode/workflows/verify.md +222 -222
  77. package/.mcp.json.example +12 -12
  78. package/.opencode/commands/dev.md +340 -314
  79. package/.opencode/commands/plan.md +521 -478
  80. package/.opencode/commands/premerge.md +176 -179
  81. package/.opencode/commands/research.md +42 -42
  82. package/.opencode/commands/review.md +442 -442
  83. package/.opencode/commands/rollback.md +721 -721
  84. package/.opencode/commands/ship.md +164 -134
  85. package/.opencode/commands/sonarcloud.md +149 -149
  86. package/.opencode/commands/status.md +48 -77
  87. package/.opencode/commands/validate.md +282 -237
  88. package/.opencode/commands/verify.md +221 -221
  89. package/.roo/commands/dev.md +341 -315
  90. package/.roo/commands/plan.md +522 -479
  91. package/.roo/commands/premerge.md +177 -180
  92. package/.roo/commands/research.md +43 -43
  93. package/.roo/commands/review.md +443 -443
  94. package/.roo/commands/rollback.md +722 -722
  95. package/.roo/commands/ship.md +165 -135
  96. package/.roo/commands/sonarcloud.md +150 -150
  97. package/.roo/commands/status.md +49 -78
  98. package/.roo/commands/validate.md +283 -238
  99. package/.roo/commands/verify.md +222 -222
  100. package/AGENTS.md +175 -169
  101. package/CLAUDE.md +100 -99
  102. package/LICENSE +21 -21
  103. package/README.md +429 -414
  104. package/bin/forge-cmd.js +313 -313
  105. package/bin/{forge-validate.js → forge-preflight.js} +309 -303
  106. package/bin/forge.js +4596 -4232
  107. package/docs/AGENT_INSTALL_PROMPT.md +342 -342
  108. package/docs/BEADS_GITHUB_SYNC.md +251 -0
  109. package/docs/ENHANCED_ONBOARDING.md +602 -602
  110. package/docs/EXAMPLES.md +482 -482
  111. package/docs/GREPTILE_SETUP.md +400 -400
  112. package/docs/MANUAL_REVIEW_GUIDE.md +106 -106
  113. package/docs/ROADMAP.md +359 -359
  114. package/docs/SETUP.md +663 -632
  115. package/docs/TOOLCHAIN.md +630 -630
  116. package/docs/VALIDATION.md +363 -363
  117. package/install.sh +40 -1058
  118. package/lefthook.yml +39 -39
  119. package/lib/agents/README.md +198 -198
  120. package/lib/agents/claude.plugin.json +28 -28
  121. package/lib/agents/cline.plugin.json +22 -22
  122. package/lib/agents/codex.plugin.json +19 -19
  123. package/lib/agents/copilot.plugin.json +24 -24
  124. package/lib/agents/cursor.plugin.json +25 -25
  125. package/lib/agents/kilocode.plugin.json +22 -22
  126. package/lib/agents/opencode.plugin.json +20 -20
  127. package/lib/agents/roo.plugin.json +23 -23
  128. package/lib/agents-config.js +2112 -2112
  129. package/lib/beads-health-check.js +143 -0
  130. package/lib/beads-setup.js +341 -0
  131. package/lib/beads-sync-scaffold.js +260 -0
  132. package/lib/commands/dev.js +513 -513
  133. package/lib/commands/plan.js +692 -692
  134. package/lib/commands/recommend.js +119 -119
  135. package/lib/commands/ship.js +377 -377
  136. package/lib/commands/status.js +378 -378
  137. package/lib/commands/validate.js +602 -602
  138. package/lib/context-merge.js +359 -359
  139. package/lib/dep-guard/analyzer.js +294 -294
  140. package/lib/dep-guard/behavior-detector.js +98 -98
  141. package/lib/dep-guard/contract-detector.js +162 -162
  142. package/lib/dep-guard/import-detector.js +498 -498
  143. package/lib/dep-guard/path-utils.js +13 -13
  144. package/lib/dep-guard/rubric.js +120 -120
  145. package/lib/dep-guard/task-parser.js +318 -318
  146. package/lib/detect-agent.js +191 -0
  147. package/lib/detect-worktree.js +47 -0
  148. package/lib/file-hash.js +26 -0
  149. package/lib/husky-migration.js +450 -0
  150. package/lib/lefthook-check.js +65 -0
  151. package/lib/pat-setup.js +207 -0
  152. package/lib/plugin-catalog.js +350 -350
  153. package/lib/plugin-manager.js +166 -166
  154. package/lib/plugin-recommender.js +141 -141
  155. package/lib/project-discovery.js +491 -491
  156. package/lib/setup-action-log.js +139 -0
  157. package/lib/setup-summary-renderer.js +106 -0
  158. package/lib/setup-utils.js +96 -0
  159. package/lib/setup.js +192 -118
  160. package/lib/smart-merge.js +64 -0
  161. package/lib/symlink-utils.js +81 -0
  162. package/lib/workflow-profiles.js +197 -197
  163. package/package.json +131 -129
  164. package/scripts/beads-context.sh +291 -0
  165. package/scripts/beads-context.test.js +563 -0
  166. package/scripts/behavioral-judge.sh +378 -0
  167. package/scripts/benchmark.js +85 -0
  168. package/scripts/branch-protection.js +183 -0
  169. package/scripts/check-agents.js +172 -0
  170. package/scripts/commitlint.js +42 -0
  171. package/scripts/conflict-detect.sh +323 -0
  172. package/scripts/dep-guard-analyze.js +71 -0
  173. package/scripts/dep-guard.sh +811 -0
  174. package/scripts/eval_win.py +249 -0
  175. package/scripts/file-index.sh +399 -0
  176. package/scripts/github-beads-sync/comment.mjs +64 -0
  177. package/scripts/github-beads-sync/config.mjs +148 -0
  178. package/scripts/github-beads-sync/github-api.mjs +131 -0
  179. package/scripts/github-beads-sync/index.mjs +332 -0
  180. package/scripts/github-beads-sync/label-mapper.mjs +54 -0
  181. package/scripts/github-beads-sync/mapping.mjs +78 -0
  182. package/scripts/github-beads-sync/reverse-sync-cli.mjs +31 -0
  183. package/scripts/github-beads-sync/reverse-sync.mjs +138 -0
  184. package/scripts/github-beads-sync/run-bd.mjs +159 -0
  185. package/scripts/github-beads-sync/sanitize.mjs +121 -0
  186. package/scripts/github-beads-sync.config.json +26 -0
  187. package/scripts/improve-command.js +375 -0
  188. package/scripts/lib/eval-runner.js +229 -0
  189. package/scripts/lib/eval-schema.js +135 -0
  190. package/scripts/lib/eval-storage.js +78 -0
  191. package/scripts/lib/grading.js +203 -0
  192. package/scripts/lib/transcript-parser.js +63 -0
  193. package/scripts/lint.js +47 -0
  194. package/scripts/migrate-to-bun-test.js +412 -0
  195. package/scripts/run-command-eval.js +236 -0
  196. package/scripts/smart-status.sh +782 -0
  197. package/scripts/sync-commands.js +571 -0
  198. package/scripts/sync-utils.sh +460 -0
  199. package/scripts/test-dashboard.js +123 -0
  200. package/scripts/test.js +44 -0
  201. package/scripts/validate.sh +94 -0
  202. package/skills/parallel-deep-research/SKILL.md +108 -108
  203. package/skills/parallel-deep-research/evals/README.md +27 -27
  204. package/skills/parallel-deep-research/evals/evals.json +62 -62
  205. package/skills/sonarcloud-analysis/SKILL.md +171 -171
  206. package/skills/sonarcloud-analysis/evals/README.md +27 -27
  207. package/skills/sonarcloud-analysis/evals/evals.json +50 -50
  208. package/skills/sonarcloud-analysis/references/api-reference.md +466 -466
  209. package/docs/WORKFLOW.md +0 -400
@@ -1,315 +1,341 @@
1
- ---
2
- description: Subagent-driven TDD implementation per task from /plan task list
3
- mode: code
4
- ---
5
-
6
- Implement each task from the /plan task list using a subagent-driven loop: implementer → spec compliance reviewer → code quality reviewer per task.
7
-
8
- # Dev
9
-
10
- This command reads the task list created by `/plan` and implements each task using a three-stage subagent loop. TDD is enforced inside each implementer subagent.
11
-
12
- ## Usage
13
-
14
- ```bash
15
- /dev
16
- ```
17
-
18
- ---
19
-
20
- ## Setup
21
-
22
- ### Step 1: Load context
23
-
24
- ```bash
25
- # Find task list and design doc
26
- ls docs/plans/
27
- ```
28
-
29
- Read:
30
- - **Task list**: `docs/plans/YYYY-MM-DD-<slug>-tasks.md` — extract ALL task text upfront
31
- - **Design doc**: `docs/plans/YYYY-MM-DD-<slug>-design.md` — including ambiguity policy section
32
-
33
- ### Step 2: Create decisions log
34
-
35
- Create an empty decisions log at the start of every /dev session:
36
-
37
- ```bash
38
- # docs/plans/YYYY-MM-DD-<slug>-decisions.md
39
- ```
40
-
41
- Format for each entry:
42
- ```
43
- ## Decision N
44
- **Date**: YYYY-MM-DD
45
- **Task**: Task N — <title>
46
- **Gap**: [what the spec didn't cover]
47
- **Score**: [filled checklist total]
48
- **Route**: PROCEED / SPEC-REVIEWER / BLOCKED
49
- **Choice made**: [if PROCEED: what was decided and why]
50
- **Status**: RESOLVED / PENDING-DEVELOPER-INPUT
51
- ```
52
-
53
- ### Step 3: Pre-flight checks
54
-
55
- ```
56
- <HARD-GATE: /dev start>
57
- Do NOT write any code until ALL confirmed:
58
- 1. git branch --show-current output is NOT main or master
59
- 2. git worktree list shows the worktree path for this feature
60
- 3. Task list file confirmed to exist (use Read tool — do not assume)
61
- 4. Decisions log file created
62
- </HARD-GATE>
63
- ```
64
-
65
- ---
66
-
67
- ## Per-Task Loop
68
-
69
- Repeat for each task in the task list, in order:
70
-
71
- ### Step A: Dispatch implementer subagent
72
-
73
- Provide the subagent with:
74
- - **Full task text** (copy the complete task content — do NOT send just the file path)
75
- - **Relevant design doc sections** for this task
76
- - **Recent git log** showing what has already been implemented
77
-
78
- The implementer subagent:
79
- 1. Asks clarifying questions before writing any code
80
- 2. Implements using RED-GREEN-REFACTOR
81
- 3. Self-reviews for correctness
82
- 4. Commits with a descriptive message
83
-
84
- ```
85
- <HARD-GATE: TDD enforcement (inside implementer subagent)>
86
- Do NOT write any production code until:
87
- 1. A FAILING test exists for that code
88
- 2. The test has been run and output shows it FAILING
89
- 3. The failure reason matches the expected missing behavior
90
-
91
- If code was written before its test: delete it. Start with the test.
92
- "The test would obviously fail" is not evidence. Run it and show the output.
93
- </HARD-GATE>
94
- ```
95
-
96
- ---
97
-
98
- ### Step B: Decision gate (when implementer hits a spec gap)
99
-
100
- If the implementer encounters something not specified in the design doc, STOP and fill this checklist BEFORE deciding how to proceed:
101
-
102
- ```
103
- Gap: [describe exactly what the spec doesn't cover]
104
-
105
- Score each dimension (0=No / 1=Possibly / 2=Yes):
106
- [ ] 1. Files affected beyond the current task?
107
- [ ] 2. Changes a function signature or public export?
108
- [ ] 3. Changes a shared module used by other tasks?
109
- [ ] 4. Changes or touches persistent data or schema?
110
- [ ] 5. Changes user-visible behavior not discussed in design doc?
111
- [ ] 6. Affects auth, permissions, or data exposure?
112
- [ ] 7. Hard to reverse without cascading changes to other files?
113
- TOTAL: ___ / 14
114
-
115
- Mandatory overrides any of these = automatically BLOCKED:
116
- [ ] Security dimension (6) scored 2
117
- [ ] Schema migration or data model change
118
- [ ] Removes or changes an existing public API endpoint
119
- [ ] Affects a task that is already implemented and committed
120
- ```
121
-
122
- **Score routing**:
123
- - **0-3**: PROCEED — make the decision, document in decisions log with full reasoning
124
- - **4-7**: SPEC-REVIEWER route this decision to spec reviewer. Continue other independent tasks while waiting
125
- - **8+, or any mandatory override triggered**: BLOCKED — document in decisions log with Status=PENDING-DEVELOPER-INPUT. Complete all other independent tasks first. Surface to developer at /dev exit
126
-
127
- Log the decision entry before continuing.
128
-
129
- ---
130
-
131
- ### Step C: Spec compliance review
132
-
133
- After the implementer finishes the task, dispatch a **spec compliance reviewer** subagent.
134
-
135
- Provide:
136
- - Full task text (what was supposed to be implemented)
137
- - Relevant design doc sections
138
- - `git diff` for this task's commits
139
-
140
- Reviewer checks:
141
- - All requirements from the task text are implemented
142
- - Nothing extra was added beyond task scope
143
- - Edge cases documented in design doc are handled
144
- - TDD evidence: test exists, test was run failing, then passing
145
-
146
- If spec issues found: implementer fixes → re-review → repeat until ✅
147
-
148
- ```
149
- <HARD-GATE: spec before quality>
150
- Do NOT dispatch code quality reviewer until spec compliance reviewer returns for this task.
151
- Running quality review before spec compliance is the wrong order.
152
- </HARD-GATE>
153
- ```
154
-
155
- ---
156
-
157
- ### Step D: Code quality review
158
-
159
- After spec ✅, dispatch a **code quality reviewer** subagent.
160
-
161
- Provide:
162
- - git SHAs for this task's commits
163
- - The changed code (`git diff`)
164
-
165
- Reviewer checks:
166
- - Naming: clear, descriptive, consistent with codebase conventions
167
- - Structure: functions not too long, proper separation of concerns
168
- - Duplication: no copy-paste that could be extracted
169
- - Test coverage: tests cover happy path and at least one error path
170
- - No magic numbers, no commented-out code, no TODO without a Beads issue
171
-
172
- If quality issues found: implementer fixes → re-review → repeat until ✅
173
-
174
- ---
175
-
176
- ### Step E: Task completion
177
-
178
- ```
179
- <HARD-GATE: task completion>
180
- NO COMPLETION CLAIMS WITHOUT FRESH VERIFICATION EVIDENCE.
181
-
182
- Do NOT mark task complete or move to next task until ALL confirmed in this session:
183
- 1. Spec compliance reviewer returned
184
- 2. Code quality reviewer returned ✅
185
- 3. Identify what command proves this task is done (e.g. `bun test`, a CLI invocation, a script run).
186
- 4. Run it fresh — show the actual output. "Last run was fine" is not evidence.
187
- 5. Tests run fresh — actual output shows passing.
188
- 6. Implementer has committed (git log shows the commit).
189
- 7. `bash scripts/beads-context.sh update-progress <id> <task-num> <total> "<title>" <commit-sha> <test-count> <gate-count>` ran successfully (exit code 0). If it fails: STOP. Show error. Do not proceed to next task.
190
-
191
- Forbidden phrases (these are not evidence):
192
- - "should pass"
193
- - "looks good"
194
- - "seems to work"
195
- </HARD-GATE>
196
- ```
197
-
198
- Mark task complete. Move to next task.
199
-
200
- ---
201
-
202
- ## /dev Completion
203
-
204
- After all tasks are complete (or BLOCKED):
205
-
206
- ### Final code review
207
-
208
- Dispatch a final code reviewer for the full implementation:
209
- - Overall coherence: does the feature hang together as a whole?
210
- - Cross-task consistency: naming, patterns, style consistent across all tasks?
211
- - Integration: do all the pieces connect correctly?
212
-
213
- ### Surface BLOCKED decisions
214
-
215
- If any decisions have Status=PENDING-DEVELOPER-INPUT:
216
-
217
- ```
218
- ⏸️ /dev blocked — developer input needed
219
-
220
- The following decisions were deferred during implementation:
221
-
222
- Decision 1: [gap description]
223
- Task: Task N — <title>
224
- Score: 11/14 (mandatory override: schema change)
225
- Options considered: [A] vs [B]
226
- Recommendation: [A] because [reason]
227
- Blocked tasks: Task 6, Task 7 (depend on this decision)
228
-
229
- Decision 2: ...
230
-
231
- Please review and respond. After decisions are resolved, the implementer
232
- will complete the blocked tasks and re-run spec + quality review.
233
- ```
234
-
235
- Wait for developer input. After decisions resolved: implement blocked tasks spec review quality review → complete.
236
-
237
- ### /dev exit gate
238
-
239
- ```
240
- <HARD-GATE: /dev exit>
241
- Do NOT declare /dev complete until:
242
- 1. All tasks are marked complete OR have BLOCKED status with PENDING-DEVELOPER-INPUT
243
- 2. BLOCKED decisions have been surfaced to developer and are awaiting input
244
- 3. Final code reviewer has approved (or issues fixed and re-reviewed)
245
- 4. All decisions in decisions log have Status of RESOLVED or PENDING-DEVELOPER-INPUT
246
- 5. No unresolved spec or quality issues remain
247
- </HARD-GATE>
248
- ```
249
-
250
- ### Beads update
251
-
252
- ```bash
253
- bash scripts/beads-context.sh stage-transition <id> dev validate
254
- ```
255
-
256
- ---
257
-
258
- ## Decision Gate Calibration
259
-
260
- The frequency of decision gates is a **plan quality metric**:
261
- - **0 gates fired**: Excellent Phase 1 Q&A covered all cases
262
- - **1-2 gates fired**: Good — minor gaps, normal
263
- - **3-5 gates fired**: Plan was incomplete — note for Phase 1 improvement next feature
264
- - **5+ gates fired**: Phase 1 Q&A was insufficient — the ambiguity policy field needed to be more specific
265
-
266
- Document the gate count in the final commit message.
267
-
268
- ---
269
-
270
- ## Example Output (all tasks complete)
271
-
272
- ```
273
- ✓ Task 1: Types and interfaces — COMPLETE
274
- Spec: ✅ Quality: ✅ Tests: 4/4 passing Commit: abc1234
275
- Decision gates: 0
276
-
277
- ✓ Task 2: Validation logic — COMPLETE
278
- Spec: ✅ Quality: ✅ Tests: 8/8 passing Commit: def5678
279
- Decision gates: 1 (PROCEED, score 2 — documented in decisions log)
280
-
281
- ✓ Task 3: API endpoint — COMPLETE
282
- Spec: ✅ Quality: ✅ Tests: 6/6 passing Commit: ghi9012
283
- Decision gates: 0
284
-
285
- ✓ Final code review: ✅ (coherent, consistent, correctly integrated)
286
-
287
- Decisions log: docs/plans/2026-02-26-stripe-billing-decisions.md
288
- - Decision 1: RESOLVED (score 2, proceeded with conservative choice)
289
- - Decision gates fired: 1 (plan quality: Good)
290
-
291
- ✓ Beads updated: forge-xyz → implementation complete
292
-
293
- Ready for /validate
294
- ```
295
-
296
- ## Integration with Workflow
297
-
298
- ```
299
- Utility: /status → Understand current context before starting
300
- Stage 1: /plan → Design intent → research → branch + worktree + task list
301
- Stage 2: /dev → Implement each task with subagent-driven TDD (you are here)
302
- Stage 3: /validate → Type check, lint, tests, security — all fresh output
303
- Stage 4: /ship → Push + create PR
304
- Stage 5: /review → Address GitHub Actions, Greptile, SonarCloud
305
- Stage 6: /premerge → Update docs, hand off PR to user
306
- Stage 7: /verify → Post-merge CI check on main
307
- ```
308
-
309
- ## Tips
310
-
311
- - **Send full task text to subagents**: Never send the file path — copy the complete task text directly into the subagent prompt
312
- - **TDD lives inside the implementer**: The implementer subagent is responsible for RED-GREEN-REFACTOR, not the orchestrating /dev session
313
- - **Spec before quality — always**: A task that passes quality review but fails spec compliance has still failed
314
- - **Decision gates are rare with a good plan**: If gates fire frequently, the Phase 1 Q&A needs more depth next time
315
- - **BLOCKED failed**: Surfacing a blocked decision with documentation and a recommendation is the correct behavior
1
+ ---
2
+ description: Subagent-driven TDD implementation per task from /plan task list
3
+ mode: code
4
+ ---
5
+
6
+ Implement each task from the /plan task list using a subagent-driven loop: implementer → spec compliance reviewer → code quality reviewer per task.
7
+
8
+ # Dev
9
+
10
+ This command reads the task list created by `/plan` and implements each task using a three-stage subagent loop. TDD is enforced inside each implementer subagent.
11
+
12
+ ## Usage
13
+
14
+ ```bash
15
+ /dev
16
+ ```
17
+
18
+ ---
19
+
20
+ ## Setup
21
+
22
+ ### Step 1: Load context
23
+
24
+ ```bash
25
+ # Find task list and design doc
26
+ ls docs/plans/
27
+ ```
28
+
29
+ Read:
30
+ - **Task list**: `docs/plans/YYYY-MM-DD-<slug>-tasks.md` — extract ALL task text upfront
31
+ - **Design doc**: `docs/plans/YYYY-MM-DD-<slug>-design.md` — including ambiguity policy section
32
+
33
+ ### Step 2: Create decisions log
34
+
35
+ Create an empty decisions log at the start of every /dev session:
36
+
37
+ ```bash
38
+ # docs/plans/YYYY-MM-DD-<slug>-decisions.md
39
+ ```
40
+
41
+ Format for each entry:
42
+ ```
43
+ ## Decision N
44
+ **Date**: YYYY-MM-DD
45
+ **Task**: Task N — <title>
46
+ **Gap**: [what the spec didn't cover]
47
+ **Score**: [filled checklist total]
48
+ **Route**: PROCEED / SPEC-REVIEWER / BLOCKED
49
+ **Choice made**: [if PROCEED: what was decided and why]
50
+ **Status**: RESOLVED / PENDING-DEVELOPER-INPUT
51
+ ```
52
+
53
+ ### Step 3: Pre-flight checks
54
+
55
+ ```
56
+ <HARD-GATE: /dev start>
57
+ Do NOT write any code until ALL confirmed:
58
+ 1. git branch --show-current output is NOT main or master
59
+ 2. git worktree list shows the worktree path for this feature
60
+ 3. Task list file confirmed to exist (use Read tool — do not assume)
61
+ 4. Decisions log file created
62
+ </HARD-GATE>
63
+ ```
64
+
65
+ ---
66
+
67
+
68
+ ### Multi-developer conflict check (soft block)
69
+
70
+ Before starting the per-task loop, check for cross-developer conflicts:
71
+
72
+ ```bash
73
+ # Auto-sync to get latest team state
74
+ bash scripts/sync-utils.sh auto-sync
75
+
76
+ # Check for conflicts with the current beads issue
77
+ bash scripts/conflict-detect.sh --issue <beads-id>
78
+ ```
79
+
80
+ If exit code 2 (validation error): show error message, abort — do not show conflict prompt.
81
+
82
+ If exit code 1 (conflicts found):
83
+ - Display the conflict output to the developer
84
+ - Ask: "Other developers are working in overlapping areas. Proceed anyway? (y/n)"
85
+ - If `n`: exit cleanly, no side effects
86
+ - If `y`: log override via `bd comments add <id> "Conflict override: proceeding despite overlap with <conflicting-issues>"`, then continue to Per-Task Loop
87
+ - Audit: record conflict override per OWASP A09
88
+
89
+ If exit code 0: proceed silently to Per-Task Loop.
90
+
91
+ ---
92
+
93
+ ## Per-Task Loop
94
+
95
+ Repeat for each task in the task list, in order:
96
+
97
+ ### Step A: Dispatch implementer subagent
98
+
99
+ Provide the subagent with:
100
+ - **Full task text** (copy the complete task content do NOT send just the file path)
101
+ - **Relevant design doc sections** for this task
102
+ - **Recent git log** showing what has already been implemented
103
+
104
+ The implementer subagent:
105
+ 1. Asks clarifying questions before writing any code
106
+ 2. Implements using RED-GREEN-REFACTOR
107
+ 3. Self-reviews for correctness
108
+ 4. Commits with a descriptive message
109
+
110
+ ```
111
+ <HARD-GATE: TDD enforcement (inside implementer subagent)>
112
+ Do NOT write any production code until:
113
+ 1. A FAILING test exists for that code
114
+ 2. The test has been run and output shows it FAILING
115
+ 3. The failure reason matches the expected missing behavior
116
+
117
+ If code was written before its test: delete it. Start with the test.
118
+ "The test would obviously fail" is not evidence. Run it and show the output.
119
+ </HARD-GATE>
120
+ ```
121
+
122
+ ---
123
+
124
+ ### Step B: Decision gate (when implementer hits a spec gap)
125
+
126
+ If the implementer encounters something not specified in the design doc, STOP and fill this checklist BEFORE deciding how to proceed:
127
+
128
+ ```
129
+ Gap: [describe exactly what the spec doesn't cover]
130
+
131
+ Score each dimension (0=No / 1=Possibly / 2=Yes):
132
+ [ ] 1. Files affected beyond the current task?
133
+ [ ] 2. Changes a function signature or public export?
134
+ [ ] 3. Changes a shared module used by other tasks?
135
+ [ ] 4. Changes or touches persistent data or schema?
136
+ [ ] 5. Changes user-visible behavior not discussed in design doc?
137
+ [ ] 6. Affects auth, permissions, or data exposure?
138
+ [ ] 7. Hard to reverse without cascading changes to other files?
139
+ TOTAL: ___ / 14
140
+
141
+ Mandatory overrides any of these = automatically BLOCKED:
142
+ [ ] Security dimension (6) scored 2
143
+ [ ] Schema migration or data model change
144
+ [ ] Removes or changes an existing public API endpoint
145
+ [ ] Affects a task that is already implemented and committed
146
+ ```
147
+
148
+ **Score routing**:
149
+ - **0-3**: PROCEED — make the decision, document in decisions log with full reasoning
150
+ - **4-7**: SPEC-REVIEWER route this decision to spec reviewer. Continue other independent tasks while waiting
151
+ - **8+, or any mandatory override triggered**: BLOCKED document in decisions log with Status=PENDING-DEVELOPER-INPUT. Complete all other independent tasks first. Surface to developer at /dev exit
152
+
153
+ Log the decision entry before continuing.
154
+
155
+ ---
156
+
157
+ ### Step C: Spec compliance review
158
+
159
+ After the implementer finishes the task, dispatch a **spec compliance reviewer** subagent.
160
+
161
+ Provide:
162
+ - Full task text (what was supposed to be implemented)
163
+ - Relevant design doc sections
164
+ - `git diff` for this task's commits
165
+
166
+ Reviewer checks:
167
+ - All requirements from the task text are implemented
168
+ - Nothing extra was added beyond task scope
169
+ - Edge cases documented in design doc are handled
170
+ - TDD evidence: test exists, test was run failing, then passing
171
+
172
+ If spec issues found: implementer fixes → re-review → repeat until ✅
173
+
174
+ ```
175
+ <HARD-GATE: spec before quality>
176
+ Do NOT dispatch code quality reviewer until spec compliance reviewer returns ✅ for this task.
177
+ Running quality review before spec compliance is the wrong order.
178
+ </HARD-GATE>
179
+ ```
180
+
181
+ ---
182
+
183
+ ### Step D: Code quality review
184
+
185
+ After spec ✅, dispatch a **code quality reviewer** subagent.
186
+
187
+ Provide:
188
+ - git SHAs for this task's commits
189
+ - The changed code (`git diff`)
190
+
191
+ Reviewer checks:
192
+ - Naming: clear, descriptive, consistent with codebase conventions
193
+ - Structure: functions not too long, proper separation of concerns
194
+ - Duplication: no copy-paste that could be extracted
195
+ - Test coverage: tests cover happy path and at least one error path
196
+ - No magic numbers, no commented-out code, no TODO without a Beads issue
197
+
198
+ If quality issues found: implementer fixes → re-review → repeat until ✅
199
+
200
+ ---
201
+
202
+ ### Step E: Task completion
203
+
204
+ ```
205
+ <HARD-GATE: task completion>
206
+ NO COMPLETION CLAIMS WITHOUT FRESH VERIFICATION EVIDENCE.
207
+
208
+ Do NOT mark task complete or move to next task until ALL confirmed in this session:
209
+ 1. Spec compliance reviewer returned
210
+ 2. Code quality reviewer returned
211
+ 3. Identify what command proves this task is done (e.g. `bun test`, a CLI invocation, a script run).
212
+ 4. Run it fresh — show the actual output. "Last run was fine" is not evidence.
213
+ 5. Tests run fresh — actual output shows passing.
214
+ 6. Implementer has committed (git log shows the commit).
215
+ 7. `bash scripts/beads-context.sh update-progress <id> <task-num> <total> "<title>" <commit-sha> <test-count> <gate-count>` ran successfully (exit code 0). If it fails: STOP. Show error. Do not proceed to next task.
216
+
217
+ Forbidden phrases (these are not evidence):
218
+ - "should pass"
219
+ - "looks good"
220
+ - "seems to work"
221
+ </HARD-GATE>
222
+ ```
223
+
224
+ Mark task complete. Move to next task.
225
+
226
+ ---
227
+
228
+ ## /dev Completion
229
+
230
+ After all tasks are complete (or BLOCKED):
231
+
232
+ ### Final code review
233
+
234
+ Dispatch a final code reviewer for the full implementation:
235
+ - Overall coherence: does the feature hang together as a whole?
236
+ - Cross-task consistency: naming, patterns, style consistent across all tasks?
237
+ - Integration: do all the pieces connect correctly?
238
+
239
+ ### Surface BLOCKED decisions
240
+
241
+ If any decisions have Status=PENDING-DEVELOPER-INPUT:
242
+
243
+ ```
244
+ ⏸️ /dev blocked developer input needed
245
+
246
+ The following decisions were deferred during implementation:
247
+
248
+ Decision 1: [gap description]
249
+ Task: Task N — <title>
250
+ Score: 11/14 (mandatory override: schema change)
251
+ Options considered: [A] vs [B]
252
+ Recommendation: [A] because [reason]
253
+ Blocked tasks: Task 6, Task 7 (depend on this decision)
254
+
255
+ Decision 2: ...
256
+
257
+ Please review and respond. After decisions are resolved, the implementer
258
+ will complete the blocked tasks and re-run spec + quality review.
259
+ ```
260
+
261
+ Wait for developer input. After decisions resolved: implement blocked tasks spec review → quality review → complete.
262
+
263
+ ### /dev exit gate
264
+
265
+ ```
266
+ <HARD-GATE: /dev exit>
267
+ Do NOT declare /dev complete until:
268
+ 1. All tasks are marked complete OR have BLOCKED status with PENDING-DEVELOPER-INPUT
269
+ 2. BLOCKED decisions have been surfaced to developer and are awaiting input
270
+ 3. Final code reviewer has approved (or issues fixed and re-reviewed)
271
+ 4. All decisions in decisions log have Status of RESOLVED or PENDING-DEVELOPER-INPUT
272
+ 5. No unresolved spec or quality issues remain
273
+ </HARD-GATE>
274
+ ```
275
+
276
+ ### Beads update
277
+
278
+ ```bash
279
+ bash scripts/beads-context.sh stage-transition <id> dev validate
280
+ ```
281
+
282
+ ---
283
+
284
+ ## Decision Gate Calibration
285
+
286
+ The frequency of decision gates is a **plan quality metric**:
287
+ - **0 gates fired**: Excellent — Phase 1 Q&A covered all cases
288
+ - **1-2 gates fired**: Good minor gaps, normal
289
+ - **3-5 gates fired**: Plan was incomplete — note for Phase 1 improvement next feature
290
+ - **5+ gates fired**: Phase 1 Q&A was insufficient — the ambiguity policy field needed to be more specific
291
+
292
+ Document the gate count in the final commit message.
293
+
294
+ ---
295
+
296
+ ## Example Output (all tasks complete)
297
+
298
+ ```
299
+ ✓ Task 1: Types and interfaces COMPLETE
300
+ Spec: ✅ Quality: ✅ Tests: 4/4 passing Commit: abc1234
301
+ Decision gates: 0
302
+
303
+ Task 2: Validation logic COMPLETE
304
+ Spec: ✅ Quality: ✅ Tests: 8/8 passing Commit: def5678
305
+ Decision gates: 1 (PROCEED, score 2 documented in decisions log)
306
+
307
+ ✓ Task 3: API endpoint — COMPLETE
308
+ Spec: ✅ Quality: ✅ Tests: 6/6 passing Commit: ghi9012
309
+ Decision gates: 0
310
+
311
+ Final code review: (coherent, consistent, correctly integrated)
312
+
313
+ Decisions log: docs/plans/2026-02-26-stripe-billing-decisions.md
314
+ - Decision 1: RESOLVED (score 2, proceeded with conservative choice)
315
+ - Decision gates fired: 1 (plan quality: Good)
316
+
317
+ ✓ Beads updated: forge-xyz → implementation complete
318
+
319
+ Ready for /validate
320
+ ```
321
+
322
+ ## Integration with Workflow
323
+
324
+ ```
325
+ Utility: /status → Understand current context before starting
326
+ Stage 1: /plan → Design intent → research → branch + worktree + task list
327
+ Stage 2: /dev → Implement each task with subagent-driven TDD (you are here)
328
+ Stage 3: /validate → Type check, lint, tests, security — all fresh output
329
+ Stage 4: /ship → Push + create PR
330
+ Stage 5: /review → Address GitHub Actions, Greptile, SonarCloud
331
+ Stage 6: /premerge → Update docs, hand off PR to user
332
+ Stage 7: /verify → Post-merge CI check on main
333
+ ```
334
+
335
+ ## Tips
336
+
337
+ - **Send full task text to subagents**: Never send the file path — copy the complete task text directly into the subagent prompt
338
+ - **TDD lives inside the implementer**: The implementer subagent is responsible for RED-GREEN-REFACTOR, not the orchestrating /dev session
339
+ - **Spec before quality — always**: A task that passes quality review but fails spec compliance has still failed
340
+ - **Decision gates are rare with a good plan**: If gates fire frequently, the Phase 1 Q&A needs more depth next time
341
+ - **BLOCKED ≠ failed**: Surfacing a blocked decision with documentation and a recommendation is the correct behavior