superpowers-mcp 6.3.10 → 6.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.ja.md +34 -33
- package/README.ko.md +34 -33
- package/README.md +34 -33
- package/README.zh-TW.md +34 -33
- package/docs/skill-compositions.ja.md +6 -2
- package/docs/skill-compositions.ko.md +6 -2
- package/docs/skill-compositions.md +5 -1
- package/docs/skill-compositions.zh-TW.md +6 -2
- package/out/server.js +66 -64
- package/out/setup-runner.js +17 -16
- package/out/setup.js +17 -16
- package/package.json +2 -2
- package/scripts/copy-skills.js +101 -26
- package/scripts/setup.js +99 -6
- package/scripts/upstream-drift.js +40 -5
- package/skills/brainstorming/visual-companion.md +6 -6
- package/skills/diagnosing-superpowers/SKILL.md +120 -0
- package/skills/diagnosing-superpowers/prompts/analyst-common.md +38 -0
- package/skills/diagnosing-superpowers/prompts/cost-and-time.md +28 -0
- package/skills/diagnosing-superpowers/prompts/plan-adherence.md +29 -0
- package/skills/diagnosing-superpowers/prompts/quality-evidence.md +26 -0
- package/skills/diagnosing-superpowers/prompts/repeated-work.md +30 -0
- package/skills/diagnosing-superpowers/prompts/request-conflicts.md +20 -0
- package/skills/diagnosing-superpowers/prompts/scrub-audit.md +33 -0
- package/skills/diagnosing-superpowers/prompts/scrub.md +29 -0
- package/skills/diagnosing-superpowers/prompts/similar-session.md +38 -0
- package/skills/diagnosing-superpowers/prompts/skill-timeline.md +30 -0
- package/skills/diagnosing-superpowers/prompts/stumbles.md +28 -0
- package/skills/diagnosing-superpowers/references/context-safety.md +22 -0
- package/skills/diagnosing-superpowers/references/github-issues.md +47 -0
- package/skills/diagnosing-superpowers/references/redaction-policy.md +34 -0
- package/skills/diagnosing-superpowers/references/session-discovery.md +31 -0
- package/skills/diagnosing-superpowers/templates/bundle-README.md +77 -0
- package/skills/diagnosing-superpowers/templates/case.md +64 -0
- package/skills/diagnosing-superpowers/templates/issue.md +51 -0
- package/skills/diagnosing-superpowers/templates/report.md +82 -0
- package/skills/executing-plans/SKILL.md +405 -58
- package/skills/executing-plans/scripts/task-done +55 -0
- package/skills/executing-plans/scripts/task-done.ps1 +83 -0
- package/skills/executing-plans/scripts/task-start +30 -0
- package/skills/executing-plans/scripts/task-start.ps1 +38 -0
- package/skills/requesting-code-review/code-reviewer.md +18 -1
- package/skills/subagent-driven-development/SKILL.md +31 -26
- package/skills/subagent-driven-development/re-review-prompt.md +1 -1
- package/skills/subagent-driven-development/scripts/review-package +4 -0
- package/skills/subagent-driven-development/scripts/review-package.ps1 +2 -0
- package/skills/subagent-driven-development/scripts/sdd-workspace +9 -5
- package/skills/subagent-driven-development/scripts/sdd-workspace.ps1 +10 -0
- package/skills/subagent-driven-development/scripts/task-brief +2 -0
- package/skills/subagent-driven-development/task-reviewer-prompt.md +2 -2
- package/skills/systematic-debugging/root-cause-tracing.md +1 -1
- package/skills/using-superpowers/SKILL.md +2 -0
- package/skills/using-superpowers/references/claude-code-tools.md +29 -0
- package/skills/using-superpowers/references/muse-tools.md +35 -0
- package/skills/writing-plans/SKILL.md +23 -12
- package/skills/writing-skills/SKILL.md +4 -2
|
@@ -1,70 +1,417 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: executing-plans
|
|
3
|
-
description: Use when
|
|
3
|
+
description: Use when executing an implementation plan in the current session as the implementer yourself — your human partner chose inline execution, or no subagent tool is available
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Executing Plans
|
|
7
7
|
|
|
8
|
-
|
|
8
|
+
Execute the plan yourself, task by task, in this session: no implementer
|
|
9
|
+
subagent per task, no reviewer per task. One fresh-context review of the
|
|
10
|
+
whole branch at the end.
|
|
9
11
|
|
|
10
|
-
|
|
12
|
+
**Why inline:** Subagent-driven development pays for a fresh implementer
|
|
13
|
+
and a fresh reviewer on every task, each re-reading the codebase from zero.
|
|
14
|
+
Inline execution pays for one context (yours) plus one reviewer at the end.
|
|
15
|
+
What it gives up is a fresh context per task and a second pair of eyes per
|
|
16
|
+
task. This skill keeps what those two things bought, by other means: the
|
|
17
|
+
brief is the spec, the ledger is your memory, TDD is the per-task gate, and
|
|
18
|
+
the final reviewer is the second pair of eyes.
|
|
11
19
|
|
|
12
|
-
**
|
|
20
|
+
**Core principle:** The plan already did the thinking. Execute it exactly,
|
|
21
|
+
prove each step with a test you watched fail and then pass, and leave a
|
|
22
|
+
record that survives your own forgetting.
|
|
13
23
|
|
|
14
|
-
**
|
|
24
|
+
**Narration:** between tool calls, narrate at most one short line — the
|
|
25
|
+
ledger and the tool results carry the record.
|
|
26
|
+
|
|
27
|
+
**Continuous execution:** Do not pause to check in with your human partner
|
|
28
|
+
between tasks. They chose inline execution to spend less, not to answer
|
|
29
|
+
"should I continue?" after every task. Execute all tasks from the plan
|
|
30
|
+
without stopping.
|
|
31
|
+
|
|
32
|
+
**Rulings, not stalls.** Conflicts, ambiguities, plan defects — decide them.
|
|
33
|
+
The spec is the binding authority, the plan is its argument, and your
|
|
34
|
+
judgment settles what neither answers. Record every decision in the ledger
|
|
35
|
+
as `Ruling: <what you decided> — <why> — <what it costs if wrong>`, and keep
|
|
36
|
+
going. Deviating from the plan without a ledgered ruling is a decision made
|
|
37
|
+
in secret.
|
|
38
|
+
|
|
39
|
+
Four things stop you, and only these: an irreversible or destructive
|
|
40
|
+
operation; a security-sensitive action; a side effect outside this worktree
|
|
41
|
+
that norms say you ask about first (a merge, a push to a shared branch, a
|
|
42
|
+
publish); and a plan so broken that every path forward is a guess. For
|
|
43
|
+
those, stop and ask.
|
|
44
|
+
|
|
45
|
+
## When to Use
|
|
46
|
+
|
|
47
|
+
- You have a plan from superpowers:writing-plans and your human partner
|
|
48
|
+
chose inline execution at the handoff.
|
|
49
|
+
- Your harness has no subagent tool (see the per-platform references in
|
|
50
|
+
`../using-superpowers/references/`). Never fabricate a dispatch; run
|
|
51
|
+
the plan here.
|
|
52
|
+
- Tasks are mostly independent — the same precondition as
|
|
53
|
+
superpowers:subagent-driven-development.
|
|
54
|
+
|
|
55
|
+
A fully specified plan makes inline execution transcription plus testing:
|
|
56
|
+
it runs well on a mid-tier session model, and the one place the most
|
|
57
|
+
capable model earns its cost is the final review, which this skill
|
|
58
|
+
dispatches separately. Tell your human partner so when they choose inline.
|
|
59
|
+
|
|
60
|
+
Prefer superpowers:subagent-driven-development when your human partner
|
|
61
|
+
wants a review gate on every task, or when the plan is long enough that
|
|
62
|
+
its later tasks would run on a compacted context. Inline execution over a
|
|
63
|
+
long plan still works — the ledger is what makes it recoverable — but the
|
|
64
|
+
last tasks get the least of you.
|
|
15
65
|
|
|
16
66
|
## The Process
|
|
17
67
|
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
-
|
|
49
|
-
-
|
|
50
|
-
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
68
|
+
```dot
|
|
69
|
+
digraph process {
|
|
70
|
+
rankdir=TB;
|
|
71
|
+
|
|
72
|
+
subgraph cluster_per_task {
|
|
73
|
+
label="Per Task";
|
|
74
|
+
"task-start: brief + BASE; read the brief" [shape=box];
|
|
75
|
+
"Work the steps in order: TDD, run every verification, read every output" [shape=box];
|
|
76
|
+
"Step output matches plan's Expected?" [shape=diamond];
|
|
77
|
+
"Plan wrong? Rule and ledger. Code wrong? systematic-debugging" [shape=box];
|
|
78
|
+
"Commit as the plan's commit steps say" [shape=box];
|
|
79
|
+
"Completion contract met?" [shape=diamond];
|
|
80
|
+
"task-done: run tests, ledger the result; mark todo complete" [shape=box];
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
"Setup: worktree, workspace + ledger, read plan + spec, pre-flight scan" [shape=box];
|
|
84
|
+
"More tasks remain?" [shape=diamond];
|
|
85
|
+
"Final whole-branch review (fresh reviewer if you have one)" [shape=box];
|
|
86
|
+
"Re-grade, then: Critical/Important → ONE fix pass, each fix RED→GREEN + green suite; Minor → ledger" [shape=box];
|
|
87
|
+
"Final review clean: delete this plan's workspace" [shape=box];
|
|
88
|
+
"Use superpowers:finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen];
|
|
89
|
+
|
|
90
|
+
"Setup: worktree, workspace + ledger, read plan + spec, pre-flight scan" -> "task-start: brief + BASE; read the brief";
|
|
91
|
+
"task-start: brief + BASE; read the brief" -> "Work the steps in order: TDD, run every verification, read every output";
|
|
92
|
+
"Work the steps in order: TDD, run every verification, read every output" -> "Step output matches plan's Expected?";
|
|
93
|
+
"Step output matches plan's Expected?" -> "Plan wrong? Rule and ledger. Code wrong? systematic-debugging" [label="no"];
|
|
94
|
+
"Plan wrong? Rule and ledger. Code wrong? systematic-debugging" -> "Work the steps in order: TDD, run every verification, read every output";
|
|
95
|
+
"Step output matches plan's Expected?" -> "Commit as the plan's commit steps say" [label="yes, last step"];
|
|
96
|
+
"Commit as the plan's commit steps say" -> "Completion contract met?";
|
|
97
|
+
"Completion contract met?" -> "Work the steps in order: TDD, run every verification, read every output" [label="no - finish the task"];
|
|
98
|
+
"Completion contract met?" -> "task-done: run tests, ledger the result; mark todo complete" [label="yes"];
|
|
99
|
+
"task-done: run tests, ledger the result; mark todo complete" -> "More tasks remain?";
|
|
100
|
+
"More tasks remain?" -> "task-start: brief + BASE; read the brief" [label="yes"];
|
|
101
|
+
"More tasks remain?" -> "Final whole-branch review (fresh reviewer if you have one)" [label="no"];
|
|
102
|
+
"Final whole-branch review (fresh reviewer if you have one)" -> "Re-grade, then: Critical/Important → ONE fix pass, each fix RED→GREEN + green suite; Minor → ledger";
|
|
103
|
+
"Re-grade, then: Critical/Important → ONE fix pass, each fix RED→GREEN + green suite; Minor → ledger" -> "Final review clean: delete this plan's workspace";
|
|
104
|
+
"Final review clean: delete this plan's workspace" -> "Use superpowers:finishing-a-development-branch";
|
|
105
|
+
}
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
## Setup
|
|
109
|
+
|
|
110
|
+
Ensure the work happens in an isolated workspace: use
|
|
111
|
+
superpowers:using-git-worktrees to create one or verify the existing one.
|
|
112
|
+
Never start implementation on a main/master branch without your human
|
|
113
|
+
partner's explicit consent.
|
|
114
|
+
|
|
115
|
+
Conversation memory does not survive compaction. An inline executor that
|
|
116
|
+
loses its place re-implements tasks whose commits already exist — the same
|
|
117
|
+
failure as a controller re-dispatching them, paid for in your own context.
|
|
118
|
+
Track progress in a ledger file, not only in todos. Harness todos are a
|
|
119
|
+
live view; the ledger is the record.
|
|
120
|
+
|
|
121
|
+
The workspace and ledger are shared with superpowers:subagent-driven-development
|
|
122
|
+
— same directory, same format — so a plan can change executors mid-flight
|
|
123
|
+
and the new one resumes from the same ledger.
|
|
124
|
+
|
|
125
|
+
- Each plan owns a workspace: at skill start, run
|
|
126
|
+
`bash ../subagent-driven-development/scripts/sdd-workspace PLAN_FILE`
|
|
127
|
+
(or `../subagent-driven-development/scripts/sdd-workspace.ps1 PLAN_FILE`
|
|
128
|
+
on Windows PowerShell) — it
|
|
129
|
+
prints the plan's git-ignored directory
|
|
130
|
+
(`<repo-root>/.superpowers/sdd/<plan-basename>/`), home to every
|
|
131
|
+
artifact for THIS plan: ledger, briefs, review packages. Another plan's
|
|
132
|
+
directory is never yours to read or write.
|
|
133
|
+
- Check for this plan's ledger at `<workspace>/progress.md`. If its first
|
|
134
|
+
line names your plan file, tasks with a `Task <N>: complete` line are
|
|
135
|
+
DONE — do not redo them; resume at the first task without one. Their
|
|
136
|
+
commits exist in git even when your context no longer remembers making
|
|
137
|
+
them: after compaction, trust the ledger and `git log` over your own
|
|
138
|
+
recollection. A ledger whose first line names a different plan file is
|
|
139
|
+
another plan's progress: leave it and start your own, fresh.
|
|
140
|
+
- Create the ledger with its identity as the first line:
|
|
141
|
+
`# SDD ledger — plan: <plan file path>`.
|
|
142
|
+
- `git clean -fdx` will destroy the workspace (it's git-ignored scratch);
|
|
143
|
+
if that happens, recover from `git log`.
|
|
144
|
+
|
|
145
|
+
Read the plan once, note its context and Global Constraints, and create a
|
|
146
|
+
todo per task. If the plan names a Spec, read that too: the spec is the
|
|
147
|
+
authority the plan argues from, and conflicts inside the plan resolve
|
|
148
|
+
against it. A plan with no reachable spec gets a ledger note saying so —
|
|
149
|
+
rulings made without one are provisional.
|
|
150
|
+
|
|
151
|
+
**REQUIRED SUB-SKILL:** load superpowers:test-driven-development now,
|
|
152
|
+
before Task 1. It governs every step of every task below; a plan whose
|
|
153
|
+
steps already say "write the failing test first" does not exempt you
|
|
154
|
+
from reading it.
|
|
155
|
+
|
|
156
|
+
Before Task 1, scan the plan for conflicts between tasks. The plan's
|
|
157
|
+
Interfaces blocks tell you where to look: for every task that consumes
|
|
158
|
+
what an earlier task produces, one ledger row — the two tasks, what one
|
|
159
|
+
produces against what the other consumes, and what you found. Tasks that
|
|
160
|
+
share nothing get no row; a plan whose tasks share nothing gets the single
|
|
161
|
+
line `Pre-flight: no shared interfaces`. Rule on each conflict a row
|
|
162
|
+
surfaces with the spec as the binding authority, record the ruling beside
|
|
163
|
+
its row, and start Task 1. Each task's own text is checked when you read
|
|
164
|
+
its brief, not here.
|
|
165
|
+
|
|
166
|
+
## The Task Loop
|
|
167
|
+
|
|
168
|
+
Everything you print, and every tool result, stays resident in your
|
|
169
|
+
context for the rest of the session. Redirect long test output to a file
|
|
170
|
+
in the workspace and read its tail; read a brief, not the whole plan.
|
|
171
|
+
|
|
172
|
+
### 1. Take the task
|
|
173
|
+
|
|
174
|
+
- Run this skill's `bash scripts/task-start PLAN_FILE N` (or
|
|
175
|
+
`scripts/task-start.ps1 PLAN_FILE N` on Windows PowerShell). It prints
|
|
176
|
+
the brief path and BASE (the commit the task's range is cut from) in
|
|
177
|
+
one call.
|
|
178
|
+
Read the brief for every task, including ones you remember from setup:
|
|
179
|
+
what you remember is a summary, the brief has the exact values,
|
|
180
|
+
signatures, and test cases.
|
|
181
|
+
- Mark the task's todo in_progress.
|
|
182
|
+
|
|
183
|
+
Every tool call is a turn that re-reads your whole context. Bookkeeping
|
|
184
|
+
rides along with work — a ledger append in the same call as the commit,
|
|
185
|
+
never in a call of its own.
|
|
186
|
+
|
|
187
|
+
### 2. Work the steps
|
|
188
|
+
|
|
189
|
+
The plan's steps are already in RED-GREEN order; follow them in that
|
|
190
|
+
order under superpowers:test-driven-development, loaded at setup. A test
|
|
191
|
+
step's code is written first and run first. Watching it fail is a step,
|
|
192
|
+
not a formality — a test that passes before the implementation exists is
|
|
193
|
+
a finding about the test.
|
|
194
|
+
|
|
195
|
+
Every step that runs a command has an `Expected:` line. Run the command,
|
|
196
|
+
read its output, and compare. Three outcomes:
|
|
197
|
+
|
|
198
|
+
- **Matches.** Next step.
|
|
199
|
+
- **The code is wrong.** Use superpowers:systematic-debugging. Find the
|
|
200
|
+
cause; never patch the symptom to make the step's output match.
|
|
201
|
+
- **The plan is wrong** — a step contradicts the spec, an interface from an
|
|
202
|
+
earlier task doesn't match what this task consumes, a command that
|
|
203
|
+
cannot work. Rule on the smallest change that satisfies the spec, ledger
|
|
204
|
+
it as `Task <N>: Ruling: <finding> — <what you decided and why>`, and
|
|
205
|
+
continue. The ruling is carried, not remembered: later tasks that touch
|
|
206
|
+
the same interface read it from the ledger.
|
|
207
|
+
|
|
208
|
+
Commit as the plan's commit steps say. A task that spans several commits
|
|
209
|
+
is fine; BASE is what the review range is cut from, never `HEAD~1`.
|
|
210
|
+
Commits stay local — no push/pull/fetch unless the plan or your human partner says so. At each task boundary glance at `git status -sb`: a branch tracking or ahead of a shared branch (main/dev/production) is a stop-and-fix, not a footnote.
|
|
211
|
+
Never rewrite a shared branch (`git push --force*` to main/dev/production) to undo a mistake — a forward-only `git revert` is the only remedy you apply yourself; anything more is your human partner's call.
|
|
212
|
+
|
|
213
|
+
### 3. The completion contract
|
|
214
|
+
|
|
215
|
+
Before a task's ledger line, all of the following are true, with evidence
|
|
216
|
+
in this session — not inferred from the diff looking right:
|
|
217
|
+
|
|
218
|
+
- Every test the brief names exists and ran in this task, and you read
|
|
219
|
+
the output.
|
|
220
|
+
- The final test run for the task passed — `task-done` is that run, and
|
|
221
|
+
it writes the command and result into the ledger line.
|
|
222
|
+
- Every `Expected:` line in the brief was compared against real output.
|
|
223
|
+
- Every deviation from the brief has a `Ruling:` line in the ledger.
|
|
224
|
+
|
|
225
|
+
**REQUIRED SUB-SKILL:** superpowers:verification-before-completion governs
|
|
226
|
+
the claim. If any item is missing, the task is not complete: finish it.
|
|
227
|
+
|
|
228
|
+
### 4. Complete the task
|
|
229
|
+
|
|
230
|
+
Run this skill's `bash scripts/task-done PLAN_FILE N BASE -- <test command>`
|
|
231
|
+
(or `scripts/task-done.ps1 PLAN_FILE N BASE -- <test command>` on Windows
|
|
232
|
+
PowerShell) with the test command the brief names for the whole task. It runs the
|
|
233
|
+
tests, keeps the full output in the workspace, prints the tail, and — only
|
|
234
|
+
if they pass — appends the completion line to the ledger:
|
|
235
|
+
|
|
236
|
+
`Task <N>: complete (commits <base7>..<head7>, tests: <command> → <result>)`
|
|
237
|
+
|
|
238
|
+
A failing run records nothing; the task is not complete. When it records,
|
|
239
|
+
mark as completed — the todo **and** the plan file: edit it and flip this
|
|
240
|
+
task's steps from `- [ ]` to `- [x]`, then take the next task. The plan is
|
|
241
|
+
the artifact a human reads to see where things stand, and nothing else in
|
|
242
|
+
this skill writes back to it. Leave a step unticked only when its
|
|
243
|
+
deliverable does not exist yet; never tick one you skipped.
|
|
244
|
+
|
|
245
|
+
## Final Review
|
|
246
|
+
|
|
247
|
+
Run `bash ../subagent-driven-development/scripts/review-package PLAN_FILE MERGE_BASE HEAD`
|
|
248
|
+
(or `../subagent-driven-development/scripts/review-package.ps1 PLAN_FILE MERGE_BASE HEAD`
|
|
249
|
+
on Windows PowerShell; MERGE_BASE = the commit the branch started from, e.g.
|
|
250
|
+
`git merge-base main HEAD`) and review from the file it prints.
|
|
251
|
+
|
|
252
|
+
**With a subagent tool:** dispatch the reviewer on the most capable
|
|
253
|
+
available model — the whole-branch review is a judgment task — using
|
|
254
|
+
superpowers:requesting-code-review's
|
|
255
|
+
[code-reviewer.md](../requesting-code-review/code-reviewer.md), with the
|
|
256
|
+
package path, the plan and spec paths, the plan's Review Focus section
|
|
257
|
+
verbatim if it has one (the five input classes and failure modes most
|
|
258
|
+
likely to bite — each has its test pinned in the owning task; the
|
|
259
|
+
reviewer checks each deliberately), and a
|
|
260
|
+
pointer to the ledger's `Ruling:` lines so it can weigh the calls you
|
|
261
|
+
made. Specify the model
|
|
262
|
+
explicitly; an omitted model inherits the session's, which may not be the
|
|
263
|
+
most capable. This is the one fresh context the whole run buys. Do not
|
|
264
|
+
skip it, and do not replace it with your own read of the diff.
|
|
265
|
+
|
|
266
|
+
**Without a subagent tool:** read code-reviewer.md and perform that review
|
|
267
|
+
yourself against the package, as a separate pass after the last task's
|
|
268
|
+
ledger line. Write `Final review: self-review (no subagent tool)` to the
|
|
269
|
+
ledger, and say so in your final message: a self-review by the author is
|
|
270
|
+
weaker than a fresh reviewer, and your human partner decides whether that
|
|
271
|
+
is enough before merge.
|
|
272
|
+
|
|
273
|
+
Sort the findings before you act on any of them. The reviewer's severity
|
|
274
|
+
labels are advice; the gate is yours. Its "Declined to judge" list is
|
|
275
|
+
yours too: every line there is a ruling you make and ledger, exactly like
|
|
276
|
+
a plan conflict — `Final: Ruling: <behavior the reviewer set aside> —
|
|
277
|
+
<what a reasonable person using this software gets, and why that stands
|
|
278
|
+
or why it is now a finding> — <cost if wrong>`. Re-grade first, by effect: the
|
|
279
|
+
spec is a vision document, and a finding's grade is what a reasonable
|
|
280
|
+
person using this software gets if it ships, not whether the spec names
|
|
281
|
+
the input that triggers it — a reviewer who set a finding at Minor
|
|
282
|
+
because the spec was silent has graded the spec, not the effect. Then:
|
|
283
|
+
|
|
284
|
+
- **Critical and Important** enter the fix pass.
|
|
285
|
+
- **Minor** goes to the ledger as `Final: minor (deferred): <one-liner>`
|
|
286
|
+
and to your final message under "Deferred minors". Minors never enter
|
|
287
|
+
the fix pass, and never become rulings — a ruling is a decision about a
|
|
288
|
+
conflict, not a note that you declined a polish suggestion.
|
|
289
|
+
|
|
290
|
+
Fix the Critical and Important findings yourself — you are the
|
|
291
|
+
implementer here — in ONE pass. Each fix is verified by TDD, not by a
|
|
292
|
+
second reviewer: write the test that reproduces the finding, watch it
|
|
293
|
+
fail, make it pass, then run the whole suite. Record each in the ledger as
|
|
294
|
+
`Final: fixed <finding> — <test name> RED→GREEN, suite <N>/<N>`. A fix
|
|
295
|
+
without a test that failed first is not verified; a suite that is not
|
|
296
|
+
green after the pass means the pass is not over. Do not dispatch a
|
|
297
|
+
re-review: it would re-read a diff whose covering tests already answer
|
|
298
|
+
"addressed" and whose suite run already answers "broke nothing".
|
|
299
|
+
|
|
300
|
+
A finding you decide not to fix is a ruling — `Final: Ruling: <finding> —
|
|
301
|
+
<why the code stands> — <cost if wrong>` — and reaches your human partner
|
|
302
|
+
in the rulings list. There is no second fix pass.
|
|
303
|
+
|
|
304
|
+
## Finish
|
|
305
|
+
|
|
306
|
+
Before you delete anything, collect every ledger line containing
|
|
307
|
+
`Ruling:` into your final message under "Rulings I made", in the order you
|
|
308
|
+
made them, each with what it costs if wrong, and every `minor (deferred)`
|
|
309
|
+
or `parked` line under "Deferred minors". Both lists are exhaustive.
|
|
310
|
+
|
|
311
|
+
Rulings are not the only content that dies with the workspace. Findings
|
|
312
|
+
you chose not to fix are the record of what was *not* done — git history
|
|
313
|
+
cannot carry them, because git records what was done. So export them
|
|
314
|
+
before anything is deleted: grep the progress ledger for its finding tags
|
|
315
|
+
|
|
316
|
+
```bash
|
|
317
|
+
grep -E 'Ruling:|^(Task [0-9]+: |Final: )?(minor \(deferred\)|parked)' <workspace>/progress.md
|
|
318
|
+
```
|
|
319
|
+
|
|
320
|
+
and carry every matching line, verbatim, into a durable,
|
|
321
|
+
human-reachable artifact. The chat roll-up does not satisfy this —
|
|
322
|
+
scrollback, compaction, and session end all eat it. Where the artifact
|
|
323
|
+
lives depends on how finishing-a-development-branch resolves (it carries
|
|
324
|
+
the same obligation from its side):
|
|
325
|
+
|
|
326
|
+
- **Option 2 (push and create PR):** append the lines to the PR description
|
|
327
|
+
under a "Deferred items" checklist. That is where your human partner
|
|
328
|
+
reviews, and checkboxes survive the merge.
|
|
329
|
+
- **Option 1 (merge locally):** write them to
|
|
330
|
+
`docs/superpowers/follow-ups/<plan-basename>.md` — append under a dated
|
|
331
|
+
heading if a re-run under the same basename already created the file — and
|
|
332
|
+
commit the file to the branch before the merge so it survives the branch
|
|
333
|
+
deletion.
|
|
334
|
+
- **Option 3 (keep as-is):** the workspace stays, so there is nothing to
|
|
335
|
+
export.
|
|
336
|
+
- **Explicit discard:** your human partner asked to throw the work away, so
|
|
337
|
+
no export is required.
|
|
338
|
+
|
|
339
|
+
Use superpowers:finishing-a-development-branch.
|
|
340
|
+
|
|
341
|
+
When the finish path has resolved and its export exists — the checklist in
|
|
342
|
+
the PR description, or the committed follow-ups file — delete this plan's
|
|
343
|
+
workspace directory, provided the final review was clean and its fixes are
|
|
344
|
+
committed. The export, not git history, is the record of the deferred
|
|
345
|
+
findings. Sibling directories belong to other plans; leave them alone. On
|
|
346
|
+
an explicit discard there is no export, but the workspace is deleted
|
|
347
|
+
together with the branch — a ledger describing thrown-away work is a
|
|
348
|
+
false record.
|
|
349
|
+
|
|
350
|
+
## Common Rationalizations
|
|
351
|
+
|
|
352
|
+
| Excuse | Reality |
|
|
353
|
+
|--------|---------|
|
|
354
|
+
| "I remember what Task N says" | You remember a summary. The brief has the exact values. Read it. |
|
|
355
|
+
| "The plan's code is right, skip watching the test fail" | A test you never saw fail proves nothing. It is one step. Run it. |
|
|
356
|
+
| "I'll run the full suite at the end instead of per step" | Per-step runs are how you learn which step broke it. The end-of-task run is the contract, not a substitute. |
|
|
357
|
+
| "The plan is wrong here, I'll just do the right thing" | Do the right thing and ledger the ruling. Unledgered deviation is a decision made in secret. |
|
|
358
|
+
| "I'll write the ledger lines after a few tasks" | Compaction does not wait for a convenient moment. One line per task, in the same message as the commit. |
|
|
359
|
+
| "Let me check in before the next task" | They chose inline to spend less. Progress prompts spend their time instead. Only the four stops stop you. |
|
|
360
|
+
| "I read my own diff carefully; the final reviewer is redundant" | Same author, same blind spots. The reviewer is the only fresh context this run buys. |
|
|
361
|
+
| "Tests should pass, the change was trivial" | "Should" is not evidence. The contract requires the command and its output. |
|
|
362
|
+
| "Subagents are slow and expensive, I'll skip the final review too" | Inline already removed the per-task reviewers. One review of the whole branch is the floor, not the ceiling. |
|
|
363
|
+
| "The reviewer said Minor, so it's Minor" | The label graded the spec's silence. Grade what the person gets. Re-grade, then gate. |
|
|
364
|
+
| "The fix is obvious, no need for a failing test first" | The failing test is the only proof the finding was real and is now gone. Without it you have a diff and a hope. |
|
|
365
|
+
| "I'll fix the minors too while I'm in there" | Every minor you fix is a test, a fix, and a suite run your partner did not ask for. Ledger them; your partner decides. |
|
|
366
|
+
|
|
367
|
+
## Example Workflow
|
|
368
|
+
|
|
369
|
+
```
|
|
370
|
+
You: I'm using the executing-plans skill to implement this plan inline.
|
|
371
|
+
|
|
372
|
+
[Setup: worktree verified]
|
|
373
|
+
[Read plan once: docs/superpowers/plans/feature-plan.md; spec read]
|
|
374
|
+
[Resolve workspace: bash ../subagent-driven-development/scripts/sdd-workspace docs/superpowers/plans/feature-plan.md — no ledger inside, fresh start]
|
|
375
|
+
[Pre-flight scan: 2 shared-interface rows, clean; written to ledger]
|
|
376
|
+
[Create todos for all tasks]
|
|
377
|
+
|
|
378
|
+
Task 1: Hook installation script
|
|
379
|
+
|
|
380
|
+
[bash scripts/task-start plan 1 → brief read; BASE a1b2c3d]
|
|
381
|
+
[Step 1: write failing test — written]
|
|
382
|
+
[Step 2: run it — FAIL: install_hook not defined. Matches Expected.]
|
|
383
|
+
[Step 3: implement — written]
|
|
384
|
+
[Step 4: run it — PASS 1/1. Matches Expected.]
|
|
385
|
+
[Step 5: commit — d4e5f6a]
|
|
386
|
+
[Contract: tests ran, output read, no deviations]
|
|
387
|
+
[bash scripts/task-done plan 1 a1b2c3d -- npm test -- hooks → ledger: Task 1: complete (commits a1b2c3d..d4e5f6a, tests: npm test -- hooks → 1/1 pass)]
|
|
388
|
+
|
|
389
|
+
Task 2: Recovery modes
|
|
390
|
+
|
|
391
|
+
[bash scripts/task-start plan 2 → brief read; BASE d4e5f6a]
|
|
392
|
+
[Step 2: run failing test — FAIL, but on an import error: Task 1 exported
|
|
393
|
+
installHook, brief consumes install_hook]
|
|
394
|
+
[Ruling: brief's consumer name is a typo against Task 1's Produces block;
|
|
395
|
+
use installHook — Ledger: Task 2: Ruling: install_hook → installHook — matches Task 1 Produces — cost if wrong: one rename]
|
|
396
|
+
[Steps 2-5 as planned; commit b7c8d9e]
|
|
397
|
+
[bash scripts/task-done plan 2 d4e5f6a -- npm test -- recovery → ledger: Task 2: complete (commits d4e5f6a..b7c8d9e, tests: npm test -- recovery → 8/8 pass)]
|
|
398
|
+
|
|
399
|
+
...
|
|
400
|
+
|
|
401
|
+
[After all tasks: bash ../subagent-driven-development/scripts/review-package plan MERGE_BASE HEAD; dispatch code-reviewer, most capable model]
|
|
402
|
+
Reviewer: One Important finding — progress reporting interval hardcoded. Two Minor.
|
|
403
|
+
[Re-grade: Important stands; minors → ledger as deferred]
|
|
404
|
+
[Fix pass: test_progress_interval_configurable RED → extract PROGRESS_INTERVAL → GREEN; suite 12/12; commit]
|
|
405
|
+
[Ledger: Final: fixed hardcoded interval — test_progress_interval_configurable RED→GREEN, suite 12/12]
|
|
406
|
+
|
|
407
|
+
Rulings I made:
|
|
408
|
+
- Task 2: install_hook → installHook (brief typo; cost if wrong: one rename)
|
|
409
|
+
|
|
410
|
+
Deferred minors:
|
|
411
|
+
- README lacks a usage example
|
|
412
|
+
- recovery.js could split verify/repair into two files
|
|
413
|
+
|
|
414
|
+
[Use superpowers:finishing-a-development-branch — Option 2: push and create PR]
|
|
415
|
+
[Export deferred findings (Ruling:, minor (deferred), parked lines) to the PR description as a "Deferred items" checklist]
|
|
416
|
+
[Delete this plan's workspace — the export, not git, is the deferred findings' record]
|
|
417
|
+
```
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Close one task of an inline plan execution in a single call: run the task's
|
|
3
|
+
# test command, keep its full output in the workspace, print the tail, and —
|
|
4
|
+
# only if the command succeeded — append the completion line to the ledger.
|
|
5
|
+
# A failing command records nothing: the task is not complete.
|
|
6
|
+
#
|
|
7
|
+
# Usage: task-done PLAN_FILE TASK_NUMBER BASE -- TEST_COMMAND [ARGS...]
|
|
8
|
+
# BASE is the SHA task-start printed; the completion line records BASE..HEAD.
|
|
9
|
+
# Exit: the test command's exit status.
|
|
10
|
+
set -euo pipefail
|
|
11
|
+
|
|
12
|
+
if [ $# -lt 5 ] || [ "$4" != "--" ]; then
|
|
13
|
+
echo "usage: task-done PLAN_FILE TASK_NUMBER BASE -- TEST_COMMAND [ARGS...]" >&2
|
|
14
|
+
exit 2
|
|
15
|
+
fi
|
|
16
|
+
|
|
17
|
+
plan=$1
|
|
18
|
+
n=$2
|
|
19
|
+
base=$3
|
|
20
|
+
shift 4
|
|
21
|
+
sdd="$(cd "$(dirname "$0")/../../subagent-driven-development/scripts" && pwd)"
|
|
22
|
+
|
|
23
|
+
git rev-parse --verify --quiet "$base" >/dev/null || { echo "bad BASE: $base" >&2; exit 2; }
|
|
24
|
+
|
|
25
|
+
# Invoke via bash rather than direct exec: some extractors (Python zipfile)
|
|
26
|
+
# strip Unix exec bits when unpacking marketplace packages (#2040).
|
|
27
|
+
dir=$("${BASH:-bash}" "$sdd/sdd-workspace" "$plan")
|
|
28
|
+
log="$dir/task-${n}-tests.log"
|
|
29
|
+
ledger="$dir/progress.md"
|
|
30
|
+
|
|
31
|
+
# Render the command the way a person would type it, for the ledger line.
|
|
32
|
+
cmd=""
|
|
33
|
+
for a in "$@"; do
|
|
34
|
+
case "$a" in
|
|
35
|
+
*[[:space:]\"\;\|\&]*) cmd="$cmd '$a'" ;;
|
|
36
|
+
*) cmd="$cmd $a" ;;
|
|
37
|
+
esac
|
|
38
|
+
done
|
|
39
|
+
cmd=${cmd# }
|
|
40
|
+
|
|
41
|
+
rc=0
|
|
42
|
+
"$@" > "$log" 2>&1 || rc=$?
|
|
43
|
+
|
|
44
|
+
tail -n 5 "$log"
|
|
45
|
+
if [ "$rc" -ne 0 ]; then
|
|
46
|
+
echo "task-done: test command exited $rc; Task $n NOT recorded (full output: $log)" >&2
|
|
47
|
+
exit "$rc"
|
|
48
|
+
fi
|
|
49
|
+
|
|
50
|
+
last=$(grep -v '^[[:space:]]*$' "$log" | tail -n 1 || true)
|
|
51
|
+
[ -n "$last" ] || last="(no output)"
|
|
52
|
+
[ -f "$ledger" ] || printf '# SDD ledger — plan: %s\n' "$plan" > "$ledger"
|
|
53
|
+
line="Task $n: complete (commits $(git rev-parse --short=7 "$base")..$(git rev-parse --short=7 HEAD), tests: $cmd → $last)"
|
|
54
|
+
printf '%s\n' "$line" >> "$ledger"
|
|
55
|
+
echo "ledger: $line"
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
#!/usr/bin/env pwsh
|
|
2
|
+
# Close one task of an inline plan execution in a single call: run the task's
|
|
3
|
+
# test command, keep its full output in the workspace, print the tail, and —
|
|
4
|
+
# only if the command succeeded — append the completion line to the ledger.
|
|
5
|
+
# A failing command records nothing: the task is not complete.
|
|
6
|
+
#
|
|
7
|
+
# Usage: ./task-done.ps1 PLAN_FILE TASK_NUMBER BASE -- TEST_COMMAND [ARGS...]
|
|
8
|
+
# BASE is the SHA task-start printed; the completion line records BASE..HEAD.
|
|
9
|
+
# Exit: the test command's exit status.
|
|
10
|
+
|
|
11
|
+
$ErrorActionPreference = "Stop"
|
|
12
|
+
|
|
13
|
+
if ($args.Count -lt 4) {
|
|
14
|
+
[Console]::Error.WriteLine("usage: task-done.ps1 PLAN_FILE TASK_NUMBER BASE -- TEST_COMMAND [ARGS...]")
|
|
15
|
+
exit 2
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
$plan = $args[0]
|
|
19
|
+
$n = $args[1]
|
|
20
|
+
$base = $args[2]
|
|
21
|
+
# PowerShell consumes an unquoted `--` as its own end-of-parameters marker,
|
|
22
|
+
# so the documented `... BASE -- CMD` arrives here WITHOUT the separator; a
|
|
23
|
+
# quoted "--" survives and is skipped below. Unlike the sh twin, this port
|
|
24
|
+
# therefore cannot reject a call that genuinely omits the separator.
|
|
25
|
+
$cmdArgs = @($args[3..($args.Count - 1)])
|
|
26
|
+
if ($cmdArgs[0] -eq "--") {
|
|
27
|
+
if ($cmdArgs.Count -lt 2) {
|
|
28
|
+
[Console]::Error.WriteLine("usage: task-done.ps1 PLAN_FILE TASK_NUMBER BASE -- TEST_COMMAND [ARGS...]")
|
|
29
|
+
exit 2
|
|
30
|
+
}
|
|
31
|
+
$cmdArgs = @($cmdArgs[1..($cmdArgs.Count - 1)])
|
|
32
|
+
}
|
|
33
|
+
$sdd = Join-Path $PSScriptRoot "../../subagent-driven-development/scripts"
|
|
34
|
+
|
|
35
|
+
& git rev-parse --verify --quiet $base >$null 2>&1
|
|
36
|
+
if ($LASTEXITCODE -ne 0) {
|
|
37
|
+
[Console]::Error.WriteLine("bad BASE: $base")
|
|
38
|
+
exit 2
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
$dir = (& (Join-Path $sdd "sdd-workspace.ps1") $plan | Select-Object -First 1).Trim()
|
|
42
|
+
$log = Join-Path $dir "task-$n-tests.log"
|
|
43
|
+
$ledger = Join-Path $dir "progress.md"
|
|
44
|
+
|
|
45
|
+
# Render the command the way a person would type it, for the ledger line.
|
|
46
|
+
$parts = foreach ($a in [string[]]$cmdArgs) {
|
|
47
|
+
if ($a -match '[\s";|&]') { "'$a'" } else { $a }
|
|
48
|
+
}
|
|
49
|
+
$cmd = $parts -join ' '
|
|
50
|
+
|
|
51
|
+
$rc = 0
|
|
52
|
+
try {
|
|
53
|
+
$exe = $cmdArgs[0]
|
|
54
|
+
$rest = @()
|
|
55
|
+
if ($cmdArgs.Count -gt 1) { $rest = $cmdArgs[1..($cmdArgs.Count - 1)] }
|
|
56
|
+
& $exe @rest > $log 2>&1
|
|
57
|
+
$rc = $LASTEXITCODE
|
|
58
|
+
} catch {
|
|
59
|
+
$rc = 127
|
|
60
|
+
if (-not (Test-Path -LiteralPath $log -PathType Leaf)) {
|
|
61
|
+
[System.IO.File]::WriteAllText($log, ($_.Exception.Message + [Environment]::NewLine), [System.Text.UTF8Encoding]::new($false))
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
if (Test-Path -LiteralPath $log -PathType Leaf) {
|
|
66
|
+
Get-Content -LiteralPath $log -Tail 5
|
|
67
|
+
}
|
|
68
|
+
if ($rc -ne 0) {
|
|
69
|
+
[Console]::Error.WriteLine("task-done: test command exited $rc; Task $n NOT recorded (full output: $log)")
|
|
70
|
+
exit $rc
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
$last = Get-Content -LiteralPath $log | Where-Object { $_ -match '\S' } | Select-Object -Last 1
|
|
74
|
+
if ([string]::IsNullOrWhiteSpace($last)) { $last = "(no output)" }
|
|
75
|
+
$utf8NoBom = [System.Text.UTF8Encoding]::new($false)
|
|
76
|
+
if (-not (Test-Path -LiteralPath $ledger -PathType Leaf)) {
|
|
77
|
+
[System.IO.File]::WriteAllText($ledger, "# SDD ledger — plan: $plan" + [Environment]::NewLine, $utf8NoBom)
|
|
78
|
+
}
|
|
79
|
+
$base7 = (& git rev-parse --short=7 $base).Trim()
|
|
80
|
+
$head7 = (& git rev-parse --short=7 HEAD).Trim()
|
|
81
|
+
$line = "Task ${n}: complete (commits $base7..$head7, tests: $cmd → $last)"
|
|
82
|
+
[System.IO.File]::AppendAllText($ledger, $line + [Environment]::NewLine, $utf8NoBom)
|
|
83
|
+
Write-Output "ledger: $line"
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Begin one task of an inline plan execution in a single call: extract the
|
|
3
|
+
# task's brief (via subagent-driven-development's task-brief, so both skills
|
|
4
|
+
# share one workspace) and record BASE, the commit the task's review range is
|
|
5
|
+
# cut from. One tool call instead of two, because every call in an inline
|
|
6
|
+
# session is a turn that re-reads the whole context.
|
|
7
|
+
#
|
|
8
|
+
# Usage: task-start PLAN_FILE TASK_NUMBER
|
|
9
|
+
# Prints:
|
|
10
|
+
# brief: <path to the task's brief file>
|
|
11
|
+
# base: <full SHA of HEAD>
|
|
12
|
+
set -euo pipefail
|
|
13
|
+
|
|
14
|
+
if [ $# -ne 2 ]; then
|
|
15
|
+
echo "usage: task-start PLAN_FILE TASK_NUMBER" >&2
|
|
16
|
+
exit 2
|
|
17
|
+
fi
|
|
18
|
+
|
|
19
|
+
plan=$1
|
|
20
|
+
n=$2
|
|
21
|
+
sdd="$(cd "$(dirname "$0")/../../subagent-driven-development/scripts" && pwd)"
|
|
22
|
+
|
|
23
|
+
# Invoke via bash rather than direct exec: some extractors (Python zipfile)
|
|
24
|
+
# strip Unix exec bits when unpacking marketplace packages (#2040).
|
|
25
|
+
out=$("${BASH:-bash}" "$sdd/task-brief" "$plan" "$n")
|
|
26
|
+
brief=$(printf '%s\n' "$out" | sed -n 's/^wrote \(.*\): [0-9][0-9]* lines$/\1/p')
|
|
27
|
+
[ -n "$brief" ] || { echo "task-brief did not report a path: $out" >&2; exit 1; }
|
|
28
|
+
|
|
29
|
+
echo "brief: $brief"
|
|
30
|
+
echo "base: $(git rev-parse HEAD)"
|