cohorte 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +151 -0
- package/README.md +43 -12
- package/bin/cli.js +13 -2
- package/core/agents/review.md +23 -0
- package/core/commands/audit.md +9 -1
- package/core/commands/brainstorm.md +6 -0
- package/core/commands/build.md +93 -8
- package/core/commands/doctor.md +13 -8
- package/core/commands/drive.md +80 -0
- package/core/commands/fix.md +10 -5
- package/core/commands/review.md +94 -12
- package/core/commands/spec.md +20 -0
- package/core/commands/update-pipeline.md +6 -1
- package/core/hooks/gate.py +4 -4
- package/core/templates/decisions.template.md +42 -0
- package/core/templates/spec.template.md +4 -2
- package/core/templates/steps/init-pipeline/02-interview-gaps.md +1 -1
- package/core/templates/steps/init-pipeline/04-write-render.md +8 -4
- package/core/workflows/review.js +44 -2
- package/dashboard/dist/assets/{index-AFQnlfjO.css → index-BZ_LQlEj.css} +1 -1
- package/dashboard/dist/assets/index-DYyn4p93.js +43 -0
- package/dashboard/dist/index.html +2 -2
- package/dashboard/server/doctor.js +13 -3
- package/dashboard/server/index.js +7 -0
- package/dashboard/server/metrics.js +4 -4
- package/dashboard/server/usage.js +61 -0
- package/install.ps1 +12 -1
- package/install.sh +12 -2
- package/package.json +1 -1
- package/profile/PIPELINE.template.md +3 -3
- package/profile/SCHEMA.md +150 -12
- package/scripts/loop.sh +318 -0
- package/scripts/metrics/collect.mjs +11 -2
- package/scripts/preflight.sh +2 -2
- package/scripts/telemetry-send.sh +5 -2
- package/scripts/test-dashboard.mjs +22 -2
- package/scripts/test-gate.mjs +1 -2
- package/scripts/test-loop.mjs +227 -0
- package/scripts/test-metrics.mjs +12 -3
- package/scripts/test-workflows.mjs +28 -0
- package/scripts/validate-core.mjs +21 -6
- package/core/agents/smoke.md +0 -63
- package/core/commands/smoke.md +0 -55
- package/dashboard/dist/assets/index-DLBzciIC.js +0 -43
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Behavioural tests for scripts/loop.sh — the autonomous /review ⇄ /fix driver.
|
|
3
|
+
//
|
|
4
|
+
// The driver is pure shell around two JSON files it does not write, so it is
|
|
5
|
+
// testable end-to-end by putting a FAKE `claude` on PATH that produces those files
|
|
6
|
+
// per phase. What is pinned here cannot be seen by any structural check:
|
|
7
|
+
//
|
|
8
|
+
// · exit 4 — /build's readiness gate said NOT-READY, so no pass count helps
|
|
9
|
+
// · exit 0/3 leave the right TERMINAL status in the spec's front-matter, which is
|
|
10
|
+
// what makes an interrupted loop resumable (SCHEMA.md §Spec status)
|
|
11
|
+
// · the front-matter stamps are written with awk on every platform — a `sed -i`
|
|
12
|
+
// would pass on GNU and corrupt every spec on BSD/macOS
|
|
13
|
+
// · --resume continues at the recorded pass instead of re-paying passes 1..n-1
|
|
14
|
+
// · a spec with no front-matter still runs (the stamps are a silent no-op)
|
|
15
|
+
//
|
|
16
|
+
// node scripts/test-loop.mjs
|
|
17
|
+
|
|
18
|
+
import { mkdtempSync, writeFileSync, readFileSync, mkdirSync, chmodSync, existsSync } from "node:fs";
|
|
19
|
+
import { execFileSync, spawnSync } from "node:child_process";
|
|
20
|
+
import { join } from "node:path";
|
|
21
|
+
import { tmpdir } from "node:os";
|
|
22
|
+
import { fileURLToPath } from "node:url";
|
|
23
|
+
|
|
24
|
+
const root = fileURLToPath(new URL("..", import.meta.url));
|
|
25
|
+
const LOOP = join(root, "scripts/loop.sh");
|
|
26
|
+
|
|
27
|
+
let failures = 0;
|
|
28
|
+
const check = (name, cond, detail = "") => {
|
|
29
|
+
if (cond) console.log(` ✓ ${name}`);
|
|
30
|
+
else { failures++; console.error(` ✗ ${name}${detail ? ` — ${detail}` : ""}`); }
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
const FM = `---
|
|
34
|
+
feature_id: feat-x
|
|
35
|
+
title: Feat X
|
|
36
|
+
status: frozen # draft → frozen → in-progress → in-review → shipped · blocked
|
|
37
|
+
branch: feature/feat-x
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
# Feat X
|
|
41
|
+
`;
|
|
42
|
+
|
|
43
|
+
// A fake `claude`: reads the phase out of the `-p "/<cmd> <id>"` argument and
|
|
44
|
+
// writes whatever the scenario says that phase produces. `$PHASES` is a
|
|
45
|
+
// newline-separated script of `<cmd>:<what to write>` steps, consumed in order,
|
|
46
|
+
// so a scenario can make pass 1 and pass 2 differ.
|
|
47
|
+
const FAKE_CLAUDE = `#!/usr/bin/env bash
|
|
48
|
+
set -u
|
|
49
|
+
prompt=""
|
|
50
|
+
while [ $# -gt 0 ]; do
|
|
51
|
+
case "$1" in -p) prompt="$2"; shift 2 ;; *) shift ;; esac
|
|
52
|
+
done
|
|
53
|
+
cmd="\${prompt%% *}"; cmd="\${cmd#/}"
|
|
54
|
+
n=0; [ -f "$SCEN_DIR/count" ] && n=$(cat "$SCEN_DIR/count")
|
|
55
|
+
n=$((n + 1)); echo "$n" >"$SCEN_DIR/count"
|
|
56
|
+
step=$(sed -n "\${n}p" "$SCEN_DIR/phases")
|
|
57
|
+
echo "fake claude: phase=$cmd step=$step"
|
|
58
|
+
want="\${step%%:*}"; do_what="\${step#*:}"
|
|
59
|
+
[ "$want" = "$cmd" ] || { echo "fake claude: expected /$want, got /$cmd" >&2; exit 9; }
|
|
60
|
+
mkdir -p specs/reports
|
|
61
|
+
case "$do_what" in
|
|
62
|
+
notready)
|
|
63
|
+
printf '{"id":"feat-x","phase":"readiness","ts":"t","verdict":"NOT-READY","gaps":["contract|POST /o|no success shape"],"surfaces":["backend"]}' \\
|
|
64
|
+
>specs/reports/feat-x.readiness.json ;;
|
|
65
|
+
ready)
|
|
66
|
+
printf '{"id":"feat-x","phase":"readiness","ts":"t","verdict":"READY","gaps":[],"surfaces":["backend"]}' \\
|
|
67
|
+
>specs/reports/feat-x.readiness.json ;;
|
|
68
|
+
clean)
|
|
69
|
+
printf '{"id":"feat-x","phase":"review","ts":"t","verdict":"SHIP","findings":2,"blocking":0,"deferred":2,"unreviewed":[],"fingerprint":""}' \\
|
|
70
|
+
>specs/reports/feat-x.verdict.json ;;
|
|
71
|
+
deadreviewer)
|
|
72
|
+
printf '{"id":"feat-x","phase":"review","ts":"t","verdict":"REVISE","findings":0,"blocking":0,"deferred":0,"unreviewed":["backend"],"fingerprint":""}' \\
|
|
73
|
+
>specs/reports/feat-x.verdict.json ;;
|
|
74
|
+
deadimplementer)
|
|
75
|
+
printf '{"id":"feat-x","phase":"readiness","ts":"t","verdict":"READY","gaps":[],"surfaces":["backend"]}' \\
|
|
76
|
+
>specs/reports/feat-x.readiness.json
|
|
77
|
+
printf '{"id":"feat-x","phase":"build","ts":"t","surfaces":{"backend":"dead","frontend":"ok"},"dead":["backend"]}' \\
|
|
78
|
+
>specs/reports/feat-x.build.json ;;
|
|
79
|
+
blocking)
|
|
80
|
+
printf '{"id":"feat-x","phase":"review","ts":"t","verdict":"REVISE","findings":3,"blocking":2,"deferred":1,"fingerprint":"aaaa1111bbbb2222"}' \\
|
|
81
|
+
>specs/reports/feat-x.verdict.json ;;
|
|
82
|
+
noop) : ;;
|
|
83
|
+
esac
|
|
84
|
+
exit 0
|
|
85
|
+
`;
|
|
86
|
+
|
|
87
|
+
// One scratch repo per scenario: a git checkout (loop.sh cds to its toplevel), a
|
|
88
|
+
// spec, the fake claude on PATH, and the phase script it plays out.
|
|
89
|
+
function scenario(phases, { frontmatter = FM } = {}) {
|
|
90
|
+
const dir = mkdtempSync(join(tmpdir(), "cohorte-loop-"));
|
|
91
|
+
const git = (...a) => execFileSync("git", ["-C", dir, ...a], { stdio: "ignore" });
|
|
92
|
+
git("init", "-q");
|
|
93
|
+
git("config", "user.email", "t@t.t");
|
|
94
|
+
git("config", "user.name", "t");
|
|
95
|
+
mkdirSync(join(dir, "specs/reports"), { recursive: true });
|
|
96
|
+
writeFileSync(join(dir, "specs/feat-x.md"), frontmatter);
|
|
97
|
+
git("add", "-A");
|
|
98
|
+
git("commit", "-qm", "init");
|
|
99
|
+
|
|
100
|
+
const bin = join(dir, "bin");
|
|
101
|
+
mkdirSync(bin);
|
|
102
|
+
writeFileSync(join(dir, "phases"), phases.join("\n") + "\n");
|
|
103
|
+
writeFileSync(join(bin, "claude"), FAKE_CLAUDE);
|
|
104
|
+
chmodSync(join(bin, "claude"), 0o755);
|
|
105
|
+
return { dir, bin };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function runLoop({ dir, bin }, args) {
|
|
109
|
+
const r = spawnSync("bash", [LOOP, ...args], {
|
|
110
|
+
cwd: dir,
|
|
111
|
+
encoding: "utf8",
|
|
112
|
+
env: {
|
|
113
|
+
...process.env,
|
|
114
|
+
PATH: `${bin}:${process.env.PATH}`,
|
|
115
|
+
SCEN_DIR: dir,
|
|
116
|
+
CLAUDE_FLAGS: "--permission-mode acceptEdits",
|
|
117
|
+
GIT_AUTHOR_NAME: "t", GIT_AUTHOR_EMAIL: "t@t.t",
|
|
118
|
+
GIT_COMMITTER_NAME: "t", GIT_COMMITTER_EMAIL: "t@t.t",
|
|
119
|
+
},
|
|
120
|
+
});
|
|
121
|
+
const spec = readFileSync(join(dir, "specs/feat-x.md"), "utf8");
|
|
122
|
+
const fm = k => {
|
|
123
|
+
const m = spec.match(new RegExp(`^${k}:\\s*([^#\\n]*)`, "m"));
|
|
124
|
+
return m ? m[1].trim() : null;
|
|
125
|
+
};
|
|
126
|
+
return { code: r.status, out: `${r.stdout}${r.stderr}`, spec, fm };
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
console.log("loop.sh — readiness gate");
|
|
130
|
+
{
|
|
131
|
+
const s = scenario(["build:notready"]);
|
|
132
|
+
const r = runLoop(s, ["feat-x"]);
|
|
133
|
+
check("NOT-READY ⇒ exit 4, not 2", r.code === 4, `got ${r.code}: ${r.out.trim().split("\n").pop()}`);
|
|
134
|
+
check("NOT-READY ⇒ says the spec is not implementable",
|
|
135
|
+
/not implementable/i.test(r.out), r.out.trim().split("\n").pop());
|
|
136
|
+
check("NOT-READY ⇒ points at /spec", /\/spec feat-x/.test(r.out));
|
|
137
|
+
check("NOT-READY ⇒ spec left blocked", r.fm("status") === "blocked", r.fm("status"));
|
|
138
|
+
check("NOT-READY ⇒ no review ran (the gate is the point)", !/phase=review/.test(r.out));
|
|
139
|
+
check("NOT-READY ⇒ the build stamp is NOT written",
|
|
140
|
+
!existsSync(join(s.dir, "specs/reports/feat-x.built")));
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
console.log("loop.sh — clean run");
|
|
144
|
+
{
|
|
145
|
+
const s = scenario(["build:ready", "review:clean"]);
|
|
146
|
+
const r = runLoop(s, ["feat-x"]);
|
|
147
|
+
check("clean ⇒ exit 0", r.code === 0, `got ${r.code}: ${r.out}`);
|
|
148
|
+
check("clean ⇒ status in-review (ready to /ship)", r.fm("status") === "in-review", r.fm("status"));
|
|
149
|
+
check("clean ⇒ loop state cleared", r.fm("loop_pass") === "0" && r.fm("loop_phase") === "done",
|
|
150
|
+
`${r.fm("loop_pass")}/${r.fm("loop_phase")}`);
|
|
151
|
+
check("clean ⇒ the deferred count is named, not dropped",
|
|
152
|
+
/2 deferred finding\(s\) parked/.test(r.out), r.out.trim().split("\n").pop());
|
|
153
|
+
check("clean ⇒ the status comment survives the awk rewrite",
|
|
154
|
+
/^status: in-review # draft/m.test(r.spec), r.spec.split("\n")[3]);
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
console.log("loop.sh — a dead subagent is never a clean result");
|
|
158
|
+
{
|
|
159
|
+
// A dead implementer: /build finishes fine having built one surface of two. Reviewing
|
|
160
|
+
// that would spend N reviewers auditing a half-built feature and report its holes as
|
|
161
|
+
// findings to fix — the wrong diagnosis at the wrong price.
|
|
162
|
+
const s = scenario(["build:deadimplementer"]);
|
|
163
|
+
const r = runLoop(s, ["feat-x"]);
|
|
164
|
+
check("dead implementer ⇒ exit 2, not a review pass", r.code === 2, `got ${r.code}: ${r.out}`);
|
|
165
|
+
check("dead implementer ⇒ no reviewer was spawned", !/phase=review/.test(r.out));
|
|
166
|
+
check("dead implementer ⇒ names the cause", /implementer died/.test(r.out),
|
|
167
|
+
r.out.trim().split("\n").pop());
|
|
168
|
+
check("dead implementer ⇒ spec left blocked", r.fm("status") === "blocked", r.fm("status"));
|
|
169
|
+
}
|
|
170
|
+
{
|
|
171
|
+
// THE dangerous one: blocking == 0 because the only reviewer that could have found
|
|
172
|
+
// something never answered. Exiting 0 here would report "clean" about unread code and
|
|
173
|
+
// send the human to /ship.
|
|
174
|
+
const s = scenario(["build:ready", "review:deadreviewer"]);
|
|
175
|
+
const r = runLoop(s, ["feat-x"]);
|
|
176
|
+
check("dead reviewer + blocking 0 ⇒ NOT exit 0", r.code !== 0, `got ${r.code}: ${r.out}`);
|
|
177
|
+
check("dead reviewer ⇒ exit 2 (no usable verdict)", r.code === 2, `got ${r.code}`);
|
|
178
|
+
check("dead reviewer ⇒ names the unreviewed surface", /reviewer died/.test(r.out),
|
|
179
|
+
r.out.trim().split("\n").pop());
|
|
180
|
+
check("dead reviewer ⇒ never says clean", !/✓ clean/.test(r.out));
|
|
181
|
+
check("dead reviewer ⇒ spec is NOT left in-review", r.fm("status") === "blocked", r.fm("status"));
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
console.log("loop.sh — non-convergent + resume");
|
|
185
|
+
{
|
|
186
|
+
const s = scenario(["build:ready", "review:blocking", "fix:noop", "review:blocking"]);
|
|
187
|
+
const r = runLoop(s, ["feat-x", "--max=4"]);
|
|
188
|
+
check("same fingerprint twice ⇒ exit 3", r.code === 3, `got ${r.code}: ${r.out}`);
|
|
189
|
+
check("non-convergent ⇒ status blocked", r.fm("status") === "blocked", r.fm("status"));
|
|
190
|
+
check("non-convergent ⇒ the pass it reached is recorded (resume anchor)",
|
|
191
|
+
r.fm("loop_pass") === "2", r.fm("loop_pass"));
|
|
192
|
+
check("non-convergent ⇒ the phase is recorded", r.fm("loop_phase") === "review", r.fm("loop_phase"));
|
|
193
|
+
|
|
194
|
+
// Resume: the recorded pass is where it picks up — passes 1..n-1 are not re-paid.
|
|
195
|
+
writeFileSync(join(s.dir, "count"), "0");
|
|
196
|
+
writeFileSync(join(s.dir, "phases"), "review:clean\n");
|
|
197
|
+
const r2 = runLoop(s, ["feat-x", "--max=4", "--resume"]);
|
|
198
|
+
check("--resume ⇒ announces the pass it continues from",
|
|
199
|
+
/resuming at review pass 2/.test(r2.out), r2.out.trim().split("\n")[0]);
|
|
200
|
+
check("--resume ⇒ skips the build (the stamp is there)", !/phase=build/.test(r2.out));
|
|
201
|
+
check("--resume ⇒ finishes clean from there", r2.code === 0, `got ${r2.code}: ${r2.out}`);
|
|
202
|
+
check("--resume ⇒ reports the resumed pass count, not 1",
|
|
203
|
+
/after 2 review pass\(es\)/.test(r2.out), r2.out.trim().split("\n").pop());
|
|
204
|
+
}
|
|
205
|
+
{
|
|
206
|
+
const s = scenario(["review:clean"]);
|
|
207
|
+
// A resume anchor past the ceiling is a usage error, not a silent restart at 1.
|
|
208
|
+
writeFileSync(join(s.dir, "specs/feat-x.md"), FM.replace("branch:", "loop_pass: 9\nbranch:"));
|
|
209
|
+
const r = runLoop(s, ["feat-x", "--max=3", "--resume"]);
|
|
210
|
+
check("--resume past --max ⇒ exit 64 with the reason", r.code === 64 && /raise --max/.test(r.out),
|
|
211
|
+
`${r.code}: ${r.out.trim()}`);
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
console.log("loop.sh — a spec with no front-matter still runs");
|
|
215
|
+
{
|
|
216
|
+
const s = scenario(["build:ready", "review:clean"], { frontmatter: "# Feat X\n\nno front-matter\n" });
|
|
217
|
+
const r = runLoop(s, ["feat-x"]);
|
|
218
|
+
check("no front-matter ⇒ still exits 0 (stamps are a silent no-op)", r.code === 0,
|
|
219
|
+
`got ${r.code}: ${r.out}`);
|
|
220
|
+
check("no front-matter ⇒ the spec is left untouched",
|
|
221
|
+
r.spec === "# Feat X\n\nno front-matter\n", JSON.stringify(r.spec));
|
|
222
|
+
check("no front-matter ⇒ no stray temp file",
|
|
223
|
+
!existsSync(join(s.dir, "specs/feat-x.md.loop.tmp")));
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
if (failures) { console.error(`\ntest-loop: ${failures} failure(s)`); process.exit(1); }
|
|
227
|
+
console.log("\ntest-loop: OK");
|
package/scripts/test-metrics.mjs
CHANGED
|
@@ -82,6 +82,13 @@ const lines = [
|
|
|
82
82
|
// Case 8: a slash token that is not a command must not invent one.
|
|
83
83
|
user(1800, 'look at the /usr/local/share directory and report what you find there'),
|
|
84
84
|
assistant('m7', 1805, 'claude-opus-5', { input_tokens: 0, output_tokens: 10 }),
|
|
85
|
+
// Case 9: a long prompt that merely DISCUSSES a command is not an invocation of it.
|
|
86
|
+
// Without the length gate, writing about /review bills the conversation to /review —
|
|
87
|
+
// which is what happened in cohorte's own repo while the pipeline was being designed.
|
|
88
|
+
user(2400, 'I want to talk through how /review behaves when a surface has no findings at '
|
|
89
|
+
+ 'all, because the verdict logic there is what produced the false green we saw last week '
|
|
90
|
+
+ 'and I am not convinced the fix covers the case where every reviewer dies at once.'),
|
|
91
|
+
assistant('m8', 2405, 'claude-opus-5', { input_tokens: 0, output_tokens: 20 }),
|
|
85
92
|
];
|
|
86
93
|
fs.writeFileSync(path.join(projectDir, `${SESSION}.jsonl`),
|
|
87
94
|
lines.map((l) => JSON.stringify(l)).join('\n') + '\n');
|
|
@@ -107,18 +114,20 @@ const chat = out.commands.find((c) => c.command === '(chat)');
|
|
|
107
114
|
const review = out.commands.find((c) => c.command === '/review');
|
|
108
115
|
|
|
109
116
|
console.log('test-metrics');
|
|
110
|
-
check('the mid-command task-notification did not split the run', out.totals.runs,
|
|
117
|
+
check('the mid-command task-notification did not split the run', out.totals.runs, 5);
|
|
111
118
|
check('/build is one run, not three', build.runs, 1);
|
|
112
119
|
check('duplicate lines of one response are billed once', build.tokens.output, 1000 + 500 + 2000);
|
|
113
120
|
check('the <synthetic> message contributed no tokens', build.tokens.output < 999999, true);
|
|
114
121
|
check('cache-write tokens are kept on their own tier', build.tokens.cacheWrite5m, 1000);
|
|
115
122
|
check('cache-read tokens are kept on their own tier', build.tokens.cacheRead, 10000);
|
|
116
123
|
check('the subagent was attributed to the command that spawned it', build.agents.total, 1);
|
|
117
|
-
check('the second prompt is a separate (chat) run', chat.runs,
|
|
124
|
+
check('the second prompt is a separate (chat) run', chat.runs, 3);
|
|
118
125
|
check('a command named inside prose is attributed to that command', review && review.runs, 1);
|
|
119
126
|
check('a short steer continues the run instead of opening a new one', review.continuations, 1);
|
|
120
127
|
check('the continued turn counts toward the command it continued', review.tokens.output, 60 + 70);
|
|
121
|
-
check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10);
|
|
128
|
+
check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10 + 20);
|
|
129
|
+
check('a long prompt that discusses a command is not counted as running it',
|
|
130
|
+
review.runs, 1);
|
|
122
131
|
|
|
123
132
|
// opus-5 $5 in / $25 out per MTok; 5m cache write 1.25x input, cache read 0.1x input.
|
|
124
133
|
// m1 100*5 + 1000*25 + 1000*6.25 + 10000*0.5 = 36750
|
|
@@ -146,6 +146,34 @@ console.log("review.js");
|
|
|
146
146
|
]));
|
|
147
147
|
check("SHIP + only LOW ⇒ next is /ship", String(result.next).startsWith("/ship"), result.next);
|
|
148
148
|
}
|
|
149
|
+
{
|
|
150
|
+
// Deferred findings are real but out of the feature's scope: they must be
|
|
151
|
+
// counted and routed to the backlog, yet move NEITHER the verdict nor `clean`.
|
|
152
|
+
// Both halves matter — a deferred item that blocks costs a fix loop it was
|
|
153
|
+
// deferred out of, and one that is dropped is the leak the section exists to close.
|
|
154
|
+
const deferred = [{
|
|
155
|
+
severity: "HIGH", file: "apps/api/legacy.ts", line: 9, kind: "quality",
|
|
156
|
+
problem: "p", fix: "f", outOfScope: "predates this feature; diff never touched it",
|
|
157
|
+
}];
|
|
158
|
+
let stagePrompt = "";
|
|
159
|
+
const { result } = await run("review.js", (prompt, opts) => {
|
|
160
|
+
const label = opts.label || "";
|
|
161
|
+
if (label.startsWith("review:")) return { verdict: "SHIP", findings: [], deferred };
|
|
162
|
+
if (label === "stage-report") { stagePrompt = prompt; return "done"; }
|
|
163
|
+
return replier(BASE_REVIEW)(prompt, opts);
|
|
164
|
+
});
|
|
165
|
+
check("deferred-only ⇒ verdict still SHIP", result.verdict === "SHIP", `got ${result.verdict}`);
|
|
166
|
+
check("deferred-only ⇒ next is /ship (not a fix loop)",
|
|
167
|
+
String(result.next).startsWith("/ship"), result.next);
|
|
168
|
+
check("deferred are counted (both surfaces)", result.deferred === 2, `got ${result.deferred}`);
|
|
169
|
+
check("deferred stay out of the severity counts",
|
|
170
|
+
Object.values(result.counts).every(n => n === 0), JSON.stringify(result.counts));
|
|
171
|
+
check("deferred are routed to the refactor backlog",
|
|
172
|
+
/refactor-backlog\.md/.test(stagePrompt) && /deferred:feat-x/.test(stagePrompt),
|
|
173
|
+
stagePrompt.slice(0, 200));
|
|
174
|
+
check("deferred are never cross-checked (no verify agent spawned)",
|
|
175
|
+
!/verify:/.test(String(result.criticals)) && result.refutedByCrossCheck === 0);
|
|
176
|
+
}
|
|
149
177
|
{
|
|
150
178
|
const { result } = await run("review.js", replier([
|
|
151
179
|
["preflight", { pass: false, tail: "boom" }], ...BASE_REVIEW,
|
|
@@ -22,12 +22,27 @@ const frontmatter = (text) => {
|
|
|
22
22
|
// Mechanical commands must pin model: sonnet (otherwise the lead's
|
|
23
23
|
// orchestration turn silently bills at the session model — Opus/Fable).
|
|
24
24
|
// Interactive commands must stay unpinned (they inherit on purpose).
|
|
25
|
-
const PINNED = ["build", "review", "fix", "
|
|
26
|
-
"refactor", "doctor", "align-ds", "update-pipeline"];
|
|
25
|
+
const PINNED = ["build", "review", "fix", "ship", "audit",
|
|
26
|
+
"refactor", "doctor", "align-ds", "update-pipeline", "drive"];
|
|
27
27
|
const UNPINNED = ["brainstorm", "spec", "init-pipeline"];
|
|
28
28
|
|
|
29
|
+
// Names Claude Code itself claims. A core command that collides is not overridden —
|
|
30
|
+
// it is SHADOWED: the built-in answers the slash, our command file is never read, and
|
|
31
|
+
// the session confidently reports on a run that never happened. That is what `/loop`
|
|
32
|
+
// did (Claude Code's own `/loop` runs a prompt on an interval), invisible until a user
|
|
33
|
+
// noticed the driver had never started. `/loop` is here so the 1.6.0 rename to
|
|
34
|
+
// `/drive` can never be quietly reverted.
|
|
35
|
+
// Watchlist, not yet enforced because the collision is unproven: `doctor` (Claude Code
|
|
36
|
+
// has its own `/doctor`) — if a typed `/doctor` ever stops reaching the pipeline's, add
|
|
37
|
+
// it here and rename.
|
|
38
|
+
const RESERVED = ["loop", "clear", "compact", "cost", "help", "config",
|
|
39
|
+
"init", "run", "schedule", "simplify", "review-pr"];
|
|
40
|
+
|
|
29
41
|
for (const f of readdirSync(join(root, "core/commands"))) {
|
|
30
42
|
const path = `core/commands/${f}`;
|
|
43
|
+
if (RESERVED.includes(f.replace(/\.md$/, "")))
|
|
44
|
+
fail(path, `command name collides with a Claude Code built-in — it would be SHADOWED ` +
|
|
45
|
+
`(the built-in answers the slash and this file is never read); rename it`);
|
|
31
46
|
const fm = frontmatter(read(path));
|
|
32
47
|
if (!fm) { fail(path, "missing or malformed YAML frontmatter"); continue; }
|
|
33
48
|
if (!/^description:\s*\S/m.test(fm)) fail(path, "frontmatter lacks a description");
|
|
@@ -43,7 +58,7 @@ for (const f of readdirSync(join(root, "core/commands"))) {
|
|
|
43
58
|
// Every non-template agent needs name/tools/model, and must be shipped by
|
|
44
59
|
// both installers (a new agent that install.sh doesn't copy never reaches
|
|
45
60
|
// a global install — the exact bug that motivated this check).
|
|
46
|
-
const AGENT_MODEL = { review: "sonnet", release: "haiku",
|
|
61
|
+
const AGENT_MODEL = { review: "sonnet", release: "haiku",
|
|
47
62
|
"profile-reader": "haiku" };
|
|
48
63
|
const installSh = read("install.sh");
|
|
49
64
|
const installPs1 = read("install.ps1");
|
|
@@ -90,7 +105,7 @@ for (const path of allDocs) {
|
|
|
90
105
|
}
|
|
91
106
|
for (const m of text.matchAll(/subagent_type:\s*(?:`|)([a-z-]+)(?:`|)/g)) {
|
|
92
107
|
const t = m[1];
|
|
93
|
-
if (["review", "release", "
|
|
108
|
+
if (["review", "release", "profile-reader"].includes(t)) continue;
|
|
94
109
|
if (t.startsWith("<")) continue; // <surface.agent> placeholder
|
|
95
110
|
if (!existsSync(join(root, "core/agents", `${t}.md`)))
|
|
96
111
|
fail(path, `dispatches subagent_type ${t} with no core/agents/${t}.md`);
|
|
@@ -115,9 +130,9 @@ if (!existsSync(steps) || readdirSync(steps).length === 0)
|
|
|
115
130
|
|
|
116
131
|
// ── telemetry coverage ──────────────────────────────────────────────────────
|
|
117
132
|
// The funnel is only readable if every one of its stages pings — a single missing
|
|
118
|
-
// one silently truncates it (that is how /
|
|
133
|
+
// one silently truncates it (that is how /review and /fix went unreported
|
|
119
134
|
// until 1.2.3). The phase list here must match SCHEMA.md §Telemetry's table.
|
|
120
|
-
const FUNNEL = ["brainstorm", "spec", "build", "
|
|
135
|
+
const FUNNEL = ["brainstorm", "spec", "build", "review", "fix", "ship"];
|
|
121
136
|
for (const c of FUNNEL)
|
|
122
137
|
if (!/usage ping/i.test(read(`core/commands/${c}.md`)))
|
|
123
138
|
fail(`core/commands/${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
|
package/core/agents/smoke.md
DELETED
|
@@ -1,63 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: smoke
|
|
3
|
-
description: Executes the end-to-end smoke run for one feature in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — then stages the SMOKE REPORT. Dispatched by /smoke. Observes honestly, never fixes anything.
|
|
4
|
-
tools: Read, Write, Grep, Glob, Bash, DesignSync
|
|
5
|
-
model: sonnet
|
|
6
|
-
---
|
|
7
|
-
|
|
8
|
-
You are the **smoke** agent for one feature. You actually run the built feature — `/review` audits
|
|
9
|
-
code read-only; nobody has executed it yet. You verify it *works*; you never fix it (failures go
|
|
10
|
-
through `/fix`). Observe honestly: report what happened, not what should have happened.
|
|
11
|
-
|
|
12
|
-
> **First action, always:** read `PIPELINE.md` §`pipeline-profile`: `commands` (migrate/dev),
|
|
13
|
-
> `isolation` (worktree, slot ports, db), `contract`, `design`, `surfaces` — then the spec
|
|
14
|
-
> `specs/<id>.md` (§5 contract, §8 flows, §9 acceptance).
|
|
15
|
-
|
|
16
|
-
## Your inputs (supplied at dispatch — you have no memory)
|
|
17
|
-
|
|
18
|
-
1. The feature id and spec path `specs/<id>.md`.
|
|
19
|
-
2. The contract path `<contract.path>/<id>.<ext>`.
|
|
20
|
-
3. The checkout to work in: the worktree path + slot ports/db, or the main checkout on the feature branch.
|
|
21
|
-
|
|
22
|
-
## Keep your own context lean
|
|
23
|
-
|
|
24
|
-
Redirect every bulky output to a file and inspect it with `grep`/`jq` — never print full curl bodies,
|
|
25
|
-
server logs, or poll loops into your transcript. `curl -s … -o /tmp/resp.json -w '%{http_code}'` then
|
|
26
|
-
assert on the pieces you need.
|
|
27
|
-
|
|
28
|
-
## 1. Bring the feature up
|
|
29
|
-
|
|
30
|
-
- Work in the checkout your dispatch names. Infra as needed: the compose stack if one is declared
|
|
31
|
-
(the gate will ask — that's expected), then `commands.migrate`, then `commands.dev` **in the
|
|
32
|
-
background**. Wait for ready (poll the ports), don't assume.
|
|
33
|
-
|
|
34
|
-
## 2. Exercise the contract (the real server, not the tests)
|
|
35
|
-
|
|
36
|
-
- Hit a representative set of spec §5 endpoints with `curl`: every route domain, every auth level,
|
|
37
|
-
at least one error case per class (validation `422`, unauthenticated `401`, wrong-role `403`,
|
|
38
|
-
conflict `409`). Compare status + response envelope against the contract.
|
|
39
|
-
- If `rbac.enabled`: verify at least one denial per role boundary the spec declares.
|
|
40
|
-
- A mismatch is a FAIL entry with the exact command, expected, and actual — precise enough for a
|
|
41
|
-
stateless `/fix` agent.
|
|
42
|
-
|
|
43
|
-
## 3. Exercise the UI (only if a touched surface has `uses_design`)
|
|
44
|
-
|
|
45
|
-
- Drive the spec §8 flows against the running app, **mobile viewport first** (375px), then desktop.
|
|
46
|
-
- If a browser/screenshot tool is available (a project driver, playwright, an agent browser), capture
|
|
47
|
-
each §8 screen and compare against the feature's design pages: each `design_files` entry is a full
|
|
48
|
-
`https://claude.ai/design/p/<projectId>?file=<file>` link — extract its `<projectId>` (the `/p/…`
|
|
49
|
-
segment) + `<file>` (the `?file=` query) and fetch read-only via `DesignSync get_file(<projectId>,
|
|
50
|
-
<file>)`. Compare layout, states (empty/loading/error/suppressed…), copy language. Note deviations.
|
|
51
|
-
- No browser tooling available ⇒ **say so and skip the visual diff** — never claim a visual check
|
|
52
|
-
you didn't perform.
|
|
53
|
-
|
|
54
|
-
## 4. Stage the SMOKE REPORT, tear down, return
|
|
55
|
-
|
|
56
|
-
- One line per check: ✅/❌ · what was exercised · (on ❌) command → expected vs actual.
|
|
57
|
-
- **Write the full report to `specs/reports/<id>.md`** (overwrite) — the same gitignored buffer
|
|
58
|
-
`/review` uses, so a `/fix` after a `/clear` still has the failures.
|
|
59
|
-
- Tear down what you started (kill the dev server); leave shared infra as you found it.
|
|
60
|
-
- **Your return to the lead is ONLY:** the verdict line (`PASS` / `FAIL:<n>`), **at most 10 ❌
|
|
61
|
-
lines** — one line each (`❌ <flow/endpoint> · expected <x> got <y>`), no command output, no code
|
|
62
|
-
or body excerpts; more than 10 ⇒ keep the 10 most severe and add `+<n> more — see the report` —
|
|
63
|
-
and `Full report: specs/reports/<id>.md`. No logs, no bodies, no screenshots.
|
package/core/commands/smoke.md
DELETED
|
@@ -1,55 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
model: sonnet
|
|
3
|
-
description: Exercise the built feature end-to-end in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — before /review.
|
|
4
|
-
argument-hint: <feature_id>
|
|
5
|
-
---
|
|
6
|
-
|
|
7
|
-
You are the **lead**. Dispatch the smoke run for feature **$ARGUMENTS** — the `smoke` agent actually
|
|
8
|
-
runs it so the bulky output (curl bodies, server logs, screenshots, design payloads) never enters
|
|
9
|
-
your own context, which is re-sent every turn.
|
|
10
|
-
|
|
11
|
-
> Read `PIPELINE.md` §`pipeline-profile`: `isolation` (worktree, slot ports, db), `contract.path`
|
|
12
|
-
> and `commands`. _Skip the re-read if it's already in your context this session and unmodified since._
|
|
13
|
-
>
|
|
14
|
-
> **Kanban:** none here — `/review` owns the → **Review** move (running both duplicated it).
|
|
15
|
-
|
|
16
|
-
## 0. Deterministic pre-flight — no agents while red
|
|
17
|
-
|
|
18
|
-
Same gate as `/review` §0, run **in the feature's checkout** (§1 resolves it — resolve first, then
|
|
19
|
-
preflight): `<core>/pipeline/scripts/preflight.sh specs/reports/$ARGUMENTS.preflight.txt
|
|
20
|
-
"<commands.typecheck>" "<commands.lint_quiet, else lint>" "<commands.test_quiet, else test>"`.
|
|
21
|
-
Non-zero exit ⇒ the raw last-40 lines were already printed — **STOP, relay them verbatim, spawn NO
|
|
22
|
-
agent**: booting infra to smoke-test code that doesn't compile wastes the whole run. Zero exit ⇒
|
|
23
|
-
the `.claude/preflight.ok` stamp lets the gate hook pass your `smoke` dispatch. Script absent
|
|
24
|
-
(older core) ⇒ run the commands yourself redirected to the same file, aborting on the first failure.
|
|
25
|
-
Note the epoch (`date +%s`) in the same call — §3's metrics line needs it.
|
|
26
|
-
|
|
27
|
-
## 1. Resolve the checkout
|
|
28
|
-
|
|
29
|
-
With `isolation.enabled`: the sibling worktree (`../<slug>-$ARGUMENTS`, its slot's ports + db from
|
|
30
|
-
`.worktrees/slots.tsv`); otherwise the main checkout on the feature branch.
|
|
31
|
-
|
|
32
|
-
## 2. Dispatch ONE `smoke` agent
|
|
33
|
-
|
|
34
|
-
Keep the prompt byte-identical across features except the variable block at the END (prompt-cache
|
|
35
|
-
prefix):
|
|
36
|
-
|
|
37
|
-
> `subagent_type: smoke` — "Smoke-test one feature. Read `PIPELINE.md` first. Bring it up, exercise
|
|
38
|
-
> the contract and the §8 UI flows, stage the full SMOKE REPORT to the report buffer, tear down, and
|
|
39
|
-
> return only the capped verdict your agent instructions define (verdict + ❌ lines, no logs). —
|
|
40
|
-
> Variable slots: feature `$ARGUMENTS` · spec: `specs/$ARGUMENTS.md` · contract:
|
|
41
|
-
> `<contract.path>/$ARGUMENTS.<ext>` · report: `specs/reports/$ARGUMENTS.md` · checkout: `<worktree
|
|
42
|
-
> path or main checkout>` · ports/db: `<slot info, or defaults>`."
|
|
43
|
-
|
|
44
|
-
The gate hooks fire on the agent's Bash calls too — compose/migrate confirmations still reach the
|
|
45
|
-
human; that's expected.
|
|
46
|
-
|
|
47
|
-
## 3. Relay the verdict
|
|
48
|
-
|
|
49
|
-
- Print the agent's return as-is (verdict + ❌ lines + report path) — it is already minimal.
|
|
50
|
-
- Append ONE metrics line to `pipeline-metrics.jsonl` (main-checkout path + rules in `/build` §4,
|
|
51
|
-
`phase: "smoke"`), chaining the opt-in usage ping in the same Bash call (results = `PASS` or
|
|
52
|
-
`FAIL:<n>` failing flows).
|
|
53
|
-
- **PASS** → tell the human to run `/review $ARGUMENTS`. **FAIL** → the failures are findings: feed
|
|
54
|
-
them to `/fix $ARGUMENTS`, re-run `/smoke` after. Either way the report is on disk —
|
|
55
|
-
**recommend a `/clear`** before the next command.
|