cohorte 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/CHANGELOG.md +151 -0
  2. package/README.md +43 -12
  3. package/bin/cli.js +13 -2
  4. package/core/agents/review.md +23 -0
  5. package/core/commands/audit.md +9 -1
  6. package/core/commands/brainstorm.md +6 -0
  7. package/core/commands/build.md +93 -8
  8. package/core/commands/doctor.md +13 -8
  9. package/core/commands/drive.md +80 -0
  10. package/core/commands/fix.md +10 -5
  11. package/core/commands/review.md +94 -12
  12. package/core/commands/spec.md +20 -0
  13. package/core/commands/update-pipeline.md +6 -1
  14. package/core/hooks/gate.py +4 -4
  15. package/core/templates/decisions.template.md +42 -0
  16. package/core/templates/spec.template.md +4 -2
  17. package/core/templates/steps/init-pipeline/02-interview-gaps.md +1 -1
  18. package/core/templates/steps/init-pipeline/04-write-render.md +8 -4
  19. package/core/workflows/review.js +44 -2
  20. package/dashboard/dist/assets/{index-AFQnlfjO.css → index-BZ_LQlEj.css} +1 -1
  21. package/dashboard/dist/assets/index-DYyn4p93.js +43 -0
  22. package/dashboard/dist/index.html +2 -2
  23. package/dashboard/server/doctor.js +13 -3
  24. package/dashboard/server/index.js +7 -0
  25. package/dashboard/server/metrics.js +4 -4
  26. package/dashboard/server/usage.js +61 -0
  27. package/install.ps1 +12 -1
  28. package/install.sh +12 -2
  29. package/package.json +1 -1
  30. package/profile/PIPELINE.template.md +3 -3
  31. package/profile/SCHEMA.md +150 -12
  32. package/scripts/loop.sh +318 -0
  33. package/scripts/metrics/collect.mjs +11 -2
  34. package/scripts/preflight.sh +2 -2
  35. package/scripts/telemetry-send.sh +5 -2
  36. package/scripts/test-dashboard.mjs +22 -2
  37. package/scripts/test-gate.mjs +1 -2
  38. package/scripts/test-loop.mjs +227 -0
  39. package/scripts/test-metrics.mjs +12 -3
  40. package/scripts/test-workflows.mjs +28 -0
  41. package/scripts/validate-core.mjs +21 -6
  42. package/core/agents/smoke.md +0 -63
  43. package/core/commands/smoke.md +0 -55
  44. package/dashboard/dist/assets/index-DLBzciIC.js +0 -43
@@ -0,0 +1,227 @@
1
+ #!/usr/bin/env node
2
+ // Behavioural tests for scripts/loop.sh — the autonomous /review ⇄ /fix driver.
3
+ //
4
+ // The driver is pure shell around two JSON files it does not write, so it is
5
+ // testable end-to-end by putting a FAKE `claude` on PATH that produces those files
6
+ // per phase. What is pinned here cannot be seen by any structural check:
7
+ //
8
+ // · exit 4 — /build's readiness gate said NOT-READY, so no pass count helps
9
+ // · exit 0/3 leave the right TERMINAL status in the spec's front-matter, which is
10
+ // what makes an interrupted loop resumable (SCHEMA.md §Spec status)
11
+ // · the front-matter stamps are written with awk on every platform — a `sed -i`
12
+ // would pass on GNU and corrupt every spec on BSD/macOS
13
+ // · --resume continues at the recorded pass instead of re-paying passes 1..n-1
14
+ // · a spec with no front-matter still runs (the stamps are a silent no-op)
15
+ //
16
+ // node scripts/test-loop.mjs
17
+
18
+ import { mkdtempSync, writeFileSync, readFileSync, mkdirSync, chmodSync, existsSync } from "node:fs";
19
+ import { execFileSync, spawnSync } from "node:child_process";
20
+ import { join } from "node:path";
21
+ import { tmpdir } from "node:os";
22
+ import { fileURLToPath } from "node:url";
23
+
24
+ const root = fileURLToPath(new URL("..", import.meta.url));
25
+ const LOOP = join(root, "scripts/loop.sh");
26
+
27
+ let failures = 0;
28
+ const check = (name, cond, detail = "") => {
29
+ if (cond) console.log(` ✓ ${name}`);
30
+ else { failures++; console.error(` ✗ ${name}${detail ? ` — ${detail}` : ""}`); }
31
+ };
32
+
33
+ const FM = `---
34
+ feature_id: feat-x
35
+ title: Feat X
36
+ status: frozen # draft → frozen → in-progress → in-review → shipped · blocked
37
+ branch: feature/feat-x
38
+ ---
39
+
40
+ # Feat X
41
+ `;
42
+
43
+ // A fake `claude`: reads the phase out of the `-p "/<cmd> <id>"` argument and
44
+ // writes whatever the scenario says that phase produces. `$PHASES` is a
45
+ // newline-separated script of `<cmd>:<what to write>` steps, consumed in order,
46
+ // so a scenario can make pass 1 and pass 2 differ.
47
+ const FAKE_CLAUDE = `#!/usr/bin/env bash
48
+ set -u
49
+ prompt=""
50
+ while [ $# -gt 0 ]; do
51
+ case "$1" in -p) prompt="$2"; shift 2 ;; *) shift ;; esac
52
+ done
53
+ cmd="\${prompt%% *}"; cmd="\${cmd#/}"
54
+ n=0; [ -f "$SCEN_DIR/count" ] && n=$(cat "$SCEN_DIR/count")
55
+ n=$((n + 1)); echo "$n" >"$SCEN_DIR/count"
56
+ step=$(sed -n "\${n}p" "$SCEN_DIR/phases")
57
+ echo "fake claude: phase=$cmd step=$step"
58
+ want="\${step%%:*}"; do_what="\${step#*:}"
59
+ [ "$want" = "$cmd" ] || { echo "fake claude: expected /$want, got /$cmd" >&2; exit 9; }
60
+ mkdir -p specs/reports
61
+ case "$do_what" in
62
+ notready)
63
+ printf '{"id":"feat-x","phase":"readiness","ts":"t","verdict":"NOT-READY","gaps":["contract|POST /o|no success shape"],"surfaces":["backend"]}' \\
64
+ >specs/reports/feat-x.readiness.json ;;
65
+ ready)
66
+ printf '{"id":"feat-x","phase":"readiness","ts":"t","verdict":"READY","gaps":[],"surfaces":["backend"]}' \\
67
+ >specs/reports/feat-x.readiness.json ;;
68
+ clean)
69
+ printf '{"id":"feat-x","phase":"review","ts":"t","verdict":"SHIP","findings":2,"blocking":0,"deferred":2,"unreviewed":[],"fingerprint":""}' \\
70
+ >specs/reports/feat-x.verdict.json ;;
71
+ deadreviewer)
72
+ printf '{"id":"feat-x","phase":"review","ts":"t","verdict":"REVISE","findings":0,"blocking":0,"deferred":0,"unreviewed":["backend"],"fingerprint":""}' \\
73
+ >specs/reports/feat-x.verdict.json ;;
74
+ deadimplementer)
75
+ printf '{"id":"feat-x","phase":"readiness","ts":"t","verdict":"READY","gaps":[],"surfaces":["backend"]}' \\
76
+ >specs/reports/feat-x.readiness.json
77
+ printf '{"id":"feat-x","phase":"build","ts":"t","surfaces":{"backend":"dead","frontend":"ok"},"dead":["backend"]}' \\
78
+ >specs/reports/feat-x.build.json ;;
79
+ blocking)
80
+ printf '{"id":"feat-x","phase":"review","ts":"t","verdict":"REVISE","findings":3,"blocking":2,"deferred":1,"fingerprint":"aaaa1111bbbb2222"}' \\
81
+ >specs/reports/feat-x.verdict.json ;;
82
+ noop) : ;;
83
+ esac
84
+ exit 0
85
+ `;
86
+
87
+ // One scratch repo per scenario: a git checkout (loop.sh cds to its toplevel), a
88
+ // spec, the fake claude on PATH, and the phase script it plays out.
89
+ function scenario(phases, { frontmatter = FM } = {}) {
90
+ const dir = mkdtempSync(join(tmpdir(), "cohorte-loop-"));
91
+ const git = (...a) => execFileSync("git", ["-C", dir, ...a], { stdio: "ignore" });
92
+ git("init", "-q");
93
+ git("config", "user.email", "t@t.t");
94
+ git("config", "user.name", "t");
95
+ mkdirSync(join(dir, "specs/reports"), { recursive: true });
96
+ writeFileSync(join(dir, "specs/feat-x.md"), frontmatter);
97
+ git("add", "-A");
98
+ git("commit", "-qm", "init");
99
+
100
+ const bin = join(dir, "bin");
101
+ mkdirSync(bin);
102
+ writeFileSync(join(dir, "phases"), phases.join("\n") + "\n");
103
+ writeFileSync(join(bin, "claude"), FAKE_CLAUDE);
104
+ chmodSync(join(bin, "claude"), 0o755);
105
+ return { dir, bin };
106
+ }
107
+
108
+ function runLoop({ dir, bin }, args) {
109
+ const r = spawnSync("bash", [LOOP, ...args], {
110
+ cwd: dir,
111
+ encoding: "utf8",
112
+ env: {
113
+ ...process.env,
114
+ PATH: `${bin}:${process.env.PATH}`,
115
+ SCEN_DIR: dir,
116
+ CLAUDE_FLAGS: "--permission-mode acceptEdits",
117
+ GIT_AUTHOR_NAME: "t", GIT_AUTHOR_EMAIL: "t@t.t",
118
+ GIT_COMMITTER_NAME: "t", GIT_COMMITTER_EMAIL: "t@t.t",
119
+ },
120
+ });
121
+ const spec = readFileSync(join(dir, "specs/feat-x.md"), "utf8");
122
+ const fm = k => {
123
+ const m = spec.match(new RegExp(`^${k}:\\s*([^#\\n]*)`, "m"));
124
+ return m ? m[1].trim() : null;
125
+ };
126
+ return { code: r.status, out: `${r.stdout}${r.stderr}`, spec, fm };
127
+ }
128
+
129
+ console.log("loop.sh — readiness gate");
130
+ {
131
+ const s = scenario(["build:notready"]);
132
+ const r = runLoop(s, ["feat-x"]);
133
+ check("NOT-READY ⇒ exit 4, not 2", r.code === 4, `got ${r.code}: ${r.out.trim().split("\n").pop()}`);
134
+ check("NOT-READY ⇒ says the spec is not implementable",
135
+ /not implementable/i.test(r.out), r.out.trim().split("\n").pop());
136
+ check("NOT-READY ⇒ points at /spec", /\/spec feat-x/.test(r.out));
137
+ check("NOT-READY ⇒ spec left blocked", r.fm("status") === "blocked", r.fm("status"));
138
+ check("NOT-READY ⇒ no review ran (the gate is the point)", !/phase=review/.test(r.out));
139
+ check("NOT-READY ⇒ the build stamp is NOT written",
140
+ !existsSync(join(s.dir, "specs/reports/feat-x.built")));
141
+ }
142
+
143
+ console.log("loop.sh — clean run");
144
+ {
145
+ const s = scenario(["build:ready", "review:clean"]);
146
+ const r = runLoop(s, ["feat-x"]);
147
+ check("clean ⇒ exit 0", r.code === 0, `got ${r.code}: ${r.out}`);
148
+ check("clean ⇒ status in-review (ready to /ship)", r.fm("status") === "in-review", r.fm("status"));
149
+ check("clean ⇒ loop state cleared", r.fm("loop_pass") === "0" && r.fm("loop_phase") === "done",
150
+ `${r.fm("loop_pass")}/${r.fm("loop_phase")}`);
151
+ check("clean ⇒ the deferred count is named, not dropped",
152
+ /2 deferred finding\(s\) parked/.test(r.out), r.out.trim().split("\n").pop());
153
+ check("clean ⇒ the status comment survives the awk rewrite",
154
+ /^status: in-review # draft/m.test(r.spec), r.spec.split("\n")[3]);
155
+ }
156
+
157
+ console.log("loop.sh — a dead subagent is never a clean result");
158
+ {
159
+ // A dead implementer: /build finishes fine having built one surface of two. Reviewing
160
+ // that would spend N reviewers auditing a half-built feature and report its holes as
161
+ // findings to fix — the wrong diagnosis at the wrong price.
162
+ const s = scenario(["build:deadimplementer"]);
163
+ const r = runLoop(s, ["feat-x"]);
164
+ check("dead implementer ⇒ exit 2, not a review pass", r.code === 2, `got ${r.code}: ${r.out}`);
165
+ check("dead implementer ⇒ no reviewer was spawned", !/phase=review/.test(r.out));
166
+ check("dead implementer ⇒ names the cause", /implementer died/.test(r.out),
167
+ r.out.trim().split("\n").pop());
168
+ check("dead implementer ⇒ spec left blocked", r.fm("status") === "blocked", r.fm("status"));
169
+ }
170
+ {
171
+ // THE dangerous one: blocking == 0 because the only reviewer that could have found
172
+ // something never answered. Exiting 0 here would report "clean" about unread code and
173
+ // send the human to /ship.
174
+ const s = scenario(["build:ready", "review:deadreviewer"]);
175
+ const r = runLoop(s, ["feat-x"]);
176
+ check("dead reviewer + blocking 0 ⇒ NOT exit 0", r.code !== 0, `got ${r.code}: ${r.out}`);
177
+ check("dead reviewer ⇒ exit 2 (no usable verdict)", r.code === 2, `got ${r.code}`);
178
+ check("dead reviewer ⇒ names the unreviewed surface", /reviewer died/.test(r.out),
179
+ r.out.trim().split("\n").pop());
180
+ check("dead reviewer ⇒ never says clean", !/✓ clean/.test(r.out));
181
+ check("dead reviewer ⇒ spec is NOT left in-review", r.fm("status") === "blocked", r.fm("status"));
182
+ }
183
+
184
+ console.log("loop.sh — non-convergent + resume");
185
+ {
186
+ const s = scenario(["build:ready", "review:blocking", "fix:noop", "review:blocking"]);
187
+ const r = runLoop(s, ["feat-x", "--max=4"]);
188
+ check("same fingerprint twice ⇒ exit 3", r.code === 3, `got ${r.code}: ${r.out}`);
189
+ check("non-convergent ⇒ status blocked", r.fm("status") === "blocked", r.fm("status"));
190
+ check("non-convergent ⇒ the pass it reached is recorded (resume anchor)",
191
+ r.fm("loop_pass") === "2", r.fm("loop_pass"));
192
+ check("non-convergent ⇒ the phase is recorded", r.fm("loop_phase") === "review", r.fm("loop_phase"));
193
+
194
+ // Resume: the recorded pass is where it picks up — passes 1..n-1 are not re-paid.
195
+ writeFileSync(join(s.dir, "count"), "0");
196
+ writeFileSync(join(s.dir, "phases"), "review:clean\n");
197
+ const r2 = runLoop(s, ["feat-x", "--max=4", "--resume"]);
198
+ check("--resume ⇒ announces the pass it continues from",
199
+ /resuming at review pass 2/.test(r2.out), r2.out.trim().split("\n")[0]);
200
+ check("--resume ⇒ skips the build (the stamp is there)", !/phase=build/.test(r2.out));
201
+ check("--resume ⇒ finishes clean from there", r2.code === 0, `got ${r2.code}: ${r2.out}`);
202
+ check("--resume ⇒ reports the resumed pass count, not 1",
203
+ /after 2 review pass\(es\)/.test(r2.out), r2.out.trim().split("\n").pop());
204
+ }
205
+ {
206
+ const s = scenario(["review:clean"]);
207
+ // A resume anchor past the ceiling is a usage error, not a silent restart at 1.
208
+ writeFileSync(join(s.dir, "specs/feat-x.md"), FM.replace("branch:", "loop_pass: 9\nbranch:"));
209
+ const r = runLoop(s, ["feat-x", "--max=3", "--resume"]);
210
+ check("--resume past --max ⇒ exit 64 with the reason", r.code === 64 && /raise --max/.test(r.out),
211
+ `${r.code}: ${r.out.trim()}`);
212
+ }
213
+
214
+ console.log("loop.sh — a spec with no front-matter still runs");
215
+ {
216
+ const s = scenario(["build:ready", "review:clean"], { frontmatter: "# Feat X\n\nno front-matter\n" });
217
+ const r = runLoop(s, ["feat-x"]);
218
+ check("no front-matter ⇒ still exits 0 (stamps are a silent no-op)", r.code === 0,
219
+ `got ${r.code}: ${r.out}`);
220
+ check("no front-matter ⇒ the spec is left untouched",
221
+ r.spec === "# Feat X\n\nno front-matter\n", JSON.stringify(r.spec));
222
+ check("no front-matter ⇒ no stray temp file",
223
+ !existsSync(join(s.dir, "specs/feat-x.md.loop.tmp")));
224
+ }
225
+
226
+ if (failures) { console.error(`\ntest-loop: ${failures} failure(s)`); process.exit(1); }
227
+ console.log("\ntest-loop: OK");
@@ -82,6 +82,13 @@ const lines = [
82
82
  // Case 8: a slash token that is not a command must not invent one.
83
83
  user(1800, 'look at the /usr/local/share directory and report what you find there'),
84
84
  assistant('m7', 1805, 'claude-opus-5', { input_tokens: 0, output_tokens: 10 }),
85
+ // Case 9: a long prompt that merely DISCUSSES a command is not an invocation of it.
86
+ // Without the length gate, writing about /review bills the conversation to /review —
87
+ // which is what happened in cohorte's own repo while the pipeline was being designed.
88
+ user(2400, 'I want to talk through how /review behaves when a surface has no findings at '
89
+ + 'all, because the verdict logic there is what produced the false green we saw last week '
90
+ + 'and I am not convinced the fix covers the case where every reviewer dies at once.'),
91
+ assistant('m8', 2405, 'claude-opus-5', { input_tokens: 0, output_tokens: 20 }),
85
92
  ];
86
93
  fs.writeFileSync(path.join(projectDir, `${SESSION}.jsonl`),
87
94
  lines.map((l) => JSON.stringify(l)).join('\n') + '\n');
@@ -107,18 +114,20 @@ const chat = out.commands.find((c) => c.command === '(chat)');
107
114
  const review = out.commands.find((c) => c.command === '/review');
108
115
 
109
116
  console.log('test-metrics');
110
- check('the mid-command task-notification did not split the run', out.totals.runs, 4);
117
+ check('the mid-command task-notification did not split the run', out.totals.runs, 5);
111
118
  check('/build is one run, not three', build.runs, 1);
112
119
  check('duplicate lines of one response are billed once', build.tokens.output, 1000 + 500 + 2000);
113
120
  check('the <synthetic> message contributed no tokens', build.tokens.output < 999999, true);
114
121
  check('cache-write tokens are kept on their own tier', build.tokens.cacheWrite5m, 1000);
115
122
  check('cache-read tokens are kept on their own tier', build.tokens.cacheRead, 10000);
116
123
  check('the subagent was attributed to the command that spawned it', build.agents.total, 1);
117
- check('the second prompt is a separate (chat) run', chat.runs, 2);
124
+ check('the second prompt is a separate (chat) run', chat.runs, 3);
118
125
  check('a command named inside prose is attributed to that command', review && review.runs, 1);
119
126
  check('a short steer continues the run instead of opening a new one', review.continuations, 1);
120
127
  check('the continued turn counts toward the command it continued', review.tokens.output, 60 + 70);
121
- check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10);
128
+ check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10 + 20);
129
+ check('a long prompt that discusses a command is not counted as running it',
130
+ review.runs, 1);
122
131
 
123
132
  // opus-5 $5 in / $25 out per MTok; 5m cache write 1.25x input, cache read 0.1x input.
124
133
  // m1 100*5 + 1000*25 + 1000*6.25 + 10000*0.5 = 36750
@@ -146,6 +146,34 @@ console.log("review.js");
146
146
  ]));
147
147
  check("SHIP + only LOW ⇒ next is /ship", String(result.next).startsWith("/ship"), result.next);
148
148
  }
149
+ {
150
+ // Deferred findings are real but out of the feature's scope: they must be
151
+ // counted and routed to the backlog, yet move NEITHER the verdict nor `clean`.
152
+ // Both halves matter — a deferred item that blocks costs a fix loop it was
153
+ // deferred out of, and one that is dropped is the leak the section exists to close.
154
+ const deferred = [{
155
+ severity: "HIGH", file: "apps/api/legacy.ts", line: 9, kind: "quality",
156
+ problem: "p", fix: "f", outOfScope: "predates this feature; diff never touched it",
157
+ }];
158
+ let stagePrompt = "";
159
+ const { result } = await run("review.js", (prompt, opts) => {
160
+ const label = opts.label || "";
161
+ if (label.startsWith("review:")) return { verdict: "SHIP", findings: [], deferred };
162
+ if (label === "stage-report") { stagePrompt = prompt; return "done"; }
163
+ return replier(BASE_REVIEW)(prompt, opts);
164
+ });
165
+ check("deferred-only ⇒ verdict still SHIP", result.verdict === "SHIP", `got ${result.verdict}`);
166
+ check("deferred-only ⇒ next is /ship (not a fix loop)",
167
+ String(result.next).startsWith("/ship"), result.next);
168
+ check("deferred are counted (both surfaces)", result.deferred === 2, `got ${result.deferred}`);
169
+ check("deferred stay out of the severity counts",
170
+ Object.values(result.counts).every(n => n === 0), JSON.stringify(result.counts));
171
+ check("deferred are routed to the refactor backlog",
172
+ /refactor-backlog\.md/.test(stagePrompt) && /deferred:feat-x/.test(stagePrompt),
173
+ stagePrompt.slice(0, 200));
174
+ check("deferred are never cross-checked (no verify agent spawned)",
175
+ !/verify:/.test(String(result.criticals)) && result.refutedByCrossCheck === 0);
176
+ }
149
177
  {
150
178
  const { result } = await run("review.js", replier([
151
179
  ["preflight", { pass: false, tail: "boom" }], ...BASE_REVIEW,
@@ -22,12 +22,27 @@ const frontmatter = (text) => {
22
22
  // Mechanical commands must pin model: sonnet (otherwise the lead's
23
23
  // orchestration turn silently bills at the session model — Opus/Fable).
24
24
  // Interactive commands must stay unpinned (they inherit on purpose).
25
- const PINNED = ["build", "review", "fix", "smoke", "ship", "audit",
26
- "refactor", "doctor", "align-ds", "update-pipeline"];
25
+ const PINNED = ["build", "review", "fix", "ship", "audit",
26
+ "refactor", "doctor", "align-ds", "update-pipeline", "drive"];
27
27
  const UNPINNED = ["brainstorm", "spec", "init-pipeline"];
28
28
 
29
+ // Names Claude Code itself claims. A core command that collides is not overridden —
30
+ // it is SHADOWED: the built-in answers the slash, our command file is never read, and
31
+ // the session confidently reports on a run that never happened. That is what `/loop`
32
+ // did (Claude Code's own `/loop` runs a prompt on an interval), invisible until a user
33
+ // noticed the driver had never started. `/loop` is here so the 1.6.0 rename to
34
+ // `/drive` can never be quietly reverted.
35
+ // Watchlist, not yet enforced because the collision is unproven: `doctor` (Claude Code
36
+ // has its own `/doctor`) — if a typed `/doctor` ever stops reaching the pipeline's, add
37
+ // it here and rename.
38
+ const RESERVED = ["loop", "clear", "compact", "cost", "help", "config",
39
+ "init", "run", "schedule", "simplify", "review-pr"];
40
+
29
41
  for (const f of readdirSync(join(root, "core/commands"))) {
30
42
  const path = `core/commands/${f}`;
43
+ if (RESERVED.includes(f.replace(/\.md$/, "")))
44
+ fail(path, `command name collides with a Claude Code built-in — it would be SHADOWED ` +
45
+ `(the built-in answers the slash and this file is never read); rename it`);
31
46
  const fm = frontmatter(read(path));
32
47
  if (!fm) { fail(path, "missing or malformed YAML frontmatter"); continue; }
33
48
  if (!/^description:\s*\S/m.test(fm)) fail(path, "frontmatter lacks a description");
@@ -43,7 +58,7 @@ for (const f of readdirSync(join(root, "core/commands"))) {
43
58
  // Every non-template agent needs name/tools/model, and must be shipped by
44
59
  // both installers (a new agent that install.sh doesn't copy never reaches
45
60
  // a global install — the exact bug that motivated this check).
46
- const AGENT_MODEL = { review: "sonnet", release: "haiku", smoke: "sonnet",
61
+ const AGENT_MODEL = { review: "sonnet", release: "haiku",
47
62
  "profile-reader": "haiku" };
48
63
  const installSh = read("install.sh");
49
64
  const installPs1 = read("install.ps1");
@@ -90,7 +105,7 @@ for (const path of allDocs) {
90
105
  }
91
106
  for (const m of text.matchAll(/subagent_type:\s*(?:`|)([a-z-]+)(?:`|)/g)) {
92
107
  const t = m[1];
93
- if (["review", "release", "smoke", "profile-reader"].includes(t)) continue;
108
+ if (["review", "release", "profile-reader"].includes(t)) continue;
94
109
  if (t.startsWith("<")) continue; // <surface.agent> placeholder
95
110
  if (!existsSync(join(root, "core/agents", `${t}.md`)))
96
111
  fail(path, `dispatches subagent_type ${t} with no core/agents/${t}.md`);
@@ -115,9 +130,9 @@ if (!existsSync(steps) || readdirSync(steps).length === 0)
115
130
 
116
131
  // ── telemetry coverage ──────────────────────────────────────────────────────
117
132
  // The funnel is only readable if every one of its stages pings — a single missing
118
- // one silently truncates it (that is how /smoke, /review and /fix went unreported
133
+ // one silently truncates it (that is how /review and /fix went unreported
119
134
  // until 1.2.3). The phase list here must match SCHEMA.md §Telemetry's table.
120
- const FUNNEL = ["brainstorm", "spec", "build", "smoke", "review", "fix", "ship"];
135
+ const FUNNEL = ["brainstorm", "spec", "build", "review", "fix", "ship"];
121
136
  for (const c of FUNNEL)
122
137
  if (!/usage ping/i.test(read(`core/commands/${c}.md`)))
123
138
  fail(`core/commands/${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
@@ -1,63 +0,0 @@
1
- ---
2
- name: smoke
3
- description: Executes the end-to-end smoke run for one feature in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — then stages the SMOKE REPORT. Dispatched by /smoke. Observes honestly, never fixes anything.
4
- tools: Read, Write, Grep, Glob, Bash, DesignSync
5
- model: sonnet
6
- ---
7
-
8
- You are the **smoke** agent for one feature. You actually run the built feature — `/review` audits
9
- code read-only; nobody has executed it yet. You verify it *works*; you never fix it (failures go
10
- through `/fix`). Observe honestly: report what happened, not what should have happened.
11
-
12
- > **First action, always:** read `PIPELINE.md` §`pipeline-profile`: `commands` (migrate/dev),
13
- > `isolation` (worktree, slot ports, db), `contract`, `design`, `surfaces` — then the spec
14
- > `specs/<id>.md` (§5 contract, §8 flows, §9 acceptance).
15
-
16
- ## Your inputs (supplied at dispatch — you have no memory)
17
-
18
- 1. The feature id and spec path `specs/<id>.md`.
19
- 2. The contract path `<contract.path>/<id>.<ext>`.
20
- 3. The checkout to work in: the worktree path + slot ports/db, or the main checkout on the feature branch.
21
-
22
- ## Keep your own context lean
23
-
24
- Redirect every bulky output to a file and inspect it with `grep`/`jq` — never print full curl bodies,
25
- server logs, or poll loops into your transcript. `curl -s … -o /tmp/resp.json -w '%{http_code}'` then
26
- assert on the pieces you need.
27
-
28
- ## 1. Bring the feature up
29
-
30
- - Work in the checkout your dispatch names. Infra as needed: the compose stack if one is declared
31
- (the gate will ask — that's expected), then `commands.migrate`, then `commands.dev` **in the
32
- background**. Wait for ready (poll the ports), don't assume.
33
-
34
- ## 2. Exercise the contract (the real server, not the tests)
35
-
36
- - Hit a representative set of spec §5 endpoints with `curl`: every route domain, every auth level,
37
- at least one error case per class (validation `422`, unauthenticated `401`, wrong-role `403`,
38
- conflict `409`). Compare status + response envelope against the contract.
39
- - If `rbac.enabled`: verify at least one denial per role boundary the spec declares.
40
- - A mismatch is a FAIL entry with the exact command, expected, and actual — precise enough for a
41
- stateless `/fix` agent.
42
-
43
- ## 3. Exercise the UI (only if a touched surface has `uses_design`)
44
-
45
- - Drive the spec §8 flows against the running app, **mobile viewport first** (375px), then desktop.
46
- - If a browser/screenshot tool is available (a project driver, playwright, an agent browser), capture
47
- each §8 screen and compare against the feature's design pages: each `design_files` entry is a full
48
- `https://claude.ai/design/p/<projectId>?file=<file>` link — extract its `<projectId>` (the `/p/…`
49
- segment) + `<file>` (the `?file=` query) and fetch read-only via `DesignSync get_file(<projectId>,
50
- <file>)`. Compare layout, states (empty/loading/error/suppressed…), copy language. Note deviations.
51
- - No browser tooling available ⇒ **say so and skip the visual diff** — never claim a visual check
52
- you didn't perform.
53
-
54
- ## 4. Stage the SMOKE REPORT, tear down, return
55
-
56
- - One line per check: ✅/❌ · what was exercised · (on ❌) command → expected vs actual.
57
- - **Write the full report to `specs/reports/<id>.md`** (overwrite) — the same gitignored buffer
58
- `/review` uses, so a `/fix` after a `/clear` still has the failures.
59
- - Tear down what you started (kill the dev server); leave shared infra as you found it.
60
- - **Your return to the lead is ONLY:** the verdict line (`PASS` / `FAIL:<n>`), **at most 10 ❌
61
- lines** — one line each (`❌ <flow/endpoint> · expected <x> got <y>`), no command output, no code
62
- or body excerpts; more than 10 ⇒ keep the 10 most severe and add `+<n> more — see the report` —
63
- and `Full report: specs/reports/<id>.md`. No logs, no bodies, no screenshots.
@@ -1,55 +0,0 @@
1
- ---
2
- model: sonnet
3
- description: Exercise the built feature end-to-end in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — before /review.
4
- argument-hint: <feature_id>
5
- ---
6
-
7
- You are the **lead**. Dispatch the smoke run for feature **$ARGUMENTS** — the `smoke` agent actually
8
- runs it so the bulky output (curl bodies, server logs, screenshots, design payloads) never enters
9
- your own context, which is re-sent every turn.
10
-
11
- > Read `PIPELINE.md` §`pipeline-profile`: `isolation` (worktree, slot ports, db), `contract.path`
12
- > and `commands`. _Skip the re-read if it's already in your context this session and unmodified since._
13
- >
14
- > **Kanban:** none here — `/review` owns the → **Review** move (running both duplicated it).
15
-
16
- ## 0. Deterministic pre-flight — no agents while red
17
-
18
- Same gate as `/review` §0, run **in the feature's checkout** (§1 resolves it — resolve first, then
19
- preflight): `<core>/pipeline/scripts/preflight.sh specs/reports/$ARGUMENTS.preflight.txt
20
- "<commands.typecheck>" "<commands.lint_quiet, else lint>" "<commands.test_quiet, else test>"`.
21
- Non-zero exit ⇒ the raw last-40 lines were already printed — **STOP, relay them verbatim, spawn NO
22
- agent**: booting infra to smoke-test code that doesn't compile wastes the whole run. Zero exit ⇒
23
- the `.claude/preflight.ok` stamp lets the gate hook pass your `smoke` dispatch. Script absent
24
- (older core) ⇒ run the commands yourself redirected to the same file, aborting on the first failure.
25
- Note the epoch (`date +%s`) in the same call — §3's metrics line needs it.
26
-
27
- ## 1. Resolve the checkout
28
-
29
- With `isolation.enabled`: the sibling worktree (`../<slug>-$ARGUMENTS`, its slot's ports + db from
30
- `.worktrees/slots.tsv`); otherwise the main checkout on the feature branch.
31
-
32
- ## 2. Dispatch ONE `smoke` agent
33
-
34
- Keep the prompt byte-identical across features except the variable block at the END (prompt-cache
35
- prefix):
36
-
37
- > `subagent_type: smoke` — "Smoke-test one feature. Read `PIPELINE.md` first. Bring it up, exercise
38
- > the contract and the §8 UI flows, stage the full SMOKE REPORT to the report buffer, tear down, and
39
- > return only the capped verdict your agent instructions define (verdict + ❌ lines, no logs). —
40
- > Variable slots: feature `$ARGUMENTS` · spec: `specs/$ARGUMENTS.md` · contract:
41
- > `<contract.path>/$ARGUMENTS.<ext>` · report: `specs/reports/$ARGUMENTS.md` · checkout: `<worktree
42
- > path or main checkout>` · ports/db: `<slot info, or defaults>`."
43
-
44
- The gate hooks fire on the agent's Bash calls too — compose/migrate confirmations still reach the
45
- human; that's expected.
46
-
47
- ## 3. Relay the verdict
48
-
49
- - Print the agent's return as-is (verdict + ❌ lines + report path) — it is already minimal.
50
- - Append ONE metrics line to `pipeline-metrics.jsonl` (main-checkout path + rules in `/build` §4,
51
- `phase: "smoke"`), chaining the opt-in usage ping in the same Bash call (results = `PASS` or
52
- `FAIL:<n>` failing flows).
53
- - **PASS** → tell the human to run `/review $ARGUMENTS`. **FAIL** → the failures are findings: feed
54
- them to `/fix $ARGUMENTS`, re-run `/smoke` after. Either way the report is on disk —
55
- **recommend a `/clear`** before the next command.