cohorte 1.6.0 → 2.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/CHANGELOG.md +117 -2
  2. package/README.md +57 -57
  3. package/bin/cli.js +23 -15
  4. package/core/agents/implementer.template.md +3 -3
  5. package/core/agents/release.md +1 -1
  6. package/core/agents/review.md +3 -3
  7. package/core/commands/{audit.md → cohorte-audit.md} +3 -3
  8. package/core/commands/{brainstorm.md → cohorte-brainstorm.md} +4 -4
  9. package/core/commands/{build.md → cohorte-build.md} +15 -15
  10. package/core/commands/{doctor.md → cohorte-doctor.md} +17 -9
  11. package/core/commands/{fix.md → cohorte-fix.md} +15 -13
  12. package/core/commands/{init-pipeline.md → cohorte-init-pipeline.md} +1 -1
  13. package/core/commands/cohorte-loop.md +110 -0
  14. package/core/commands/{refactor.md → cohorte-refactor.md} +3 -3
  15. package/core/commands/{review.md → cohorte-review.md} +20 -19
  16. package/core/commands/{ship.md → cohorte-ship.md} +5 -5
  17. package/core/commands/{spec.md → cohorte-spec.md} +13 -13
  18. package/core/commands/{update-pipeline.md → cohorte-update-pipeline.md} +11 -6
  19. package/core/hooks/gate.py +101 -6
  20. package/core/templates/brainstorm-return.md +4 -4
  21. package/core/templates/decisions.template.md +1 -1
  22. package/core/templates/design-brief.md +1 -1
  23. package/core/templates/spec.template.md +7 -7
  24. package/core/templates/steps/init-pipeline/01-detect-stack.md +1 -1
  25. package/core/templates/steps/init-pipeline/02-interview-gaps.md +6 -6
  26. package/core/templates/steps/init-pipeline/03-draft-profile.md +1 -1
  27. package/core/templates/steps/init-pipeline/04-write-render.md +16 -12
  28. package/core/templates/steps/init-pipeline/05-report.md +5 -5
  29. package/core/workflows/audit.js +6 -6
  30. package/core/workflows/refactor.js +14 -14
  31. package/core/workflows/review.js +22 -22
  32. package/dashboard/README.md +2 -2
  33. package/dashboard/dist/assets/{index-DYyn4p93.js → index-P1I1JGtj.js} +2 -2
  34. package/dashboard/dist/index.html +1 -1
  35. package/dashboard/server/doctor.js +69 -19
  36. package/dashboard/server/index.js +5 -5
  37. package/dashboard/server/metrics.js +1 -1
  38. package/install.ps1 +23 -14
  39. package/install.sh +24 -14
  40. package/package.json +2 -2
  41. package/profile/PIPELINE.template.md +17 -16
  42. package/profile/SCHEMA.md +89 -77
  43. package/profile/cohorte.config.template.yaml +8 -8
  44. package/scripts/loop-detach.sh +153 -0
  45. package/scripts/loop.sh +110 -29
  46. package/scripts/metrics/collect.mjs +17 -8
  47. package/scripts/new-feature.sh.template +3 -3
  48. package/scripts/preflight.sh +40 -4
  49. package/scripts/remove-feature.sh.template +2 -2
  50. package/scripts/test-dashboard.mjs +34 -7
  51. package/scripts/test-gate.mjs +58 -0
  52. package/scripts/test-loop.mjs +123 -20
  53. package/scripts/test-metrics.mjs +23 -11
  54. package/scripts/test-workflows.mjs +7 -7
  55. package/scripts/validate-core.mjs +45 -23
  56. package/core/commands/drive.md +0 -80
  57. /package/core/commands/{align-ds.md → cohorte-align-ds.md} +0 -0
@@ -2,8 +2,8 @@
2
2
  // Tests for the dashboard's server modules (dashboard/server/*.js).
3
3
  //
4
4
  // These are shipped runtime code with real logic and zero coverage until now:
5
- // a hand-rolled YAML parser that every /doctor check is derived from, a metrics
6
- // aggregator, the JS port of /doctor, an Obsidian board parser, the fleet
5
+ // a hand-rolled YAML parser that every /cohorte-doctor check is derived from, a metrics
6
+ // aggregator, the JS port of /cohorte-doctor, an Obsidian board parser, the fleet
7
7
  // registry, and an HTTP layer whose guards are the dashboard's only defence
8
8
  // against a web page driving the local agent.
9
9
  //
@@ -127,7 +127,7 @@ console.log("usage.js — the collector bridge");
127
127
  }
128
128
 
129
129
  // ── doctor.js ────────────────────────────────────────────────────────────────
130
- console.log("doctor.js — the /doctor port");
130
+ console.log("doctor.js — the /cohorte-doctor port");
131
131
  {
132
132
  const spec = (fm) => `---\n${fm}\n---\n\n# x\n`;
133
133
  const d = scratch();
@@ -136,13 +136,13 @@ console.log("doctor.js — the /doctor port");
136
136
  writeFileSync(join(d, "specs", "b.md"), spec("feature_id: b\nstatus: shipped # done"));
137
137
  writeFileSync(join(d, "specs", "c.md"), "no front-matter at all");
138
138
  writeFileSync(join(d, "specs", "_template.md"), spec("status: draft"));
139
- // /audit writes this file by design and it has no front-matter. Scanning it as a
140
- // spec made /doctor warn about a file cohorte itself had just created — it fired in
141
- // every project that had ever run /audit.
139
+ // /cohorte-audit writes this file by design and it has no front-matter. Scanning it as a
140
+ // spec made /cohorte-doctor warn about a file cohorte itself had just created — it fired in
141
+ // every project that had ever run /cohorte-audit.
142
142
  writeFileSync(join(d, "specs", "refactor-backlog.md"), "# Refactor Backlog\n\n## backend\n- [ ] x\n");
143
143
  const specs = scanSpecs(d);
144
144
  eq("_template.md is excluded", specs.length, 3);
145
- eq("the /audit backlog is not scanned as a spec",
145
+ eq("the /cohorte-audit backlog is not scanned as a spec",
146
146
  specs.some(s => s.file === "refactor-backlog.md"), false);
147
147
  eq("front-matter fields are read", specs.find(s => s.id === "a").title, "A");
148
148
  eq("a trailing comment is stripped from status", specs.find(s => s.id === "b").status, "shipped");
@@ -201,6 +201,21 @@ console.log("doctor.js — the /doctor port");
201
201
  eq("retrieval wired in .mcp.json ⇒ ok", by(s.checks, "retrieval").status, "ok");
202
202
  eq("workflows + profile-reader ⇒ ok", by(s.checks, "workflows").status, "ok");
203
203
 
204
+ // Local artifacts: a versioned preflight stamp is what made the phase gate ask on
205
+ // every review dispatch forever, so its absence from .gitignore is a hard failure.
206
+ check("no .gitignore ⇒ local artifacts flagged bad (the stamp is the breaking one)",
207
+ by(s.checks, "artifacts").status === "bad"
208
+ && /preflight\.ok/.test(by(s.checks, "artifacts").detail),
209
+ by(s.checks, "artifacts").detail);
210
+ writeFileSync(join(d, ".gitignore"),
211
+ "node_modules/\n.claude/preflight.ok\n.claude/pipeline-metrics.jsonl\nspecs/reports/\n");
212
+ s = await state({ projectRoot: d, globalDir: g, cliVersion: "9.9.9" });
213
+ eq("all local artifacts gitignored ⇒ ok", by(s.checks, "artifacts").status, "ok");
214
+ writeFileSync(join(d, ".gitignore"), "node_modules/\n.claude/\nspecs/reports/\n");
215
+ s = await state({ projectRoot: d, globalDir: g, cliVersion: "9.9.9" });
216
+ eq("a `.claude/` directory rule covers the files inside it",
217
+ by(s.checks, "artifacts").status, "ok");
218
+
204
219
  // …and each check must actually FAIL when its precondition breaks.
205
220
  writeFileSync(join(d, ".claude", "gate-config.json"),
206
221
  JSON.stringify({ ...gate, preflight: { enabled: false } }));
@@ -368,6 +383,18 @@ console.log("index.js — HTTP guards");
368
383
  const badCmd = await post({ action: "claude", command: "/evil", project: proj });
369
384
  eq("a non-whitelisted slash command is rejected", badCmd.status, 400);
370
385
 
386
+ // Both directions, because testing only the rejection missed a real bug: 2.0.0 prefixed
387
+ // every command, the error message was updated to say `/cohorte-audit`, but the allowlist
388
+ // regex still matched the bare names — so the server accepted the one command that no
389
+ // longer exists and rejected the only one the UI can send. A rejection-only test is blind
390
+ // to an allowlist that drifts away from the client.
391
+ const staleCmd = await post({ action: "claude", command: "/audit", project: proj });
392
+ eq("the pre-2.0.0 unprefixed command is rejected", staleCmd.status, 400);
393
+
394
+ const goodCmd = await post({ action: "claude", command: "/cohorte-audit", project: proj });
395
+ check("a prefixed whitelisted command passes the allowlist",
396
+ goodCmd.status !== 400 || !/unsupported command/.test((await goodCmd.json()).error || ""));
397
+
371
398
  eq("a missing hashed asset 404s (never index.html)",
372
399
  (await fetch(`${base}/assets/index-DEADBEEF.js`)).status, 404);
373
400
  eq("a malformed percent-escape is a 400, not a 500",
@@ -231,6 +231,64 @@ console.log("gate.py — preflight phase gate");
231
231
  run(task("review"), { projectDir: noblock }).decision === null);
232
232
  }
233
233
 
234
+ // ── the content digest (2.0.0): freshness keyed on code, not on HEAD ─────────
235
+ // Before this, the stamp recorded the HEAD sha — backwards on both sides. The
236
+ // reviewed tree is normally DIRTY, so committing already-verified code made the
237
+ // gate ask on a clean tree (and a committed stamp made it ask forever), while an
238
+ // implementer's edit between preflight and dispatch invalidated nothing.
239
+ console.log("gate.py — preflight content digest");
240
+ {
241
+ const pf = { enabled: true, agents: ["review"], max_age_minutes: 30 };
242
+ const d = scratch(); writeConfig(d, { ...GATE_CFG, preflight: pf });
243
+ gitRepo(d, "main");
244
+ mkdirSync(join(d, "specs", "reports"), { recursive: true });
245
+ writeFileSync(join(d, "specs", "s.md"), "spec\n");
246
+ writeFileSync(join(d, "src.txt"), "code v1\n"); // uncommitted feature work
247
+ const git = (...a) => execFileSync("git", a, { cwd: d, stdio: "ignore" });
248
+ const at = { projectDir: d };
249
+ const runPreflight = () =>
250
+ spawnSync("sh", [join(root, "scripts", "preflight.sh"), join(d, "specs", "reports", "r.txt"), "true"],
251
+ { cwd: d, encoding: "utf8" });
252
+
253
+ const pre = runPreflight();
254
+ const raw = execFileSync("cat", [join(d, ".claude", "preflight.ok")], { encoding: "utf8" }).trim();
255
+ check("preflight.sh stamps three fields (epoch, sha, digest)",
256
+ raw.split(/\s+/).length === 3, `${pre.status}: ${raw}`);
257
+ check("fresh stamp on a dirty tree ⇒ passes", run(task("review"), at).decision === null);
258
+
259
+ // The regression that started this: commit the very code the preflight verified.
260
+ git("add", "-A"); git("commit", "-qm", "wip");
261
+ const afterCommit = run(task("review"), at);
262
+ check("committing the verified code ⇒ still passes (HEAD moved, code did not)",
263
+ afterCommit.decision === null, `got ${afterCommit.decision} — ${afterCommit.reason}`);
264
+
265
+ // The pipeline's own writes must never invalidate its own stamp.
266
+ writeFileSync(join(d, "specs", "s.md"), "spec + DoD ticks\n");
267
+ writeFileSync(join(d, "specs", "reports", "r2.txt"), "report\n");
268
+ writeFileSync(join(d, ".claude", "pipeline-metrics.jsonl"), "{}\n");
269
+ check("spec ticks, report buffer and metrics writes ⇒ still passes",
270
+ run(task("review"), at).decision === null);
271
+
272
+ // …and a real edit must.
273
+ writeFileSync(join(d, "src.txt"), "code v2\n");
274
+ const edited = run(task("review"), at);
275
+ check("an uncommitted code edit ⇒ ask", edited.decision === "ask", edited.decision);
276
+ check("…and the reason says the code changed", /code changed/.test(edited.reason || ""));
277
+
278
+ // A brand-new untracked source file is a code change too (the sha never saw these).
279
+ writeFileSync(join(d, "src.txt"), "code v1\n");
280
+ writeFileSync(join(d, "extra.txt"), "new surface\n");
281
+ check("a new untracked source file ⇒ ask", run(task("review"), at).decision === "ask");
282
+ rmSync(join(d, "extra.txt"));
283
+ check("reverting to the verified content ⇒ passes again",
284
+ run(task("review"), at).decision === null);
285
+
286
+ // The hook must never touch the caller's index — it computes in a throwaway one.
287
+ const status = execFileSync("git", ["status", "--porcelain"], { cwd: d, encoding: "utf8" });
288
+ check("the gate leaves the real index untouched (nothing staged)",
289
+ !/^[MARCD]/m.test(status), status.trim());
290
+ }
291
+
234
292
  // ── worktree awareness (the 1.3.3 known_heads fix) ───────────────────────────
235
293
  console.log("gate.py — worktree awareness");
236
294
  {
@@ -1,11 +1,11 @@
1
1
  #!/usr/bin/env node
2
- // Behavioural tests for scripts/loop.sh — the autonomous /review ⇄ /fix driver.
2
+ // Behavioural tests for scripts/loop.sh — the autonomous /cohorte-review ⇄ /cohorte-fix driver.
3
3
  //
4
4
  // The driver is pure shell around two JSON files it does not write, so it is
5
5
  // testable end-to-end by putting a FAKE `claude` on PATH that produces those files
6
6
  // per phase. What is pinned here cannot be seen by any structural check:
7
7
  //
8
- // · exit 4 — /build's readiness gate said NOT-READY, so no pass count helps
8
+ // · exit 4 — /cohorte-build's readiness gate said NOT-READY, so no pass count helps
9
9
  // · exit 0/3 leave the right TERMINAL status in the spec's front-matter, which is
10
10
  // what makes an interrupted loop resumable (SCHEMA.md §Spec status)
11
11
  // · the front-matter stamps are written with awk on every platform — a `sed -i`
@@ -50,7 +50,18 @@ prompt=""
50
50
  while [ $# -gt 0 ]; do
51
51
  case "$1" in -p) prompt="$2"; shift 2 ;; *) shift ;; esac
52
52
  done
53
- cmd="\${prompt%% *}"; cmd="\${cmd#/}"
53
+ cmd="\${prompt%% *}"
54
+ # The driver must dispatch the PREFIXED command (2.0.0) — an unprefixed /build would be
55
+ # shadowed by Claude Code's own built-in and never reach the pipeline, so fail loudly
56
+ # rather than let a regression pass by being lenient here.
57
+ case "$cmd" in
58
+ /cohorte-*) ;;
59
+ *) echo "fake claude: expected a /cohorte-* command, got '$cmd'" >&2; exit 9 ;;
60
+ esac
61
+ cmd="\${cmd#/cohorte-}" # scenarios are keyed on the PHASE, which stays unprefixed
62
+ # Echoed so the tests can assert what the driver hands its children: an unattended child
63
+ # that cannot answer a permission prompt, and a background ceiling that must not fire.
64
+ echo "fake claude: bgceil=\${CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS:-unset}"
54
65
  n=0; [ -f "$SCEN_DIR/count" ] && n=$(cat "$SCEN_DIR/count")
55
66
  n=$((n + 1)); echo "$n" >"$SCEN_DIR/count"
56
67
  step=$(sed -n "\${n}p" "$SCEN_DIR/phases")
@@ -64,6 +75,13 @@ case "$do_what" in
64
75
  >specs/reports/feat-x.readiness.json ;;
65
76
  ready)
66
77
  printf '{"id":"feat-x","phase":"readiness","ts":"t","verdict":"READY","gaps":[],"surfaces":["backend"]}' \\
78
+ >specs/reports/feat-x.readiness.json
79
+ printf '{"id":"feat-x","phase":"build","ts":"t","surfaces":{"backend":"ok"},"dead":[]}' \\
80
+ >specs/reports/feat-x.build.json ;;
81
+ # READY, dispatched, and then cut short before §3's report — the harness terminating
82
+ # background implementers, a teardown, a crash. No build.json, and exit 0 anyway.
83
+ cutshort)
84
+ printf '{"id":"feat-x","phase":"readiness","ts":"t","verdict":"READY","gaps":[],"surfaces":["backend","frontend"]}' \\
67
85
  >specs/reports/feat-x.readiness.json ;;
68
86
  clean)
69
87
  printf '{"id":"feat-x","phase":"review","ts":"t","verdict":"SHIP","findings":2,"blocking":0,"deferred":2,"unreviewed":[],"fingerprint":""}' \\
@@ -106,24 +124,26 @@ function scenario(phases, { frontmatter = FM } = {}) {
106
124
  }
107
125
 
108
126
  function runLoop({ dir, bin }, args) {
109
- const r = spawnSync("bash", [LOOP, ...args], {
110
- cwd: dir,
111
- encoding: "utf8",
112
- env: {
113
- ...process.env,
114
- PATH: `${bin}:${process.env.PATH}`,
115
- SCEN_DIR: dir,
116
- CLAUDE_FLAGS: "--permission-mode acceptEdits",
117
- GIT_AUTHOR_NAME: "t", GIT_AUTHOR_EMAIL: "t@t.t",
118
- GIT_COMMITTER_NAME: "t", GIT_COMMITTER_EMAIL: "t@t.t",
119
- },
120
- });
127
+ const env = {
128
+ ...process.env,
129
+ PATH: `${bin}:${process.env.PATH}`,
130
+ SCEN_DIR: dir,
131
+ GIT_AUTHOR_NAME: "t", GIT_AUTHOR_EMAIL: "t@t.t",
132
+ GIT_COMMITTER_NAME: "t", GIT_COMMITTER_EMAIL: "t@t.t",
133
+ };
134
+ // Never inherit these from whoever runs the suite: the default child flags and the
135
+ // background ceiling are exactly what the assertions below are about.
136
+ delete env.CLAUDE_FLAGS;
137
+ delete env.CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS;
138
+ const r = spawnSync("bash", [LOOP, ...args], { cwd: dir, encoding: "utf8", env });
121
139
  const spec = readFileSync(join(dir, "specs/feat-x.md"), "utf8");
122
140
  const fm = k => {
123
141
  const m = spec.match(new RegExp(`^${k}:\\s*([^#\\n]*)`, "m"));
124
142
  return m ? m[1].trim() : null;
125
143
  };
126
- return { code: r.status, out: `${r.stdout}${r.stderr}`, spec, fm };
144
+ const logPath = join(dir, "specs/reports/feat-x.loop.log");
145
+ const log = existsSync(logPath) ? readFileSync(logPath, "utf8") : "";
146
+ return { code: r.status, out: `${r.stdout}${r.stderr}`, spec, fm, log };
127
147
  }
128
148
 
129
149
  console.log("loop.sh — readiness gate");
@@ -133,7 +153,7 @@ console.log("loop.sh — readiness gate");
133
153
  check("NOT-READY ⇒ exit 4, not 2", r.code === 4, `got ${r.code}: ${r.out.trim().split("\n").pop()}`);
134
154
  check("NOT-READY ⇒ says the spec is not implementable",
135
155
  /not implementable/i.test(r.out), r.out.trim().split("\n").pop());
136
- check("NOT-READY ⇒ points at /spec", /\/spec feat-x/.test(r.out));
156
+ check("NOT-READY ⇒ points at /cohorte-spec", /\/cohorte-spec feat-x/.test(r.out));
137
157
  check("NOT-READY ⇒ spec left blocked", r.fm("status") === "blocked", r.fm("status"));
138
158
  check("NOT-READY ⇒ no review ran (the gate is the point)", !/phase=review/.test(r.out));
139
159
  check("NOT-READY ⇒ the build stamp is NOT written",
@@ -145,7 +165,7 @@ console.log("loop.sh — clean run");
145
165
  const s = scenario(["build:ready", "review:clean"]);
146
166
  const r = runLoop(s, ["feat-x"]);
147
167
  check("clean ⇒ exit 0", r.code === 0, `got ${r.code}: ${r.out}`);
148
- check("clean ⇒ status in-review (ready to /ship)", r.fm("status") === "in-review", r.fm("status"));
168
+ check("clean ⇒ status in-review (ready to /cohorte-ship)", r.fm("status") === "in-review", r.fm("status"));
149
169
  check("clean ⇒ loop state cleared", r.fm("loop_pass") === "0" && r.fm("loop_phase") === "done",
150
170
  `${r.fm("loop_pass")}/${r.fm("loop_phase")}`);
151
171
  check("clean ⇒ the deferred count is named, not dropped",
@@ -156,7 +176,7 @@ console.log("loop.sh — clean run");
156
176
 
157
177
  console.log("loop.sh — a dead subagent is never a clean result");
158
178
  {
159
- // A dead implementer: /build finishes fine having built one surface of two. Reviewing
179
+ // A dead implementer: /cohorte-build finishes fine having built one surface of two. Reviewing
160
180
  // that would spend N reviewers auditing a half-built feature and report its holes as
161
181
  // findings to fix — the wrong diagnosis at the wrong price.
162
182
  const s = scenario(["build:deadimplementer"]);
@@ -170,7 +190,7 @@ console.log("loop.sh — a dead subagent is never a clean result");
170
190
  {
171
191
  // THE dangerous one: blocking == 0 because the only reviewer that could have found
172
192
  // something never answered. Exiting 0 here would report "clean" about unread code and
173
- // send the human to /ship.
193
+ // send the human to /cohorte-ship.
174
194
  const s = scenario(["build:ready", "review:deadreviewer"]);
175
195
  const r = runLoop(s, ["feat-x"]);
176
196
  check("dead reviewer + blocking 0 ⇒ NOT exit 0", r.code !== 0, `got ${r.code}: ${r.out}`);
@@ -181,6 +201,55 @@ console.log("loop.sh — a dead subagent is never a clean result");
181
201
  check("dead reviewer ⇒ spec is NOT left in-review", r.fm("status") === "blocked", r.fm("status"));
182
202
  }
183
203
 
204
+ console.log("loop.sh — a build that never reported is never a built build");
205
+ {
206
+ // The absent-file twin of the dead implementer, and the one the `dead[]` grep cannot
207
+ // see: a phase cut short never reaches the step that writes build.json, so there is no
208
+ // file to read and no surface to name — while the child still exits 0. Scoring that as
209
+ // a clean build stamps `.built` over a half-written tree and sends reviewers at it.
210
+ const s = scenario(["build:cutshort"]);
211
+ const r = runLoop(s, ["feat-x"]);
212
+ check("no build.json ⇒ exit 2, not a review pass", r.code === 2, `got ${r.code}: ${r.out}`);
213
+ check("no build.json ⇒ no reviewer was spawned", !/phase=review/.test(r.out));
214
+ check("no build.json ⇒ names the cut-short phase", /wrote no .*build\.json/.test(r.out),
215
+ r.out.trim().split("\n").pop());
216
+ check("no build.json ⇒ the build stamp is NOT written (a re-run must rebuild)",
217
+ !existsSync(join(s.dir, "specs/reports/feat-x.built")));
218
+ check("no build.json ⇒ spec left blocked", r.fm("status") === "blocked", r.fm("status"));
219
+ }
220
+
221
+ console.log("loop.sh — what the children are handed");
222
+ {
223
+ // acceptEdits auto-approves Write/Edit and NOTHING else, so the first child Bash call no
224
+ // `allow` rule covers raises a prompt no `claude -p` can answer: the child stalls, asks
225
+ // the human in prose, and exits 0 — which the driver scores `ok`. Seen on a real run,
226
+ // where the review child hung on its own preflight.sh call. gate.py is built for the
227
+ // other mode: it escalates `ask` to a hard deny under bypassPermissions.
228
+ const s = scenario(["build:ready", "review:clean"]);
229
+ const r = runLoop(s, ["feat-x"]);
230
+ check("default child flags are bypassPermissions, not acceptEdits",
231
+ /# flags: --permission-mode bypassPermissions/.test(r.log) && !/acceptEdits/.test(r.log),
232
+ r.log.split("\n")[1]);
233
+ // Print mode TERMINATES still-running background tasks at its ceiling ("Background tasks
234
+ // still running after 600s"), which cuts a 25–40 min implementer batch off mid-write.
235
+ check("children inherit an unbounded background-task ceiling",
236
+ /fake claude: bgceil=0/.test(r.log), (r.log.match(/bgceil=\S*/) || ["absent"])[0]);
237
+ }
238
+ {
239
+ // The override is the escape hatch for a watched run — it must not have been hard-coded away.
240
+ const s = scenario(["review:clean"]);
241
+ const r = spawnSync("bash", [LOOP, "feat-x", "--no-build"], {
242
+ cwd: s.dir, encoding: "utf8",
243
+ env: { ...process.env, PATH: `${s.bin}:${process.env.PATH}`, SCEN_DIR: s.dir,
244
+ CLAUDE_FLAGS: "--permission-mode acceptEdits",
245
+ GIT_AUTHOR_NAME: "t", GIT_AUTHOR_EMAIL: "t@t.t",
246
+ GIT_COMMITTER_NAME: "t", GIT_COMMITTER_EMAIL: "t@t.t" },
247
+ });
248
+ const log = readFileSync(join(s.dir, "specs/reports/feat-x.loop.log"), "utf8");
249
+ check("CLAUDE_FLAGS overrides the default", /# flags: --permission-mode acceptEdits/.test(log),
250
+ `${r.status}: ${log.split("\n")[1]}`);
251
+ }
252
+
184
253
  console.log("loop.sh — non-convergent + resume");
185
254
  {
186
255
  const s = scenario(["build:ready", "review:blocking", "fix:noop", "review:blocking"]);
@@ -223,5 +292,39 @@ console.log("loop.sh — a spec with no front-matter still runs");
223
292
  !existsSync(join(s.dir, "specs/feat-x.md.loop.tmp")));
224
293
  }
225
294
 
295
+ // ── the sleep inhibitor must never be able to fail the run ───────────────────
296
+ // loop.sh re-execs itself under caffeinate/systemd-inhibit to hold a power assertion.
297
+ // `exec` replaces the shell, so an inhibitor that EXISTS but is refused makes its own
298
+ // failure the driver's exit code and the run never starts. CI found this the hard way:
299
+ // GitHub's Linux runners ship systemd-inhibit and answer "Failed to inhibit: Access
300
+ // denied", which turned all 24 loop tests red at once.
301
+ console.log("loop.sh — the sleep inhibitor is best-effort, never fatal");
302
+ {
303
+ const s = scenario(["build:ready", "review:clean"]);
304
+ // Both inhibitors present on PATH and both failing — the CI shape.
305
+ writeFileSync(join(s.bin, "systemd-inhibit"),
306
+ '#!/bin/sh\necho "Failed to inhibit: Access denied" >&2\nexit 1\n');
307
+ chmodSync(join(s.bin, "systemd-inhibit"), 0o755);
308
+ writeFileSync(join(s.bin, "caffeinate"), "#!/bin/sh\nexit 127\n");
309
+ chmodSync(join(s.bin, "caffeinate"), 0o755);
310
+ const r = runLoop(s, ["feat-x"]);
311
+ check("a refused inhibitor ⇒ the run still completes clean", r.code === 0,
312
+ `got ${r.code}: ${r.out.trim().split("\n").pop()}`);
313
+ check("a refused inhibitor ⇒ its error never reaches the driver's output",
314
+ !/Access denied/.test(r.out), r.out.trim().split("\n").pop());
315
+
316
+ // A WORKING inhibitor must still be used (or the probe would have disabled the feature).
317
+ const s2 = scenario(["build:ready", "review:clean"]);
318
+ writeFileSync(join(s2.bin, "systemd-inhibit"),
319
+ '#!/bin/sh\nwhile [ $# -gt 0 ]; do case "$1" in --*) shift ;; *) break ;; esac; done\n'
320
+ + 'echo "INHIBIT-HELD" >&2\nexec "$@"\n');
321
+ chmodSync(join(s2.bin, "systemd-inhibit"), 0o755);
322
+ writeFileSync(join(s2.bin, "caffeinate"), "#!/bin/sh\nexit 127\n");
323
+ chmodSync(join(s2.bin, "caffeinate"), 0o755);
324
+ const r2 = runLoop(s2, ["feat-x"]);
325
+ check("a usable inhibitor is still exec'd (the probe didn't kill the feature)",
326
+ /INHIBIT-HELD/.test(r2.out) && r2.code === 0, `${r2.code}: ${r2.out.trim().split("\n").pop()}`);
327
+ }
328
+
226
329
  if (failures) { console.error(`\ntest-loop: ${failures} failure(s)`); process.exit(1); }
227
330
  console.log("\ntest-loop: OK");
@@ -59,7 +59,7 @@ const usageOpus = {
59
59
  };
60
60
 
61
61
  const lines = [
62
- user(0, '<command-message>build</command-message>\n<command-name>/build</command-name>'),
62
+ user(0, '<command-message>build</command-message>\n<command-name>/cohorte-build</command-name>'),
63
63
  // Case 1: one response, three lines, identical usage on each. Only one should be billed.
64
64
  assistant('m1', 5, 'claude-opus-5', usageOpus, [{ type: 'thinking', thinking: '...' }]),
65
65
  assistant('m1', 5, 'claude-opus-5', usageOpus, [{ type: 'text', text: 'hello' }]),
@@ -74,26 +74,34 @@ const lines = [
74
74
  assistant('m4', 605, 'claude-opus-5', { input_tokens: 0, output_tokens: 40 }),
75
75
  // Case 6: a command named inside ordinary prose. The harness emits no <command-name>
76
76
  // for this, but it is the way commands actually get invoked in practice.
77
- user(1200, 'move on branding-ramp and /review'),
77
+ user(1200, 'move on branding-ramp and /cohorte-review'),
78
78
  assistant('m5', 1205, 'claude-opus-5', { input_tokens: 0, output_tokens: 60 }),
79
- // Case 7: a short steer continues the /review rather than opening an anonymous run.
79
+ // Case 7: a short steer continues the /cohorte-review rather than opening an anonymous run.
80
80
  user(1260, 'continue'),
81
81
  assistant('m6', 1265, 'claude-opus-5', { input_tokens: 0, output_tokens: 70 }),
82
82
  // Case 8: a slash token that is not a command must not invent one.
83
83
  user(1800, 'look at the /usr/local/share directory and report what you find there'),
84
84
  assistant('m7', 1805, 'claude-opus-5', { input_tokens: 0, output_tokens: 10 }),
85
85
  // Case 9: a long prompt that merely DISCUSSES a command is not an invocation of it.
86
- // Without the length gate, writing about /review bills the conversation to /review —
86
+ // Without the length gate, writing about /cohorte-review bills the conversation to /cohorte-review —
87
87
  // which is what happened in cohorte's own repo while the pipeline was being designed.
88
- user(2400, 'I want to talk through how /review behaves when a surface has no findings at '
88
+ user(2400, 'I want to talk through how /cohorte-review behaves when a surface has no findings at '
89
89
  + 'all, because the verdict logic there is what produced the false green we saw last week '
90
90
  + 'and I am not convinced the fix covers the case where every reviewer dies at once.'),
91
91
  assistant('m8', 2405, 'claude-opus-5', { input_tokens: 0, output_tokens: 20 }),
92
+ // Case 10: a RETIRED command name still attributes to itself. 2.0.0 prefixed every
93
+ // command, so months of existing transcripts say `/build` — and the collector reads its
94
+ // known names off the shipped core, where `build.md` no longer exists. Without the
95
+ // retired list every one of those runs silently reclassifies to (chat), rewriting spend
96
+ // history and inflating the catch-all. This is the largest instance of that bug class,
97
+ // so it gets pinned rather than trusted to a comment.
98
+ user(3000, '/build branding-ramp'),
99
+ assistant('m9', 3005, 'claude-opus-5', { input_tokens: 0, output_tokens: 90 }),
92
100
  ];
93
101
  fs.writeFileSync(path.join(projectDir, `${SESSION}.jsonl`),
94
102
  lines.map((l) => JSON.stringify(l)).join('\n') + '\n');
95
103
 
96
- // Case 3: subagent spend, linked back to /build by the Task tool_use id.
104
+ // Case 3: subagent spend, linked back to /cohorte-build by the Task tool_use id.
97
105
  const agentDir = path.join(projectDir, SESSION, 'subagents');
98
106
  fs.writeFileSync(path.join(agentDir, 'agent-a1.meta.json'),
99
107
  JSON.stringify({ agentType: 'core', description: 'Build core surface', toolUseId: 'toolu_A', spawnDepth: 1 }));
@@ -109,13 +117,14 @@ if (run.status !== 0) {
109
117
  process.exit(1);
110
118
  }
111
119
  const out = JSON.parse(run.stdout);
112
- const build = out.commands.find((c) => c.command === '/build');
120
+ const build = out.commands.find((c) => c.command === '/cohorte-build');
113
121
  const chat = out.commands.find((c) => c.command === '(chat)');
114
- const review = out.commands.find((c) => c.command === '/review');
122
+ const retired = out.commands.find((c) => c.command === '/build');
123
+ const review = out.commands.find((c) => c.command === '/cohorte-review');
115
124
 
116
125
  console.log('test-metrics');
117
- check('the mid-command task-notification did not split the run', out.totals.runs, 5);
118
- check('/build is one run, not three', build.runs, 1);
126
+ check('the mid-command task-notification did not split the run', out.totals.runs, 6);
127
+ check('/cohorte-build is one run, not three', build.runs, 1);
119
128
  check('duplicate lines of one response are billed once', build.tokens.output, 1000 + 500 + 2000);
120
129
  check('the <synthetic> message contributed no tokens', build.tokens.output < 999999, true);
121
130
  check('cache-write tokens are kept on their own tier', build.tokens.cacheWrite5m, 1000);
@@ -128,6 +137,9 @@ check('the continued turn counts toward the command it continued', review.tokens
128
137
  check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10 + 20);
129
138
  check('a long prompt that discusses a command is not counted as running it',
130
139
  review.runs, 1);
140
+ check('a retired unprefixed command stays attributed to itself, not (chat)',
141
+ retired && retired.runs, 1);
142
+ check('…and keeps its own spend', retired && retired.tokens.output, 90);
131
143
 
132
144
  // opus-5 $5 in / $25 out per MTok; 5m cache write 1.25x input, cache read 0.1x input.
133
145
  // m1 100*5 + 1000*25 + 1000*6.25 + 10000*0.5 = 36750
@@ -136,7 +148,7 @@ check('a long prompt that discusses a command is not counted as running it',
136
148
  check('cost sums the cache tiers at their own rates', Number(build.cost.total.toFixed(6)), 0.07925);
137
149
  check('the unpriced list stays empty for known models', build.unpriced, []);
138
150
 
139
- const detail = out.runs.find((r) => r.command === '/build');
151
+ const detail = out.runs.find((r) => r.command === '/cohorte-build');
140
152
  check('per-run detail carries the subagent', detail.agents.map((a) => a.type), ['core']);
141
153
 
142
154
  fs.rmSync(tmp, { recursive: true, force: true });
@@ -105,7 +105,7 @@ console.log("review.js");
105
105
  ]));
106
106
  check("clean run ⇒ SHIP", result.verdict === "SHIP", `got ${result.verdict}`);
107
107
  check("clean run ⇒ no unreviewed surfaces", (result.unreviewedSurfaces || []).length === 0);
108
- check("clean run ⇒ next is /ship", String(result.next).startsWith("/ship"), result.next);
108
+ check("clean run ⇒ next is /cohorte-ship", String(result.next).startsWith("/cohorte-ship"), result.next);
109
109
  }
110
110
  {
111
111
  // THE regression: every reviewer dies ⇒ zero findings ⇒ must NOT read as SHIP.
@@ -132,19 +132,19 @@ console.log("review.js");
132
132
  }
133
133
  {
134
134
  // A SHIP carrying HIGH findings is a real verdict, but it is not "go ship it":
135
- // the conversational /review routes any surviving HIGH to /fix.
135
+ // the conversational /cohorte-review routes any surviving HIGH to /cohorte-fix.
136
136
  const { result } = await run("review.js", replier([
137
137
  ["review:", { verdict: "SHIP", findings: [finding()] }], ...BASE_REVIEW,
138
138
  ]));
139
139
  check("SHIP + HIGH findings ⇒ verdict still SHIP", result.verdict === "SHIP");
140
- check("SHIP + HIGH findings ⇒ next routes to /fix, not /ship",
141
- String(result.next).startsWith("/fix"), result.next);
140
+ check("SHIP + HIGH findings ⇒ next routes to /cohorte-fix, not /cohorte-ship",
141
+ String(result.next).startsWith("/cohorte-fix"), result.next);
142
142
  }
143
143
  {
144
144
  const { result } = await run("review.js", replier([
145
145
  ["review:", { verdict: "SHIP", findings: [finding({ severity: "LOW" })] }], ...BASE_REVIEW,
146
146
  ]));
147
- check("SHIP + only LOW ⇒ next is /ship", String(result.next).startsWith("/ship"), result.next);
147
+ check("SHIP + only LOW ⇒ next is /cohorte-ship", String(result.next).startsWith("/cohorte-ship"), result.next);
148
148
  }
149
149
  {
150
150
  // Deferred findings are real but out of the feature's scope: they must be
@@ -163,8 +163,8 @@ console.log("review.js");
163
163
  return replier(BASE_REVIEW)(prompt, opts);
164
164
  });
165
165
  check("deferred-only ⇒ verdict still SHIP", result.verdict === "SHIP", `got ${result.verdict}`);
166
- check("deferred-only ⇒ next is /ship (not a fix loop)",
167
- String(result.next).startsWith("/ship"), result.next);
166
+ check("deferred-only ⇒ next is /cohorte-ship (not a fix loop)",
167
+ String(result.next).startsWith("/cohorte-ship"), result.next);
168
168
  check("deferred are counted (both surfaces)", result.deferred === 2, `got ${result.deferred}`);
169
169
  check("deferred stay out of the severity counts",
170
170
  Object.values(result.counts).every(n => n === 0), JSON.stringify(result.counts));
@@ -22,27 +22,28 @@ const frontmatter = (text) => {
22
22
  // Mechanical commands must pin model: sonnet (otherwise the lead's
23
23
  // orchestration turn silently bills at the session model — Opus/Fable).
24
24
  // Interactive commands must stay unpinned (they inherit on purpose).
25
- const PINNED = ["build", "review", "fix", "ship", "audit",
26
- "refactor", "doctor", "align-ds", "update-pipeline", "drive"];
27
- const UNPINNED = ["brainstorm", "spec", "init-pipeline"];
25
+ const PINNED = ["cohorte-build", "cohorte-review", "cohorte-fix", "cohorte-ship",
26
+ "cohorte-audit", "cohorte-refactor", "cohorte-doctor", "cohorte-align-ds",
27
+ "cohorte-update-pipeline", "cohorte-loop"];
28
+ const UNPINNED = ["cohorte-brainstorm", "cohorte-spec", "cohorte-init-pipeline"];
28
29
 
29
- // Names Claude Code itself claims. A core command that collides is not overridden —
30
- // it is SHADOWED: the built-in answers the slash, our command file is never read, and
31
- // the session confidently reports on a run that never happened. That is what `/loop`
32
- // did (Claude Code's own `/loop` runs a prompt on an interval), invisible until a user
33
- // noticed the driver had never started. `/loop` is here so the 1.6.0 rename to
34
- // `/drive` can never be quietly reverted.
35
- // Watchlist, not yet enforced because the collision is unproven: `doctor` (Claude Code
36
- // has its own `/doctor`) — if a typed `/doctor` ever stops reaching the pipeline's, add
37
- // it here and rename.
38
- const RESERVED = ["loop", "clear", "compact", "cost", "help", "config",
39
- "init", "run", "schedule", "simplify", "review-pr"];
30
+ // Every command must carry the `cohorte-` prefix. This replaces the old RESERVED
31
+ // blocklist, which chased collisions one name at a time and always lagged: a command
32
+ // that collides with a Claude Code built-in is not overridden, it is SHADOWED — the
33
+ // built-in answers the slash, our file is never read, and the session confidently
34
+ // reports on a run that never happened. `/loop` did exactly that (Claude Code's own
35
+ // `/loop` runs a prompt on an interval) and went unnoticed until a user found the
36
+ // driver had never started; `/doctor` sat on a watchlist waiting to do the same.
37
+ // A blocklist can only forbid the collisions we already know about. The prefix makes
38
+ // the whole class unreachable, so this check is structural, not a list to maintain.
39
+ const PREFIX = "cohorte-";
40
40
 
41
41
  for (const f of readdirSync(join(root, "core/commands"))) {
42
42
  const path = `core/commands/${f}`;
43
- if (RESERVED.includes(f.replace(/\.md$/, "")))
44
- fail(path, `command name collides with a Claude Code built-in — it would be SHADOWED ` +
45
- `(the built-in answers the slash and this file is never read); rename it`);
43
+ if (!f.startsWith(PREFIX))
44
+ fail(path, `command name lacks the \`${PREFIX}\` prefix — an unprefixed command can be ` +
45
+ `SHADOWED by a Claude Code built-in of the same name (the built-in answers the slash ` +
46
+ `and this file is never read); rename it to ${PREFIX}${f}`);
46
47
  const fm = frontmatter(read(path));
47
48
  if (!fm) { fail(path, "missing or malformed YAML frontmatter"); continue; }
48
49
  if (!/^description:\s*\S/m.test(fm)) fail(path, "frontmatter lacks a description");
@@ -130,16 +131,20 @@ if (!existsSync(steps) || readdirSync(steps).length === 0)
130
131
 
131
132
  // ── telemetry coverage ──────────────────────────────────────────────────────
132
133
  // The funnel is only readable if every one of its stages pings — a single missing
133
- // one silently truncates it (that is how /review and /fix went unreported
134
+ // one silently truncates it (that is how /cohorte-review and /cohorte-fix went unreported
134
135
  // until 1.2.3). The phase list here must match SCHEMA.md §Telemetry's table.
136
+ // These are telemetry PHASE names, not command names — they stay unprefixed even though
137
+ // the commands that emit them are now `/cohorte-*`. The phase is a wire field allowlisted
138
+ // in telemetry-send.sh and keyed on by the collector's existing dataset; prefixing it would
139
+ // orphan every ping ever sent. Command file = PREFIX + phase.
135
140
  const FUNNEL = ["brainstorm", "spec", "build", "review", "fix", "ship"];
136
141
  for (const c of FUNNEL)
137
- if (!/usage ping/i.test(read(`core/commands/${c}.md`)))
138
- fail(`core/commands/${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
142
+ if (!/usage ping/i.test(read(`core/commands/${PREFIX}${c}.md`)))
143
+ fail(`core/commands/${PREFIX}${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
139
144
  // …and nothing outside the funnel may ping (consent text scopes it to the funnel).
140
145
  for (const f of readdirSync(join(root, "core/commands"))) {
141
- const c = f.replace(/\.md$/, "");
142
- // `telemetry-send.sh` + an argument = a call site; the bare filename (e.g. /doctor
146
+ const c = f.replace(/\.md$/, "").replace(new RegExp(`^${PREFIX}`), "");
147
+ // `telemetry-send.sh` + an argument = a call site; the bare filename (e.g. /cohorte-doctor
143
148
  // listing the scripts it checks for) is a mention, not a ping.
144
149
  if (!FUNNEL.includes(c) && /telemetry-send\.sh +\S|usage ping/i.test(read(`core/commands/${f}`)))
145
150
  fail(`core/commands/${f}`, "non-funnel command pings telemetry — outside the consented scope");
@@ -154,7 +159,7 @@ for (const f of readdirSync(join(root, "core/commands"))) {
154
159
  // scratch HOME and asserts the same postconditions instead. Both are needed: this
155
160
  // check catches a forgotten name, that one catches a drifted rule.
156
161
  // A `<name>.sh` with a `<name>.sh.template` sibling is a locally-rendered artifact
157
- // (this repo dogfoods its own /init-pipeline), not a core asset — skip those.
162
+ // (this repo dogfoods its own /cohorte-init-pipeline), not a core asset — skip those.
158
163
  const installers = { "install.sh": read("install.sh"), "install.ps1": read("install.ps1") };
159
164
  const shipped = readdirSync(join(root, "scripts"));
160
165
  for (const f of shipped.filter((f) => f.endsWith(".sh") && !shipped.includes(`${f}.template`)))
@@ -218,6 +223,23 @@ for (const f of workflowNames) {
218
223
  fail("dashboard/server/doctor.js", `checkWorkflows() does not list ${f}`);
219
224
  }
220
225
 
226
+ // ── every test suite must run in BOTH workflows ──────────────────────────────
227
+ // publish.yml re-runs the test suites under the comment "same gate as CI", because
228
+ // it has no dependency on the CI workflow's conclusion — a merge whose CI failed
229
+ // would otherwise still ship to npm. That only holds if the two lists agree, and
230
+ // they drift the moment a suite is added to one: test-loop.mjs landed in ci.yml and
231
+ // publish.yml kept publishing without it. Neither list is the source of truth —
232
+ // the directory is.
233
+ const ciYml = existsSync(join(root, ".github/workflows/ci.yml")) ? read(".github/workflows/ci.yml") : "";
234
+ const publishYml = existsSync(join(root, ".github/workflows/publish.yml"))
235
+ ? read(".github/workflows/publish.yml") : "";
236
+ for (const f of readdirSync(join(root, "scripts")).filter((f) => /^test-.*\.mjs$/.test(f))) {
237
+ if (ciYml && !ciYml.includes(`scripts/${f}`))
238
+ fail(".github/workflows/ci.yml", `never runs scripts/${f} — a suite CI does not run is a suite that does not exist`);
239
+ if (publishYml && !publishYml.includes(`scripts/${f}`))
240
+ fail(".github/workflows/publish.yml", `never runs scripts/${f} — publish would ship past a failure that gate is meant to catch`);
241
+ }
242
+
221
243
  // ── dashboard: the metrics phase list is duplicated server/client ────────────
222
244
  // A phase present in one and not the other parses fine and renders in no column —
223
245
  // silently invisible data, which is how a phase batch once went unnoticed.