cohorte 1.5.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/CHANGELOG.md +169 -3
  2. package/README.md +65 -57
  3. package/bin/cli.js +31 -15
  4. package/core/agents/implementer.template.md +3 -3
  5. package/core/agents/release.md +1 -1
  6. package/core/agents/review.md +25 -2
  7. package/core/commands/{audit.md → cohorte-audit.md} +11 -3
  8. package/core/commands/{brainstorm.md → cohorte-brainstorm.md} +9 -3
  9. package/core/commands/{build.md → cohorte-build.md} +95 -10
  10. package/core/commands/{doctor.md → cohorte-doctor.md} +22 -9
  11. package/core/commands/{fix.md → cohorte-fix.md} +20 -13
  12. package/core/commands/{init-pipeline.md → cohorte-init-pipeline.md} +1 -1
  13. package/core/commands/cohorte-loop.md +110 -0
  14. package/core/commands/{refactor.md → cohorte-refactor.md} +3 -3
  15. package/core/commands/{review.md → cohorte-review.md} +70 -20
  16. package/core/commands/{ship.md → cohorte-ship.md} +5 -5
  17. package/core/commands/{spec.md → cohorte-spec.md} +32 -12
  18. package/core/commands/{update-pipeline.md → cohorte-update-pipeline.md} +16 -6
  19. package/core/hooks/gate.py +101 -6
  20. package/core/templates/brainstorm-return.md +4 -4
  21. package/core/templates/decisions.template.md +42 -0
  22. package/core/templates/design-brief.md +1 -1
  23. package/core/templates/spec.template.md +8 -6
  24. package/core/templates/steps/init-pipeline/01-detect-stack.md +1 -1
  25. package/core/templates/steps/init-pipeline/02-interview-gaps.md +6 -6
  26. package/core/templates/steps/init-pipeline/03-draft-profile.md +1 -1
  27. package/core/templates/steps/init-pipeline/04-write-render.md +16 -12
  28. package/core/templates/steps/init-pipeline/05-report.md +5 -5
  29. package/core/workflows/audit.js +6 -6
  30. package/core/workflows/refactor.js +14 -14
  31. package/core/workflows/review.js +62 -20
  32. package/dashboard/README.md +2 -2
  33. package/dashboard/dist/assets/{index-dkO8UUVl.css → index-BZ_LQlEj.css} +1 -1
  34. package/dashboard/dist/assets/{index-8owBnqyv.js → index-P1I1JGtj.js} +11 -11
  35. package/dashboard/dist/index.html +2 -2
  36. package/dashboard/server/doctor.js +75 -18
  37. package/dashboard/server/index.js +5 -5
  38. package/dashboard/server/metrics.js +1 -1
  39. package/install.ps1 +31 -14
  40. package/install.sh +31 -14
  41. package/package.json +2 -2
  42. package/profile/PIPELINE.template.md +17 -16
  43. package/profile/SCHEMA.md +199 -48
  44. package/profile/cohorte.config.template.yaml +8 -8
  45. package/scripts/loop-detach.sh +153 -0
  46. package/scripts/loop.sh +202 -25
  47. package/scripts/metrics/collect.mjs +17 -8
  48. package/scripts/new-feature.sh.template +3 -3
  49. package/scripts/preflight.sh +40 -4
  50. package/scripts/remove-feature.sh.template +2 -2
  51. package/scripts/test-dashboard.mjs +34 -7
  52. package/scripts/test-gate.mjs +58 -0
  53. package/scripts/test-loop.mjs +269 -0
  54. package/scripts/test-metrics.mjs +23 -11
  55. package/scripts/test-workflows.mjs +33 -5
  56. package/scripts/validate-core.mjs +46 -9
  57. package/core/commands/loop.md +0 -61
  58. /package/core/commands/{align-ds.md → cohorte-align-ds.md} +0 -0
@@ -22,12 +22,28 @@ const frontmatter = (text) => {
22
22
  // Mechanical commands must pin model: sonnet (otherwise the lead's
23
23
  // orchestration turn silently bills at the session model — Opus/Fable).
24
24
  // Interactive commands must stay unpinned (they inherit on purpose).
25
- const PINNED = ["build", "review", "fix", "ship", "audit",
26
- "refactor", "doctor", "align-ds", "update-pipeline"];
27
- const UNPINNED = ["brainstorm", "spec", "init-pipeline"];
25
+ const PINNED = ["cohorte-build", "cohorte-review", "cohorte-fix", "cohorte-ship",
26
+ "cohorte-audit", "cohorte-refactor", "cohorte-doctor", "cohorte-align-ds",
27
+ "cohorte-update-pipeline", "cohorte-loop"];
28
+ const UNPINNED = ["cohorte-brainstorm", "cohorte-spec", "cohorte-init-pipeline"];
29
+
30
+ // Every command must carry the `cohorte-` prefix. This replaces the old RESERVED
31
+ // blocklist, which chased collisions one name at a time and always lagged: a command
32
+ // that collides with a Claude Code built-in is not overridden, it is SHADOWED — the
33
+ // built-in answers the slash, our file is never read, and the session confidently
34
+ // reports on a run that never happened. `/loop` did exactly that (Claude Code's own
35
+ // `/loop` runs a prompt on an interval) and went unnoticed until a user found the
36
+ // driver had never started; `/doctor` sat on a watchlist waiting to do the same.
37
+ // A blocklist can only forbid the collisions we already know about. The prefix makes
38
+ // the whole class unreachable, so this check is structural, not a list to maintain.
39
+ const PREFIX = "cohorte-";
28
40
 
29
41
  for (const f of readdirSync(join(root, "core/commands"))) {
30
42
  const path = `core/commands/${f}`;
43
+ if (!f.startsWith(PREFIX))
44
+ fail(path, `command name lacks the \`${PREFIX}\` prefix — an unprefixed command can be ` +
45
+ `SHADOWED by a Claude Code built-in of the same name (the built-in answers the slash ` +
46
+ `and this file is never read); rename it to ${PREFIX}${f}`);
31
47
  const fm = frontmatter(read(path));
32
48
  if (!fm) { fail(path, "missing or malformed YAML frontmatter"); continue; }
33
49
  if (!/^description:\s*\S/m.test(fm)) fail(path, "frontmatter lacks a description");
@@ -115,16 +131,20 @@ if (!existsSync(steps) || readdirSync(steps).length === 0)
115
131
 
116
132
  // ── telemetry coverage ──────────────────────────────────────────────────────
117
133
  // The funnel is only readable if every one of its stages pings — a single missing
118
- // one silently truncates it (that is how /review and /fix went unreported
134
+ // one silently truncates it (that is how /cohorte-review and /cohorte-fix went unreported
119
135
  // until 1.2.3). The phase list here must match SCHEMA.md §Telemetry's table.
136
+ // These are telemetry PHASE names, not command names — they stay unprefixed even though
137
+ // the commands that emit them are now `/cohorte-*`. The phase is a wire field allowlisted
138
+ // in telemetry-send.sh and keyed on by the collector's existing dataset; prefixing it would
139
+ // orphan every ping ever sent. Command file = PREFIX + phase.
120
140
  const FUNNEL = ["brainstorm", "spec", "build", "review", "fix", "ship"];
121
141
  for (const c of FUNNEL)
122
- if (!/usage ping/i.test(read(`core/commands/${c}.md`)))
123
- fail(`core/commands/${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
142
+ if (!/usage ping/i.test(read(`core/commands/${PREFIX}${c}.md`)))
143
+ fail(`core/commands/${PREFIX}${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
124
144
  // …and nothing outside the funnel may ping (consent text scopes it to the funnel).
125
145
  for (const f of readdirSync(join(root, "core/commands"))) {
126
- const c = f.replace(/\.md$/, "");
127
- // `telemetry-send.sh` + an argument = a call site; the bare filename (e.g. /doctor
146
+ const c = f.replace(/\.md$/, "").replace(new RegExp(`^${PREFIX}`), "");
147
+ // `telemetry-send.sh` + an argument = a call site; the bare filename (e.g. /cohorte-doctor
128
148
  // listing the scripts it checks for) is a mention, not a ping.
129
149
  if (!FUNNEL.includes(c) && /telemetry-send\.sh +\S|usage ping/i.test(read(`core/commands/${f}`)))
130
150
  fail(`core/commands/${f}`, "non-funnel command pings telemetry — outside the consented scope");
@@ -139,7 +159,7 @@ for (const f of readdirSync(join(root, "core/commands"))) {
139
159
  // scratch HOME and asserts the same postconditions instead. Both are needed: this
140
160
  // check catches a forgotten name, that one catches a drifted rule.
141
161
  // A `<name>.sh` with a `<name>.sh.template` sibling is a locally-rendered artifact
142
- // (this repo dogfoods its own /init-pipeline), not a core asset — skip those.
162
+ // (this repo dogfoods its own /cohorte-init-pipeline), not a core asset — skip those.
143
163
  const installers = { "install.sh": read("install.sh"), "install.ps1": read("install.ps1") };
144
164
  const shipped = readdirSync(join(root, "scripts"));
145
165
  for (const f of shipped.filter((f) => f.endsWith(".sh") && !shipped.includes(`${f}.template`)))
@@ -203,6 +223,23 @@ for (const f of workflowNames) {
203
223
  fail("dashboard/server/doctor.js", `checkWorkflows() does not list ${f}`);
204
224
  }
205
225
 
226
+ // ── every test suite must run in BOTH workflows ──────────────────────────────
227
+ // publish.yml re-runs the test suites under the comment "same gate as CI", because
228
+ // it has no dependency on the CI workflow's conclusion — a merge whose CI failed
229
+ // would otherwise still ship to npm. That only holds if the two lists agree, and
230
+ // they drift the moment a suite is added to one: test-loop.mjs landed in ci.yml and
231
+ // publish.yml kept publishing without it. Neither list is the source of truth —
232
+ // the directory is.
233
+ const ciYml = existsSync(join(root, ".github/workflows/ci.yml")) ? read(".github/workflows/ci.yml") : "";
234
+ const publishYml = existsSync(join(root, ".github/workflows/publish.yml"))
235
+ ? read(".github/workflows/publish.yml") : "";
236
+ for (const f of readdirSync(join(root, "scripts")).filter((f) => /^test-.*\.mjs$/.test(f))) {
237
+ if (ciYml && !ciYml.includes(`scripts/${f}`))
238
+ fail(".github/workflows/ci.yml", `never runs scripts/${f} — a suite CI does not run is a suite that does not exist`);
239
+ if (publishYml && !publishYml.includes(`scripts/${f}`))
240
+ fail(".github/workflows/publish.yml", `never runs scripts/${f} — publish would ship past a failure that gate is meant to catch`);
241
+ }
242
+
206
243
  // ── dashboard: the metrics phase list is duplicated server/client ────────────
207
244
  // A phase present in one and not the other parses fine and renders in no column —
208
245
  // silently invisible data, which is how a phase batch once went unnoticed.
@@ -1,61 +0,0 @@
1
- ---
2
- model: sonnet
3
- description: Autonomous /build → /review → /fix → /review loop for one feature, until no blocking finding remains.
4
- argument-hint: <feature_id> [--max=N] [--no-build] [--rebuild]
5
- allowed-tools: Bash(bash ~/.claude/pipeline/scripts/loop.sh:*), Bash(bash .claude/pipeline/scripts/loop.sh:*), Bash(test:*), Read(specs/reports/**)
6
- disable-model-invocation: true
7
- ---
8
-
9
- You are the **launcher**, not the loop. Run the driver for **$ARGUMENTS** and relay three lines.
10
-
11
- > This command exists because a slash command cannot `/clear` itself. Every phase of the loop runs
12
- > as a **separate `claude -p` child session** with its own fresh context, driven by a bash script —
13
- > so the diff, the N review reports and the N contracts never accumulate in YOUR history, which is
14
- > re-sent at input price on every turn. Running the loop conversationally here would cost more than
15
- > the loop saves.
16
-
17
- ## 1. Launch
18
-
19
- Probe the core, then run the script — ONE Bash call, and let it run to completion:
20
-
21
- ```
22
- test -f .claude/pipeline/scripts/loop.sh \
23
- && bash .claude/pipeline/scripts/loop.sh $ARGUMENTS \
24
- || bash ~/.claude/pipeline/scripts/loop.sh $ARGUMENTS
25
- ```
26
-
27
- Pass `$ARGUMENTS` through untouched — the script owns its own flag parsing (`--max=N`,
28
- `--no-build`, `--rebuild`) and exits 64 on anything it doesn't know. Don't validate flags yourself,
29
- don't rewrite them, don't add any.
30
-
31
- **Never read `specs/reports/<id>.loop.log`.** It holds the full transcript of every child session —
32
- the entire diff, every review report, every fix handoff. Pulling it into this session re-imports
33
- exactly the context the loop was built to keep out, and it is the one mistake that turns this
34
- command into the most expensive one in the pipeline. Point the human at the path instead; they can
35
- open it in an editor for free. The same goes for the per-surface `.diff` and `.preflight.txt` files.
36
-
37
- ## 2. Report — three lines, from the exit code
38
-
39
- The script prints one line per phase and one closing line; that is your raw material. For exit
40
- **1** or **3** only, also Read `specs/reports/<id>.verdict.json` (small, structured, safe) to name
41
- the remaining findings — never the markdown report, which is the findings body in full.
42
-
43
- | exit | meaning | what to say |
44
- | ---- | ------- | ----------- |
45
- | `0` | clean | no blocking findings left; the human can `/ship <id>` |
46
- | `1` | ceiling hit | the fix was progressing but ran out of passes ⇒ re-run with a higher `--max` |
47
- | `2` | no usable verdict | `/review` produced nothing, or aborted on a red preflight — the closing line says which; point at `specs/reports/<id>.preflight.txt` |
48
- | `3` | non-convergent | the same blocking findings survived a fix pass; a higher `--max` will NOT help — the human needs to look at them (list them from the verdict) |
49
- | `64` | usage | relay the script's own message verbatim |
50
-
51
- Then print exactly three lines and nothing else:
52
-
53
- ```
54
- outcome: <one clause — clean / ceiling / no verdict / non-convergent / usage>
55
- iterations: <n> review pass(es)<, m fix pass(es) committed>
56
- remaining: <blocking count + one short phrase per blocking item, or "none">
57
- ```
58
-
59
- Add at most one follow-up sentence: the next command to run. Never restate a finding's fix, never
60
- summarize the log, never open the diff. Each fix pass is already committed
61
- (`loop(<id>): fix pass <i>`) — say so on a non-zero exit, since those commits are the way back.