cohorte 1.5.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +169 -3
- package/README.md +65 -57
- package/bin/cli.js +31 -15
- package/core/agents/implementer.template.md +3 -3
- package/core/agents/release.md +1 -1
- package/core/agents/review.md +25 -2
- package/core/commands/{audit.md → cohorte-audit.md} +11 -3
- package/core/commands/{brainstorm.md → cohorte-brainstorm.md} +9 -3
- package/core/commands/{build.md → cohorte-build.md} +95 -10
- package/core/commands/{doctor.md → cohorte-doctor.md} +22 -9
- package/core/commands/{fix.md → cohorte-fix.md} +20 -13
- package/core/commands/{init-pipeline.md → cohorte-init-pipeline.md} +1 -1
- package/core/commands/cohorte-loop.md +110 -0
- package/core/commands/{refactor.md → cohorte-refactor.md} +3 -3
- package/core/commands/{review.md → cohorte-review.md} +70 -20
- package/core/commands/{ship.md → cohorte-ship.md} +5 -5
- package/core/commands/{spec.md → cohorte-spec.md} +32 -12
- package/core/commands/{update-pipeline.md → cohorte-update-pipeline.md} +16 -6
- package/core/hooks/gate.py +101 -6
- package/core/templates/brainstorm-return.md +4 -4
- package/core/templates/decisions.template.md +42 -0
- package/core/templates/design-brief.md +1 -1
- package/core/templates/spec.template.md +8 -6
- package/core/templates/steps/init-pipeline/01-detect-stack.md +1 -1
- package/core/templates/steps/init-pipeline/02-interview-gaps.md +6 -6
- package/core/templates/steps/init-pipeline/03-draft-profile.md +1 -1
- package/core/templates/steps/init-pipeline/04-write-render.md +16 -12
- package/core/templates/steps/init-pipeline/05-report.md +5 -5
- package/core/workflows/audit.js +6 -6
- package/core/workflows/refactor.js +14 -14
- package/core/workflows/review.js +62 -20
- package/dashboard/README.md +2 -2
- package/dashboard/dist/assets/{index-dkO8UUVl.css → index-BZ_LQlEj.css} +1 -1
- package/dashboard/dist/assets/{index-8owBnqyv.js → index-P1I1JGtj.js} +11 -11
- package/dashboard/dist/index.html +2 -2
- package/dashboard/server/doctor.js +75 -18
- package/dashboard/server/index.js +5 -5
- package/dashboard/server/metrics.js +1 -1
- package/install.ps1 +31 -14
- package/install.sh +31 -14
- package/package.json +2 -2
- package/profile/PIPELINE.template.md +17 -16
- package/profile/SCHEMA.md +199 -48
- package/profile/cohorte.config.template.yaml +8 -8
- package/scripts/loop-detach.sh +153 -0
- package/scripts/loop.sh +202 -25
- package/scripts/metrics/collect.mjs +17 -8
- package/scripts/new-feature.sh.template +3 -3
- package/scripts/preflight.sh +40 -4
- package/scripts/remove-feature.sh.template +2 -2
- package/scripts/test-dashboard.mjs +34 -7
- package/scripts/test-gate.mjs +58 -0
- package/scripts/test-loop.mjs +269 -0
- package/scripts/test-metrics.mjs +23 -11
- package/scripts/test-workflows.mjs +33 -5
- package/scripts/validate-core.mjs +46 -9
- package/core/commands/loop.md +0 -61
- /package/core/commands/{align-ds.md → cohorte-align-ds.md} +0 -0
|
@@ -22,12 +22,28 @@ const frontmatter = (text) => {
|
|
|
22
22
|
// Mechanical commands must pin model: sonnet (otherwise the lead's
|
|
23
23
|
// orchestration turn silently bills at the session model — Opus/Fable).
|
|
24
24
|
// Interactive commands must stay unpinned (they inherit on purpose).
|
|
25
|
-
const PINNED = ["build", "review", "fix", "ship",
|
|
26
|
-
"refactor", "doctor", "align-ds",
|
|
27
|
-
|
|
25
|
+
const PINNED = ["cohorte-build", "cohorte-review", "cohorte-fix", "cohorte-ship",
|
|
26
|
+
"cohorte-audit", "cohorte-refactor", "cohorte-doctor", "cohorte-align-ds",
|
|
27
|
+
"cohorte-update-pipeline", "cohorte-loop"];
|
|
28
|
+
const UNPINNED = ["cohorte-brainstorm", "cohorte-spec", "cohorte-init-pipeline"];
|
|
29
|
+
|
|
30
|
+
// Every command must carry the `cohorte-` prefix. This replaces the old RESERVED
|
|
31
|
+
// blocklist, which chased collisions one name at a time and always lagged: a command
|
|
32
|
+
// that collides with a Claude Code built-in is not overridden, it is SHADOWED — the
|
|
33
|
+
// built-in answers the slash, our file is never read, and the session confidently
|
|
34
|
+
// reports on a run that never happened. `/loop` did exactly that (Claude Code's own
|
|
35
|
+
// `/loop` runs a prompt on an interval) and went unnoticed until a user found the
|
|
36
|
+
// driver had never started; `/doctor` sat on a watchlist waiting to do the same.
|
|
37
|
+
// A blocklist can only forbid the collisions we already know about. The prefix makes
|
|
38
|
+
// the whole class unreachable, so this check is structural, not a list to maintain.
|
|
39
|
+
const PREFIX = "cohorte-";
|
|
28
40
|
|
|
29
41
|
for (const f of readdirSync(join(root, "core/commands"))) {
|
|
30
42
|
const path = `core/commands/${f}`;
|
|
43
|
+
if (!f.startsWith(PREFIX))
|
|
44
|
+
fail(path, `command name lacks the \`${PREFIX}\` prefix — an unprefixed command can be ` +
|
|
45
|
+
`SHADOWED by a Claude Code built-in of the same name (the built-in answers the slash ` +
|
|
46
|
+
`and this file is never read); rename it to ${PREFIX}${f}`);
|
|
31
47
|
const fm = frontmatter(read(path));
|
|
32
48
|
if (!fm) { fail(path, "missing or malformed YAML frontmatter"); continue; }
|
|
33
49
|
if (!/^description:\s*\S/m.test(fm)) fail(path, "frontmatter lacks a description");
|
|
@@ -115,16 +131,20 @@ if (!existsSync(steps) || readdirSync(steps).length === 0)
|
|
|
115
131
|
|
|
116
132
|
// ── telemetry coverage ──────────────────────────────────────────────────────
|
|
117
133
|
// The funnel is only readable if every one of its stages pings — a single missing
|
|
118
|
-
// one silently truncates it (that is how /review and /fix went unreported
|
|
134
|
+
// one silently truncates it (that is how /cohorte-review and /cohorte-fix went unreported
|
|
119
135
|
// until 1.2.3). The phase list here must match SCHEMA.md §Telemetry's table.
|
|
136
|
+
// These are telemetry PHASE names, not command names — they stay unprefixed even though
|
|
137
|
+
// the commands that emit them are now `/cohorte-*`. The phase is a wire field allowlisted
|
|
138
|
+
// in telemetry-send.sh and keyed on by the collector's existing dataset; prefixing it would
|
|
139
|
+
// orphan every ping ever sent. Command file = PREFIX + phase.
|
|
120
140
|
const FUNNEL = ["brainstorm", "spec", "build", "review", "fix", "ship"];
|
|
121
141
|
for (const c of FUNNEL)
|
|
122
|
-
if (!/usage ping/i.test(read(`core/commands/${c}.md`)))
|
|
123
|
-
fail(`core/commands/${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
|
|
142
|
+
if (!/usage ping/i.test(read(`core/commands/${PREFIX}${c}.md`)))
|
|
143
|
+
fail(`core/commands/${PREFIX}${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
|
|
124
144
|
// …and nothing outside the funnel may ping (consent text scopes it to the funnel).
|
|
125
145
|
for (const f of readdirSync(join(root, "core/commands"))) {
|
|
126
|
-
const c = f.replace(/\.md$/, "");
|
|
127
|
-
// `telemetry-send.sh` + an argument = a call site; the bare filename (e.g. /doctor
|
|
146
|
+
const c = f.replace(/\.md$/, "").replace(new RegExp(`^${PREFIX}`), "");
|
|
147
|
+
// `telemetry-send.sh` + an argument = a call site; the bare filename (e.g. /cohorte-doctor
|
|
128
148
|
// listing the scripts it checks for) is a mention, not a ping.
|
|
129
149
|
if (!FUNNEL.includes(c) && /telemetry-send\.sh +\S|usage ping/i.test(read(`core/commands/${f}`)))
|
|
130
150
|
fail(`core/commands/${f}`, "non-funnel command pings telemetry — outside the consented scope");
|
|
@@ -139,7 +159,7 @@ for (const f of readdirSync(join(root, "core/commands"))) {
|
|
|
139
159
|
// scratch HOME and asserts the same postconditions instead. Both are needed: this
|
|
140
160
|
// check catches a forgotten name, that one catches a drifted rule.
|
|
141
161
|
// A `<name>.sh` with a `<name>.sh.template` sibling is a locally-rendered artifact
|
|
142
|
-
// (this repo dogfoods its own /init-pipeline), not a core asset — skip those.
|
|
162
|
+
// (this repo dogfoods its own /cohorte-init-pipeline), not a core asset — skip those.
|
|
143
163
|
const installers = { "install.sh": read("install.sh"), "install.ps1": read("install.ps1") };
|
|
144
164
|
const shipped = readdirSync(join(root, "scripts"));
|
|
145
165
|
for (const f of shipped.filter((f) => f.endsWith(".sh") && !shipped.includes(`${f}.template`)))
|
|
@@ -203,6 +223,23 @@ for (const f of workflowNames) {
|
|
|
203
223
|
fail("dashboard/server/doctor.js", `checkWorkflows() does not list ${f}`);
|
|
204
224
|
}
|
|
205
225
|
|
|
226
|
+
// ── every test suite must run in BOTH workflows ──────────────────────────────
|
|
227
|
+
// publish.yml re-runs the test suites under the comment "same gate as CI", because
|
|
228
|
+
// it has no dependency on the CI workflow's conclusion — a merge whose CI failed
|
|
229
|
+
// would otherwise still ship to npm. That only holds if the two lists agree, and
|
|
230
|
+
// they drift the moment a suite is added to one: test-loop.mjs landed in ci.yml and
|
|
231
|
+
// publish.yml kept publishing without it. Neither list is the source of truth —
|
|
232
|
+
// the directory is.
|
|
233
|
+
const ciYml = existsSync(join(root, ".github/workflows/ci.yml")) ? read(".github/workflows/ci.yml") : "";
|
|
234
|
+
const publishYml = existsSync(join(root, ".github/workflows/publish.yml"))
|
|
235
|
+
? read(".github/workflows/publish.yml") : "";
|
|
236
|
+
for (const f of readdirSync(join(root, "scripts")).filter((f) => /^test-.*\.mjs$/.test(f))) {
|
|
237
|
+
if (ciYml && !ciYml.includes(`scripts/${f}`))
|
|
238
|
+
fail(".github/workflows/ci.yml", `never runs scripts/${f} — a suite CI does not run is a suite that does not exist`);
|
|
239
|
+
if (publishYml && !publishYml.includes(`scripts/${f}`))
|
|
240
|
+
fail(".github/workflows/publish.yml", `never runs scripts/${f} — publish would ship past a failure that gate is meant to catch`);
|
|
241
|
+
}
|
|
242
|
+
|
|
206
243
|
// ── dashboard: the metrics phase list is duplicated server/client ────────────
|
|
207
244
|
// A phase present in one and not the other parses fine and renders in no column —
|
|
208
245
|
// silently invisible data, which is how a phase batch once went unnoticed.
|
package/core/commands/loop.md
DELETED
|
@@ -1,61 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
model: sonnet
|
|
3
|
-
description: Autonomous /build → /review → /fix → /review loop for one feature, until no blocking finding remains.
|
|
4
|
-
argument-hint: <feature_id> [--max=N] [--no-build] [--rebuild]
|
|
5
|
-
allowed-tools: Bash(bash ~/.claude/pipeline/scripts/loop.sh:*), Bash(bash .claude/pipeline/scripts/loop.sh:*), Bash(test:*), Read(specs/reports/**)
|
|
6
|
-
disable-model-invocation: true
|
|
7
|
-
---
|
|
8
|
-
|
|
9
|
-
You are the **launcher**, not the loop. Run the driver for **$ARGUMENTS** and relay three lines.
|
|
10
|
-
|
|
11
|
-
> This command exists because a slash command cannot `/clear` itself. Every phase of the loop runs
|
|
12
|
-
> as a **separate `claude -p` child session** with its own fresh context, driven by a bash script —
|
|
13
|
-
> so the diff, the N review reports and the N contracts never accumulate in YOUR history, which is
|
|
14
|
-
> re-sent at input price on every turn. Running the loop conversationally here would cost more than
|
|
15
|
-
> the loop saves.
|
|
16
|
-
|
|
17
|
-
## 1. Launch
|
|
18
|
-
|
|
19
|
-
Probe the core, then run the script — ONE Bash call, and let it run to completion:
|
|
20
|
-
|
|
21
|
-
```
|
|
22
|
-
test -f .claude/pipeline/scripts/loop.sh \
|
|
23
|
-
&& bash .claude/pipeline/scripts/loop.sh $ARGUMENTS \
|
|
24
|
-
|| bash ~/.claude/pipeline/scripts/loop.sh $ARGUMENTS
|
|
25
|
-
```
|
|
26
|
-
|
|
27
|
-
Pass `$ARGUMENTS` through untouched — the script owns its own flag parsing (`--max=N`,
|
|
28
|
-
`--no-build`, `--rebuild`) and exits 64 on anything it doesn't know. Don't validate flags yourself,
|
|
29
|
-
don't rewrite them, don't add any.
|
|
30
|
-
|
|
31
|
-
**Never read `specs/reports/<id>.loop.log`.** It holds the full transcript of every child session —
|
|
32
|
-
the entire diff, every review report, every fix handoff. Pulling it into this session re-imports
|
|
33
|
-
exactly the context the loop was built to keep out, and it is the one mistake that turns this
|
|
34
|
-
command into the most expensive one in the pipeline. Point the human at the path instead; they can
|
|
35
|
-
open it in an editor for free. The same goes for the per-surface `.diff` and `.preflight.txt` files.
|
|
36
|
-
|
|
37
|
-
## 2. Report — three lines, from the exit code
|
|
38
|
-
|
|
39
|
-
The script prints one line per phase and one closing line; that is your raw material. For exit
|
|
40
|
-
**1** or **3** only, also Read `specs/reports/<id>.verdict.json` (small, structured, safe) to name
|
|
41
|
-
the remaining findings — never the markdown report, which is the findings body in full.
|
|
42
|
-
|
|
43
|
-
| exit | meaning | what to say |
|
|
44
|
-
| ---- | ------- | ----------- |
|
|
45
|
-
| `0` | clean | no blocking findings left; the human can `/ship <id>` |
|
|
46
|
-
| `1` | ceiling hit | the fix was progressing but ran out of passes ⇒ re-run with a higher `--max` |
|
|
47
|
-
| `2` | no usable verdict | `/review` produced nothing, or aborted on a red preflight — the closing line says which; point at `specs/reports/<id>.preflight.txt` |
|
|
48
|
-
| `3` | non-convergent | the same blocking findings survived a fix pass; a higher `--max` will NOT help — the human needs to look at them (list them from the verdict) |
|
|
49
|
-
| `64` | usage | relay the script's own message verbatim |
|
|
50
|
-
|
|
51
|
-
Then print exactly three lines and nothing else:
|
|
52
|
-
|
|
53
|
-
```
|
|
54
|
-
outcome: <one clause — clean / ceiling / no verdict / non-convergent / usage>
|
|
55
|
-
iterations: <n> review pass(es)<, m fix pass(es) committed>
|
|
56
|
-
remaining: <blocking count + one short phrase per blocking item, or "none">
|
|
57
|
-
```
|
|
58
|
-
|
|
59
|
-
Add at most one follow-up sentence: the next command to run. Never restate a finding's fix, never
|
|
60
|
-
summarize the log, never open the diff. Each fix pass is already committed
|
|
61
|
-
(`loop(<id>): fix pass <i>`) — say so on a non-zero exit, since those commits are the way back.
|
|
File without changes
|