cohorte 1.4.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +67 -0
- package/README.md +32 -9
- package/bin/cli.js +5 -2
- package/core/commands/build.md +4 -4
- package/core/commands/doctor.md +5 -5
- package/core/commands/fix.md +5 -5
- package/core/commands/loop.md +61 -0
- package/core/commands/review.md +38 -5
- package/core/hooks/gate.py +4 -4
- package/core/templates/spec.template.md +1 -1
- package/core/templates/steps/init-pipeline/02-interview-gaps.md +1 -1
- package/core/templates/steps/init-pipeline/04-write-render.md +8 -4
- package/dashboard/dist/assets/index-8owBnqyv.js +43 -0
- package/dashboard/dist/assets/{index-AFQnlfjO.css → index-dkO8UUVl.css} +1 -1
- package/dashboard/dist/index.html +2 -2
- package/dashboard/server/doctor.js +5 -2
- package/dashboard/server/index.js +7 -0
- package/dashboard/server/metrics.js +4 -4
- package/dashboard/server/usage.js +61 -0
- package/install.ps1 +4 -1
- package/install.sh +5 -2
- package/package.json +1 -1
- package/profile/PIPELINE.template.md +3 -3
- package/profile/SCHEMA.md +10 -11
- package/scripts/loop.sh +189 -0
- package/scripts/metrics/collect.mjs +11 -2
- package/scripts/preflight.sh +2 -2
- package/scripts/telemetry-send.sh +5 -2
- package/scripts/test-dashboard.mjs +22 -2
- package/scripts/test-gate.mjs +1 -2
- package/scripts/test-metrics.mjs +12 -3
- package/scripts/validate-core.mjs +5 -5
- package/core/agents/smoke.md +0 -63
- package/core/commands/smoke.md +0 -55
- package/dashboard/dist/assets/index-DLBzciIC.js +0 -43
package/scripts/test-metrics.mjs
CHANGED
|
@@ -82,6 +82,13 @@ const lines = [
|
|
|
82
82
|
// Case 8: a slash token that is not a command must not invent one.
|
|
83
83
|
user(1800, 'look at the /usr/local/share directory and report what you find there'),
|
|
84
84
|
assistant('m7', 1805, 'claude-opus-5', { input_tokens: 0, output_tokens: 10 }),
|
|
85
|
+
// Case 9: a long prompt that merely DISCUSSES a command is not an invocation of it.
|
|
86
|
+
// Without the length gate, writing about /review bills the conversation to /review —
|
|
87
|
+
// which is what happened in cohorte's own repo while the pipeline was being designed.
|
|
88
|
+
user(2400, 'I want to talk through how /review behaves when a surface has no findings at '
|
|
89
|
+
+ 'all, because the verdict logic there is what produced the false green we saw last week '
|
|
90
|
+
+ 'and I am not convinced the fix covers the case where every reviewer dies at once.'),
|
|
91
|
+
assistant('m8', 2405, 'claude-opus-5', { input_tokens: 0, output_tokens: 20 }),
|
|
85
92
|
];
|
|
86
93
|
fs.writeFileSync(path.join(projectDir, `${SESSION}.jsonl`),
|
|
87
94
|
lines.map((l) => JSON.stringify(l)).join('\n') + '\n');
|
|
@@ -107,18 +114,20 @@ const chat = out.commands.find((c) => c.command === '(chat)');
|
|
|
107
114
|
const review = out.commands.find((c) => c.command === '/review');
|
|
108
115
|
|
|
109
116
|
console.log('test-metrics');
|
|
110
|
-
check('the mid-command task-notification did not split the run', out.totals.runs,
|
|
117
|
+
check('the mid-command task-notification did not split the run', out.totals.runs, 5);
|
|
111
118
|
check('/build is one run, not three', build.runs, 1);
|
|
112
119
|
check('duplicate lines of one response are billed once', build.tokens.output, 1000 + 500 + 2000);
|
|
113
120
|
check('the <synthetic> message contributed no tokens', build.tokens.output < 999999, true);
|
|
114
121
|
check('cache-write tokens are kept on their own tier', build.tokens.cacheWrite5m, 1000);
|
|
115
122
|
check('cache-read tokens are kept on their own tier', build.tokens.cacheRead, 10000);
|
|
116
123
|
check('the subagent was attributed to the command that spawned it', build.agents.total, 1);
|
|
117
|
-
check('the second prompt is a separate (chat) run', chat.runs,
|
|
124
|
+
check('the second prompt is a separate (chat) run', chat.runs, 3);
|
|
118
125
|
check('a command named inside prose is attributed to that command', review && review.runs, 1);
|
|
119
126
|
check('a short steer continues the run instead of opening a new one', review.continuations, 1);
|
|
120
127
|
check('the continued turn counts toward the command it continued', review.tokens.output, 60 + 70);
|
|
121
|
-
check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10);
|
|
128
|
+
check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10 + 20);
|
|
129
|
+
check('a long prompt that discusses a command is not counted as running it',
|
|
130
|
+
review.runs, 1);
|
|
122
131
|
|
|
123
132
|
// opus-5 $5 in / $25 out per MTok; 5m cache write 1.25x input, cache read 0.1x input.
|
|
124
133
|
// m1 100*5 + 1000*25 + 1000*6.25 + 10000*0.5 = 36750
|
|
@@ -22,7 +22,7 @@ const frontmatter = (text) => {
|
|
|
22
22
|
// Mechanical commands must pin model: sonnet (otherwise the lead's
|
|
23
23
|
// orchestration turn silently bills at the session model — Opus/Fable).
|
|
24
24
|
// Interactive commands must stay unpinned (they inherit on purpose).
|
|
25
|
-
const PINNED = ["build", "review", "fix", "
|
|
25
|
+
const PINNED = ["build", "review", "fix", "ship", "audit",
|
|
26
26
|
"refactor", "doctor", "align-ds", "update-pipeline"];
|
|
27
27
|
const UNPINNED = ["brainstorm", "spec", "init-pipeline"];
|
|
28
28
|
|
|
@@ -43,7 +43,7 @@ for (const f of readdirSync(join(root, "core/commands"))) {
|
|
|
43
43
|
// Every non-template agent needs name/tools/model, and must be shipped by
|
|
44
44
|
// both installers (a new agent that install.sh doesn't copy never reaches
|
|
45
45
|
// a global install — the exact bug that motivated this check).
|
|
46
|
-
const AGENT_MODEL = { review: "sonnet", release: "haiku",
|
|
46
|
+
const AGENT_MODEL = { review: "sonnet", release: "haiku",
|
|
47
47
|
"profile-reader": "haiku" };
|
|
48
48
|
const installSh = read("install.sh");
|
|
49
49
|
const installPs1 = read("install.ps1");
|
|
@@ -90,7 +90,7 @@ for (const path of allDocs) {
|
|
|
90
90
|
}
|
|
91
91
|
for (const m of text.matchAll(/subagent_type:\s*(?:`|)([a-z-]+)(?:`|)/g)) {
|
|
92
92
|
const t = m[1];
|
|
93
|
-
if (["review", "release", "
|
|
93
|
+
if (["review", "release", "profile-reader"].includes(t)) continue;
|
|
94
94
|
if (t.startsWith("<")) continue; // <surface.agent> placeholder
|
|
95
95
|
if (!existsSync(join(root, "core/agents", `${t}.md`)))
|
|
96
96
|
fail(path, `dispatches subagent_type ${t} with no core/agents/${t}.md`);
|
|
@@ -115,9 +115,9 @@ if (!existsSync(steps) || readdirSync(steps).length === 0)
|
|
|
115
115
|
|
|
116
116
|
// ── telemetry coverage ──────────────────────────────────────────────────────
|
|
117
117
|
// The funnel is only readable if every one of its stages pings — a single missing
|
|
118
|
-
// one silently truncates it (that is how /
|
|
118
|
+
// one silently truncates it (that is how /review and /fix went unreported
|
|
119
119
|
// until 1.2.3). The phase list here must match SCHEMA.md §Telemetry's table.
|
|
120
|
-
const FUNNEL = ["brainstorm", "spec", "build", "
|
|
120
|
+
const FUNNEL = ["brainstorm", "spec", "build", "review", "fix", "ship"];
|
|
121
121
|
for (const c of FUNNEL)
|
|
122
122
|
if (!/usage ping/i.test(read(`core/commands/${c}.md`)))
|
|
123
123
|
fail(`core/commands/${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
|
package/core/agents/smoke.md
DELETED
|
@@ -1,63 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: smoke
|
|
3
|
-
description: Executes the end-to-end smoke run for one feature in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — then stages the SMOKE REPORT. Dispatched by /smoke. Observes honestly, never fixes anything.
|
|
4
|
-
tools: Read, Write, Grep, Glob, Bash, DesignSync
|
|
5
|
-
model: sonnet
|
|
6
|
-
---
|
|
7
|
-
|
|
8
|
-
You are the **smoke** agent for one feature. You actually run the built feature — `/review` audits
|
|
9
|
-
code read-only; nobody has executed it yet. You verify it *works*; you never fix it (failures go
|
|
10
|
-
through `/fix`). Observe honestly: report what happened, not what should have happened.
|
|
11
|
-
|
|
12
|
-
> **First action, always:** read `PIPELINE.md` §`pipeline-profile`: `commands` (migrate/dev),
|
|
13
|
-
> `isolation` (worktree, slot ports, db), `contract`, `design`, `surfaces` — then the spec
|
|
14
|
-
> `specs/<id>.md` (§5 contract, §8 flows, §9 acceptance).
|
|
15
|
-
|
|
16
|
-
## Your inputs (supplied at dispatch — you have no memory)
|
|
17
|
-
|
|
18
|
-
1. The feature id and spec path `specs/<id>.md`.
|
|
19
|
-
2. The contract path `<contract.path>/<id>.<ext>`.
|
|
20
|
-
3. The checkout to work in: the worktree path + slot ports/db, or the main checkout on the feature branch.
|
|
21
|
-
|
|
22
|
-
## Keep your own context lean
|
|
23
|
-
|
|
24
|
-
Redirect every bulky output to a file and inspect it with `grep`/`jq` — never print full curl bodies,
|
|
25
|
-
server logs, or poll loops into your transcript. `curl -s … -o /tmp/resp.json -w '%{http_code}'` then
|
|
26
|
-
assert on the pieces you need.
|
|
27
|
-
|
|
28
|
-
## 1. Bring the feature up
|
|
29
|
-
|
|
30
|
-
- Work in the checkout your dispatch names. Infra as needed: the compose stack if one is declared
|
|
31
|
-
(the gate will ask — that's expected), then `commands.migrate`, then `commands.dev` **in the
|
|
32
|
-
background**. Wait for ready (poll the ports), don't assume.
|
|
33
|
-
|
|
34
|
-
## 2. Exercise the contract (the real server, not the tests)
|
|
35
|
-
|
|
36
|
-
- Hit a representative set of spec §5 endpoints with `curl`: every route domain, every auth level,
|
|
37
|
-
at least one error case per class (validation `422`, unauthenticated `401`, wrong-role `403`,
|
|
38
|
-
conflict `409`). Compare status + response envelope against the contract.
|
|
39
|
-
- If `rbac.enabled`: verify at least one denial per role boundary the spec declares.
|
|
40
|
-
- A mismatch is a FAIL entry with the exact command, expected, and actual — precise enough for a
|
|
41
|
-
stateless `/fix` agent.
|
|
42
|
-
|
|
43
|
-
## 3. Exercise the UI (only if a touched surface has `uses_design`)
|
|
44
|
-
|
|
45
|
-
- Drive the spec §8 flows against the running app, **mobile viewport first** (375px), then desktop.
|
|
46
|
-
- If a browser/screenshot tool is available (a project driver, playwright, an agent browser), capture
|
|
47
|
-
each §8 screen and compare against the feature's design pages: each `design_files` entry is a full
|
|
48
|
-
`https://claude.ai/design/p/<projectId>?file=<file>` link — extract its `<projectId>` (the `/p/…`
|
|
49
|
-
segment) + `<file>` (the `?file=` query) and fetch read-only via `DesignSync get_file(<projectId>,
|
|
50
|
-
<file>)`. Compare layout, states (empty/loading/error/suppressed…), copy language. Note deviations.
|
|
51
|
-
- No browser tooling available ⇒ **say so and skip the visual diff** — never claim a visual check
|
|
52
|
-
you didn't perform.
|
|
53
|
-
|
|
54
|
-
## 4. Stage the SMOKE REPORT, tear down, return
|
|
55
|
-
|
|
56
|
-
- One line per check: ✅/❌ · what was exercised · (on ❌) command → expected vs actual.
|
|
57
|
-
- **Write the full report to `specs/reports/<id>.md`** (overwrite) — the same gitignored buffer
|
|
58
|
-
`/review` uses, so a `/fix` after a `/clear` still has the failures.
|
|
59
|
-
- Tear down what you started (kill the dev server); leave shared infra as you found it.
|
|
60
|
-
- **Your return to the lead is ONLY:** the verdict line (`PASS` / `FAIL:<n>`), **at most 10 ❌
|
|
61
|
-
lines** — one line each (`❌ <flow/endpoint> · expected <x> got <y>`), no command output, no code
|
|
62
|
-
or body excerpts; more than 10 ⇒ keep the 10 most severe and add `+<n> more — see the report` —
|
|
63
|
-
and `Full report: specs/reports/<id>.md`. No logs, no bodies, no screenshots.
|
package/core/commands/smoke.md
DELETED
|
@@ -1,55 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
model: sonnet
|
|
3
|
-
description: Exercise the built feature end-to-end in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — before /review.
|
|
4
|
-
argument-hint: <feature_id>
|
|
5
|
-
---
|
|
6
|
-
|
|
7
|
-
You are the **lead**. Dispatch the smoke run for feature **$ARGUMENTS** — the `smoke` agent actually
|
|
8
|
-
runs it so the bulky output (curl bodies, server logs, screenshots, design payloads) never enters
|
|
9
|
-
your own context, which is re-sent every turn.
|
|
10
|
-
|
|
11
|
-
> Read `PIPELINE.md` §`pipeline-profile`: `isolation` (worktree, slot ports, db), `contract.path`
|
|
12
|
-
> and `commands`. _Skip the re-read if it's already in your context this session and unmodified since._
|
|
13
|
-
>
|
|
14
|
-
> **Kanban:** none here — `/review` owns the → **Review** move (running both duplicated it).
|
|
15
|
-
|
|
16
|
-
## 0. Deterministic pre-flight — no agents while red
|
|
17
|
-
|
|
18
|
-
Same gate as `/review` §0, run **in the feature's checkout** (§1 resolves it — resolve first, then
|
|
19
|
-
preflight): `<core>/pipeline/scripts/preflight.sh specs/reports/$ARGUMENTS.preflight.txt
|
|
20
|
-
"<commands.typecheck>" "<commands.lint_quiet, else lint>" "<commands.test_quiet, else test>"`.
|
|
21
|
-
Non-zero exit ⇒ the raw last-40 lines were already printed — **STOP, relay them verbatim, spawn NO
|
|
22
|
-
agent**: booting infra to smoke-test code that doesn't compile wastes the whole run. Zero exit ⇒
|
|
23
|
-
the `.claude/preflight.ok` stamp lets the gate hook pass your `smoke` dispatch. Script absent
|
|
24
|
-
(older core) ⇒ run the commands yourself redirected to the same file, aborting on the first failure.
|
|
25
|
-
Note the epoch (`date +%s`) in the same call — §3's metrics line needs it.
|
|
26
|
-
|
|
27
|
-
## 1. Resolve the checkout
|
|
28
|
-
|
|
29
|
-
With `isolation.enabled`: the sibling worktree (`../<slug>-$ARGUMENTS`, its slot's ports + db from
|
|
30
|
-
`.worktrees/slots.tsv`); otherwise the main checkout on the feature branch.
|
|
31
|
-
|
|
32
|
-
## 2. Dispatch ONE `smoke` agent
|
|
33
|
-
|
|
34
|
-
Keep the prompt byte-identical across features except the variable block at the END (prompt-cache
|
|
35
|
-
prefix):
|
|
36
|
-
|
|
37
|
-
> `subagent_type: smoke` — "Smoke-test one feature. Read `PIPELINE.md` first. Bring it up, exercise
|
|
38
|
-
> the contract and the §8 UI flows, stage the full SMOKE REPORT to the report buffer, tear down, and
|
|
39
|
-
> return only the capped verdict your agent instructions define (verdict + ❌ lines, no logs). —
|
|
40
|
-
> Variable slots: feature `$ARGUMENTS` · spec: `specs/$ARGUMENTS.md` · contract:
|
|
41
|
-
> `<contract.path>/$ARGUMENTS.<ext>` · report: `specs/reports/$ARGUMENTS.md` · checkout: `<worktree
|
|
42
|
-
> path or main checkout>` · ports/db: `<slot info, or defaults>`."
|
|
43
|
-
|
|
44
|
-
The gate hooks fire on the agent's Bash calls too — compose/migrate confirmations still reach the
|
|
45
|
-
human; that's expected.
|
|
46
|
-
|
|
47
|
-
## 3. Relay the verdict
|
|
48
|
-
|
|
49
|
-
- Print the agent's return as-is (verdict + ❌ lines + report path) — it is already minimal.
|
|
50
|
-
- Append ONE metrics line to `pipeline-metrics.jsonl` (main-checkout path + rules in `/build` §4,
|
|
51
|
-
`phase: "smoke"`), chaining the opt-in usage ping in the same Bash call (results = `PASS` or
|
|
52
|
-
`FAIL:<n>` failing flows).
|
|
53
|
-
- **PASS** → tell the human to run `/review $ARGUMENTS`. **FAIL** → the failures are findings: feed
|
|
54
|
-
them to `/fix $ARGUMENTS`, re-run `/smoke` after. Either way the report is on disk —
|
|
55
|
-
**recommend a `/clear`** before the next command.
|