cohorte 1.4.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -82,6 +82,13 @@ const lines = [
82
82
  // Case 8: a slash token that is not a command must not invent one.
83
83
  user(1800, 'look at the /usr/local/share directory and report what you find there'),
84
84
  assistant('m7', 1805, 'claude-opus-5', { input_tokens: 0, output_tokens: 10 }),
85
+ // Case 9: a long prompt that merely DISCUSSES a command is not an invocation of it.
86
+ // Without the length gate, writing about /review bills the conversation to /review —
87
+ // which is what happened in cohorte's own repo while the pipeline was being designed.
88
+ user(2400, 'I want to talk through how /review behaves when a surface has no findings at '
89
+ + 'all, because the verdict logic there is what produced the false green we saw last week '
90
+ + 'and I am not convinced the fix covers the case where every reviewer dies at once.'),
91
+ assistant('m8', 2405, 'claude-opus-5', { input_tokens: 0, output_tokens: 20 }),
85
92
  ];
86
93
  fs.writeFileSync(path.join(projectDir, `${SESSION}.jsonl`),
87
94
  lines.map((l) => JSON.stringify(l)).join('\n') + '\n');
@@ -107,18 +114,20 @@ const chat = out.commands.find((c) => c.command === '(chat)');
107
114
  const review = out.commands.find((c) => c.command === '/review');
108
115
 
109
116
  console.log('test-metrics');
110
- check('the mid-command task-notification did not split the run', out.totals.runs, 4);
117
+ check('the mid-command task-notification did not split the run', out.totals.runs, 5);
111
118
  check('/build is one run, not three', build.runs, 1);
112
119
  check('duplicate lines of one response are billed once', build.tokens.output, 1000 + 500 + 2000);
113
120
  check('the <synthetic> message contributed no tokens', build.tokens.output < 999999, true);
114
121
  check('cache-write tokens are kept on their own tier', build.tokens.cacheWrite5m, 1000);
115
122
  check('cache-read tokens are kept on their own tier', build.tokens.cacheRead, 10000);
116
123
  check('the subagent was attributed to the command that spawned it', build.agents.total, 1);
117
- check('the second prompt is a separate (chat) run', chat.runs, 2);
124
+ check('the second prompt is a separate (chat) run', chat.runs, 3);
118
125
  check('a command named inside prose is attributed to that command', review && review.runs, 1);
119
126
  check('a short steer continues the run instead of opening a new one', review.continuations, 1);
120
127
  check('the continued turn counts toward the command it continued', review.tokens.output, 60 + 70);
121
- check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10);
128
+ check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10 + 20);
129
+ check('a long prompt that discusses a command is not counted as running it',
130
+ review.runs, 1);
122
131
 
123
132
  // opus-5 $5 in / $25 out per MTok; 5m cache write 1.25x input, cache read 0.1x input.
124
133
  // m1 100*5 + 1000*25 + 1000*6.25 + 10000*0.5 = 36750
@@ -22,7 +22,7 @@ const frontmatter = (text) => {
22
22
  // Mechanical commands must pin model: sonnet (otherwise the lead's
23
23
  // orchestration turn silently bills at the session model — Opus/Fable).
24
24
  // Interactive commands must stay unpinned (they inherit on purpose).
25
- const PINNED = ["build", "review", "fix", "smoke", "ship", "audit",
25
+ const PINNED = ["build", "review", "fix", "ship", "audit",
26
26
  "refactor", "doctor", "align-ds", "update-pipeline"];
27
27
  const UNPINNED = ["brainstorm", "spec", "init-pipeline"];
28
28
 
@@ -43,7 +43,7 @@ for (const f of readdirSync(join(root, "core/commands"))) {
43
43
  // Every non-template agent needs name/tools/model, and must be shipped by
44
44
  // both installers (a new agent that install.sh doesn't copy never reaches
45
45
  // a global install — the exact bug that motivated this check).
46
- const AGENT_MODEL = { review: "sonnet", release: "haiku", smoke: "sonnet",
46
+ const AGENT_MODEL = { review: "sonnet", release: "haiku",
47
47
  "profile-reader": "haiku" };
48
48
  const installSh = read("install.sh");
49
49
  const installPs1 = read("install.ps1");
@@ -90,7 +90,7 @@ for (const path of allDocs) {
90
90
  }
91
91
  for (const m of text.matchAll(/subagent_type:\s*(?:`|)([a-z-]+)(?:`|)/g)) {
92
92
  const t = m[1];
93
- if (["review", "release", "smoke", "profile-reader"].includes(t)) continue;
93
+ if (["review", "release", "profile-reader"].includes(t)) continue;
94
94
  if (t.startsWith("<")) continue; // <surface.agent> placeholder
95
95
  if (!existsSync(join(root, "core/agents", `${t}.md`)))
96
96
  fail(path, `dispatches subagent_type ${t} with no core/agents/${t}.md`);
@@ -115,9 +115,9 @@ if (!existsSync(steps) || readdirSync(steps).length === 0)
115
115
 
116
116
  // ── telemetry coverage ──────────────────────────────────────────────────────
117
117
  // The funnel is only readable if every one of its stages pings — a single missing
118
- // one silently truncates it (that is how /smoke, /review and /fix went unreported
118
+ // one silently truncates it (that is how /review and /fix went unreported
119
119
  // until 1.2.3). The phase list here must match SCHEMA.md §Telemetry's table.
120
- const FUNNEL = ["brainstorm", "spec", "build", "smoke", "review", "fix", "ship"];
120
+ const FUNNEL = ["brainstorm", "spec", "build", "review", "fix", "ship"];
121
121
  for (const c of FUNNEL)
122
122
  if (!/usage ping/i.test(read(`core/commands/${c}.md`)))
123
123
  fail(`core/commands/${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
@@ -1,63 +0,0 @@
1
- ---
2
- name: smoke
3
- description: Executes the end-to-end smoke run for one feature in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — then stages the SMOKE REPORT. Dispatched by /smoke. Observes honestly, never fixes anything.
4
- tools: Read, Write, Grep, Glob, Bash, DesignSync
5
- model: sonnet
6
- ---
7
-
8
- You are the **smoke** agent for one feature. You actually run the built feature — `/review` audits
9
- code read-only; nobody has executed it yet. You verify it *works*; you never fix it (failures go
10
- through `/fix`). Observe honestly: report what happened, not what should have happened.
11
-
12
- > **First action, always:** read `PIPELINE.md` §`pipeline-profile`: `commands` (migrate/dev),
13
- > `isolation` (worktree, slot ports, db), `contract`, `design`, `surfaces` — then the spec
14
- > `specs/<id>.md` (§5 contract, §8 flows, §9 acceptance).
15
-
16
- ## Your inputs (supplied at dispatch — you have no memory)
17
-
18
- 1. The feature id and spec path `specs/<id>.md`.
19
- 2. The contract path `<contract.path>/<id>.<ext>`.
20
- 3. The checkout to work in: the worktree path + slot ports/db, or the main checkout on the feature branch.
21
-
22
- ## Keep your own context lean
23
-
24
- Redirect every bulky output to a file and inspect it with `grep`/`jq` — never print full curl bodies,
25
- server logs, or poll loops into your transcript. `curl -s … -o /tmp/resp.json -w '%{http_code}'` then
26
- assert on the pieces you need.
27
-
28
- ## 1. Bring the feature up
29
-
30
- - Work in the checkout your dispatch names. Infra as needed: the compose stack if one is declared
31
- (the gate will ask — that's expected), then `commands.migrate`, then `commands.dev` **in the
32
- background**. Wait for ready (poll the ports), don't assume.
33
-
34
- ## 2. Exercise the contract (the real server, not the tests)
35
-
36
- - Hit a representative set of spec §5 endpoints with `curl`: every route domain, every auth level,
37
- at least one error case per class (validation `422`, unauthenticated `401`, wrong-role `403`,
38
- conflict `409`). Compare status + response envelope against the contract.
39
- - If `rbac.enabled`: verify at least one denial per role boundary the spec declares.
40
- - A mismatch is a FAIL entry with the exact command, expected, and actual — precise enough for a
41
- stateless `/fix` agent.
42
-
43
- ## 3. Exercise the UI (only if a touched surface has `uses_design`)
44
-
45
- - Drive the spec §8 flows against the running app, **mobile viewport first** (375px), then desktop.
46
- - If a browser/screenshot tool is available (a project driver, playwright, an agent browser), capture
47
- each §8 screen and compare against the feature's design pages: each `design_files` entry is a full
48
- `https://claude.ai/design/p/<projectId>?file=<file>` link — extract its `<projectId>` (the `/p/…`
49
- segment) + `<file>` (the `?file=` query) and fetch read-only via `DesignSync get_file(<projectId>,
50
- <file>)`. Compare layout, states (empty/loading/error/suppressed…), copy language. Note deviations.
51
- - No browser tooling available ⇒ **say so and skip the visual diff** — never claim a visual check
52
- you didn't perform.
53
-
54
- ## 4. Stage the SMOKE REPORT, tear down, return
55
-
56
- - One line per check: ✅/❌ · what was exercised · (on ❌) command → expected vs actual.
57
- - **Write the full report to `specs/reports/<id>.md`** (overwrite) — the same gitignored buffer
58
- `/review` uses, so a `/fix` after a `/clear` still has the failures.
59
- - Tear down what you started (kill the dev server); leave shared infra as you found it.
60
- - **Your return to the lead is ONLY:** the verdict line (`PASS` / `FAIL:<n>`), **at most 10 ❌
61
- lines** — one line each (`❌ <flow/endpoint> · expected <x> got <y>`), no command output, no code
62
- or body excerpts; more than 10 ⇒ keep the 10 most severe and add `+<n> more — see the report` —
63
- and `Full report: specs/reports/<id>.md`. No logs, no bodies, no screenshots.
@@ -1,55 +0,0 @@
1
- ---
2
- model: sonnet
3
- description: Exercise the built feature end-to-end in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — before /review.
4
- argument-hint: <feature_id>
5
- ---
6
-
7
- You are the **lead**. Dispatch the smoke run for feature **$ARGUMENTS** — the `smoke` agent actually
8
- runs it so the bulky output (curl bodies, server logs, screenshots, design payloads) never enters
9
- your own context, which is re-sent every turn.
10
-
11
- > Read `PIPELINE.md` §`pipeline-profile`: `isolation` (worktree, slot ports, db), `contract.path`
12
- > and `commands`. _Skip the re-read if it's already in your context this session and unmodified since._
13
- >
14
- > **Kanban:** none here — `/review` owns the → **Review** move (running both duplicated it).
15
-
16
- ## 0. Deterministic pre-flight — no agents while red
17
-
18
- Same gate as `/review` §0, run **in the feature's checkout** (§1 resolves it — resolve first, then
19
- preflight): `<core>/pipeline/scripts/preflight.sh specs/reports/$ARGUMENTS.preflight.txt
20
- "<commands.typecheck>" "<commands.lint_quiet, else lint>" "<commands.test_quiet, else test>"`.
21
- Non-zero exit ⇒ the raw last-40 lines were already printed — **STOP, relay them verbatim, spawn NO
22
- agent**: booting infra to smoke-test code that doesn't compile wastes the whole run. Zero exit ⇒
23
- the `.claude/preflight.ok` stamp lets the gate hook pass your `smoke` dispatch. Script absent
24
- (older core) ⇒ run the commands yourself redirected to the same file, aborting on the first failure.
25
- Note the epoch (`date +%s`) in the same call — §3's metrics line needs it.
26
-
27
- ## 1. Resolve the checkout
28
-
29
- With `isolation.enabled`: the sibling worktree (`../<slug>-$ARGUMENTS`, its slot's ports + db from
30
- `.worktrees/slots.tsv`); otherwise the main checkout on the feature branch.
31
-
32
- ## 2. Dispatch ONE `smoke` agent
33
-
34
- Keep the prompt byte-identical across features except the variable block at the END (prompt-cache
35
- prefix):
36
-
37
- > `subagent_type: smoke` — "Smoke-test one feature. Read `PIPELINE.md` first. Bring it up, exercise
38
- > the contract and the §8 UI flows, stage the full SMOKE REPORT to the report buffer, tear down, and
39
- > return only the capped verdict your agent instructions define (verdict + ❌ lines, no logs). —
40
- > Variable slots: feature `$ARGUMENTS` · spec: `specs/$ARGUMENTS.md` · contract:
41
- > `<contract.path>/$ARGUMENTS.<ext>` · report: `specs/reports/$ARGUMENTS.md` · checkout: `<worktree
42
- > path or main checkout>` · ports/db: `<slot info, or defaults>`."
43
-
44
- The gate hooks fire on the agent's Bash calls too — compose/migrate confirmations still reach the
45
- human; that's expected.
46
-
47
- ## 3. Relay the verdict
48
-
49
- - Print the agent's return as-is (verdict + ❌ lines + report path) — it is already minimal.
50
- - Append ONE metrics line to `pipeline-metrics.jsonl` (main-checkout path + rules in `/build` §4,
51
- `phase: "smoke"`), chaining the opt-in usage ping in the same Bash call (results = `PASS` or
52
- `FAIL:<n>` failing flows).
53
- - **PASS** → tell the human to run `/review $ARGUMENTS`. **FAIL** → the failures are findings: feed
54
- them to `/fix $ARGUMENTS`, re-run `/smoke` after. Either way the report is on disk —
55
- **recommend a `/clear`** before the next command.