cohorte 1.3.4 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/CHANGELOG.md +110 -0
  2. package/README.md +36 -16
  3. package/bin/cli.js +34 -4
  4. package/core/agents/profile-reader.md +22 -0
  5. package/core/commands/build.md +4 -4
  6. package/core/commands/doctor.md +10 -8
  7. package/core/commands/fix.md +5 -5
  8. package/core/commands/loop.md +61 -0
  9. package/core/commands/review.md +38 -5
  10. package/core/hooks/gate.py +4 -4
  11. package/core/templates/spec.template.md +1 -1
  12. package/core/templates/steps/init-pipeline/02-interview-gaps.md +1 -1
  13. package/core/templates/steps/init-pipeline/04-write-render.md +8 -4
  14. package/core/workflows/audit.js +68 -4
  15. package/core/workflows/refactor.js +69 -4
  16. package/core/workflows/review.js +73 -4
  17. package/dashboard/dist/assets/index-8owBnqyv.js +43 -0
  18. package/dashboard/dist/assets/{index-AFQnlfjO.css → index-dkO8UUVl.css} +1 -1
  19. package/dashboard/dist/index.html +2 -2
  20. package/dashboard/server/doctor.js +15 -5
  21. package/dashboard/server/index.js +7 -0
  22. package/dashboard/server/metrics.js +6 -5
  23. package/dashboard/server/usage.js +61 -0
  24. package/install.ps1 +4 -1
  25. package/install.sh +5 -2
  26. package/package.json +1 -1
  27. package/profile/PIPELINE.template.md +3 -3
  28. package/profile/SCHEMA.md +21 -44
  29. package/scripts/loop.sh +189 -0
  30. package/scripts/metrics/collect.mjs +504 -0
  31. package/scripts/metrics/prices.json +39 -0
  32. package/scripts/preflight.sh +2 -2
  33. package/scripts/telemetry-send.sh +5 -2
  34. package/scripts/test-dashboard.mjs +29 -3
  35. package/scripts/test-gate.mjs +1 -2
  36. package/scripts/test-metrics.mjs +144 -0
  37. package/scripts/test-workflows.mjs +56 -178
  38. package/scripts/validate-core.mjs +10 -9
  39. package/core/agents/smoke.md +0 -63
  40. package/core/commands/cycle.md +0 -61
  41. package/core/commands/smoke.md +0 -55
  42. package/core/workflows/cycle.js +0 -513
  43. package/dashboard/dist/assets/index-DLBzciIC.js +0 -43
@@ -0,0 +1,504 @@
1
+ #!/usr/bin/env node
2
+ // collect.mjs — reconstruct per-command cost and runtime from Claude Code's own transcripts.
3
+ //
4
+ // node scripts/metrics/collect.mjs [projectRoot] [--json] [--runs] [--since=ISO] [--days=N]
5
+ //
6
+ // WHY THIS EXISTS
7
+ // `.claude/pipeline-metrics.jsonl` is written by the model itself (each command file tells
8
+ // it to append a line). That makes it unreliable — a command that ends early, errors, or
9
+ // simply forgets writes nothing, and it can never report tokens because the model does not
10
+ // know its own usage. Claude Code, meanwhile, already logs every API response it makes to
11
+ // ~/.claude/projects/<slug>/<sessionId>.jsonl with exact `usage` and timestamps, and every
12
+ // subagent to <sessionId>/subagents/agent-<id>.jsonl. That is ground truth, it needs no
13
+ // cooperation from the model, and it is retroactive: this script works on runs that already
14
+ // happened. Nothing here writes to the pipeline — it is a pure reader.
15
+ //
16
+ // THREE THINGS THAT ARE EASY TO GET WRONG, HANDLED HERE
17
+ // 1. One API response is written as SEVERAL transcript lines (one per content block:
18
+ // thinking, text, tool_use, tool_use), and EACH line repeats the full `usage` object.
19
+ // Summing lines inflates tokens ~1.8x. We dedupe by message.id.
20
+ // 2. Feature work happens in git worktrees, whose cwd hashes to a DIFFERENT project slug.
21
+ // Scanning only the main checkout's slug silently drops most of a multi-surface run. We match
22
+ // sessions by their recorded `cwd` against `git worktree list`.
23
+ // 3. Subagent spend lives in a separate file tree and is invisible in the parent
24
+ // transcript. For cohorte that is the majority of the cost, so we walk subagents/ and
25
+ // attribute each agent to the command segment that spawned it (via meta.toolUseId,
26
+ // falling back to timestamp containment for agents spawned by the Workflow runner).
27
+
28
+ import fs from 'node:fs';
29
+ import path from 'node:path';
30
+ import os from 'node:os';
31
+ import { execFileSync } from 'node:child_process';
32
+ import { fileURLToPath } from 'node:url';
33
+
34
+ const HERE = path.dirname(fileURLToPath(import.meta.url));
35
+ const PRICES = JSON.parse(fs.readFileSync(path.join(HERE, 'prices.json'), 'utf8'));
36
+
37
+ // A gap longer than this between two API responses is the human thinking, reading, or away
38
+ // — not the command working. `wall` keeps it, `active` drops it. Without the split, a
39
+ // command left open over lunch reports a two-hour runtime and poisons every median.
40
+ const IDLE_GAP_S = 120;
41
+
42
+ // A prompt this short with no command in it ("continue", "go", "ok next") is the human
43
+ // steering a run that is already going, not starting a new one. Without this, a single
44
+ // /review driven by three "continue"s reports as one /review plus three anonymous chat
45
+ // runs, and three quarters of its cost lands under (chat).
46
+ const CONTINUATION_MAX_CHARS = 40;
47
+
48
+ // Commands are recognised two ways. `<command-name>` is emitted only when the whole prompt
49
+ // IS the slash command; in practice people write "move on branding-ramp and /review", which
50
+ // the harness records as ordinary prose. So we also look for an inline mention, checked
51
+ // against the real command list rather than any /token — otherwise a file path like
52
+ // /usr/bin or a URL fragment would invent commands that were never run.
53
+ function knownCommands() {
54
+ const names = new Set();
55
+ for (const dir of [path.join(HERE, '..', '..', 'core', 'commands'),
56
+ path.join(process.env.CLAUDE_CONFIG_DIR || path.join(os.homedir(), '.claude'), 'pipeline', 'commands')]) {
57
+ try {
58
+ for (const f of fs.readdirSync(dir)) if (f.endsWith('.md')) names.add(f.slice(0, -3));
59
+ } catch {}
60
+ }
61
+ // Fallback for a collector run outside the package (e.g. copied into a repo on its own).
62
+ if (!names.size) {
63
+ for (const n of ['brainstorm', 'spec', 'build', 'review', 'fix', 'ship',
64
+ 'audit', 'refactor', 'align-ds', 'doctor', 'init-pipeline', 'update-pipeline']) names.add(n);
65
+ }
66
+ // Retired commands. The list above is read from the shipped core, so a command that is
67
+ // removed stops being recognised — and every run of it already in the transcripts silently
68
+ // reclassifies as (chat), rewriting history and inflating the catch-all bucket. Keep the
69
+ // names here so past runs stay attributed to what actually ran.
70
+ for (const n of ['cycle', 'smoke']) names.add(n);
71
+ return names;
72
+ }
73
+ const COMMANDS = knownCommands();
74
+
75
+ // An invocation is a short instruction that is mostly the command ("move on branding-ramp
76
+ // and /review"). A long prompt that happens to name one is someone TALKING ABOUT the
77
+ // command — a bug report, a design discussion, a pasted transcript. Counting those as runs
78
+ // inflates a command's run count and cost with conversation that never invoked it, which is
79
+ // exactly what happened in cohorte's own repo while this pipeline was being discussed.
80
+ // Only inline mentions are length-gated; an explicit <command-name> is always an invocation.
81
+ const MENTION_MAX_CHARS = 120;
82
+
83
+ function commandIn(text) {
84
+ const explicit = /<command-name>\s*(\/?[\w:-]+)\s*<\/command-name>/.exec(text);
85
+ if (explicit) return explicit[1].replace(/^\//, '');
86
+ if (text.trim().length > MENTION_MAX_CHARS) return null;
87
+ // Last mention wins: "finish /build then /review" ends on the one being asked for.
88
+ let found = null;
89
+ for (const m of text.matchAll(/(?:^|\s)\/([a-z][a-z0-9-]{2,})\b/g)) {
90
+ if (COMMANDS.has(m[1])) found = m[1];
91
+ }
92
+ return found;
93
+ }
94
+
95
+ // ── paths ────────────────────────────────────────────────────────────────────────────────
96
+
97
+ const norm = (p) => String(p || '').replace(/\\/g, '/').replace(/\/+$/, '').toLowerCase();
98
+
99
+ function git(args, cwd) {
100
+ try {
101
+ return execFileSync('git', args, { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] });
102
+ } catch { return ''; }
103
+ }
104
+
105
+ // Every checkout that belongs to this repo: the main one plus every live feature worktree.
106
+ // A cohorte run spreads its surfaces across worktrees, so this set is what makes a feature
107
+ // add up instead of reporting only whatever the human happened to type in the main window.
108
+ function repoCheckouts(root) {
109
+ const out = new Set([norm(root)]);
110
+ const common = git(['rev-parse', '--git-common-dir'], root).trim();
111
+ if (common) out.add(norm(path.resolve(root, common, '..')));
112
+ for (const line of git(['worktree', 'list', '--porcelain'], root).split(/\r?\n/)) {
113
+ if (line.startsWith('worktree ')) out.add(norm(line.slice(9)));
114
+ }
115
+ return out;
116
+ }
117
+
118
+ const projectsDir = () =>
119
+ path.join(process.env.CLAUDE_CONFIG_DIR || path.join(os.homedir(), '.claude'), 'projects');
120
+
121
+ // Read only the head of a transcript to decide whether it belongs to this repo. Transcripts
122
+ // run to tens of MB and most of them belong to other projects; fully parsing every file to
123
+ // find the handful that match turns a 2-second command into a 2-minute one.
124
+ function sessionCwd(file) {
125
+ let fd;
126
+ try {
127
+ fd = fs.openSync(file, 'r');
128
+ const buf = Buffer.alloc(65536);
129
+ const n = fs.readSync(fd, buf, 0, buf.length, 0);
130
+ const m = /"cwd"\s*:\s*"((?:[^"\\]|\\.)*)"/.exec(buf.subarray(0, n).toString('utf8'));
131
+ return m ? JSON.parse(`"${m[1]}"`) : null;
132
+ } catch { return null; }
133
+ finally { if (fd !== undefined) try { fs.closeSync(fd); } catch {} }
134
+ }
135
+
136
+ function findSessions(checkouts) {
137
+ const root = projectsDir();
138
+ let dirs;
139
+ try { dirs = fs.readdirSync(root, { withFileTypes: true }); }
140
+ catch { return []; }
141
+
142
+ const found = [];
143
+ for (const d of dirs) {
144
+ if (!d.isDirectory()) continue;
145
+ const dir = path.join(root, d.name);
146
+ let files;
147
+ try { files = fs.readdirSync(dir); } catch { continue; }
148
+ for (const f of files) {
149
+ if (!f.endsWith('.jsonl')) continue;
150
+ const file = path.join(dir, f);
151
+ const cwd = sessionCwd(file);
152
+ if (!cwd) continue;
153
+ // A worktree's cwd can be a SUBDIRECTORY of the checkout (an agent cd'd into a
154
+ // package), so prefix-match rather than compare for equality.
155
+ const c = norm(cwd);
156
+ if (![...checkouts].some((k) => c === k || c.startsWith(k + '/'))) continue;
157
+ found.push({ file, dir: path.join(dir, path.basename(f, '.jsonl')), cwd });
158
+ }
159
+ }
160
+ return found;
161
+ }
162
+
163
+ // ── pricing ──────────────────────────────────────────────────────────────────────────────
164
+
165
+ function rates(model, speed) {
166
+ let best = null;
167
+ for (const key of Object.keys(PRICES.models)) {
168
+ if (String(model || '').startsWith(key) && (!best || key.length > best.length)) best = key;
169
+ }
170
+ if (!best) return null;
171
+ const entry = PRICES.models[best];
172
+ return (speed === 'fast' && entry.fast) ? entry.fast : entry;
173
+ }
174
+
175
+ const emptyTokens = () => ({ input: 0, output: 0, cacheWrite5m: 0, cacheWrite1h: 0, cacheRead: 0 });
176
+
177
+ function addUsage(acc, usage, model, speed) {
178
+ // `<synthetic>` messages are harness-authored (API error notices, interrupt markers).
179
+ // They carry a usage block but cost nothing — billing them invents spend.
180
+ if (model === '<synthetic>') return;
181
+ const cc = usage.cache_creation || {};
182
+ let w5 = cc.ephemeral_5m_input_tokens || 0;
183
+ const w1 = cc.ephemeral_1h_input_tokens || 0;
184
+ // Older transcripts carry only the flat total with no TTL breakdown. Bill it as 5m —
185
+ // the cheaper of the two, so an unknown-TTL run under-reports rather than inflates.
186
+ if (!w5 && !w1) w5 = usage.cache_creation_input_tokens || 0;
187
+
188
+ const t = { input: usage.input_tokens || 0, output: usage.output_tokens || 0,
189
+ cacheWrite5m: w5, cacheWrite1h: w1, cacheRead: usage.cache_read_input_tokens || 0 };
190
+ for (const k of Object.keys(t)) acc.tokens[k] += t[k];
191
+
192
+ const r = rates(model, speed);
193
+ if (!r) { acc.unpriced.add(model || 'unknown'); return; }
194
+ const m = PRICES.multipliers;
195
+ acc.cost += (
196
+ t.input * r.input +
197
+ t.output * r.output +
198
+ t.cacheWrite5m * r.input * m.cacheWrite5m +
199
+ t.cacheWrite1h * r.input * m.cacheWrite1h +
200
+ t.cacheRead * r.input * m.cacheRead
201
+ ) / 1e6;
202
+
203
+ const byModel = acc.models[model] || (acc.models[model] = emptyTokens());
204
+ for (const k of Object.keys(t)) byModel[k] += t[k];
205
+ }
206
+
207
+ // ── transcript parsing ───────────────────────────────────────────────────────────────────
208
+
209
+ const userText = (msg) => {
210
+ const c = msg && msg.content;
211
+ if (typeof c === 'string') return c;
212
+ if (!Array.isArray(c)) return '';
213
+ return c.filter((b) => b && b.type === 'text').map((b) => b.text || '').join('\n');
214
+ };
215
+
216
+ const isToolResultTurn = (msg) =>
217
+ Array.isArray(msg && msg.content) && msg.content.some((b) => b && b.type === 'tool_result');
218
+
219
+ function newSegment(label, ts) {
220
+ return {
221
+ label, startTs: ts, endTs: ts,
222
+ tokens: emptyTokens(), cost: 0, models: {}, unpriced: new Set(),
223
+ activeS: 0, lastTs: ts, turns: 0, continuations: 0,
224
+ toolUseIds: new Set(), agents: [],
225
+ };
226
+ }
227
+
228
+ // One segment = one command invocation (or one free-form chat turn). Boundaries are user
229
+ // turns that are NOT tool results — a tool result is the harness feeding the loop, not the
230
+ // human starting something new.
231
+ function parseSession(file) {
232
+ const segments = [];
233
+ let seg = null;
234
+ const seenMessageIds = new Set();
235
+
236
+ let raw;
237
+ try { raw = fs.readFileSync(file, 'utf8'); } catch { return segments; }
238
+
239
+ for (const line of raw.split(/\r?\n/)) {
240
+ if (!line) continue;
241
+ let e;
242
+ try { e = JSON.parse(line); } catch { continue; }
243
+ const ts = e.timestamp ? Date.parse(e.timestamp) : NaN;
244
+
245
+ if (e.type === 'user') {
246
+ const msg = e.message || {};
247
+ if (isToolResultTurn(msg) || e.isMeta || e.isSidechain) continue;
248
+ const text = userText(msg);
249
+ const cmd = commandIn(text);
250
+ // Not every user-role turn is the human starting something. The harness injects
251
+ // turns mid-run — a local-command echo, and (critically for cohorte) a
252
+ // <task-notification> when a background agent finishes. Those arrive DURING a
253
+ // a /build; treating them as boundaries chops one command into several
254
+ // cheap-looking fragments and strands the agent spend in the wrong segment.
255
+ if (!cmd && /<(local-command-(stdout|stderr)|task-notification|system-reminder)>/.test(text)) continue;
256
+ // A short steer with no command keeps the current run open rather than opening a new
257
+ // one — see CONTINUATION_MAX_CHARS.
258
+ if (!cmd && seg && text.trim().length <= CONTINUATION_MAX_CHARS) { seg.continuations += 1; continue; }
259
+ seg = newSegment(cmd ? '/' + cmd : '(chat)', ts);
260
+ segments.push(seg);
261
+ continue;
262
+ }
263
+
264
+ if (e.type !== 'assistant' || !seg) continue;
265
+ const msg = e.message || {};
266
+
267
+ // Dedupe: the same API response is written once per content block, each copy carrying
268
+ // the full usage. See header note 1 — this is the single biggest correctness trap.
269
+ const id = msg.id;
270
+ if (id && seenMessageIds.has(id)) {
271
+ for (const b of msg.content || []) if (b && b.type === 'tool_use' && b.id) seg.toolUseIds.add(b.id);
272
+ if (!Number.isNaN(ts)) seg.endTs = Math.max(seg.endTs, ts);
273
+ continue;
274
+ }
275
+ if (id) seenMessageIds.add(id);
276
+
277
+ if (msg.usage) addUsage(seg, msg.usage, msg.model, msg.usage.speed);
278
+ for (const b of msg.content || []) if (b && b.type === 'tool_use' && b.id) seg.toolUseIds.add(b.id);
279
+
280
+ seg.turns += 1;
281
+ if (!Number.isNaN(ts)) {
282
+ const gap = (ts - seg.lastTs) / 1000;
283
+ if (gap > 0 && gap <= IDLE_GAP_S) seg.activeS += gap;
284
+ seg.lastTs = ts;
285
+ seg.endTs = Math.max(seg.endTs, ts);
286
+ }
287
+ }
288
+ return segments;
289
+ }
290
+
291
+ // ── subagents ────────────────────────────────────────────────────────────────────────────
292
+
293
+ function readSubagents(sessionDir) {
294
+ const dir = path.join(sessionDir, 'subagents');
295
+ let files;
296
+ try { files = fs.readdirSync(dir); } catch { return []; }
297
+
298
+ const agents = [];
299
+ for (const f of files) {
300
+ if (!f.endsWith('.jsonl')) continue;
301
+ const base = f.slice(0, -'.jsonl'.length);
302
+ let meta = {};
303
+ try { meta = JSON.parse(fs.readFileSync(path.join(dir, base + '.meta.json'), 'utf8')); } catch {}
304
+
305
+ const acc = { tokens: emptyTokens(), cost: 0, models: {}, unpriced: new Set() };
306
+ let startTs = Infinity, endTs = -Infinity, turns = 0;
307
+ const seen = new Set();
308
+
309
+ let raw;
310
+ try { raw = fs.readFileSync(path.join(dir, f), 'utf8'); } catch { continue; }
311
+ for (const line of raw.split(/\r?\n/)) {
312
+ if (!line) continue;
313
+ let e;
314
+ try { e = JSON.parse(line); } catch { continue; }
315
+ if (e.type !== 'assistant') continue;
316
+ const msg = e.message || {};
317
+ if (msg.id && seen.has(msg.id)) continue;
318
+ if (msg.id) seen.add(msg.id);
319
+ if (msg.usage) addUsage(acc, msg.usage, msg.model, msg.usage.speed);
320
+ turns += 1;
321
+ const ts = e.timestamp ? Date.parse(e.timestamp) : NaN;
322
+ if (!Number.isNaN(ts)) { startTs = Math.min(startTs, ts); endTs = Math.max(endTs, ts); }
323
+ }
324
+
325
+ agents.push({
326
+ id: base.replace(/^agent-/, ''),
327
+ agentType: meta.agentType || 'unknown',
328
+ description: meta.description || '',
329
+ toolUseId: meta.toolUseId || null,
330
+ spawnDepth: meta.spawnDepth ?? null,
331
+ turns, startTs, endTs, ...acc,
332
+ });
333
+ }
334
+ return agents;
335
+ }
336
+
337
+ function attachAgents(segments, agents) {
338
+ for (const a of agents) {
339
+ // Preferred link: the Task/Agent tool_use that spawned it. Workflow-spawned agents can
340
+ // carry a toolUseId the parent transcript never recorded, so fall back to which segment
341
+ // was running when the agent started.
342
+ let seg = a.toolUseId ? segments.find((s) => s.toolUseIds.has(a.toolUseId)) : null;
343
+ if (!seg && Number.isFinite(a.startTs)) {
344
+ seg = segments.find((s) => a.startTs >= s.startTs && a.startTs <= s.endTs + 5 * 60_000);
345
+ }
346
+ if (!seg) continue;
347
+ seg.agents.push(a);
348
+ for (const k of Object.keys(seg.tokens)) seg.tokens[k] += a.tokens[k];
349
+ seg.cost += a.cost;
350
+ for (const [m, t] of Object.entries(a.models)) {
351
+ const dst = seg.models[m] || (seg.models[m] = emptyTokens());
352
+ for (const k of Object.keys(t)) dst[k] += t[k];
353
+ }
354
+ for (const m of a.unpriced) seg.unpriced.add(m);
355
+ if (Number.isFinite(a.endTs)) seg.endTs = Math.max(seg.endTs, a.endTs);
356
+ // Agents run in parallel, so their wall time is not additive and cannot be folded into
357
+ // the parent's `active`. Segment `wall` already covers them via endTs; agent runtime is
358
+ // reported separately per command as agentWallS.
359
+ }
360
+ }
361
+
362
+ // ── rollup ───────────────────────────────────────────────────────────────────────────────
363
+
364
+ const pct = (arr, p) => {
365
+ if (!arr.length) return 0;
366
+ const s = [...arr].sort((a, b) => a - b);
367
+ return s[Math.min(s.length - 1, Math.floor((p / 100) * s.length))];
368
+ };
369
+
370
+ function rollup(segments) {
371
+ const by = new Map();
372
+ for (const s of segments) {
373
+ let g = by.get(s.label);
374
+ if (!g) {
375
+ g = { command: s.label, runs: 0, continuations: 0, tokens: emptyTokens(), cost: 0, models: {},
376
+ wall: [], active: [], agentCounts: [], agentWall: [], turns: [], unpriced: new Set() };
377
+ by.set(s.label, g);
378
+ }
379
+ g.runs += 1;
380
+ g.continuations += s.continuations;
381
+ for (const k of Object.keys(s.tokens)) g.tokens[k] += s.tokens[k];
382
+ g.cost += s.cost;
383
+ for (const [m, t] of Object.entries(s.models)) {
384
+ const dst = g.models[m] || (g.models[m] = emptyTokens());
385
+ for (const k of Object.keys(t)) dst[k] += t[k];
386
+ }
387
+ for (const m of s.unpriced) g.unpriced.add(m);
388
+ g.wall.push(Math.max(0, (s.endTs - s.startTs) / 1000) || 0);
389
+ g.active.push(s.activeS);
390
+ g.turns.push(s.turns);
391
+ g.agentCounts.push(s.agents.length);
392
+ g.agentWall.push(s.agents.reduce((n, a) => n + (Number.isFinite(a.endTs) ? (a.endTs - a.startTs) / 1000 : 0), 0));
393
+ }
394
+
395
+ return [...by.values()]
396
+ .map((g) => ({
397
+ command: g.command,
398
+ runs: g.runs,
399
+ continuations: g.continuations,
400
+ cost: { total: g.cost, perRun: g.cost / g.runs },
401
+ tokens: g.tokens,
402
+ tokensPerRun: Object.fromEntries(Object.entries(g.tokens).map(([k, v]) => [k, Math.round(v / g.runs)])),
403
+ wallS: { p50: pct(g.wall, 50), p90: pct(g.wall, 90), total: g.wall.reduce((a, b) => a + b, 0) },
404
+ activeS: { p50: pct(g.active, 50), p90: pct(g.active, 90) },
405
+ agents: { perRunP50: pct(g.agentCounts, 50), total: g.agentCounts.reduce((a, b) => a + b, 0),
406
+ serialWallP50S: pct(g.agentWall, 50) },
407
+ turnsP50: pct(g.turns, 50),
408
+ models: g.models,
409
+ unpriced: [...g.unpriced],
410
+ }))
411
+ .sort((a, b) => b.cost.total - a.cost.total);
412
+ }
413
+
414
+ // ── output ───────────────────────────────────────────────────────────────────────────────
415
+
416
+ const fmtUsd = (n) => (n < 0.01 && n > 0 ? '<$0.01' : '$' + n.toFixed(2));
417
+ const fmtTok = (n) => (n >= 1e6 ? (n / 1e6).toFixed(1) + 'M' : n >= 1e3 ? Math.round(n / 1e3) + 'k' : String(n));
418
+ const fmtDur = (s) => (s >= 3600 ? (s / 3600).toFixed(1) + 'h' : s >= 60 ? Math.round(s / 60) + 'm' : Math.round(s) + 's');
419
+
420
+ function table(rows) {
421
+ const head = ['COMMAND', 'RUNS', '$/RUN', '$ TOTAL', 'TOK/RUN', 'OUT/RUN', 'WALL p50', 'ACTIVE p50', 'AGENTS p50'];
422
+ const body = rows.map((r) => [
423
+ r.command,
424
+ String(r.runs),
425
+ fmtUsd(r.cost.perRun),
426
+ fmtUsd(r.cost.total),
427
+ fmtTok(Object.values(r.tokensPerRun).reduce((a, b) => a + b, 0)),
428
+ fmtTok(r.tokensPerRun.output),
429
+ fmtDur(r.wallS.p50),
430
+ fmtDur(r.activeS.p50),
431
+ String(r.agents.perRunP50),
432
+ ]);
433
+ const w = head.map((h, i) => Math.max(h.length, ...body.map((b) => b[i].length)));
434
+ const line = (cells) => cells.map((c, i) => (i === 0 ? c.padEnd(w[i]) : c.padStart(w[i]))).join(' ');
435
+ return [line(head), w.map((n) => '-'.repeat(n)).join(' '), ...body.map(line)].join('\n');
436
+ }
437
+
438
+ // ── main ─────────────────────────────────────────────────────────────────────────────────
439
+
440
+ function main(argv) {
441
+ const args = argv.slice(2);
442
+ const flag = (name) => args.includes('--' + name);
443
+ const opt = (name) => {
444
+ const hit = args.find((a) => a.startsWith(`--${name}=`));
445
+ return hit ? hit.slice(name.length + 3) : null;
446
+ };
447
+ const root = path.resolve(args.find((a) => !a.startsWith('--')) || process.cwd());
448
+
449
+ let since = opt('since') ? Date.parse(opt('since')) : null;
450
+ if (opt('days')) since = Date.now() - Number(opt('days')) * 86400_000;
451
+
452
+ const checkouts = repoCheckouts(root);
453
+ const sessions = findSessions(checkouts);
454
+
455
+ let segments = [];
456
+ for (const s of sessions) {
457
+ const segs = parseSession(s.file);
458
+ attachAgents(segs, readSubagents(s.dir));
459
+ segments.push(...segs);
460
+ }
461
+ if (since) segments = segments.filter((s) => s.startTs >= since);
462
+ // A segment with no API call is a typo or an interrupted prompt, not a run.
463
+ segments = segments.filter((s) => s.turns > 0);
464
+
465
+ const rows = rollup(segments);
466
+ const totals = {
467
+ sessions: sessions.length,
468
+ runs: segments.length,
469
+ cost: rows.reduce((n, r) => n + r.cost.total, 0),
470
+ agents: rows.reduce((n, r) => n + r.agents.total, 0),
471
+ };
472
+
473
+ if (flag('json')) {
474
+ const out = { generatedAt: new Date().toISOString(), projectRoot: root,
475
+ checkouts: [...checkouts], pricesUpdated: PRICES.updated, totals, commands: rows };
476
+ if (flag('runs')) {
477
+ out.runs = segments.map((s) => ({
478
+ command: s.label, startedAt: new Date(s.startTs).toISOString(),
479
+ wallS: Math.max(0, (s.endTs - s.startTs) / 1000), activeS: s.activeS,
480
+ turns: s.turns, continuations: s.continuations, cost: s.cost, tokens: s.tokens,
481
+ agents: s.agents.map((a) => ({ type: a.agentType, description: a.description,
482
+ cost: a.cost, tokens: a.tokens, turns: a.turns })),
483
+ }));
484
+ }
485
+ process.stdout.write(JSON.stringify(out, null, 2) + '\n');
486
+ return 0;
487
+ }
488
+
489
+ if (!sessions.length) {
490
+ console.log(`No Claude Code transcripts found for ${root} under ${projectsDir()}.`);
491
+ return 0;
492
+ }
493
+ console.log(`cohorte metrics — ${root}`);
494
+ console.log(`${totals.runs} runs across ${totals.sessions} sessions, ${totals.agents} subagents, ${fmtUsd(totals.cost)} total`
495
+ + (since ? ` (since ${new Date(since).toISOString().slice(0, 10)})` : ''));
496
+ console.log(`prices as of ${PRICES.updated}; wall excludes nothing, active drops gaps > ${IDLE_GAP_S}s\n`);
497
+ console.log(table(rows));
498
+
499
+ const unpriced = [...new Set(rows.flatMap((r) => r.unpriced))];
500
+ if (unpriced.length) console.log(`\nnot in prices.json (counted, not costed): ${unpriced.join(', ')}`);
501
+ return 0;
502
+ }
503
+
504
+ process.exit(main(process.argv));
@@ -0,0 +1,39 @@
1
+ {
2
+ "_comment": [
3
+ "USD per million tokens, Anthropic first-party API rates. Bedrock/Vertex are",
4
+ "partner-priced and are NOT covered here — a run on those platforms will be",
5
+ "costed at first-party rates and slightly misreported.",
6
+ "Cache rates are derived from `input` via `multipliers` (write 1.25x for the 5m",
7
+ "TTL, 2x for 1h, read 0.1x) rather than restated per model, so a price change",
8
+ "is a one-line edit. `fast` is the fast-mode premium (usage.speed === 'fast').",
9
+ "Model lookup is longest-prefix, so dated ids (claude-haiku-4-5-20251001) hit",
10
+ "their base entry without needing a row of their own."
11
+ ],
12
+ "updated": "2026-07-31",
13
+ "multipliers": {
14
+ "cacheWrite5m": 1.25,
15
+ "cacheWrite1h": 2.0,
16
+ "cacheRead": 0.1
17
+ },
18
+ "models": {
19
+ "claude-fable-5": { "input": 10, "output": 50 },
20
+ "claude-mythos-5": { "input": 10, "output": 50 },
21
+ "claude-opus-5": {
22
+ "input": 5,
23
+ "output": 25,
24
+ "fast": { "input": 10, "output": 50 }
25
+ },
26
+ "claude-opus-4-8": {
27
+ "input": 5,
28
+ "output": 25,
29
+ "fast": { "input": 10, "output": 50 }
30
+ },
31
+ "claude-opus-4-7": { "input": 5, "output": 25 },
32
+ "claude-opus-4-6": { "input": 5, "output": 25 },
33
+ "claude-opus-4-5": { "input": 5, "output": 25 },
34
+ "claude-sonnet-5": { "input": 3, "output": 15 },
35
+ "claude-sonnet-4-6": { "input": 3, "output": 15 },
36
+ "claude-sonnet-4-5": { "input": 3, "output": 15 },
37
+ "claude-haiku-4-5": { "input": 1, "output": 5 }
38
+ }
39
+ }
@@ -1,6 +1,6 @@
1
1
  #!/bin/sh
2
2
  #
3
- # preflight.sh — deterministic phase gate for /review and /smoke.
3
+ # preflight.sh — deterministic phase gate for /review.
4
4
  #
5
5
  # Runs the profile's mechanical checks (typecheck, lint, tests — whatever the caller
6
6
  # passes) BEFORE any agent is spawned. A red gate means the caller aborts and relays
@@ -15,7 +15,7 @@
15
15
  # The caller must stop there — no agents.
16
16
  # - All green: writes `<project>/.claude/preflight.ok` ("<epoch> <HEAD sha>") — the
17
17
  # stamp `hooks/gate.py` checks (gate-config.json `preflight` block) before letting
18
- # review/smoke agents dispatch.
18
+ # review agents dispatch.
19
19
 
20
20
  set -u
21
21
 
@@ -2,7 +2,7 @@
2
2
  # telemetry-send.sh — fire-and-forget anonymous usage ping (SCHEMA.md §Telemetry).
3
3
  #
4
4
  # telemetry-send.sh <phase> <feature_id> <seconds> [results]
5
- # phase brainstorm|spec|build|smoke|review|fix|ship — the feature funnel, and only it.
5
+ # phase brainstorm|spec|build|review|fix|ship — the feature funnel, and only it.
6
6
  # Setup/maintenance commands never ping (SCHEMA.md §Telemetry).
7
7
  # feature the feature id — NEVER sent raw; SHA-256-hashed to 12 hex chars
8
8
  # seconds batch wall-clock
@@ -39,7 +39,10 @@ phase="${1:-}"; feature="${2:-}"; seconds="${3:-0}"; results="${4:-}"
39
39
  # Allowlist the phase here — the collector accepts any string, so a typo in a command
40
40
  # file would silently pollute the dataset with a phantom phase nobody notices.
41
41
  case "$phase" in
42
- brainstorm|spec|build|smoke|review|fix|ship) ;;
42
+ brainstorm|spec|build|review|fix|ship) ;;
43
+ # `smoke` is a RETIRED phase (removed in 1.5.0) — still accepted so a stale install
44
+ # pinging it lands in its own bucket instead of being silently dropped.
45
+ smoke) ;;
43
46
  *) exit 0 ;;
44
47
  esac
45
48
 
@@ -20,6 +20,7 @@ const require = createRequire(import.meta.url);
20
20
  const root = fileURLToPath(new URL("..", import.meta.url));
21
21
  const { parse, parseProfileBlock } = require(join(root, "dashboard/server/yaml.js"));
22
22
  const { metrics } = require(join(root, "dashboard/server/metrics.js"));
23
+ const { usage } = require(join(root, "dashboard/server/usage.js"));
23
24
  const { state, scanSpecs } = require(join(root, "dashboard/server/doctor.js"));
24
25
  const { kanban } = require(join(root, "dashboard/server/kanban.js"));
25
26
  const fleet = require(join(root, "dashboard/server/fleet.js"));
@@ -106,6 +107,25 @@ console.log("metrics.js — the funnel aggregate");
106
107
  check("no metrics file ⇒ present:false", metrics({ projectRoot: scratch() }).present === false);
107
108
  }
108
109
 
110
+ // ── usage.js ─────────────────────────────────────────────────────────────────
111
+ // Wraps the ESM metrics collector for the CJS server. The failure that matters is
112
+ // not a crash: a project with no transcripts must say so, because rendering zeros
113
+ // reads as "this pipeline costs nothing" rather than "nothing was measured".
114
+ console.log("usage.js — the collector bridge");
115
+ {
116
+ const empty = usage({ projectRoot: scratch() });
117
+ check("a project with no transcripts reports present:false", empty.present === false);
118
+ check("…and says why rather than returning silent zeros",
119
+ typeof empty.error === "string" && empty.error.length > 0, JSON.stringify(empty));
120
+
121
+ // Same project twice: the second call must come from cache, or the panel's polling
122
+ // would re-parse tens of MB of transcripts on every refresh.
123
+ const d = scratch();
124
+ const t0 = Date.now(); usage({ projectRoot: d });
125
+ const t1 = Date.now(); usage({ projectRoot: d }); const cached = Date.now() - t1;
126
+ check("a repeated read is served from cache", cached <= Math.max(50, (t1 - t0)), `${cached}ms`);
127
+ }
128
+
109
129
  // ── doctor.js ────────────────────────────────────────────────────────────────
110
130
  console.log("doctor.js — the /doctor port");
111
131
  {
@@ -116,8 +136,14 @@ console.log("doctor.js — the /doctor port");
116
136
  writeFileSync(join(d, "specs", "b.md"), spec("feature_id: b\nstatus: shipped # done"));
117
137
  writeFileSync(join(d, "specs", "c.md"), "no front-matter at all");
118
138
  writeFileSync(join(d, "specs", "_template.md"), spec("status: draft"));
139
+ // /audit writes this file by design and it has no front-matter. Scanning it as a
140
+ // spec made /doctor warn about a file cohorte itself had just created — it fired in
141
+ // every project that had ever run /audit.
142
+ writeFileSync(join(d, "specs", "refactor-backlog.md"), "# Refactor Backlog\n\n## backend\n- [ ] x\n");
119
143
  const specs = scanSpecs(d);
120
144
  eq("_template.md is excluded", specs.length, 3);
145
+ eq("the /audit backlog is not scanned as a spec",
146
+ specs.some(s => s.file === "refactor-backlog.md"), false);
121
147
  eq("front-matter fields are read", specs.find(s => s.id === "a").title, "A");
122
148
  eq("a trailing comment is stripped from status", specs.find(s => s.id === "b").status, "shipped");
123
149
  eq("no front-matter ⇒ id falls back to the filename", specs.find(s => s.file === "c.md").id, "c");
@@ -129,7 +155,7 @@ console.log("doctor.js — the /doctor port");
129
155
  const d = scratch();
130
156
  const gate = {
131
157
  deny: ["x"], ask: ["y"], ask_on_default_branch: ["git push"], default_branch: "main",
132
- preflight: { enabled: true, agents: ["review", "smoke"], max_age_minutes: 30 },
158
+ preflight: { enabled: true, agents: ["review"], max_age_minutes: 30 },
133
159
  };
134
160
  writeFileSync(join(d, "PIPELINE.md"), [
135
161
  "```yaml pipeline-profile",
@@ -147,7 +173,7 @@ console.log("doctor.js — the /doctor port");
147
173
  ' ask_on_default_branch: ["git push"]',
148
174
  " preflight:",
149
175
  " enabled: true",
150
- " agents: [review, smoke]",
176
+ " agents: [review]",
151
177
  " max_age_minutes: 30",
152
178
  "```",
153
179
  ].join("\n"));
@@ -158,7 +184,7 @@ console.log("doctor.js — the /doctor port");
158
184
  writeFileSync(join(d, ".claude", "gate-config.json"), JSON.stringify(gate));
159
185
  writeFileSync(join(d, ".mcp.json"), JSON.stringify({ mcpServers: { serena: {} } }));
160
186
  mkdirSync(join(d, ".claude", "workflows"), { recursive: true });
161
- for (const w of ["review.js", "audit.js", "refactor.js", "cycle.js"]) {
187
+ for (const w of ["review.js", "audit.js", "refactor.js"]) {
162
188
  writeFileSync(join(d, ".claude", "workflows", w), "x");
163
189
  }
164
190
  writeFileSync(join(d, ".claude", "agents", "profile-reader.md"), "x");