cohorte 1.3.4 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +110 -0
- package/README.md +36 -16
- package/bin/cli.js +34 -4
- package/core/agents/profile-reader.md +22 -0
- package/core/commands/build.md +4 -4
- package/core/commands/doctor.md +10 -8
- package/core/commands/fix.md +5 -5
- package/core/commands/loop.md +61 -0
- package/core/commands/review.md +38 -5
- package/core/hooks/gate.py +4 -4
- package/core/templates/spec.template.md +1 -1
- package/core/templates/steps/init-pipeline/02-interview-gaps.md +1 -1
- package/core/templates/steps/init-pipeline/04-write-render.md +8 -4
- package/core/workflows/audit.js +68 -4
- package/core/workflows/refactor.js +69 -4
- package/core/workflows/review.js +73 -4
- package/dashboard/dist/assets/index-8owBnqyv.js +43 -0
- package/dashboard/dist/assets/{index-AFQnlfjO.css → index-dkO8UUVl.css} +1 -1
- package/dashboard/dist/index.html +2 -2
- package/dashboard/server/doctor.js +15 -5
- package/dashboard/server/index.js +7 -0
- package/dashboard/server/metrics.js +6 -5
- package/dashboard/server/usage.js +61 -0
- package/install.ps1 +4 -1
- package/install.sh +5 -2
- package/package.json +1 -1
- package/profile/PIPELINE.template.md +3 -3
- package/profile/SCHEMA.md +21 -44
- package/scripts/loop.sh +189 -0
- package/scripts/metrics/collect.mjs +504 -0
- package/scripts/metrics/prices.json +39 -0
- package/scripts/preflight.sh +2 -2
- package/scripts/telemetry-send.sh +5 -2
- package/scripts/test-dashboard.mjs +29 -3
- package/scripts/test-gate.mjs +1 -2
- package/scripts/test-metrics.mjs +144 -0
- package/scripts/test-workflows.mjs +56 -178
- package/scripts/validate-core.mjs +10 -9
- package/core/agents/smoke.md +0 -63
- package/core/commands/cycle.md +0 -61
- package/core/commands/smoke.md +0 -55
- package/core/workflows/cycle.js +0 -513
- package/dashboard/dist/assets/index-DLBzciIC.js +0 -43
|
@@ -0,0 +1,504 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// collect.mjs — reconstruct per-command cost and runtime from Claude Code's own transcripts.
|
|
3
|
+
//
|
|
4
|
+
// node scripts/metrics/collect.mjs [projectRoot] [--json] [--runs] [--since=ISO] [--days=N]
|
|
5
|
+
//
|
|
6
|
+
// WHY THIS EXISTS
|
|
7
|
+
// `.claude/pipeline-metrics.jsonl` is written by the model itself (each command file tells
|
|
8
|
+
// it to append a line). That makes it unreliable — a command that ends early, errors, or
|
|
9
|
+
// simply forgets writes nothing, and it can never report tokens because the model does not
|
|
10
|
+
// know its own usage. Claude Code, meanwhile, already logs every API response it makes to
|
|
11
|
+
// ~/.claude/projects/<slug>/<sessionId>.jsonl with exact `usage` and timestamps, and every
|
|
12
|
+
// subagent to <sessionId>/subagents/agent-<id>.jsonl. That is ground truth, it needs no
|
|
13
|
+
// cooperation from the model, and it is retroactive: this script works on runs that already
|
|
14
|
+
// happened. Nothing here writes to the pipeline — it is a pure reader.
|
|
15
|
+
//
|
|
16
|
+
// THREE THINGS THAT ARE EASY TO GET WRONG, HANDLED HERE
|
|
17
|
+
// 1. One API response is written as SEVERAL transcript lines (one per content block:
|
|
18
|
+
// thinking, text, tool_use, tool_use), and EACH line repeats the full `usage` object.
|
|
19
|
+
// Summing lines inflates tokens ~1.8x. We dedupe by message.id.
|
|
20
|
+
// 2. Feature work happens in git worktrees, whose cwd hashes to a DIFFERENT project slug.
|
|
21
|
+
// Scanning only the main checkout's slug silently drops most of a multi-surface run. We match
|
|
22
|
+
// sessions by their recorded `cwd` against `git worktree list`.
|
|
23
|
+
// 3. Subagent spend lives in a separate file tree and is invisible in the parent
|
|
24
|
+
// transcript. For cohorte that is the majority of the cost, so we walk subagents/ and
|
|
25
|
+
// attribute each agent to the command segment that spawned it (via meta.toolUseId,
|
|
26
|
+
// falling back to timestamp containment for agents spawned by the Workflow runner).
|
|
27
|
+
|
|
28
|
+
import fs from 'node:fs';
|
|
29
|
+
import path from 'node:path';
|
|
30
|
+
import os from 'node:os';
|
|
31
|
+
import { execFileSync } from 'node:child_process';
|
|
32
|
+
import { fileURLToPath } from 'node:url';
|
|
33
|
+
|
|
34
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
35
|
+
const PRICES = JSON.parse(fs.readFileSync(path.join(HERE, 'prices.json'), 'utf8'));
|
|
36
|
+
|
|
37
|
+
// A gap longer than this between two API responses is the human thinking, reading, or away
|
|
38
|
+
// — not the command working. `wall` keeps it, `active` drops it. Without the split, a
|
|
39
|
+
// command left open over lunch reports a two-hour runtime and poisons every median.
|
|
40
|
+
const IDLE_GAP_S = 120;
|
|
41
|
+
|
|
42
|
+
// A prompt this short with no command in it ("continue", "go", "ok next") is the human
|
|
43
|
+
// steering a run that is already going, not starting a new one. Without this, a single
|
|
44
|
+
// /review driven by three "continue"s reports as one /review plus three anonymous chat
|
|
45
|
+
// runs, and three quarters of its cost lands under (chat).
|
|
46
|
+
const CONTINUATION_MAX_CHARS = 40;
|
|
47
|
+
|
|
48
|
+
// Commands are recognised two ways. `<command-name>` is emitted only when the whole prompt
|
|
49
|
+
// IS the slash command; in practice people write "move on branding-ramp and /review", which
|
|
50
|
+
// the harness records as ordinary prose. So we also look for an inline mention, checked
|
|
51
|
+
// against the real command list rather than any /token — otherwise a file path like
|
|
52
|
+
// /usr/bin or a URL fragment would invent commands that were never run.
|
|
53
|
+
function knownCommands() {
|
|
54
|
+
const names = new Set();
|
|
55
|
+
for (const dir of [path.join(HERE, '..', '..', 'core', 'commands'),
|
|
56
|
+
path.join(process.env.CLAUDE_CONFIG_DIR || path.join(os.homedir(), '.claude'), 'pipeline', 'commands')]) {
|
|
57
|
+
try {
|
|
58
|
+
for (const f of fs.readdirSync(dir)) if (f.endsWith('.md')) names.add(f.slice(0, -3));
|
|
59
|
+
} catch {}
|
|
60
|
+
}
|
|
61
|
+
// Fallback for a collector run outside the package (e.g. copied into a repo on its own).
|
|
62
|
+
if (!names.size) {
|
|
63
|
+
for (const n of ['brainstorm', 'spec', 'build', 'review', 'fix', 'ship',
|
|
64
|
+
'audit', 'refactor', 'align-ds', 'doctor', 'init-pipeline', 'update-pipeline']) names.add(n);
|
|
65
|
+
}
|
|
66
|
+
// Retired commands. The list above is read from the shipped core, so a command that is
|
|
67
|
+
// removed stops being recognised — and every run of it already in the transcripts silently
|
|
68
|
+
// reclassifies as (chat), rewriting history and inflating the catch-all bucket. Keep the
|
|
69
|
+
// names here so past runs stay attributed to what actually ran.
|
|
70
|
+
for (const n of ['cycle', 'smoke']) names.add(n);
|
|
71
|
+
return names;
|
|
72
|
+
}
|
|
73
|
+
const COMMANDS = knownCommands();
|
|
74
|
+
|
|
75
|
+
// An invocation is a short instruction that is mostly the command ("move on branding-ramp
|
|
76
|
+
// and /review"). A long prompt that happens to name one is someone TALKING ABOUT the
|
|
77
|
+
// command — a bug report, a design discussion, a pasted transcript. Counting those as runs
|
|
78
|
+
// inflates a command's run count and cost with conversation that never invoked it, which is
|
|
79
|
+
// exactly what happened in cohorte's own repo while this pipeline was being discussed.
|
|
80
|
+
// Only inline mentions are length-gated; an explicit <command-name> is always an invocation.
|
|
81
|
+
const MENTION_MAX_CHARS = 120;
|
|
82
|
+
|
|
83
|
+
function commandIn(text) {
|
|
84
|
+
const explicit = /<command-name>\s*(\/?[\w:-]+)\s*<\/command-name>/.exec(text);
|
|
85
|
+
if (explicit) return explicit[1].replace(/^\//, '');
|
|
86
|
+
if (text.trim().length > MENTION_MAX_CHARS) return null;
|
|
87
|
+
// Last mention wins: "finish /build then /review" ends on the one being asked for.
|
|
88
|
+
let found = null;
|
|
89
|
+
for (const m of text.matchAll(/(?:^|\s)\/([a-z][a-z0-9-]{2,})\b/g)) {
|
|
90
|
+
if (COMMANDS.has(m[1])) found = m[1];
|
|
91
|
+
}
|
|
92
|
+
return found;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// ── paths ────────────────────────────────────────────────────────────────────────────────
|
|
96
|
+
|
|
97
|
+
const norm = (p) => String(p || '').replace(/\\/g, '/').replace(/\/+$/, '').toLowerCase();
|
|
98
|
+
|
|
99
|
+
function git(args, cwd) {
|
|
100
|
+
try {
|
|
101
|
+
return execFileSync('git', args, { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] });
|
|
102
|
+
} catch { return ''; }
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// Every checkout that belongs to this repo: the main one plus every live feature worktree.
|
|
106
|
+
// A cohorte run spreads its surfaces across worktrees, so this set is what makes a feature
|
|
107
|
+
// add up instead of reporting only whatever the human happened to type in the main window.
|
|
108
|
+
function repoCheckouts(root) {
|
|
109
|
+
const out = new Set([norm(root)]);
|
|
110
|
+
const common = git(['rev-parse', '--git-common-dir'], root).trim();
|
|
111
|
+
if (common) out.add(norm(path.resolve(root, common, '..')));
|
|
112
|
+
for (const line of git(['worktree', 'list', '--porcelain'], root).split(/\r?\n/)) {
|
|
113
|
+
if (line.startsWith('worktree ')) out.add(norm(line.slice(9)));
|
|
114
|
+
}
|
|
115
|
+
return out;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const projectsDir = () =>
|
|
119
|
+
path.join(process.env.CLAUDE_CONFIG_DIR || path.join(os.homedir(), '.claude'), 'projects');
|
|
120
|
+
|
|
121
|
+
// Read only the head of a transcript to decide whether it belongs to this repo. Transcripts
|
|
122
|
+
// run to tens of MB and most of them belong to other projects; fully parsing every file to
|
|
123
|
+
// find the handful that match turns a 2-second command into a 2-minute one.
|
|
124
|
+
function sessionCwd(file) {
|
|
125
|
+
let fd;
|
|
126
|
+
try {
|
|
127
|
+
fd = fs.openSync(file, 'r');
|
|
128
|
+
const buf = Buffer.alloc(65536);
|
|
129
|
+
const n = fs.readSync(fd, buf, 0, buf.length, 0);
|
|
130
|
+
const m = /"cwd"\s*:\s*"((?:[^"\\]|\\.)*)"/.exec(buf.subarray(0, n).toString('utf8'));
|
|
131
|
+
return m ? JSON.parse(`"${m[1]}"`) : null;
|
|
132
|
+
} catch { return null; }
|
|
133
|
+
finally { if (fd !== undefined) try { fs.closeSync(fd); } catch {} }
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
function findSessions(checkouts) {
|
|
137
|
+
const root = projectsDir();
|
|
138
|
+
let dirs;
|
|
139
|
+
try { dirs = fs.readdirSync(root, { withFileTypes: true }); }
|
|
140
|
+
catch { return []; }
|
|
141
|
+
|
|
142
|
+
const found = [];
|
|
143
|
+
for (const d of dirs) {
|
|
144
|
+
if (!d.isDirectory()) continue;
|
|
145
|
+
const dir = path.join(root, d.name);
|
|
146
|
+
let files;
|
|
147
|
+
try { files = fs.readdirSync(dir); } catch { continue; }
|
|
148
|
+
for (const f of files) {
|
|
149
|
+
if (!f.endsWith('.jsonl')) continue;
|
|
150
|
+
const file = path.join(dir, f);
|
|
151
|
+
const cwd = sessionCwd(file);
|
|
152
|
+
if (!cwd) continue;
|
|
153
|
+
// A worktree's cwd can be a SUBDIRECTORY of the checkout (an agent cd'd into a
|
|
154
|
+
// package), so prefix-match rather than compare for equality.
|
|
155
|
+
const c = norm(cwd);
|
|
156
|
+
if (![...checkouts].some((k) => c === k || c.startsWith(k + '/'))) continue;
|
|
157
|
+
found.push({ file, dir: path.join(dir, path.basename(f, '.jsonl')), cwd });
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
return found;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
// ── pricing ──────────────────────────────────────────────────────────────────────────────
|
|
164
|
+
|
|
165
|
+
function rates(model, speed) {
|
|
166
|
+
let best = null;
|
|
167
|
+
for (const key of Object.keys(PRICES.models)) {
|
|
168
|
+
if (String(model || '').startsWith(key) && (!best || key.length > best.length)) best = key;
|
|
169
|
+
}
|
|
170
|
+
if (!best) return null;
|
|
171
|
+
const entry = PRICES.models[best];
|
|
172
|
+
return (speed === 'fast' && entry.fast) ? entry.fast : entry;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
const emptyTokens = () => ({ input: 0, output: 0, cacheWrite5m: 0, cacheWrite1h: 0, cacheRead: 0 });
|
|
176
|
+
|
|
177
|
+
function addUsage(acc, usage, model, speed) {
|
|
178
|
+
// `<synthetic>` messages are harness-authored (API error notices, interrupt markers).
|
|
179
|
+
// They carry a usage block but cost nothing — billing them invents spend.
|
|
180
|
+
if (model === '<synthetic>') return;
|
|
181
|
+
const cc = usage.cache_creation || {};
|
|
182
|
+
let w5 = cc.ephemeral_5m_input_tokens || 0;
|
|
183
|
+
const w1 = cc.ephemeral_1h_input_tokens || 0;
|
|
184
|
+
// Older transcripts carry only the flat total with no TTL breakdown. Bill it as 5m —
|
|
185
|
+
// the cheaper of the two, so an unknown-TTL run under-reports rather than inflates.
|
|
186
|
+
if (!w5 && !w1) w5 = usage.cache_creation_input_tokens || 0;
|
|
187
|
+
|
|
188
|
+
const t = { input: usage.input_tokens || 0, output: usage.output_tokens || 0,
|
|
189
|
+
cacheWrite5m: w5, cacheWrite1h: w1, cacheRead: usage.cache_read_input_tokens || 0 };
|
|
190
|
+
for (const k of Object.keys(t)) acc.tokens[k] += t[k];
|
|
191
|
+
|
|
192
|
+
const r = rates(model, speed);
|
|
193
|
+
if (!r) { acc.unpriced.add(model || 'unknown'); return; }
|
|
194
|
+
const m = PRICES.multipliers;
|
|
195
|
+
acc.cost += (
|
|
196
|
+
t.input * r.input +
|
|
197
|
+
t.output * r.output +
|
|
198
|
+
t.cacheWrite5m * r.input * m.cacheWrite5m +
|
|
199
|
+
t.cacheWrite1h * r.input * m.cacheWrite1h +
|
|
200
|
+
t.cacheRead * r.input * m.cacheRead
|
|
201
|
+
) / 1e6;
|
|
202
|
+
|
|
203
|
+
const byModel = acc.models[model] || (acc.models[model] = emptyTokens());
|
|
204
|
+
for (const k of Object.keys(t)) byModel[k] += t[k];
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
// ── transcript parsing ───────────────────────────────────────────────────────────────────
|
|
208
|
+
|
|
209
|
+
const userText = (msg) => {
|
|
210
|
+
const c = msg && msg.content;
|
|
211
|
+
if (typeof c === 'string') return c;
|
|
212
|
+
if (!Array.isArray(c)) return '';
|
|
213
|
+
return c.filter((b) => b && b.type === 'text').map((b) => b.text || '').join('\n');
|
|
214
|
+
};
|
|
215
|
+
|
|
216
|
+
const isToolResultTurn = (msg) =>
|
|
217
|
+
Array.isArray(msg && msg.content) && msg.content.some((b) => b && b.type === 'tool_result');
|
|
218
|
+
|
|
219
|
+
function newSegment(label, ts) {
|
|
220
|
+
return {
|
|
221
|
+
label, startTs: ts, endTs: ts,
|
|
222
|
+
tokens: emptyTokens(), cost: 0, models: {}, unpriced: new Set(),
|
|
223
|
+
activeS: 0, lastTs: ts, turns: 0, continuations: 0,
|
|
224
|
+
toolUseIds: new Set(), agents: [],
|
|
225
|
+
};
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
// One segment = one command invocation (or one free-form chat turn). Boundaries are user
|
|
229
|
+
// turns that are NOT tool results — a tool result is the harness feeding the loop, not the
|
|
230
|
+
// human starting something new.
|
|
231
|
+
function parseSession(file) {
|
|
232
|
+
const segments = [];
|
|
233
|
+
let seg = null;
|
|
234
|
+
const seenMessageIds = new Set();
|
|
235
|
+
|
|
236
|
+
let raw;
|
|
237
|
+
try { raw = fs.readFileSync(file, 'utf8'); } catch { return segments; }
|
|
238
|
+
|
|
239
|
+
for (const line of raw.split(/\r?\n/)) {
|
|
240
|
+
if (!line) continue;
|
|
241
|
+
let e;
|
|
242
|
+
try { e = JSON.parse(line); } catch { continue; }
|
|
243
|
+
const ts = e.timestamp ? Date.parse(e.timestamp) : NaN;
|
|
244
|
+
|
|
245
|
+
if (e.type === 'user') {
|
|
246
|
+
const msg = e.message || {};
|
|
247
|
+
if (isToolResultTurn(msg) || e.isMeta || e.isSidechain) continue;
|
|
248
|
+
const text = userText(msg);
|
|
249
|
+
const cmd = commandIn(text);
|
|
250
|
+
// Not every user-role turn is the human starting something. The harness injects
|
|
251
|
+
// turns mid-run — a local-command echo, and (critically for cohorte) a
|
|
252
|
+
// <task-notification> when a background agent finishes. Those arrive DURING a
|
|
253
|
+
// a /build; treating them as boundaries chops one command into several
|
|
254
|
+
// cheap-looking fragments and strands the agent spend in the wrong segment.
|
|
255
|
+
if (!cmd && /<(local-command-(stdout|stderr)|task-notification|system-reminder)>/.test(text)) continue;
|
|
256
|
+
// A short steer with no command keeps the current run open rather than opening a new
|
|
257
|
+
// one — see CONTINUATION_MAX_CHARS.
|
|
258
|
+
if (!cmd && seg && text.trim().length <= CONTINUATION_MAX_CHARS) { seg.continuations += 1; continue; }
|
|
259
|
+
seg = newSegment(cmd ? '/' + cmd : '(chat)', ts);
|
|
260
|
+
segments.push(seg);
|
|
261
|
+
continue;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
if (e.type !== 'assistant' || !seg) continue;
|
|
265
|
+
const msg = e.message || {};
|
|
266
|
+
|
|
267
|
+
// Dedupe: the same API response is written once per content block, each copy carrying
|
|
268
|
+
// the full usage. See header note 1 — this is the single biggest correctness trap.
|
|
269
|
+
const id = msg.id;
|
|
270
|
+
if (id && seenMessageIds.has(id)) {
|
|
271
|
+
for (const b of msg.content || []) if (b && b.type === 'tool_use' && b.id) seg.toolUseIds.add(b.id);
|
|
272
|
+
if (!Number.isNaN(ts)) seg.endTs = Math.max(seg.endTs, ts);
|
|
273
|
+
continue;
|
|
274
|
+
}
|
|
275
|
+
if (id) seenMessageIds.add(id);
|
|
276
|
+
|
|
277
|
+
if (msg.usage) addUsage(seg, msg.usage, msg.model, msg.usage.speed);
|
|
278
|
+
for (const b of msg.content || []) if (b && b.type === 'tool_use' && b.id) seg.toolUseIds.add(b.id);
|
|
279
|
+
|
|
280
|
+
seg.turns += 1;
|
|
281
|
+
if (!Number.isNaN(ts)) {
|
|
282
|
+
const gap = (ts - seg.lastTs) / 1000;
|
|
283
|
+
if (gap > 0 && gap <= IDLE_GAP_S) seg.activeS += gap;
|
|
284
|
+
seg.lastTs = ts;
|
|
285
|
+
seg.endTs = Math.max(seg.endTs, ts);
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
return segments;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
// ── subagents ────────────────────────────────────────────────────────────────────────────
|
|
292
|
+
|
|
293
|
+
function readSubagents(sessionDir) {
|
|
294
|
+
const dir = path.join(sessionDir, 'subagents');
|
|
295
|
+
let files;
|
|
296
|
+
try { files = fs.readdirSync(dir); } catch { return []; }
|
|
297
|
+
|
|
298
|
+
const agents = [];
|
|
299
|
+
for (const f of files) {
|
|
300
|
+
if (!f.endsWith('.jsonl')) continue;
|
|
301
|
+
const base = f.slice(0, -'.jsonl'.length);
|
|
302
|
+
let meta = {};
|
|
303
|
+
try { meta = JSON.parse(fs.readFileSync(path.join(dir, base + '.meta.json'), 'utf8')); } catch {}
|
|
304
|
+
|
|
305
|
+
const acc = { tokens: emptyTokens(), cost: 0, models: {}, unpriced: new Set() };
|
|
306
|
+
let startTs = Infinity, endTs = -Infinity, turns = 0;
|
|
307
|
+
const seen = new Set();
|
|
308
|
+
|
|
309
|
+
let raw;
|
|
310
|
+
try { raw = fs.readFileSync(path.join(dir, f), 'utf8'); } catch { continue; }
|
|
311
|
+
for (const line of raw.split(/\r?\n/)) {
|
|
312
|
+
if (!line) continue;
|
|
313
|
+
let e;
|
|
314
|
+
try { e = JSON.parse(line); } catch { continue; }
|
|
315
|
+
if (e.type !== 'assistant') continue;
|
|
316
|
+
const msg = e.message || {};
|
|
317
|
+
if (msg.id && seen.has(msg.id)) continue;
|
|
318
|
+
if (msg.id) seen.add(msg.id);
|
|
319
|
+
if (msg.usage) addUsage(acc, msg.usage, msg.model, msg.usage.speed);
|
|
320
|
+
turns += 1;
|
|
321
|
+
const ts = e.timestamp ? Date.parse(e.timestamp) : NaN;
|
|
322
|
+
if (!Number.isNaN(ts)) { startTs = Math.min(startTs, ts); endTs = Math.max(endTs, ts); }
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
agents.push({
|
|
326
|
+
id: base.replace(/^agent-/, ''),
|
|
327
|
+
agentType: meta.agentType || 'unknown',
|
|
328
|
+
description: meta.description || '',
|
|
329
|
+
toolUseId: meta.toolUseId || null,
|
|
330
|
+
spawnDepth: meta.spawnDepth ?? null,
|
|
331
|
+
turns, startTs, endTs, ...acc,
|
|
332
|
+
});
|
|
333
|
+
}
|
|
334
|
+
return agents;
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
function attachAgents(segments, agents) {
|
|
338
|
+
for (const a of agents) {
|
|
339
|
+
// Preferred link: the Task/Agent tool_use that spawned it. Workflow-spawned agents can
|
|
340
|
+
// carry a toolUseId the parent transcript never recorded, so fall back to which segment
|
|
341
|
+
// was running when the agent started.
|
|
342
|
+
let seg = a.toolUseId ? segments.find((s) => s.toolUseIds.has(a.toolUseId)) : null;
|
|
343
|
+
if (!seg && Number.isFinite(a.startTs)) {
|
|
344
|
+
seg = segments.find((s) => a.startTs >= s.startTs && a.startTs <= s.endTs + 5 * 60_000);
|
|
345
|
+
}
|
|
346
|
+
if (!seg) continue;
|
|
347
|
+
seg.agents.push(a);
|
|
348
|
+
for (const k of Object.keys(seg.tokens)) seg.tokens[k] += a.tokens[k];
|
|
349
|
+
seg.cost += a.cost;
|
|
350
|
+
for (const [m, t] of Object.entries(a.models)) {
|
|
351
|
+
const dst = seg.models[m] || (seg.models[m] = emptyTokens());
|
|
352
|
+
for (const k of Object.keys(t)) dst[k] += t[k];
|
|
353
|
+
}
|
|
354
|
+
for (const m of a.unpriced) seg.unpriced.add(m);
|
|
355
|
+
if (Number.isFinite(a.endTs)) seg.endTs = Math.max(seg.endTs, a.endTs);
|
|
356
|
+
// Agents run in parallel, so their wall time is not additive and cannot be folded into
|
|
357
|
+
// the parent's `active`. Segment `wall` already covers them via endTs; agent runtime is
|
|
358
|
+
// reported separately per command as agentWallS.
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
// ── rollup ───────────────────────────────────────────────────────────────────────────────
|
|
363
|
+
|
|
364
|
+
const pct = (arr, p) => {
|
|
365
|
+
if (!arr.length) return 0;
|
|
366
|
+
const s = [...arr].sort((a, b) => a - b);
|
|
367
|
+
return s[Math.min(s.length - 1, Math.floor((p / 100) * s.length))];
|
|
368
|
+
};
|
|
369
|
+
|
|
370
|
+
function rollup(segments) {
|
|
371
|
+
const by = new Map();
|
|
372
|
+
for (const s of segments) {
|
|
373
|
+
let g = by.get(s.label);
|
|
374
|
+
if (!g) {
|
|
375
|
+
g = { command: s.label, runs: 0, continuations: 0, tokens: emptyTokens(), cost: 0, models: {},
|
|
376
|
+
wall: [], active: [], agentCounts: [], agentWall: [], turns: [], unpriced: new Set() };
|
|
377
|
+
by.set(s.label, g);
|
|
378
|
+
}
|
|
379
|
+
g.runs += 1;
|
|
380
|
+
g.continuations += s.continuations;
|
|
381
|
+
for (const k of Object.keys(s.tokens)) g.tokens[k] += s.tokens[k];
|
|
382
|
+
g.cost += s.cost;
|
|
383
|
+
for (const [m, t] of Object.entries(s.models)) {
|
|
384
|
+
const dst = g.models[m] || (g.models[m] = emptyTokens());
|
|
385
|
+
for (const k of Object.keys(t)) dst[k] += t[k];
|
|
386
|
+
}
|
|
387
|
+
for (const m of s.unpriced) g.unpriced.add(m);
|
|
388
|
+
g.wall.push(Math.max(0, (s.endTs - s.startTs) / 1000) || 0);
|
|
389
|
+
g.active.push(s.activeS);
|
|
390
|
+
g.turns.push(s.turns);
|
|
391
|
+
g.agentCounts.push(s.agents.length);
|
|
392
|
+
g.agentWall.push(s.agents.reduce((n, a) => n + (Number.isFinite(a.endTs) ? (a.endTs - a.startTs) / 1000 : 0), 0));
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
return [...by.values()]
|
|
396
|
+
.map((g) => ({
|
|
397
|
+
command: g.command,
|
|
398
|
+
runs: g.runs,
|
|
399
|
+
continuations: g.continuations,
|
|
400
|
+
cost: { total: g.cost, perRun: g.cost / g.runs },
|
|
401
|
+
tokens: g.tokens,
|
|
402
|
+
tokensPerRun: Object.fromEntries(Object.entries(g.tokens).map(([k, v]) => [k, Math.round(v / g.runs)])),
|
|
403
|
+
wallS: { p50: pct(g.wall, 50), p90: pct(g.wall, 90), total: g.wall.reduce((a, b) => a + b, 0) },
|
|
404
|
+
activeS: { p50: pct(g.active, 50), p90: pct(g.active, 90) },
|
|
405
|
+
agents: { perRunP50: pct(g.agentCounts, 50), total: g.agentCounts.reduce((a, b) => a + b, 0),
|
|
406
|
+
serialWallP50S: pct(g.agentWall, 50) },
|
|
407
|
+
turnsP50: pct(g.turns, 50),
|
|
408
|
+
models: g.models,
|
|
409
|
+
unpriced: [...g.unpriced],
|
|
410
|
+
}))
|
|
411
|
+
.sort((a, b) => b.cost.total - a.cost.total);
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
// ── output ───────────────────────────────────────────────────────────────────────────────
|
|
415
|
+
|
|
416
|
+
const fmtUsd = (n) => (n < 0.01 && n > 0 ? '<$0.01' : '$' + n.toFixed(2));
|
|
417
|
+
const fmtTok = (n) => (n >= 1e6 ? (n / 1e6).toFixed(1) + 'M' : n >= 1e3 ? Math.round(n / 1e3) + 'k' : String(n));
|
|
418
|
+
const fmtDur = (s) => (s >= 3600 ? (s / 3600).toFixed(1) + 'h' : s >= 60 ? Math.round(s / 60) + 'm' : Math.round(s) + 's');
|
|
419
|
+
|
|
420
|
+
function table(rows) {
|
|
421
|
+
const head = ['COMMAND', 'RUNS', '$/RUN', '$ TOTAL', 'TOK/RUN', 'OUT/RUN', 'WALL p50', 'ACTIVE p50', 'AGENTS p50'];
|
|
422
|
+
const body = rows.map((r) => [
|
|
423
|
+
r.command,
|
|
424
|
+
String(r.runs),
|
|
425
|
+
fmtUsd(r.cost.perRun),
|
|
426
|
+
fmtUsd(r.cost.total),
|
|
427
|
+
fmtTok(Object.values(r.tokensPerRun).reduce((a, b) => a + b, 0)),
|
|
428
|
+
fmtTok(r.tokensPerRun.output),
|
|
429
|
+
fmtDur(r.wallS.p50),
|
|
430
|
+
fmtDur(r.activeS.p50),
|
|
431
|
+
String(r.agents.perRunP50),
|
|
432
|
+
]);
|
|
433
|
+
const w = head.map((h, i) => Math.max(h.length, ...body.map((b) => b[i].length)));
|
|
434
|
+
const line = (cells) => cells.map((c, i) => (i === 0 ? c.padEnd(w[i]) : c.padStart(w[i]))).join(' ');
|
|
435
|
+
return [line(head), w.map((n) => '-'.repeat(n)).join(' '), ...body.map(line)].join('\n');
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
// ── main ─────────────────────────────────────────────────────────────────────────────────
|
|
439
|
+
|
|
440
|
+
function main(argv) {
|
|
441
|
+
const args = argv.slice(2);
|
|
442
|
+
const flag = (name) => args.includes('--' + name);
|
|
443
|
+
const opt = (name) => {
|
|
444
|
+
const hit = args.find((a) => a.startsWith(`--${name}=`));
|
|
445
|
+
return hit ? hit.slice(name.length + 3) : null;
|
|
446
|
+
};
|
|
447
|
+
const root = path.resolve(args.find((a) => !a.startsWith('--')) || process.cwd());
|
|
448
|
+
|
|
449
|
+
let since = opt('since') ? Date.parse(opt('since')) : null;
|
|
450
|
+
if (opt('days')) since = Date.now() - Number(opt('days')) * 86400_000;
|
|
451
|
+
|
|
452
|
+
const checkouts = repoCheckouts(root);
|
|
453
|
+
const sessions = findSessions(checkouts);
|
|
454
|
+
|
|
455
|
+
let segments = [];
|
|
456
|
+
for (const s of sessions) {
|
|
457
|
+
const segs = parseSession(s.file);
|
|
458
|
+
attachAgents(segs, readSubagents(s.dir));
|
|
459
|
+
segments.push(...segs);
|
|
460
|
+
}
|
|
461
|
+
if (since) segments = segments.filter((s) => s.startTs >= since);
|
|
462
|
+
// A segment with no API call is a typo or an interrupted prompt, not a run.
|
|
463
|
+
segments = segments.filter((s) => s.turns > 0);
|
|
464
|
+
|
|
465
|
+
const rows = rollup(segments);
|
|
466
|
+
const totals = {
|
|
467
|
+
sessions: sessions.length,
|
|
468
|
+
runs: segments.length,
|
|
469
|
+
cost: rows.reduce((n, r) => n + r.cost.total, 0),
|
|
470
|
+
agents: rows.reduce((n, r) => n + r.agents.total, 0),
|
|
471
|
+
};
|
|
472
|
+
|
|
473
|
+
if (flag('json')) {
|
|
474
|
+
const out = { generatedAt: new Date().toISOString(), projectRoot: root,
|
|
475
|
+
checkouts: [...checkouts], pricesUpdated: PRICES.updated, totals, commands: rows };
|
|
476
|
+
if (flag('runs')) {
|
|
477
|
+
out.runs = segments.map((s) => ({
|
|
478
|
+
command: s.label, startedAt: new Date(s.startTs).toISOString(),
|
|
479
|
+
wallS: Math.max(0, (s.endTs - s.startTs) / 1000), activeS: s.activeS,
|
|
480
|
+
turns: s.turns, continuations: s.continuations, cost: s.cost, tokens: s.tokens,
|
|
481
|
+
agents: s.agents.map((a) => ({ type: a.agentType, description: a.description,
|
|
482
|
+
cost: a.cost, tokens: a.tokens, turns: a.turns })),
|
|
483
|
+
}));
|
|
484
|
+
}
|
|
485
|
+
process.stdout.write(JSON.stringify(out, null, 2) + '\n');
|
|
486
|
+
return 0;
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
if (!sessions.length) {
|
|
490
|
+
console.log(`No Claude Code transcripts found for ${root} under ${projectsDir()}.`);
|
|
491
|
+
return 0;
|
|
492
|
+
}
|
|
493
|
+
console.log(`cohorte metrics — ${root}`);
|
|
494
|
+
console.log(`${totals.runs} runs across ${totals.sessions} sessions, ${totals.agents} subagents, ${fmtUsd(totals.cost)} total`
|
|
495
|
+
+ (since ? ` (since ${new Date(since).toISOString().slice(0, 10)})` : ''));
|
|
496
|
+
console.log(`prices as of ${PRICES.updated}; wall excludes nothing, active drops gaps > ${IDLE_GAP_S}s\n`);
|
|
497
|
+
console.log(table(rows));
|
|
498
|
+
|
|
499
|
+
const unpriced = [...new Set(rows.flatMap((r) => r.unpriced))];
|
|
500
|
+
if (unpriced.length) console.log(`\nnot in prices.json (counted, not costed): ${unpriced.join(', ')}`);
|
|
501
|
+
return 0;
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
process.exit(main(process.argv));
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": [
|
|
3
|
+
"USD per million tokens, Anthropic first-party API rates. Bedrock/Vertex are",
|
|
4
|
+
"partner-priced and are NOT covered here — a run on those platforms will be",
|
|
5
|
+
"costed at first-party rates and slightly misreported.",
|
|
6
|
+
"Cache rates are derived from `input` via `multipliers` (write 1.25x for the 5m",
|
|
7
|
+
"TTL, 2x for 1h, read 0.1x) rather than restated per model, so a price change",
|
|
8
|
+
"is a one-line edit. `fast` is the fast-mode premium (usage.speed === 'fast').",
|
|
9
|
+
"Model lookup is longest-prefix, so dated ids (claude-haiku-4-5-20251001) hit",
|
|
10
|
+
"their base entry without needing a row of their own."
|
|
11
|
+
],
|
|
12
|
+
"updated": "2026-07-31",
|
|
13
|
+
"multipliers": {
|
|
14
|
+
"cacheWrite5m": 1.25,
|
|
15
|
+
"cacheWrite1h": 2.0,
|
|
16
|
+
"cacheRead": 0.1
|
|
17
|
+
},
|
|
18
|
+
"models": {
|
|
19
|
+
"claude-fable-5": { "input": 10, "output": 50 },
|
|
20
|
+
"claude-mythos-5": { "input": 10, "output": 50 },
|
|
21
|
+
"claude-opus-5": {
|
|
22
|
+
"input": 5,
|
|
23
|
+
"output": 25,
|
|
24
|
+
"fast": { "input": 10, "output": 50 }
|
|
25
|
+
},
|
|
26
|
+
"claude-opus-4-8": {
|
|
27
|
+
"input": 5,
|
|
28
|
+
"output": 25,
|
|
29
|
+
"fast": { "input": 10, "output": 50 }
|
|
30
|
+
},
|
|
31
|
+
"claude-opus-4-7": { "input": 5, "output": 25 },
|
|
32
|
+
"claude-opus-4-6": { "input": 5, "output": 25 },
|
|
33
|
+
"claude-opus-4-5": { "input": 5, "output": 25 },
|
|
34
|
+
"claude-sonnet-5": { "input": 3, "output": 15 },
|
|
35
|
+
"claude-sonnet-4-6": { "input": 3, "output": 15 },
|
|
36
|
+
"claude-sonnet-4-5": { "input": 3, "output": 15 },
|
|
37
|
+
"claude-haiku-4-5": { "input": 1, "output": 5 }
|
|
38
|
+
}
|
|
39
|
+
}
|
package/scripts/preflight.sh
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/bin/sh
|
|
2
2
|
#
|
|
3
|
-
# preflight.sh — deterministic phase gate for /review
|
|
3
|
+
# preflight.sh — deterministic phase gate for /review.
|
|
4
4
|
#
|
|
5
5
|
# Runs the profile's mechanical checks (typecheck, lint, tests — whatever the caller
|
|
6
6
|
# passes) BEFORE any agent is spawned. A red gate means the caller aborts and relays
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
# The caller must stop there — no agents.
|
|
16
16
|
# - All green: writes `<project>/.claude/preflight.ok` ("<epoch> <HEAD sha>") — the
|
|
17
17
|
# stamp `hooks/gate.py` checks (gate-config.json `preflight` block) before letting
|
|
18
|
-
# review
|
|
18
|
+
# review agents dispatch.
|
|
19
19
|
|
|
20
20
|
set -u
|
|
21
21
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
# telemetry-send.sh — fire-and-forget anonymous usage ping (SCHEMA.md §Telemetry).
|
|
3
3
|
#
|
|
4
4
|
# telemetry-send.sh <phase> <feature_id> <seconds> [results]
|
|
5
|
-
# phase brainstorm|spec|build|
|
|
5
|
+
# phase brainstorm|spec|build|review|fix|ship — the feature funnel, and only it.
|
|
6
6
|
# Setup/maintenance commands never ping (SCHEMA.md §Telemetry).
|
|
7
7
|
# feature the feature id — NEVER sent raw; SHA-256-hashed to 12 hex chars
|
|
8
8
|
# seconds batch wall-clock
|
|
@@ -39,7 +39,10 @@ phase="${1:-}"; feature="${2:-}"; seconds="${3:-0}"; results="${4:-}"
|
|
|
39
39
|
# Allowlist the phase here — the collector accepts any string, so a typo in a command
|
|
40
40
|
# file would silently pollute the dataset with a phantom phase nobody notices.
|
|
41
41
|
case "$phase" in
|
|
42
|
-
brainstorm|spec|build|
|
|
42
|
+
brainstorm|spec|build|review|fix|ship) ;;
|
|
43
|
+
# `smoke` is a RETIRED phase (removed in 1.5.0) — still accepted so a stale install
|
|
44
|
+
# pinging it lands in its own bucket instead of being silently dropped.
|
|
45
|
+
smoke) ;;
|
|
43
46
|
*) exit 0 ;;
|
|
44
47
|
esac
|
|
45
48
|
|
|
@@ -20,6 +20,7 @@ const require = createRequire(import.meta.url);
|
|
|
20
20
|
const root = fileURLToPath(new URL("..", import.meta.url));
|
|
21
21
|
const { parse, parseProfileBlock } = require(join(root, "dashboard/server/yaml.js"));
|
|
22
22
|
const { metrics } = require(join(root, "dashboard/server/metrics.js"));
|
|
23
|
+
const { usage } = require(join(root, "dashboard/server/usage.js"));
|
|
23
24
|
const { state, scanSpecs } = require(join(root, "dashboard/server/doctor.js"));
|
|
24
25
|
const { kanban } = require(join(root, "dashboard/server/kanban.js"));
|
|
25
26
|
const fleet = require(join(root, "dashboard/server/fleet.js"));
|
|
@@ -106,6 +107,25 @@ console.log("metrics.js — the funnel aggregate");
|
|
|
106
107
|
check("no metrics file ⇒ present:false", metrics({ projectRoot: scratch() }).present === false);
|
|
107
108
|
}
|
|
108
109
|
|
|
110
|
+
// ── usage.js ─────────────────────────────────────────────────────────────────
|
|
111
|
+
// Wraps the ESM metrics collector for the CJS server. The failure that matters is
|
|
112
|
+
// not a crash: a project with no transcripts must say so, because rendering zeros
|
|
113
|
+
// reads as "this pipeline costs nothing" rather than "nothing was measured".
|
|
114
|
+
console.log("usage.js — the collector bridge");
|
|
115
|
+
{
|
|
116
|
+
const empty = usage({ projectRoot: scratch() });
|
|
117
|
+
check("a project with no transcripts reports present:false", empty.present === false);
|
|
118
|
+
check("…and says why rather than returning silent zeros",
|
|
119
|
+
typeof empty.error === "string" && empty.error.length > 0, JSON.stringify(empty));
|
|
120
|
+
|
|
121
|
+
// Same project twice: the second call must come from cache, or the panel's polling
|
|
122
|
+
// would re-parse tens of MB of transcripts on every refresh.
|
|
123
|
+
const d = scratch();
|
|
124
|
+
const t0 = Date.now(); usage({ projectRoot: d });
|
|
125
|
+
const t1 = Date.now(); usage({ projectRoot: d }); const cached = Date.now() - t1;
|
|
126
|
+
check("a repeated read is served from cache", cached <= Math.max(50, (t1 - t0)), `${cached}ms`);
|
|
127
|
+
}
|
|
128
|
+
|
|
109
129
|
// ── doctor.js ────────────────────────────────────────────────────────────────
|
|
110
130
|
console.log("doctor.js — the /doctor port");
|
|
111
131
|
{
|
|
@@ -116,8 +136,14 @@ console.log("doctor.js — the /doctor port");
|
|
|
116
136
|
writeFileSync(join(d, "specs", "b.md"), spec("feature_id: b\nstatus: shipped # done"));
|
|
117
137
|
writeFileSync(join(d, "specs", "c.md"), "no front-matter at all");
|
|
118
138
|
writeFileSync(join(d, "specs", "_template.md"), spec("status: draft"));
|
|
139
|
+
// /audit writes this file by design and it has no front-matter. Scanning it as a
|
|
140
|
+
// spec made /doctor warn about a file cohorte itself had just created — it fired in
|
|
141
|
+
// every project that had ever run /audit.
|
|
142
|
+
writeFileSync(join(d, "specs", "refactor-backlog.md"), "# Refactor Backlog\n\n## backend\n- [ ] x\n");
|
|
119
143
|
const specs = scanSpecs(d);
|
|
120
144
|
eq("_template.md is excluded", specs.length, 3);
|
|
145
|
+
eq("the /audit backlog is not scanned as a spec",
|
|
146
|
+
specs.some(s => s.file === "refactor-backlog.md"), false);
|
|
121
147
|
eq("front-matter fields are read", specs.find(s => s.id === "a").title, "A");
|
|
122
148
|
eq("a trailing comment is stripped from status", specs.find(s => s.id === "b").status, "shipped");
|
|
123
149
|
eq("no front-matter ⇒ id falls back to the filename", specs.find(s => s.file === "c.md").id, "c");
|
|
@@ -129,7 +155,7 @@ console.log("doctor.js — the /doctor port");
|
|
|
129
155
|
const d = scratch();
|
|
130
156
|
const gate = {
|
|
131
157
|
deny: ["x"], ask: ["y"], ask_on_default_branch: ["git push"], default_branch: "main",
|
|
132
|
-
preflight: { enabled: true, agents: ["review"
|
|
158
|
+
preflight: { enabled: true, agents: ["review"], max_age_minutes: 30 },
|
|
133
159
|
};
|
|
134
160
|
writeFileSync(join(d, "PIPELINE.md"), [
|
|
135
161
|
"```yaml pipeline-profile",
|
|
@@ -147,7 +173,7 @@ console.log("doctor.js — the /doctor port");
|
|
|
147
173
|
' ask_on_default_branch: ["git push"]',
|
|
148
174
|
" preflight:",
|
|
149
175
|
" enabled: true",
|
|
150
|
-
" agents: [review
|
|
176
|
+
" agents: [review]",
|
|
151
177
|
" max_age_minutes: 30",
|
|
152
178
|
"```",
|
|
153
179
|
].join("\n"));
|
|
@@ -158,7 +184,7 @@ console.log("doctor.js — the /doctor port");
|
|
|
158
184
|
writeFileSync(join(d, ".claude", "gate-config.json"), JSON.stringify(gate));
|
|
159
185
|
writeFileSync(join(d, ".mcp.json"), JSON.stringify({ mcpServers: { serena: {} } }));
|
|
160
186
|
mkdirSync(join(d, ".claude", "workflows"), { recursive: true });
|
|
161
|
-
for (const w of ["review.js", "audit.js", "refactor.js"
|
|
187
|
+
for (const w of ["review.js", "audit.js", "refactor.js"]) {
|
|
162
188
|
writeFileSync(join(d, ".claude", "workflows", w), "x");
|
|
163
189
|
}
|
|
164
190
|
writeFileSync(join(d, ".claude", "agents", "profile-reader.md"), "x");
|