cchubber 0.5.8 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,481 @@
1
+ import { shortName, categorize, labelName } from './drift-privacy.js';
2
+ import { homedir } from 'os';
3
+
4
+ /**
5
+ * Drift: how much of your agent time went to what you said you'd do.
6
+ *
7
+ * The method, in the order it runs. The report prints the same steps in plain words.
8
+ * 1. Turns. Each message you typed starts a turn; the agent's steps until your next message belong to it.
9
+ * Agent time is the time between the agent's own steps, and one wait longer than 20 minutes counts as 20.
10
+ * 2. Sittings. Coming back to a session after 2+ quiet hours with a new ask starts a new sitting ("continue" doesn't).
11
+ * 3. What you said. The written plan for that day or folder if there is one, otherwise your first real ask of the
12
+ * sitting. A /goal command replaces it for the rest of the sitting.
13
+ * 4. Threads. An ask that shares few words with the last few turns (your words, the agent's replies, the files it
14
+ * touched) starts a new thread. Short replies ("yes", "do it") stay on the thread they answer.
15
+ * 5. Labels. A thread whose opening ask shares enough words or files with what you said is on it. Anything else is
16
+ * a detour. Rare words count for more than common ones (weighted by how often they appear in your own week).
17
+ * No model is called. Everything here is word and file overlap, so it is rough, and the report says so.
18
+ */
19
+
20
+ const MIN = 60000;
21
+ const HOUR = 60 * MIN;
22
+
23
+ export const DRIFT = {
24
+ idleCapMs: 20 * MIN, // one step longer than this (a stuck prompt, a sleeping laptop) counts as this long
25
+ sittingGapMs: 2 * HOUR, // quiet this long, then a new ask, is a new sitting with its own "what you said"
26
+ continueAt: 0.34, // share of a new ask's word weight already in the last few turns that keeps it on the thread
27
+ onAt: 0.4, // share of a new ask's weight that one plan item (or the running focus) must cover for "on"
28
+ warmupMs: 15 * MIN, // people front-load a task's context in the first minutes; only then is an ask "still setup"
29
+ substantiveAt: 3, // total word weight an ask needs to be able to change the subject
30
+ replyTerms: 30, // the agent's rarest words per turn that join the thread's context
31
+ clusterAt: 0.5, // two detours sharing this much are the same detour, come back to
32
+ };
33
+
34
+ // Words that never say what a request is about: function words, chat filler, and verbs every request uses
35
+ const STOP = new Set(`a about above after again against all almost along already also although always am among an and
36
+ another any anybody anyone anything anyway anywhere are aren around as at away back be became because become been before
37
+ being below beside besides between beyond both but by came can cannot cant could couldnt did didnt do does doesnt doing
38
+ done dont down during each either else enough etc even ever every everybody everyone everything everywhere few for from
39
+ further get gets getting give given gives go goes going gone got gotten had hadnt has hasnt have havent having he her
40
+ here hers herself him himself his how however i if im in inside instead into is isnt it its itself ive just keep kept
41
+ last least less let lets like likely made make makes making many may maybe me might mine more most much must my myself
42
+ need needed needs neither never next no nobody none nor not nothing now of off often on once one only onto or other
43
+ others otherwise our ours ourselves out over own per perhaps please put quite rather really said same say says see seem
44
+ seems seen several shall she should shouldnt since so some somebody someone something sometimes somewhere soon still such
45
+ sure take taken takes taking tell than that thats the their theirs them themselves then there theres these they theyre
46
+ thing things think this those though through thus till to together too took toward towards try trying under unless
47
+ until up upon us use used uses using very via want wanted wants was wasnt way ways we well went were werent what whats
48
+ whatever when whenever where whereas wherever whether which while who whoever whole whom whose why will with within
49
+ without wont would wouldnt yet you youd youll your youre yours yourself yourselves youve
50
+ yo yh yeah yea yep yes ya nah nope ok okay alr alright bro bruh broski brudda bruv mate man gng innit lol lmao haha tbh
51
+ ngl icl idk yk ik lowk lowkey rn btw pls plz thx thanks thank cheers hey hi hello cool nice great good bad fine damn
52
+ wow oh uh um hmm huh actually basically genuinely literally honestly kinda kind sort sorta gonna wanna gotta ain aint
53
+ dunno cuz cause coz tho though anyways anyway also again already yall ye yup ofc imo imho fr frfr deadass bet
54
+ ask asked asking check checked checking continue continued create created doing find found fix fixed help know knew look
55
+ looked looking run running see show showed start started stop tell told thought work worked working works mean meant
56
+ means let give gave go went come came call called read wrote write writing added add update updated change changed
57
+ try tried seems seemed feel felt happen happened happening sure right wrong new old big small little lot lots bit
58
+ able best better way ways kind things stuff something anything everything nothing today tomorrow yesterday now later
59
+ time times day days week weeks first second third next last one two three four five also still even
60
+ claude codex agent ai model assistant chat session code file files folder message messages prompt reply answer`.split(/\s+/).filter(Boolean));
61
+
62
+ // Path parts that name the plumbing rather than the work
63
+ const PATH_STOP = new Set(`users user home documents desktop downloads library private tmp var opt src lib dist build
64
+ node modules index main app test tests spec html css js mjs cjs ts tsx jsx json md mdx txt yaml yml toml png jpg jpeg
65
+ svg gif webp pdf py sh log logs dev obsidian projects project readme claude codex`.split(/\s+/));
66
+
67
+ // Two-letter words worth keeping
68
+ const SHORT = new Set(['ui', 'ux', 'db', 'qa', 'ci', 'pr', 'js', 'ts', 'go', 'ml', 'os', 'uk', 'us', 'tv', 'pc', 'vr', 'ar']);
69
+
70
+ export function stem(w) {
71
+ if (w.length <= 3) return w;
72
+ let s = w;
73
+ if (s.endsWith('ies') && s.length > 4) s = s.slice(0, -3) + 'y';
74
+ else if (s.endsWith('sses')) s = s.slice(0, -2);
75
+ else if (s.endsWith('s') && !/(ss|us|is)$/.test(s)) s = s.slice(0, -1);
76
+ if (s.length > 5 && s.endsWith('ing')) s = s.slice(0, -3);
77
+ else if (s.length > 4 && s.endsWith('ed')) s = s.slice(0, -2);
78
+ if (/([b-df-hj-np-tv-z])\1$/.test(s) && !/(ll|ss|zz|ff)$/.test(s)) s = s.slice(0, -1);
79
+ if (s.length > 3 && s.endsWith('e')) s = s.slice(0, -1);
80
+ return s;
81
+ }
82
+
83
+ // Content words of a text, stemmed. Links keep only their site name.
84
+ export function tokenize(text, stop = STOP) {
85
+ const out = [];
86
+ const s = String(text || '').toLowerCase()
87
+ .replace(/https?:\/\/(?:www\.)?([a-z0-9-]+)[^\s]*/g, ' $1 ')
88
+ .replace(/['’]/g, '');
89
+ for (const raw of s.split(/[^a-z0-9]+/)) {
90
+ if (!raw || (raw.length < 3 && !SHORT.has(raw))) continue;
91
+ if (/^\d+$/.test(raw) || raw.length > 24) continue;
92
+ if (/\d/.test(raw) && raw.length >= 8) continue; // ids and hashes
93
+ if (stop.has(raw)) continue;
94
+ const st = stem(raw);
95
+ if (st.length < 2 || stop.has(st)) continue;
96
+ out.push(st);
97
+ }
98
+ return out;
99
+ }
100
+
101
+ function pathTerms(file, cwd) {
102
+ let p = String(file);
103
+ if (cwd && p.startsWith(cwd)) p = p.slice(cwd.length);
104
+ else p = p.split(/[\\/]/).slice(-3).join('/');
105
+ return tokenize(p.replace(/([a-z])([A-Z])/g, '$1 $2'), new Set([...STOP, ...PATH_STOP]));
106
+ }
107
+
108
+ const CONTINUE = /\b(continue|carry on|keep going|keep at it|go on|go ahead|proceed|resume|where you left off|as you were|do it|go for it|do so|do that|finish (it|up|off))\b/i;
109
+ // A command or prefix that states the plan outright
110
+ const PLAN_COMMANDS = new Set(['/goal', '/ignite', '/plan', '/focus']);
111
+ const PLAN_PREFIX = /^\s*(goal|plan|focus|today'?s? (plan|goal|focus))\s*:/i;
112
+
113
+ export function dayKey(t) {
114
+ const d = new Date(t);
115
+ return `${d.getFullYear()}-${String(d.getMonth() + 1).padStart(2, '0')}-${String(d.getDate()).padStart(2, '0')}`;
116
+ }
117
+
118
+ // Share of the weight of `terms` found in `bag` (at weight `min` or more): how much of an ask is already there
119
+ export function coverage(terms, bag, idf, min = 0.2) {
120
+ let all = 0, hit = 0;
121
+ for (const w of new Set(terms)) {
122
+ const x = idf(w);
123
+ all += x;
124
+ if ((bag.get(w) || 0) >= min) hit += x;
125
+ }
126
+ return all ? hit / all : 0;
127
+ }
128
+
129
+ // The best a single plan item covers this ask. Scoring against one item at a time, not one merged bag of every
130
+ // item, is what stops a sprawling plan from marking everything on-plan: an ask is on the plan only when one thing
131
+ // you wrote is really about it, not when its words happen to be scattered across fifty unrelated tasks.
132
+ export function bestItemCoverage(terms, itemSets, idf) {
133
+ const uniq = [...new Set(terms)];
134
+ const total = uniq.reduce((a, w) => a + idf(w), 0);
135
+ if (!total) return 0;
136
+ let best = 0;
137
+ for (const set of itemSets) {
138
+ let hit = 0;
139
+ for (const w of uniq) if (set.has(w)) hit += idf(w);
140
+ if (hit > best) best = hit;
141
+ }
142
+ return best / total;
143
+ }
144
+
145
+ function addTerms(bag, terms, weight) {
146
+ for (const w of terms) bag.set(w, Math.max(bag.get(w) || 0, weight));
147
+ }
148
+
149
+ function fade(bag, f) {
150
+ for (const [k, v] of bag) {
151
+ if (v * f < 0.05) bag.delete(k);
152
+ else bag.set(k, v * f);
153
+ }
154
+ }
155
+
156
+ function buildTurns(session) {
157
+ const turns = [];
158
+ let cur = null;
159
+ let last = null;
160
+ for (const e of session.events) {
161
+ if (e.kind === 'prompt' || e.kind === 'auto') {
162
+ cur = {
163
+ human: e.kind === 'prompt', t: e.t, text: e.text || '', command: e.command || null,
164
+ cwd: e.cwd || session.cwd, idle: last === null ? Infinity : e.t - last, steps: [], session,
165
+ };
166
+ turns.push(cur);
167
+ } else if (cur) {
168
+ cur.steps.push(e);
169
+ }
170
+ last = e.t;
171
+ }
172
+ return turns;
173
+ }
174
+
175
+ // Agent time inside the window, split by local day
176
+ function agentTime(turn, since, until) {
177
+ let prev = turn.t;
178
+ let ms = 0;
179
+ const byDay = new Map();
180
+ for (const s of turn.steps) {
181
+ const dt = s.t - prev;
182
+ prev = s.t;
183
+ if (dt <= 0 || s.t < since || s.t > until) continue;
184
+ const add = Math.min(dt, DRIFT.idleCapMs);
185
+ ms += add;
186
+ const k = dayKey(s.t);
187
+ byDay.set(k, (byDay.get(k) || 0) + add);
188
+ }
189
+ return { ms, byDay, end: prev };
190
+ }
191
+
192
+ /**
193
+ * sessions: from readTranscripts(). planFor(cwd, dayKey): from readStatedPlans(), or null when there is no plan.
194
+ */
195
+ export function analyzeDrift(sessions, { since, until = Date.now(), planFor = () => null, planNotes = [] } = {}) {
196
+ const perSession = sessions.map(s => buildTurns(s));
197
+ const turns = perSession.flat();
198
+ const human = turns.filter(t => t.human);
199
+
200
+ // Features per turn, then word weights across the whole span (a word in every turn tells you nothing)
201
+ const fileCount = new Map();
202
+ for (const t of human) {
203
+ t.words = tokenize(t.text.slice(0, 2000));
204
+ t.replyAll = tokenize(t.steps.map(s => s.words || '').join(' ').slice(0, 8000));
205
+ t.files = new Set(t.steps.flatMap(s => s.files || []));
206
+ t.fileWords = [...t.files].flatMap(f => pathTerms(f, t.cwd));
207
+ for (const f of t.files) fileCount.set(f, (fileCount.get(f) || 0) + 1);
208
+ }
209
+ const df = new Map();
210
+ for (const t of human) for (const w of new Set([...t.words, ...t.replyAll, ...t.fileWords])) df.set(w, (df.get(w) || 0) + 1);
211
+ const n = human.length;
212
+ const idf = (w) => Math.log((n + 1) / ((df.get(w) || 0) + 1)) + 1;
213
+ // Files most turns touch (logs, plan files, notes) say nothing about the topic
214
+ const everywhere = new Set([...fileCount].filter(([, c]) => c >= 5 && c > n * 0.2).map(([f]) => f));
215
+ for (const t of human) {
216
+ t.replyTop = [...new Set(t.replyAll)].sort((a, b) => idf(b) - idf(a)).slice(0, DRIFT.replyTerms);
217
+ t.topicFiles = new Set([...t.files].filter(f => !everywhere.has(f)));
218
+ const weight = [...new Set(t.words)].reduce((a, w) => a + idf(w), 0);
219
+ // How much a turn says about its topic, used to name a detour after its most descriptive turn, not its first
220
+ t.richness = weight * (shortName(t.text) ? 1 : 0);
221
+ t.substantive = weight >= DRIFT.substantiveAt && !(CONTINUE.test(t.text) && t.words.length <= 5);
222
+ t.states = t.substantive && (Boolean(t.command && PLAN_COMMANDS.has(t.command)) || PLAN_PREFIX.test(t.text));
223
+ }
224
+
225
+ const detours = [];
226
+ const sources = { plan: 0, goal: 0, ask: 0 };
227
+ const planNames = new Set();
228
+ const scopes = new Set();
229
+ let sittingId = 0;
230
+
231
+ // "What you said" for a turn: the best any single plan item covers it, OR the opening ask, OR the topics that
232
+ // earlier on-plan work has already established this sitting (so related follow-ups stay on)
233
+ const onScore = (t, intent) => {
234
+ const fromPlan = intent.items ? bestItemCoverage(t.words, intent.items, idf) : 0;
235
+ const fromFocus = coverage(t.words, intent.focus, idf);
236
+ const shared = [...t.topicFiles].filter(f => intent.files.has(f)).length;
237
+ const fromFiles = t.topicFiles.size ? shared / t.topicFiles.size : 0;
238
+ return Math.max(fromPlan, fromFocus, fromFiles);
239
+ };
240
+
241
+ const newThread = (t, lbl, score) => ({ label: lbl, bag: new Map(), head: t, best: t, ms: 0, score });
242
+ const join = (t, th, score) => {
243
+ t.label = th.label;
244
+ t.thread = th;
245
+ t.score = score;
246
+ th.ms += t.time.ms;
247
+ if (t.richness > th.best.richness) th.best = t; // name the thread after its most descriptive turn
248
+ fade(th.bag, 0.6);
249
+ addTerms(th.bag, t.words, 1);
250
+ addTerms(th.bag, t.replyTop, 0.7);
251
+ addTerms(th.bag, t.fileWords, 0.6);
252
+ };
253
+
254
+ // Walk each session in order: sittings, then threads, then labels
255
+ for (const list of perSession) {
256
+ let sitting = null;
257
+ let thread = null;
258
+ let pending = []; // "continue" and the like, before the sitting has said what it is for
259
+ for (const t of list) {
260
+ t.time = agentTime(t, since, until);
261
+ if (!t.human) { t.label = 'auto'; continue; }
262
+
263
+ if (!sitting || (t.idle >= DRIFT.sittingGapMs && t.substantive)) {
264
+ sitting = { id: ++sittingId, intent: null };
265
+ thread = null;
266
+ pending = [];
267
+ const plan = planFor(t.cwd, dayKey(t.t));
268
+ if (plan && plan.items.length) {
269
+ const items = plan.items.map(x => new Set(tokenize(x))).filter(s => s.size);
270
+ sitting.intent = { source: 'plan', items, focus: new Map(), files: new Set() };
271
+ plan.sources.forEach(x => planNames.add(x));
272
+ scopes.add(plan.scope);
273
+ sources.plan++;
274
+ }
275
+ }
276
+ t.sitting = sitting.id;
277
+
278
+ // A stated goal always resets what you said; with nothing written down, so does the sitting's first real ask.
279
+ // The opening ask becomes the plan's one "item", so later work is matched against it the same way. When there
280
+ // is no written plan, the next couple of asks are treated as part of laying out the task (people front-load
281
+ // context), so a thin opening does not make the rest of the same task read as drift.
282
+ if (t.states || (!sitting.intent && t.substantive)) {
283
+ const focus = new Map();
284
+ addTerms(focus, t.words, 1);
285
+ addTerms(focus, t.replyTop, 0.5);
286
+ addTerms(focus, t.fileWords, 0.5);
287
+ sitting.intent = { source: t.states ? 'goal' : 'ask', items: [new Set(t.words)], focus, files: new Set(t.topicFiles), warm: 2, startT: t.t };
288
+ sources[sitting.intent.source]++;
289
+ thread = newThread(t, 'on', 1);
290
+ for (const p of [...pending, t]) join(p, thread, 1);
291
+ pending = [];
292
+ continue;
293
+ }
294
+ if (!sitting.intent) { pending.push(t); continue; }
295
+
296
+ // Warmup only counts while the task is still being laid out: substantive, budget left, and soon after it opened
297
+ const warming = sitting.intent.warm > 0 && t.substantive && (t.t - sitting.intent.startT) <= DRIFT.warmupMs;
298
+ // Non-substantive turns ("yes", "continue") ride the current thread; a real ask is re-judged against the plan
299
+ if (!warming && thread && (!t.substantive || coverage(t.words, thread.bag, idf) >= DRIFT.continueAt)) {
300
+ join(t, thread, thread.score);
301
+ } else {
302
+ const score = warming ? 1 : onScore(t, sitting.intent);
303
+ thread = newThread(t, warming || score >= DRIFT.onAt ? 'on' : 'detour', score);
304
+ if (thread.label === 'detour') detours.push(thread);
305
+ join(t, thread, score);
306
+ if (warming) { sitting.intent.items.push(new Set(t.words)); sitting.intent.warm--; }
307
+ }
308
+ // Work that is on what you said establishes its topic, so a related follow-up a few turns later is still on
309
+ if (t.label === 'on') {
310
+ fade(sitting.intent.focus, 0.85);
311
+ addTerms(sitting.intent.focus, t.words, 1);
312
+ addTerms(sitting.intent.focus, t.replyTop, 0.5);
313
+ for (const f of t.topicFiles) sitting.intent.files.add(f);
314
+ }
315
+ }
316
+ // A sitting that never said anything substantive counts as on whatever it was already doing
317
+ for (const p of pending) p.label = 'on';
318
+ }
319
+
320
+ return summarize({ turns, human, detours, since, until, idf, sources, planNames, scopes: [...scopes], planNotes });
321
+ }
322
+
323
+ // What a run of turns touched, for naming a detour whose own words do not make a readable label
324
+ function touchedBy(turns) {
325
+ return { files: turns.flatMap(t => [...(t.topicFiles || t.files || [])]), cwd: turns[0]?.cwd || '', home: homedir() };
326
+ }
327
+
328
+ function summarize({ turns, human, detours, since, until, idf, sources, planNames, scopes, planNotes }) {
329
+ const counted = turns.filter(t => t.time && t.time.ms > 0 && (t.label === 'on' || t.label === 'detour'));
330
+ const onMs = counted.filter(t => t.label === 'on').reduce((a, t) => a + t.time.ms, 0);
331
+ const detourMs = counted.filter(t => t.label === 'detour').reduce((a, t) => a + t.time.ms, 0);
332
+ const autoMs = turns.filter(t => t.label === 'auto' && t.time).reduce((a, t) => a + t.time.ms, 0);
333
+ const agentMs = onMs + detourMs;
334
+
335
+ // Days in the window, oldest first, including quiet ones
336
+ const days = [];
337
+ for (let d = new Date(since); d.getTime() <= until; d.setDate(d.getDate() + 1)) {
338
+ const k = dayKey(d.getTime());
339
+ if (!days.find(x => x.day === k)) days.push({ day: k, onMs: 0, detourMs: 0 });
340
+ }
341
+ const lastKey = dayKey(until);
342
+ if (!days.find(x => x.day === lastKey)) days.push({ day: lastKey, onMs: 0, detourMs: 0 });
343
+ for (const t of counted) {
344
+ for (const [k, ms] of t.time.byDay) {
345
+ const d = days.find(x => x.day === k);
346
+ if (d) d[t.label === 'on' ? 'onMs' : 'detourMs'] += ms;
347
+ }
348
+ }
349
+ const busiest = Math.max(0, ...days.map(d => d.onMs + d.detourMs));
350
+ const worst = days
351
+ .filter(d => d.onMs + d.detourMs >= Math.max(30 * MIN, busiest * 0.15) && d.detourMs > 0)
352
+ .map(d => ({ ...d, share: d.detourMs / (d.onMs + d.detourMs) }))
353
+ .sort((a, b) => b.share - a.share || b.detourMs - a.detourMs)[0] || null;
354
+
355
+ // A detour episode is a run of off-plan turns you did in one go before coming back. Grouping the contiguous run,
356
+ // then naming it after its most descriptive turn, keeps "bin emptied btw" inside the footage-move it belongs to.
357
+ const episodes = [];
358
+ const bySitting = new Map();
359
+ for (const t of human) {
360
+ if (!t.sitting) continue;
361
+ if (!bySitting.has(t.sitting)) bySitting.set(t.sitting, []);
362
+ bySitting.get(t.sitting).push(t);
363
+ }
364
+ for (const list of bySitting.values()) {
365
+ let run = null;
366
+ const close = () => { if (run && run.ms > 0) episodes.push(run); run = null; };
367
+ for (const t of list.sort((a, b) => a.t - b.t)) {
368
+ if (t.label === 'detour') {
369
+ if (!run) run = { ms: 0, start: t.t, end: t.t, turns: [] };
370
+ run.ms += t.time.ms;
371
+ run.end = Math.max(run.end, t.time.end);
372
+ run.turns.push(t);
373
+ } else if (t.label === 'on') {
374
+ close();
375
+ }
376
+ }
377
+ close();
378
+ }
379
+ for (const e of episodes) e.best = e.turns.reduce((a, b) => (b.richness > a.richness ? b : a), e.turns[0]);
380
+
381
+ // The longest single episode is the day's deepest rabbit hole (named below, from the detour row it belongs to)
382
+ const longest = episodes.slice().sort((a, b) => b.ms - a.ms)[0] || null;
383
+
384
+ // Episodes on the same subject across the week are one named detour, counted each time it pulled you away
385
+ const clusters = [];
386
+ for (const e of episodes.filter(x => x.best.richness > 0).sort((a, b) => a.start - b.start)) {
387
+ const own = bagOf(e.best.words);
388
+ const c = clusters.find(x => coverage(e.best.words, x.bag, idf) >= DRIFT.clusterAt || coverage(x.best.words, own, idf) >= DRIFT.clusterAt);
389
+ if (c) {
390
+ c.ms += e.ms;
391
+ c.episodes.push(e);
392
+ if (e.best.richness > c.best.richness) c.best = e.best;
393
+ addTerms(c.bag, e.best.words, 1);
394
+ } else {
395
+ clusters.push({ best: e.best, bag: own, ms: e.ms, episodes: [e] });
396
+ }
397
+ }
398
+ // Two detours that end up with the same label (several fall back to "code") are one row
399
+ const mergeSame = (list) => {
400
+ const by = new Map();
401
+ for (const d of list) {
402
+ const k = `${d.name}|${d.category}`;
403
+ const c = by.get(k);
404
+ if (c) { c.ms += d.ms; c.times += d.times; c.first = Math.min(c.first, d.first); } else by.set(k, { ...d });
405
+ }
406
+ return [...by.values()];
407
+ };
408
+ // One name and one category per detour, decided once here. The list and the rabbit-hole card both read them, so they cannot disagree.
409
+ const rowOf = (c) => {
410
+ const category = categorize(c.episodes.map(e => e.best.text).join(' '));
411
+ return {
412
+ name: labelName(c.best.text, touchedBy(c.episodes.flatMap(e => e.turns)), category),
413
+ category,
414
+ ms: c.ms,
415
+ times: c.episodes.length,
416
+ first: Math.min(...c.episodes.map(e => e.start)),
417
+ };
418
+ };
419
+ const named = mergeSame(clusters.map(rowOf).filter(d => d.name)).sort((a, b) => b.ms - a.ms);
420
+
421
+ // The rabbit hole is the longest episode, labelled as the detour it belongs to. An episode that joined no cluster (no
422
+ // readable words) falls back to labelling itself.
423
+ const home = longest ? clusters.find(c => c.episodes.includes(longest)) : null;
424
+ const homeRow = home ? rowOf(home) : null;
425
+ const ownCategory = longest ? categorize(longest.turns.map(t => t.text).join(' ')) : null;
426
+ const streak = longest ? {
427
+ ms: longest.ms,
428
+ wallMs: longest.end - longest.start,
429
+ start: longest.start,
430
+ detours: longest.turns.filter(t => t.thread && t.thread.head === t).length,
431
+ name: homeRow ? homeRow.name : labelName(longest.best.text, touchedBy(longest.turns), ownCategory),
432
+ category: homeRow ? homeRow.category : ownCategory,
433
+ } : null;
434
+
435
+ const active = new Set(counted.map(t => t.session));
436
+ const inSpan = human.filter(t => t.t >= since && t.t <= until);
437
+ const agentsUsed = { claude: 0, codex: 0 };
438
+ for (const s of active) agentsUsed[s.agent] = (agentsUsed[s.agent] || 0) + 1;
439
+
440
+ return {
441
+ available: agentMs > 0,
442
+ since, until,
443
+ share: agentMs ? onMs / agentMs : 0,
444
+ totals: {
445
+ agentMs, onMs, detourMs, autoMs,
446
+ prompts: inSpan.length,
447
+ sessions: active.size,
448
+ sittings: new Set(counted.map(t => t.sitting)).size,
449
+ },
450
+ agents: agentsUsed,
451
+ days,
452
+ worstDay: worst,
453
+ streak,
454
+ detours: named,
455
+ detourCount: named.length,
456
+ source: {
457
+ plan: sources.plan, goal: sources.goal, ask: sources.ask,
458
+ planNames: [...planNames],
459
+ scope: scopes.includes('list') ? 'list' : scopes.includes('project') ? 'project' : 'session',
460
+ notes: planNotes,
461
+ },
462
+ // Every labelled ask, for --json: when, how long, which label, and how the score came out
463
+ turns: human.filter(t => t.time && t.t >= since && t.t <= until).map(t => ({
464
+ t: new Date(t.t).toISOString(),
465
+ agent: t.session.agent,
466
+ session: String(t.session.id).slice(0, 8),
467
+ sitting: t.sitting,
468
+ label: t.label,
469
+ score: t.score == null ? null : Math.round(t.score * 100) / 100,
470
+ opens: Boolean(t.thread && t.thread.head === t),
471
+ minutes: Math.round(t.time.ms / MIN),
472
+ ask: shortName(t.text, { maxWords: 14, maxChars: 90 }),
473
+ })),
474
+ };
475
+ }
476
+
477
+ function bagOf(words) {
478
+ const bag = new Map();
479
+ addTerms(bag, words, 1);
480
+ return bag;
481
+ }
@@ -0,0 +1,154 @@
1
+ import { readFileSync } from 'fs';
2
+ import { join, dirname } from 'path';
3
+ import { fileURLToPath } from 'url';
4
+ import { FRONTIER_MODELS } from '../data/frontier-models.js';
5
+ import { PRICE_OVERRIDES } from '../data/price-overrides.js';
6
+
7
+ const __dirname = dirname(fileURLToPath(import.meta.url));
8
+
9
+ // Bundled offline prices (30 Sep 2026). Read lazily so a missing file only costs the offline path.
10
+ let snapshotCache = null;
11
+ export function loadSnapshot() {
12
+ if (!snapshotCache) snapshotCache = JSON.parse(readFileSync(join(__dirname, '..', 'data', 'price-snapshot.json'), 'utf-8'));
13
+ return snapshotCache;
14
+ }
15
+
16
+ const PARTS = ['input', 'output', 'cacheRead', 'cacheWrite'];
17
+
18
+ // LiteLLM lists dollars per token; 1.32e-6 * 1e6 is 1.3199999999999998 in floating point, so round to 10 significant digits.
19
+ const perMillion = (perToken) => Number((perToken * 1e6).toPrecision(10));
20
+
21
+ /** One LiteLLM entry to per-million prices exactly as listed. A price LiteLLM does not list is null. */
22
+ export function priceFromLiteLLM(entry) {
23
+ const f = (v) => (v == null ? null : perMillion(v));
24
+ return {
25
+ input: f(entry.input_cost_per_token),
26
+ output: f(entry.output_cost_per_token),
27
+ cacheRead: f(entry.cache_read_input_token_cost),
28
+ cacheWrite: f(entry.cache_creation_input_token_cost),
29
+ };
30
+ }
31
+
32
+ /**
33
+ * Apply the two fill rules and record which fields they touched.
34
+ * cache read missing -> the input price (no discount assumed)
35
+ * cache write missing or zero -> the input price (the provider bills cache writes as ordinary input)
36
+ */
37
+ export function resolvePrice(p) {
38
+ const filled = [];
39
+ let cacheRead = p.cacheRead;
40
+ let cacheWrite = p.cacheWrite;
41
+ if (cacheRead == null) { cacheRead = p.input; filled.push('cacheRead'); }
42
+ if (cacheWrite == null || cacheWrite === 0) { cacheWrite = p.input; filled.push('cacheWrite'); }
43
+ return { price: { input: p.input, output: p.output, cacheRead, cacheWrite }, filled };
44
+ }
45
+
46
+ const usable = (p) => p && p.input > 0 && p.output > 0;
47
+
48
+ function costParts(tokens, price) {
49
+ const parts = {
50
+ input: tokens.input / 1e6 * price.input,
51
+ output: tokens.output / 1e6 * price.output,
52
+ cacheRead: tokens.cacheRead / 1e6 * price.cacheRead,
53
+ cacheWrite: tokens.cacheWrite / 1e6 * price.cacheWrite,
54
+ };
55
+ return { parts, total: parts.input + parts.output + parts.cacheRead + parts.cacheWrite };
56
+ }
57
+
58
+ /**
59
+ * Find a model's raw per-million prices. LiteLLM live data wins; a manual override is used only when LiteLLM has no
60
+ * entry for the id; the bundled snapshot stands in when the fetch failed (and as the last resort online).
61
+ */
62
+ function findPrice(id, { raw, snapshot, overrides }) {
63
+ const live = raw && raw[id] ? priceFromLiteLLM(raw[id]) : null;
64
+ if (usable(live)) return { source: 'litellm', raw: live };
65
+ const snap = snapshot?.models?.[id];
66
+ const ovr = overrides?.[id];
67
+ const fromOverride = ovr?.intro ? { source: 'override', raw: { cacheRead: null, cacheWrite: null, ...ovr.intro }, ovr } : null;
68
+ const fromSnapshot = usable(snap) ? { source: 'snapshot', raw: { cacheRead: null, cacheWrite: null, ...snap } } : null;
69
+ const order = raw ? [fromOverride, fromSnapshot] : [fromSnapshot, fromOverride];
70
+ return order.find(Boolean) || null;
71
+ }
72
+
73
+ /**
74
+ * Reprice the user's own token totals on each listed model's list price.
75
+ * `opts` exists so tests can pass a fixed price table: { raw (LiteLLM JSON or null = offline), fetchedAt, snapshot, overrides, models }.
76
+ */
77
+ export function reprice(costAnalysis, opts = {}) {
78
+ const raw = opts.raw || null;
79
+ const models = opts.models || FRONTIER_MODELS;
80
+ const overrides = opts.overrides || PRICE_OVERRIDES;
81
+ const lookup = { raw, overrides, snapshot: opts.snapshot || safeSnapshot() };
82
+
83
+ const t = costAnalysis.totals || {};
84
+ const tokens = {
85
+ input: t.inputTokens || 0,
86
+ output: t.outputTokens || 0,
87
+ cacheRead: t.cacheReadTokens || 0,
88
+ cacheWrite: t.cacheWriteTokens || 0,
89
+ };
90
+
91
+ const rows = [];
92
+ const dailyDays = costAnalysis.dailyCosts || [];
93
+ for (const m of models) {
94
+ const found = findPrice(m.id, lookup);
95
+ if (!found) continue;
96
+ const { price, filled } = resolvePrice(found.raw);
97
+ const { parts, total } = costParts(tokens, price);
98
+ const row = {
99
+ id: m.id, label: m.label, maker: m.maker,
100
+ price, parts, total,
101
+ cacheShare: total > 0 ? (parts.cacheRead + parts.cacheWrite) / total : 0,
102
+ source: found.source,
103
+ filled,
104
+ };
105
+ if (found.ovr) {
106
+ row.sourceUrl = found.ovr.sourceUrl;
107
+ row.sourceName = found.ovr.sourceName;
108
+ row.checked = found.ovr.checked;
109
+ row.note = found.ovr.note;
110
+ row.assumption = found.ovr.assumption;
111
+ if (found.ovr.after) {
112
+ const a = resolvePrice({ cacheRead: null, cacheWrite: null, ...found.ovr.after });
113
+ const c = costParts(tokens, a.price);
114
+ row.after = { price: a.price, parts: c.parts, total: c.total, filled: a.filled };
115
+ }
116
+ }
117
+ rows.push(row);
118
+ }
119
+ rows.sort((a, b) => b.total - a.total);
120
+
121
+ // Running total per day, from each day's per-model token counts priced at this model's rates.
122
+ const dates = dailyDays.map(d => d.date);
123
+ const perModel = {};
124
+ for (const r of rows) {
125
+ let cum = 0;
126
+ perModel[r.id] = dailyDays.map(d => {
127
+ for (const dm of d.models || []) {
128
+ const tk = dm.tokens;
129
+ if (!tk || typeof tk !== 'object') continue;
130
+ cum += costParts({ input: tk.input || 0, output: tk.output || 0, cacheRead: tk.cacheRead || 0, cacheWrite: tk.cacheWrite || 0 }, r.price).total;
131
+ }
132
+ return round2(cum);
133
+ });
134
+ }
135
+ let actualCum = 0;
136
+ const actual = dailyDays.map(d => { actualCum += d.cost || 0; return round2(actualCum); });
137
+
138
+ return {
139
+ models: rows,
140
+ actualListCost: costAnalysis.totalCost || 0,
141
+ tokens,
142
+ readPerWrite: tokens.output > 0 ? (tokens.input + tokens.cacheRead + tokens.cacheWrite) / tokens.output : 0,
143
+ dailyCum: { dates, perModel, actual },
144
+ fetchedAt: opts.fetchedAt ?? null,
145
+ offline: !raw,
146
+ ...(raw ? {} : { snapshotDate: (lookup.snapshot || {}).date || null }),
147
+ };
148
+ }
149
+
150
+ function safeSnapshot() {
151
+ try { return loadSnapshot(); } catch { return null; }
152
+ }
153
+
154
+ const round2 = (n) => Math.round(n * 100) / 100;