claude-token-saver 3.0.1 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/route-scan.js CHANGED
@@ -2,38 +2,74 @@
2
2
  * route-scan — detect recurring "easy" work running on expensive models and
3
3
  * propose model-delegation ratchet rules.
4
4
  *
5
- * Runs the frugon-style difficulty idea at the EPISODE level (one user
6
- * request = the consecutive API calls it triggered), because per-call scoring
7
- * saturates on Claude Code's large session contexts. An episode is easy when
8
- * the whole request finished in few calls with little generation — exactly
9
- * the work a haiku subagent could take.
5
+ * Difficulty is judged at the EPISODE level (one user request = the
6
+ * consecutive API calls it triggered), because per-call scoring saturates on
7
+ * Claude Code's large session contexts — prompt size carries no signal when
8
+ * every call ships a 100k+ cached prefix. An episode is easy when the whole
9
+ * request finished in few calls with little generation — exactly the work a
10
+ * haiku subagent could take.
10
11
  *
11
12
  * Fully local, zero token cost. Results are cached (24h) so the SessionStart
12
13
  * hook can read them without re-parsing a month of transcripts.
13
14
  *
14
- * Pipeline position (per design discussion): frugon/this scan is NOT a
15
- * real-time router — it is a session-boundary calibrator that feeds the
16
- * existing ratchet promote flow (`harness promote R<N> --project|--global`).
15
+ * Pipeline position (per design discussion): this scan is NOT a real-time
16
+ * router — it is a session-boundary calibrator that feeds the existing
17
+ * ratchet promote flow (`harness promote R<N> --project|--global`).
17
18
  */
18
19
 
19
- import { readFileSync, writeFileSync, existsSync, mkdirSync } from 'node:fs';
20
+ import { readFileSync, writeFileSync, existsSync, mkdirSync, statSync } from 'node:fs';
20
21
  import { join } from 'node:path';
21
22
  import { homedir } from 'node:os';
22
23
  import { discoverSessionFiles } from './parser.js';
23
- import { collectSessionRecords } from './frugon-export.js';
24
+ import { collectSessionRecords } from './session-records.js';
24
25
 
25
- // Episode is "easy" when the whole user request finished within these bounds.
26
- // Calibrated on real data (2026-07): 27% of episodes, 3-6% of tokens.
27
- export const EASY_MAX_CALLS = 6;
28
- export const EASY_MAX_OUT_TOKENS = 1500;
26
+ // ── Tier bands (docs/TIER_CRITERIA.md §3) ────────────────────────────────
27
+ // T2 (haiku): finished in few calls, tiny output, near-zero mutation, no
28
+ // errors. T1 (sonnet): moderate output/mutation, at most one tool error.
29
+ // T0: everything else stays on the session model. Output-token thresholds
30
+ // are calibrated per-user from their own 14-day distribution (fixed
31
+ // thresholds drift with workload — RouteLLM's stated limitation), clamped
32
+ // to sane ranges so a skewed window can't stretch them absurdly.
33
+ export const T2_MAX_CALLS = 6;
34
+ export const T2_MAX_MUTATING = 2;
35
+ export const T2_OUT_CLAMP = [1000, 3000]; // default 1500 pre-calibration
36
+ export const T1_MAX_MUTATING = 6;
37
+ export const T1_MAX_ERRORS = 1;
38
+ export const T1_OUT_CLAMP = [5000, 15000]; // default 8000 pre-calibration
39
+ export const T0_MIN_ERRORS = 2; // repeated tool errors = hard, by outcome
40
+ export const T0_MIN_MUTATING = 7;
41
+ // Episodes below this output size carry no delegable work (conversational
42
+ // acks, feedback) — skip entirely.
43
+ export const MIN_DELEGABLE_OUT = 100;
44
+ // Escalation keywords: design/analysis judgement stays on the top tier.
45
+ export const ESCALATE_RE = /설계|아키텍처|리팩토링|원인 분석|개선할|검토해보|비교|왜 |analyze|compare|evaluate|architect|refactor/i;
29
46
  // A pattern must recur this often before we nag about it.
30
47
  export const MIN_RECURRENCE = 3;
31
- // Cache is fresh for a day — the SessionStart hook never rescans inline.
32
- export const CACHE_TTL_MS = 24 * 60 * 60 * 1000;
33
48
 
34
- // Category → recommended haiku subagent. First match wins; order matters
35
- // (translate before read: "번역해줘" also matches the read keywords).
49
+ // ── Rescan gate (data-driven, not time-driven) ───────────────────────────
50
+ // A scan over unchanged transcripts is deterministic — identical output —
51
+ // so time alone is a bad trigger: it wastes scans on idle days and lags a
52
+ // full day behind heavy ones. Instead we rescan when enough NEW transcript
53
+ // data accumulated (~5MB ≈ 20-40 episodes on measured data — enough for a
54
+ // pattern to newly cross MIN_RECURRENCE), with guardrails: a minimum
55
+ // interval against session-churn thrash, a daily fallback so small trickles
56
+ // still refresh rule-health, and a hard skip when nothing changed at all.
57
+ export const RESCAN_MIN_INTERVAL_MS = 60 * 60 * 1000; // never more than hourly
58
+ export const RESCAN_BIG_DELTA_BYTES = 5 * 1024 * 1024; // this much new data → rescan now
59
+ export const RESCAN_MAX_AGE_MS = 24 * 60 * 60 * 1000; // any new data + a day old → rescan
60
+
61
+ // Category → recommended subagent. First match wins; order matters
62
+ // (paste before everything: keywords inside pasted UI/log text would
63
+ // otherwise mislabel the episode; translate before read: "번역해줘" also
64
+ // matches the read keywords).
65
+ const PASTE_MIN_LEN = 400;
36
66
  const CATEGORIES = [
67
+ {
68
+ id: 'paste',
69
+ label: '붙여넣은 화면·로그 질문',
70
+ agent: 'haiku-explore',
71
+ re: null, // matched by length, see categorize()
72
+ },
37
73
  {
38
74
  id: 'translate',
39
75
  label: '배치 번역·정형 텍스트 변환',
@@ -92,7 +128,8 @@ export function mungeProjectPath(p) {
92
128
  }
93
129
 
94
130
  function categorize(text) {
95
- for (const c of CATEGORIES) if (c.re.test(text)) return c;
131
+ if (text.length >= PASTE_MIN_LEN) return CATEGORIES.find((c) => c.id === 'paste');
132
+ for (const c of CATEGORIES) if (c.re && c.re.test(text)) return c;
96
133
  return null;
97
134
  }
98
135
 
@@ -109,11 +146,14 @@ function toEpisodes(records) {
109
146
  for (const r of records) {
110
147
  const text = (r.userText || '').trim();
111
148
  if (!cur || cur.text !== text) {
112
- cur = { text, calls: 0, out: 0, models: new Set(), cwd: '' };
149
+ cur = { text, calls: 0, out: 0, mutating: 0, errors: 0, delegated: 0, models: new Set(), cwd: '' };
113
150
  episodes.push(cur);
114
151
  }
115
152
  cur.calls += 1;
116
153
  cur.out += r.completion_tokens;
154
+ cur.mutating += r.mutatingToolCalls || 0;
155
+ cur.errors += r.toolErrors || 0;
156
+ cur.delegated += r.delegationCalls || 0;
117
157
  cur.models.add(r.model);
118
158
  if (!cur.cwd && r.cwd) cur.cwd = r.cwd;
119
159
  }
@@ -124,67 +164,139 @@ function isExpensiveModel(model) {
124
164
  return !/haiku/i.test(model);
125
165
  }
126
166
 
167
+ const clamp = (v, [lo, hi]) => Math.min(hi, Math.max(lo, v));
168
+ const percentile = (sorted, p) => sorted[Math.min(sorted.length - 1, Math.floor(p * sorted.length))];
169
+
170
+ /**
171
+ * Per-user output-token thresholds from this window's episode distribution.
172
+ * Falls back to mid-clamp defaults when the sample is too small to trust.
173
+ */
174
+ export function calibrateThresholds(episodeOuts) {
175
+ if (episodeOuts.length < 100) return { t2Out: 1500, t1Out: 8000, calibrated: false };
176
+ const sorted = [...episodeOuts].sort((a, b) => a - b);
177
+ return {
178
+ t2Out: clamp(Math.max(percentile(sorted, 0.25), 1500), T2_OUT_CLAMP),
179
+ t1Out: clamp(percentile(sorted, 0.75), T1_OUT_CLAMP),
180
+ calibrated: true,
181
+ };
182
+ }
183
+
184
+ /**
185
+ * Tier classification (docs/TIER_CRITERIA.md §3). Returns 'T0'|'T1'|'T2',
186
+ * or null when the episode carries nothing delegable. Order matters: hard
187
+ * evidence (errors, heavy mutation, big output, judgement keywords) wins
188
+ * before any cheap-band check.
189
+ */
190
+ export function tierOf(ep, category, th) {
191
+ if (ep.out < MIN_DELEGABLE_OUT) return null;
192
+ if (
193
+ ep.errors >= T0_MIN_ERRORS ||
194
+ ep.mutating >= T0_MIN_MUTATING ||
195
+ ep.out > th.t1Out ||
196
+ ESCALATE_RE.test(ep.text)
197
+ ) return 'T0';
198
+ if (!category || ep.delegated > 0) return 'T0';
199
+ if (ep.calls <= T2_MAX_CALLS && ep.out <= th.t2Out && ep.mutating <= T2_MAX_MUTATING && ep.errors === 0) return 'T2';
200
+ if (ep.out <= th.t1Out && ep.mutating <= T1_MAX_MUTATING && ep.errors <= T1_MAX_ERRORS) return 'T1';
201
+ return 'T0';
202
+ }
203
+
127
204
  /**
128
205
  * Scan transcripts and build delegation candidates.
129
206
  * Returns the cache object (also written to disk).
130
207
  */
131
208
  export async function runRouteScan({ days = 14 } = {}) {
132
209
  const files = await discoverSessionFiles({ days });
133
- const groups = new Map(); // "category|project" → aggregate
134
- let totalEpisodes = 0;
135
- let easyEpisodes = 0;
136
210
 
211
+ // Pass 1 — collect episodes (needed up front: thresholds are calibrated
212
+ // from the full window's output distribution before any tiering).
213
+ const all = []; // { ep, projectDir }
214
+ let dataBytes = 0; // window size snapshot — the rescan gate diffs against it
137
215
  for (const f of files) {
138
216
  let records;
139
217
  try {
140
- // Raw counts are irrelevant here (we classify by output size), and
141
- // content is required for categorization.
142
- records = await collectSessionRecords(f.path, { cacheWeighted: false, includeContent: true });
218
+ dataBytes += statSync(f.path).size;
219
+ records = await collectSessionRecords(f.path, { includeContent: true });
143
220
  } catch {
144
221
  continue;
145
222
  }
146
223
  for (const ep of toEpisodes(records)) {
147
224
  if (!ep.text) continue;
148
- totalEpisodes += 1;
149
- const easy = ep.calls <= EASY_MAX_CALLS && ep.out <= EASY_MAX_OUT_TOKENS;
150
- if (!easy) continue;
151
- easyEpisodes += 1;
152
- if (isSkippable(ep.text)) continue;
153
- if (![...ep.models].some(isExpensiveModel)) continue; // already cheap
154
- const cat = categorize(ep.text);
155
- if (!cat) continue;
156
- const key = `${cat.id}|${f.projectDir}`;
157
- const g = groups.get(key) || {
158
- category: cat.id,
159
- label: cat.label,
160
- agent: cat.agent,
161
- project: f.projectDir,
162
- projectPath: '',
163
- count: 0,
164
- models: new Set(),
165
- example: '',
166
- };
167
- g.count += 1;
168
- for (const m of ep.models) g.models.add(m);
169
- if (!g.projectPath && ep.cwd) g.projectPath = ep.cwd;
170
- if (!g.example || (ep.text.length < g.example.length && ep.text.length > 10)) {
171
- g.example = ep.text.slice(0, 80).replace(/\s+/g, ' ');
225
+ all.push({ ep, projectDir: f.projectDir });
226
+ }
227
+ }
228
+ const totalEpisodes = all.length;
229
+ const thresholds = calibrateThresholds(all.map((x) => x.ep.out));
230
+
231
+ // Pass 2 — tier, group by tier×category×project, and accumulate the
232
+ // per-category outcome stats that keep promoted model rules fresh.
233
+ const groups = new Map(); // "tier|category|project" → aggregate
234
+ const episodeStats = new Map(); // "category|project" (+ "category|*") → outcome stats
235
+ let tieredEpisodes = 0;
236
+ const bumpStats = (key, ep) => {
237
+ const s = episodeStats.get(key) || { count: 0, errCount: 0, epCount: 0 };
238
+ s.count += 1;
239
+ s.epCount += 1;
240
+ if (ep.errors > 0) s.errCount += 1;
241
+ episodeStats.set(key, s);
242
+ };
243
+
244
+ for (const { ep, projectDir } of all) {
245
+ if (isSkippable(ep.text)) continue;
246
+ if (![...ep.models].some(isExpensiveModel)) continue; // already cheap
247
+ const cat = categorize(ep.text);
248
+ if (cat) {
249
+ // rule-health denominator: episodes that LOOK delegable by shape
250
+ // (tier judged with the error signal zeroed — using real errors here
251
+ // would be circular, since T2 requires errors=0 by definition). The
252
+ // numerator is those that still hit errors: exactly the "light-looking
253
+ // work in this category keeps failing" risk a delegation rule cares about.
254
+ const shapeTier = tierOf({ ...ep, errors: 0 }, cat, thresholds);
255
+ if (shapeTier === 'T1' || shapeTier === 'T2') {
256
+ bumpStats(`${cat.id}|${projectDir}`, ep);
257
+ bumpStats(`${cat.id}|*`, ep);
172
258
  }
173
- groups.set(key, g);
174
259
  }
260
+ const tier = tierOf(ep, cat, thresholds);
261
+ if (tier !== 'T1' && tier !== 'T2') continue;
262
+ tieredEpisodes += 1;
263
+ const key = `${tier}|${cat.id}|${projectDir}`;
264
+ const g = groups.get(key) || {
265
+ tier,
266
+ category: cat.id,
267
+ label: cat.label,
268
+ agent: tier === 'T2' ? cat.agent : 'sonnet',
269
+ project: projectDir,
270
+ projectPath: '',
271
+ count: 0,
272
+ models: new Set(),
273
+ example: '',
274
+ };
275
+ g.count += 1;
276
+ for (const m of ep.models) g.models.add(m);
277
+ if (!g.projectPath && ep.cwd) g.projectPath = ep.cwd;
278
+ if (!g.example || (ep.text.length < g.example.length && ep.text.length > 10)) {
279
+ g.example = ep.text.slice(0, 80).replace(/\s+/g, ' ');
280
+ }
281
+ groups.set(key, g);
175
282
  }
176
283
 
177
284
  // Keep prior dismissed/promoted signatures across rescans.
178
285
  const prev = readRouteScan();
179
286
  const resolved = new Set(prev?.resolved || []);
180
287
 
288
+ const ruleText = (g) => g.tier === 'T2'
289
+ ? `"${g.label}" 유형의 단순 요청(예: "${g.example}")은 ${g.agent}(haiku) 서브에이전트로 위임한다`
290
+ : `"${g.label}" 유형의 중간 난도 요청(예: "${g.example}")은 model: sonnet 서브에이전트로 위임한다 (설계 판단·반복 에러 발생 시 메인 모델이 이어받음)`;
291
+
181
292
  const candidates = [...groups.values()]
182
293
  .filter((g) => g.count >= MIN_RECURRENCE)
183
294
  .sort((a, b) => b.count - a.count)
184
- .slice(0, 5)
295
+ .slice(0, 8)
185
296
  .map((g, i) => ({
186
297
  id: i + 1,
187
- signature: `${g.category}|${g.project}`,
298
+ signature: `${g.tier}|${g.category}|${g.project}`,
299
+ tier: g.tier,
188
300
  category: g.category,
189
301
  label: g.label,
190
302
  agent: g.agent,
@@ -200,7 +312,7 @@ export async function runRouteScan({ days = 14 } = {}) {
200
312
  // project already, so scope suggestion is per-candidate 'project' unless
201
313
  // the same category recurs across 2+ projects (then 'global').
202
314
  suggestedScope: 'project',
203
- rule: `"${g.label}" 유형의 단순 요청(예: "${g.example}")은 ${g.agent}(haiku) 서브에이전트로 위임한다`,
315
+ rule: ruleText(g),
204
316
  }));
205
317
 
206
318
  // Same category appearing in 2+ projects → suggest global for each.
@@ -216,7 +328,10 @@ export async function runRouteScan({ days = 14 } = {}) {
216
328
  scannedAt: new Date().toISOString(),
217
329
  days,
218
330
  totalEpisodes,
219
- easyEpisodes,
331
+ dataBytes,
332
+ // kept as `easyEpisodes` for statusline/back-compat; now counts T1+T2.
333
+ easyEpisodes: tieredEpisodes,
334
+ thresholds,
220
335
  candidates,
221
336
  resolved: [...resolved],
222
337
  };
@@ -227,6 +342,15 @@ export async function runRouteScan({ days = 14 } = {}) {
227
342
  } catch {
228
343
  // best-effort — scan results are still returned
229
344
  }
345
+
346
+ // Continuous update (user requirement): every rescan refreshes promoted
347
+ // model-fitting rules from the new window — recurrence counts, error
348
+ // rates, and rule-health flags — and rewrites their managed blocks.
349
+ try {
350
+ const { refreshModelRules } = await import('./model-rules.js');
351
+ refreshModelRules(episodeStats, { now: cache.scannedAt });
352
+ } catch { /* registry unwritable — scan result still valid */ }
353
+
230
354
  return cache;
231
355
  }
232
356
 
@@ -239,10 +363,34 @@ export function readRouteScan() {
239
363
  }
240
364
  }
241
365
 
242
- export function isCacheFresh(cache) {
243
- if (!cache?.scannedAt) return false;
366
+ /**
367
+ * Data-driven rescan gate (see constants above). Cheap: one stat() per
368
+ * transcript file (~32 files on measured data) — a few milliseconds.
369
+ */
370
+ export async function shouldRescan(cache, { days = 14 } = {}) {
371
+ if (!cache?.scannedAt) return true;
244
372
  const ts = Date.parse(cache.scannedAt);
245
- return Number.isFinite(ts) && Date.now() - ts < CACHE_TTL_MS;
373
+ if (!Number.isFinite(ts)) return true;
374
+ const age = Date.now() - ts;
375
+ if (age < RESCAN_MIN_INTERVAL_MS) return false;
376
+
377
+ let total = 0;
378
+ let anyNew = false;
379
+ try {
380
+ for (const f of await discoverSessionFiles({ days })) {
381
+ const s = statSync(f.path);
382
+ total += s.size;
383
+ if (s.mtimeMs > ts) anyNew = true;
384
+ }
385
+ } catch {
386
+ return age >= RESCAN_MAX_AGE_MS; // can't stat — degrade to daily
387
+ }
388
+ if (!anyNew) return false; // nothing changed → identical scan, skip forever
389
+ // Append-only transcripts: window growth ≈ new data. Files aging out of
390
+ // the window shrink the total, making this estimate conservative.
391
+ const newBytes = Math.max(0, total - (cache.dataBytes || 0));
392
+ if (newBytes >= RESCAN_BIG_DELTA_BYTES) return true;
393
+ return age >= RESCAN_MAX_AGE_MS;
246
394
  }
247
395
 
248
396
  /** Candidates not yet promoted/dismissed. */
@@ -0,0 +1,112 @@
1
+ /**
2
+ * session-records — parse a Claude Code session transcript into per-API-call
3
+ * records (model, tokens, depth, triggering user prompt, session cwd).
4
+ *
5
+ * This is the shared substrate for episode-level analysis (route-scan and the
6
+ * 3.x tier-classification work): one record per API call, deduplicated by
7
+ * requestId (last-write-wins, matching parser.js).
8
+ */
9
+
10
+ import { createReadStream } from 'node:fs';
11
+ import { createInterface } from 'node:readline';
12
+
13
+ /** Strip context-window suffixes like "[1m]" so model ids compare cleanly. */
14
+ export function normalizeModelId(model) {
15
+ return String(model || 'unknown').replace(/\[[^\]]*\]$/, '');
16
+ }
17
+
18
+ /** Extract plain text from a Claude transcript message content field. */
19
+ function contentText(content) {
20
+ if (typeof content === 'string') return content;
21
+ if (!Array.isArray(content)) return '';
22
+ return content
23
+ .filter((b) => b && b.type === 'text' && typeof b.text === 'string')
24
+ .map((b) => b.text)
25
+ .join('\n');
26
+ }
27
+
28
+ /**
29
+ * Parse one session transcript into call records.
30
+ * @returns {Promise<Array<{model, timestamp, prompt_tokens, completion_tokens, depth, userText, assistantText, cwd}>>}
31
+ */
32
+ const MUTATING_TOOLS = new Set(['Edit', 'Write', 'NotebookEdit', 'Bash']);
33
+ const DELEGATION_TOOLS = new Set(['Task', 'Agent']);
34
+
35
+ export async function collectSessionRecords(filePath, { includeContent = true } = {}) {
36
+ const records = new Map();
37
+ let depth = 0;
38
+ let lastUserText = '';
39
+ let lastCwd = '';
40
+ let lastRecord = null;
41
+
42
+ const rl = createInterface({
43
+ input: createReadStream(filePath, { encoding: 'utf8' }),
44
+ crlfDelay: Infinity,
45
+ });
46
+
47
+ for await (const line of rl) {
48
+ let entry;
49
+ try {
50
+ entry = JSON.parse(line);
51
+ } catch {
52
+ continue;
53
+ }
54
+
55
+ const msg = entry.message;
56
+ if (typeof entry.cwd === 'string' && entry.cwd) lastCwd = entry.cwd;
57
+ if (entry.type === 'user' && msg) {
58
+ depth += 1;
59
+ const text = contentText(msg.content);
60
+ if (text) lastUserText = text;
61
+ // Tool errors arrive as tool_result blocks in the user entry that
62
+ // follows the assistant call — attribute them to that call's record.
63
+ if (lastRecord && Array.isArray(msg.content)) {
64
+ for (const b of msg.content) {
65
+ if (b && b.type === 'tool_result' && b.is_error) lastRecord.toolErrors += 1;
66
+ }
67
+ }
68
+ continue;
69
+ }
70
+ if (entry.type !== 'assistant' || !msg) continue;
71
+ depth += 1;
72
+
73
+ if (!msg.usage || !msg.id) continue;
74
+ // "<synthetic>" is Claude Code's placeholder for locally-generated
75
+ // entries (e.g. error stubs) — no real API call, nothing to record.
76
+ if (msg.model === '<synthetic>') continue;
77
+ const usage = msg.usage;
78
+ const reqId = entry.requestId || msg.id;
79
+
80
+ let mutatingToolCalls = 0;
81
+ let delegationCalls = 0;
82
+ if (Array.isArray(msg.content)) {
83
+ for (const b of msg.content) {
84
+ if (!b || b.type !== 'tool_use') continue;
85
+ if (MUTATING_TOOLS.has(b.name)) mutatingToolCalls += 1;
86
+ if (DELEGATION_TOOLS.has(b.name)) delegationCalls += 1;
87
+ }
88
+ }
89
+ const prev = records.get(reqId);
90
+
91
+ lastRecord = {
92
+ model: normalizeModelId(msg.model),
93
+ timestamp: entry.timestamp || null,
94
+ prompt_tokens:
95
+ (usage.input_tokens || 0) +
96
+ (usage.cache_creation_input_tokens || 0) +
97
+ (usage.cache_read_input_tokens || 0),
98
+ completion_tokens: usage.output_tokens || 0,
99
+ depth,
100
+ userText: includeContent ? lastUserText : '',
101
+ assistantText: includeContent ? contentText(msg.content) : '',
102
+ cwd: lastCwd,
103
+ // Entries of the same request accumulate tool blocks and errors.
104
+ mutatingToolCalls: (prev?.mutatingToolCalls || 0) + mutatingToolCalls,
105
+ delegationCalls: (prev?.delegationCalls || 0) + delegationCalls,
106
+ toolErrors: prev?.toolErrors || 0,
107
+ };
108
+ records.set(reqId, lastRecord);
109
+ }
110
+
111
+ return [...records.values()];
112
+ }