claude-token-saver 2.17.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/advice.js CHANGED
@@ -39,25 +39,33 @@ function toggleShortcut() {
39
39
  */
40
40
  export const ISSUE_MESSAGES = {
41
41
  LARGE_INPUT_PER_REQUEST: {
42
- title: 'Per-request input tokens are unusually large (1M context suspected)',
43
- titleKo: '요청당 입력 토큰이 비정상적으로 큼 (1M 컨텍스트 의심)',
42
+ title: 'Per-request input tokens are unusually large (context past 200k)',
43
+ titleKo: '요청당 입력 토큰이 비정상적으로 큼 (컨텍스트 200k 초과)',
44
+ // Since Opus 4.7 there is NO long-context price premium — 1M is standard
45
+ // rate and the default window on current models. The cost driver is the
46
+ // token volume itself: a 500k-token context re-reads ~500k tokens every
47
+ // turn (cache-read billed) and burns the subscription 5H/7D windows
48
+ // several times faster. That's what this warning is about.
44
49
  explain:
45
- 'Since Opus 4.7, 1M context is priced at the standard rate, and Max plans auto-promote ' +
46
- 'sessions to 1M. Once context goes past 200k, long-context pricing kicks in and cache reuse drops.',
50
+ 'Current models default to a 1M window with no long-context premium — but the token ' +
51
+ 'volume itself is the cost: every turn re-reads the whole context (billed as cache reads) ' +
52
+ 'and drains the 5H/7D rate-limit windows several times faster. Cache reuse also drops.',
47
53
  explainKo:
48
- 'Opus 4.7부터 1M 컨텍스트가 표준 요금이 되었고, Max 플랜은 세션을 자동으로 1M으로 승격합니다. ' +
49
- '200k를 넘는 순간 장기 컨텍스트 요금이 적용되며 캐시 재사용률도 떨어집니다.',
54
+ '현재 모델은 1M 윈도가 기본이고 장기 컨텍스트 프리미엄도 없습니다 — 하지만 토큰량 자체가 비용입니다. ' +
55
+ '매 턴 컨텍스트 전체를 다시 읽고(캐시 읽기 과금) 5H/7D 한도도 몇 배 빠르게 소모됩니다. ' +
56
+ '캐시 재사용률도 떨어집니다.',
50
57
  actions: () => [
51
58
  {
52
- label: 'Disable 1M context (env var)',
53
- labelKo: '1M 컨텍스트 끄기 (환경변수)',
54
- commands: disable1mEnvSnippet(),
55
- },
56
- {
57
- label: 'In-session toggle',
58
- labelKo: '세션 내 즉시 토글',
59
- commands: [`Press ${toggleShortcut()} to toggle on/off instantly`],
60
- commandsKo: [`${toggleShortcut()} 누르면 즉시 on/off 토글`],
59
+ label: 'Compact or clear when context grows (first lever)',
60
+ labelKo: '컨텍스트가 커지면 /compact 또는 /clear (1순위)',
61
+ commands: [
62
+ '/compact — summarize the session history in place',
63
+ '/clear — drop history entirely at a clean task boundary',
64
+ ],
65
+ commandsKo: [
66
+ '/compact — 세션 히스토리를 그 자리에서 요약',
67
+ '/clear — 작업 분기점에서 히스토리를 통째로 비우기',
68
+ ],
61
69
  },
62
70
  {
63
71
  label: 'Cap extended-thinking budget (⚠ check /effort first)',
@@ -90,16 +98,14 @@ export const ISSUE_MESSAGES = {
90
98
  ],
91
99
  },
92
100
  {
93
- label: 'Compact or clear when context grows',
94
- labelKo: '컨텍스트가 커지면 /compact 또는 /clear',
95
- commands: [
96
- '/compact — summarize the session history in place',
97
- '/clear — drop history entirely at a clean task boundary',
98
- ],
99
- commandsKo: [
100
- '/compact — 세션 히스토리를 그 자리에서 요약',
101
- '/clear — 작업 분기점에서 히스토리를 통째로 비우기',
102
- ],
101
+ label: 'Cap the window at 200k if you never need more (optional)',
102
+ labelKo: '더 큰 윈도가 필요 없으면 200k로 제한 (선택)',
103
+ commands: disable1mEnvSnippet().concat([
104
+ `Or press ${toggleShortcut()} to toggle in-session`,
105
+ ]),
106
+ commandsKo: disable1mEnvSnippet().concat([
107
+ `또는 ${toggleShortcut()}로 세션 내 즉시 토글`,
108
+ ]),
103
109
  },
104
110
  {
105
111
  label: '⚠ Known bug #31640',
@@ -337,11 +343,13 @@ export const ISSUE_MESSAGES = {
337
343
  title: 'Context window is approaching the limit',
338
344
  titleKo: '컨텍스트 창이 한계에 근접',
339
345
  explain:
340
- 'As context grows toward 200k (or 1M), each turn becomes more expensive and cache ' +
341
- 'reuse efficiency drops. Past 200k, long-context pricing applies.',
346
+ 'As context grows, each turn becomes more expensive (the whole context is re-read every ' +
347
+ 'turn) and cache reuse efficiency drops. There is no price premium past 200k on current ' +
348
+ 'models — the token volume itself is the cost, and it drains the 5H/7D caps faster.',
342
349
  explainKo:
343
- '컨텍스트가 200k(또는 1M)에 가까워질수록 매 턴 비용이 높아지고 캐시 재사용 효율이 떨어집니다. ' +
344
- '200k를 넘으면 장기 컨텍스트 요금이 적용됩니다.',
350
+ '컨텍스트가 커질수록 매 턴 비용이 높아지고(전체 컨텍스트를 매 턴 다시 읽음) 캐시 재사용 효율이 ' +
351
+ '떨어집니다. 현재 모델은 200k 초과 프리미엄이 없습니다 — 토큰량 자체가 비용이며 5H/7D 한도도 ' +
352
+ '더 빠르게 소모됩니다.',
345
353
  actions: () => [
346
354
  {
347
355
  label: '/compact at the next natural break',
@@ -435,8 +443,8 @@ export const ISSUE_MESSAGES = {
435
443
  */
436
444
  export const ISSUE_TIPS = {
437
445
  LARGE_INPUT_PER_REQUEST: {
438
- en: 'Check `/effort` — `xhigh` is the #1 cap killer; switch to `/effort medium` (or `low`); disable 1M context; `/compact` when context grows',
439
- ko: '`/effort` 확인 — `xhigh`가 캡 소진 1순위 원인, `medium`(또는 `low`)으로 복귀; 1M 컨텍스트 끄기; 컨텍스트 커지면 `/compact`',
446
+ en: '`/compact` or `/clear` — big contexts re-bill every turn and burn the 5H/7D caps; check `/effort` (`xhigh` is the #1 cap killer)',
447
+ ko: '`/compact` 또는 `/clear` — 큰 컨텍스트는 매 턴 재과금되고 5H/7D 한도를 태움; `/effort` 확인 (`xhigh`가 캡 소진 1순위)',
440
448
  },
441
449
  LOW_HIT_RATE: {
442
450
  en: 'Continue with `claude --continue`; keep CLAUDE.md trim — every line ships every turn',
@@ -463,8 +471,8 @@ export const ISSUE_TIPS = {
463
471
  ko: '지금 바로 아무 프롬프트나 보내 TTL 타이머 리셋; 또는 자리 비우기 전 `/compact` (Pro = 5분, Max = 1시간)',
464
472
  },
465
473
  CONTEXT_NEAR_LIMIT: {
466
- en: '`/compact` before hitting the limit; `/clear` + re-attach only needed files; disable 1M: `export CLAUDE_MODEL_CONTEXT=200000`',
467
- ko: '한계 도달 전 `/compact`; `/clear` 후 필요한 파일만 재첨부; 1M 끄기: `export CLAUDE_MODEL_CONTEXT=200000`',
474
+ en: '`/compact` before hitting the limit; `/clear` + re-attach only needed files',
475
+ ko: '한계 도달 전 `/compact`; `/clear` 후 필요한 파일만 재첨부',
468
476
  },
469
477
  };
470
478
 
@@ -474,6 +482,9 @@ export const ISSUE_TIPS = {
474
482
  * need a fallback so history.js can still surface the right tip.
475
483
  */
476
484
  export const CHIP_TO_CODES = {
485
+ '⚠ Ctx 200k+': ['LARGE_INPUT_PER_REQUEST'],
486
+ // Legacy chip name (pre-v2.18) — kept so `last`/`history` can still resolve
487
+ // codes from history files written by older versions.
477
488
  '⚠ 1M ON': ['LARGE_INPUT_PER_REQUEST'],
478
489
  '⚠ Cache miss': ['LOW_HIT_RATE'],
479
490
  '⚠ Input spike': ['LARGE_INPUT_PER_REQUEST'],
@@ -502,7 +513,11 @@ export const CAP_TIPS = {
502
513
  * shown to a global audience.
503
514
  */
504
515
  export function chipForIssues(issues, contextWindow) {
505
- if (contextWindow?.size === '1M') return '⚠ 1M ON';
516
+ // Fires on *actual usage* (a real request carried >210k input tokens), not
517
+ // on the model merely supporting 1M — current models are all 1M by default
518
+ // with no price premium, so "1M ON" stopped being a meaningful alarm. The
519
+ // meaningful signal is "your context genuinely exceeded 200k".
520
+ if (contextWindow?.size === '1M') return '⚠ Ctx 200k+';
506
521
  const codes = issues.map((i) => i.code);
507
522
  if (codes.includes('LARGE_INPUT_PER_REQUEST')) return '⚠ Input spike';
508
523
  if (codes.includes('BUCKET_5M_DOMINANT')) return '⚠ 5m TTL';
package/src/demo.js CHANGED
@@ -26,6 +26,7 @@ const SCENARIOS = [
26
26
  savings: 2123,
27
27
  elapsedSec: 30,
28
28
  contextSize: '200k',
29
+ ctxUsedPct: 34,
29
30
  spikeChip: null,
30
31
  caps: HEALTHY_CAPS,
31
32
  },
@@ -98,14 +99,15 @@ const SCENARIOS = [
98
99
  },
99
100
  {
100
101
  name: 'ctx-1m',
101
- label: '⚠ 1M context auto-on',
102
+ label: '⚠ Context past 200k',
102
103
  data: {
103
104
  hitRate: 0.78,
104
105
  pct1h: 0.92,
105
106
  savings: 1340,
106
107
  elapsedSec: 30,
107
108
  contextSize: '1M',
108
- spikeChip: '⚠ 1M ON',
109
+ ctxUsedPct: 28, // 28% of 1M ≈ 280k actually in context
110
+ spikeChip: '⚠ Ctx 200k+',
109
111
  caps: HEALTHY_CAPS,
110
112
  },
111
113
  },
@@ -271,7 +273,7 @@ export function buildTableDemoData(options = {}) {
271
273
  spikeReport: { spikes, baseline: { p95: 940_000 } },
272
274
  contextWindow: { size: '1M', maxContext: 280_000 },
273
275
  lastActivity: Date.now() - 60 * 1000,
274
- spikeChip: '⚠ 1M ON',
276
+ spikeChip: '⚠ Ctx 200k+',
275
277
  };
276
278
  }
277
279
 
@@ -301,7 +303,7 @@ export function buildScenarioData(scenarioName, options) {
301
303
  if (!scenario) return null;
302
304
  }
303
305
 
304
- const { hitRate, pct1h, pct5m, savings, elapsedSec, contextSize, spikeChip, caps } = scenario.data;
306
+ const { hitRate, pct1h, pct5m, savings, elapsedSec, contextSize, ctxUsedPct, spikeChip, caps } = scenario.data;
305
307
  return {
306
308
  summary: { hitRate },
307
309
  ttl: { pct1h, pct5m: pct5m ?? (1 - pct1h) },
@@ -314,6 +316,11 @@ export function buildScenarioData(scenarioName, options) {
314
316
  },
315
317
  lastActivity: Date.now() - elapsedSec * 1000,
316
318
  contextWindow: { size: contextSize },
319
+ // Live fill level (`📦 68%`) — scenarios that set ctxUsedPct exercise the
320
+ // stdin-driven segment; the rest fall back to the size-based chip.
321
+ ctxLive: ctxUsedPct != null
322
+ ? { usedPct: ctxUsedPct, size: contextSize === '1M' ? 1_000_000 : 200_000 }
323
+ : undefined,
317
324
  spikeChip,
318
325
  caps: buildCapsShape(caps),
319
326
  model: DEFAULT_MODEL,
@@ -214,7 +214,7 @@ export function formatNoSession({ caps = null, model = null, windowLabel = '' }
214
214
  * @param {string[]|null} [opts.segments] - whitelist of segments to render. Names: cap-warn, spike, harness, model, hit, ttl, saved, ctx, period, plus per-window keys (`five_hour`, `seven_day`, …). `5h`/`7d` are kept as aliases for back-compat. Null/undefined = all.
215
215
  */
216
216
  export function formatReport(data, { color = true, verbose = false, timer = true, mode = 'text', segments = null } = {}) {
217
- const { summary, ttl, cost, options, lastActivity, contextWindow, spikeChip, caps, model } = data;
217
+ const { summary, ttl, cost, options, lastActivity, contextWindow, ctxLive, spikeChip, caps, model } = data;
218
218
  const { hitRate } = summary;
219
219
 
220
220
  // Hit rate → color signal
@@ -321,15 +321,36 @@ export function formatReport(data, { color = true, verbose = false, timer = true
321
321
  }
322
322
  }
323
323
 
324
- // Context window chip (e.g. "📦 1M" or "📦 200k"). 1M gets a warning color
325
- // because it's the expensive default on Max plans after Opus 4.7.
324
+ // Context chip. Two data sources, best first:
325
+ //
326
+ // 1. Live fill level from Claude Code's stdin (`context_window.used_percentage`)
327
+ // — the current session's actual usage, refreshed every render. Rendered
328
+ // as `📦 68%` and colored by fill (green <70, yellow 70–89, red 90+),
329
+ // matching the cap-segment tone scale.
330
+ // 2. Fallback (table view / older Claude Code): transcript-inferred window
331
+ // size. Note the semantics: `size === '1M'` means a real request already
332
+ // carried >210k input tokens — actual heavy usage, not just the model
333
+ // supporting 1M. Current models are all 1M by default with no price
334
+ // premium, so this renders yellow ("your context is genuinely big"),
335
+ // not red ("expensive mode on") like it used to.
326
336
  let ctxSeg = null;
327
- if (contextWindow && contextWindow.size && contextWindow.size !== 'unknown') {
337
+ if (ctxLive && Number.isFinite(ctxLive.usedPct)) {
338
+ const pct = Math.max(0, Math.round(ctxLive.usedPct));
339
+ const tone = pct >= 90 ? RED : pct >= 70 ? YELLOW : GREEN;
340
+ const sizeLabel = ctxLive.size
341
+ ? (ctxLive.size >= 900_000 ? '1M' : `${Math.round(ctxLive.size / 1000)}k`)
342
+ : null;
343
+ const longLabel = sizeLabel ? `${pct}% of ${sizeLabel}` : `${pct}%`;
344
+ if (isIcon && verbose) {
345
+ ctxSeg = `${c(tone)}📦 Ctx ${longLabel}${c(RESET)}`;
346
+ } else if (isIcon) {
347
+ ctxSeg = `${c(tone)}📦 ${pct}%${c(RESET)}`;
348
+ } else {
349
+ ctxSeg = `${c(tone)}Ctx ${longLabel}${c(RESET)}`;
350
+ }
351
+ } else if (contextWindow && contextWindow.size && contextWindow.size !== 'unknown') {
328
352
  const label = contextWindow.size === '1M' ? '1M' : '200k';
329
- const ctxColor = contextWindow.size === '1M' ? RED : GREEN;
330
- // `Ctx` (not the full word `Context`) across every mode — the icon already
331
- // tells the eye what the chip is, and the short form fits the same cadence
332
- // as `Hit`/`Saved` peers when we eventually shorten those too.
353
+ const ctxColor = contextWindow.size === '1M' ? YELLOW : GREEN;
333
354
  if (isIcon && verbose) {
334
355
  ctxSeg = `${c(ctxColor)}📦 Ctx ${label}${c(RESET)}`;
335
356
  } else if (isIcon) {
@@ -86,7 +86,7 @@ function renderSpikeSection(spikes, contextWindow) {
86
86
  lines.push(r(` ${'─'.repeat(50)}`));
87
87
  if (contextWindow && contextWindow.size === '1M') {
88
88
  lines.push(
89
- r(` Context mode: 1M (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`),
89
+ r(` Context usage exceeded 200k (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`),
90
90
  );
91
91
  lines.push('');
92
92
  }
@@ -197,7 +197,7 @@ export function formatReport({ summary: sum, trend, ttl, anomalies, cost, option
197
197
  if (contextWindow && contextWindow.size !== 'unknown') {
198
198
  const note =
199
199
  contextWindow.size === '1M'
200
- ? '⚠ 1M context active (Opus 4.7+ Max default). Disable with CLAUDE_CODE_DISABLE_1M_CONTEXT=1'
200
+ ? '⚠ Context exceeded 200k in recent requests — big contexts re-bill every turn and drain the 5H/7D caps. Use /compact or /clear.'
201
201
  : '✓ 200k context (standard)';
202
202
  lines.push(` Context window: ${contextWindow.size} ${note}`);
203
203
  lines.push(` (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`);
@@ -0,0 +1,211 @@
1
+ /**
2
+ * frugon export — convert Claude Code session transcripts into the
3
+ * OpenAI-compatible JSONL log format frugon analyzes.
4
+ * (frugon: local LLM cost analyzer — github.com/Rodiun/frugon)
5
+ *
6
+ * One output line per API call:
7
+ * {
8
+ * "model": "claude-opus-4-8",
9
+ * "timestamp": "2026-07-12T02:11:05.123Z",
10
+ * "usage": { "prompt_tokens": 1234, "completion_tokens": 56 },
11
+ * "request": { "messages": [ ...stubs..., { "role": "user", "content": "<last user prompt>" } ] },
12
+ * "response": { "choices": [ { "message": { "role": "assistant", "content": "<reply>" } } ] }
13
+ * }
14
+ *
15
+ * Design notes (kept in sync with frugon 0.2.x internals):
16
+ * - frugon prefers the usage block for token counts, so message content is
17
+ * never re-tokenized — stubs with empty content are safe.
18
+ * - frugon's easy/hard difficulty score reads prompt_tokens, completion_tokens
19
+ * and conversation depth (len(messages) - 1, saturating at 6 turns). We emit
20
+ * up to MAX_STUB_MESSAGES role-alternating stubs so depth survives the
21
+ * export without duplicating the whole conversation into every record.
22
+ * - frugon has no notion of prompt caching: every prompt token is priced at
23
+ * the base input rate. Claude Code sessions are cache-read heavy (~90%+),
24
+ * so raw totals would overstate spend ~10x. By default we fold Anthropic's
25
+ * cache multipliers (5m write 1.25x, 1h write 2x, read 0.1x) into an
26
+ * "effective" prompt_tokens so frugon's dollar figures match reality.
27
+ * Pass cacheWeighted: false for raw physical token counts.
28
+ */
29
+
30
+ import { createReadStream, createWriteStream } from 'node:fs';
31
+ import { createInterface } from 'node:readline';
32
+ import { discoverSessionFiles } from './parser.js';
33
+
34
+ // Depth cap: frugon's turn signal saturates at 6 turns (len(messages)-1 >= 6),
35
+ // so 7 messages carry the maximum-depth signal at minimum size.
36
+ const MAX_STUB_MESSAGES = 7;
37
+
38
+ // Anthropic cache multipliers relative to the base input rate — uniform
39
+ // across model tiers (see src/cost.js PRICING).
40
+ const CACHE_WEIGHTS = { write5m: 1.25, write1h: 2, read: 0.1 };
41
+
42
+ /** Strip context-window suffixes like "[1m]" so frugon's pricing table matches. */
43
+ export function normalizeModelId(model) {
44
+ return String(model || 'unknown').replace(/\[[^\]]*\]$/, '');
45
+ }
46
+
47
+ /**
48
+ * Effective prompt tokens: what the call *costs* expressed in base-rate
49
+ * input tokens, so frugon (which prices all prompt tokens at the input rate)
50
+ * reproduces the real cache-discounted spend.
51
+ */
52
+ export function effectivePromptTokens(r) {
53
+ const tracked = (r.ephemeral5mTokens || 0) + (r.ephemeral1hTokens || 0);
54
+ const untracked = Math.max(0, (r.cacheCreationTokens || 0) - tracked);
55
+ return Math.round(
56
+ (r.inputTokens || 0) +
57
+ ((r.ephemeral5mTokens || 0) + untracked) * CACHE_WEIGHTS.write5m +
58
+ (r.ephemeral1hTokens || 0) * CACHE_WEIGHTS.write1h +
59
+ (r.cacheReadTokens || 0) * CACHE_WEIGHTS.read,
60
+ );
61
+ }
62
+
63
+ /** Raw physical prompt tokens (input + cache writes + cache reads). */
64
+ export function rawPromptTokens(r) {
65
+ return (r.inputTokens || 0) + (r.cacheCreationTokens || 0) + (r.cacheReadTokens || 0);
66
+ }
67
+
68
+ /** Extract plain text from a Claude transcript message content field. */
69
+ function contentText(content) {
70
+ if (typeof content === 'string') return content;
71
+ if (!Array.isArray(content)) return '';
72
+ return content
73
+ .filter((b) => b && b.type === 'text' && typeof b.text === 'string')
74
+ .map((b) => b.text)
75
+ .join('\n');
76
+ }
77
+
78
+ /**
79
+ * Parse one session transcript into frugon records.
80
+ * Deduplicates by requestId (last-write-wins, matching parser.js) while
81
+ * tracking the conversation depth and last user prompt at each call.
82
+ */
83
+ export async function collectSessionRecords(filePath, { cacheWeighted = true, includeContent = true } = {}) {
84
+ const records = new Map();
85
+ let depth = 0;
86
+ let lastUserText = '';
87
+ let lastCwd = '';
88
+
89
+ const rl = createInterface({
90
+ input: createReadStream(filePath, { encoding: 'utf8' }),
91
+ crlfDelay: Infinity,
92
+ });
93
+
94
+ for await (const line of rl) {
95
+ let entry;
96
+ try {
97
+ entry = JSON.parse(line);
98
+ } catch {
99
+ continue;
100
+ }
101
+
102
+ const msg = entry.message;
103
+ if (typeof entry.cwd === 'string' && entry.cwd) lastCwd = entry.cwd;
104
+ if (entry.type === 'user' && msg) {
105
+ depth += 1;
106
+ const text = contentText(msg.content);
107
+ if (text) lastUserText = text;
108
+ continue;
109
+ }
110
+ if (entry.type !== 'assistant' || !msg) continue;
111
+ depth += 1;
112
+
113
+ if (!msg.usage || !msg.id) continue;
114
+ // "<synthetic>" is Claude Code's placeholder for locally-generated
115
+ // entries (e.g. error stubs) — no real API call, nothing to price.
116
+ if (msg.model === '<synthetic>') continue;
117
+ const usage = msg.usage;
118
+ const reqId = entry.requestId || msg.id;
119
+ const r = {
120
+ inputTokens: usage.input_tokens || 0,
121
+ cacheCreationTokens: usage.cache_creation_input_tokens || 0,
122
+ cacheReadTokens: usage.cache_read_input_tokens || 0,
123
+ ephemeral5mTokens: usage.cache_creation?.ephemeral_5m_input_tokens || 0,
124
+ ephemeral1hTokens: usage.cache_creation?.ephemeral_1h_input_tokens || 0,
125
+ outputTokens: usage.output_tokens || 0,
126
+ };
127
+
128
+ records.set(reqId, {
129
+ model: normalizeModelId(msg.model),
130
+ timestamp: entry.timestamp || null,
131
+ prompt_tokens: cacheWeighted ? effectivePromptTokens(r) : rawPromptTokens(r),
132
+ completion_tokens: r.outputTokens,
133
+ depth,
134
+ userText: includeContent ? lastUserText : '',
135
+ assistantText: includeContent ? contentText(msg.content) : '',
136
+ cwd: lastCwd,
137
+ });
138
+ }
139
+
140
+ return [...records.values()];
141
+ }
142
+
143
+ /** Build the frugon JSONL object for one collected record. */
144
+ export function toFrugonRecord(rec) {
145
+ const msgCount = Math.max(1, Math.min(rec.depth, MAX_STUB_MESSAGES));
146
+ const messages = [];
147
+ for (let i = 0; i < msgCount - 1; i++) {
148
+ messages.push({ role: i % 2 === 0 ? 'user' : 'assistant', content: '' });
149
+ }
150
+ messages.push({ role: 'user', content: rec.userText || '' });
151
+
152
+ const out = {
153
+ model: rec.model,
154
+ request: { messages },
155
+ response: {
156
+ choices: [{ message: { role: 'assistant', content: rec.assistantText || '' } }],
157
+ },
158
+ usage: {
159
+ prompt_tokens: rec.prompt_tokens,
160
+ completion_tokens: rec.completion_tokens,
161
+ },
162
+ };
163
+ if (rec.timestamp) out.timestamp = rec.timestamp;
164
+ return out;
165
+ }
166
+
167
+ /**
168
+ * Export Claude Code transcripts to a frugon-compatible JSONL file.
169
+ *
170
+ * @param {object} options
171
+ * days lookback window (default 30)
172
+ * projectFilter substring match on the project dir name
173
+ * outPath output JSONL path
174
+ * cacheWeighted fold cache pricing into prompt_tokens (default true)
175
+ * includeContent include user prompt / assistant reply text (default true)
176
+ * @returns {Promise<{records:number, sessions:number, models:Object, outPath:string}>}
177
+ */
178
+ export async function exportFrugonLogs({
179
+ days = 30,
180
+ projectFilter,
181
+ outPath,
182
+ cacheWeighted = true,
183
+ includeContent = true,
184
+ } = {}) {
185
+ const files = await discoverSessionFiles({ days, projectFilter });
186
+ const models = {};
187
+ let recordCount = 0;
188
+ let sessionCount = 0;
189
+
190
+ const stream = createWriteStream(outPath, { encoding: 'utf8' });
191
+ for (const f of files) {
192
+ let recs;
193
+ try {
194
+ recs = await collectSessionRecords(f.path, { cacheWeighted, includeContent });
195
+ } catch {
196
+ continue;
197
+ }
198
+ if (recs.length === 0) continue;
199
+ sessionCount += 1;
200
+ for (const rec of recs) {
201
+ stream.write(JSON.stringify(toFrugonRecord(rec)) + '\n');
202
+ models[rec.model] = (models[rec.model] || 0) + 1;
203
+ recordCount += 1;
204
+ }
205
+ }
206
+ await new Promise((resolve, reject) => {
207
+ stream.end((err) => (err ? reject(err) : resolve()));
208
+ });
209
+
210
+ return { records: recordCount, sessions: sessionCount, models, outPath };
211
+ }
package/src/harness.js CHANGED
@@ -19,6 +19,7 @@ import {
19
19
  harnessRatchetMdInitial,
20
20
  appendRatchetRule,
21
21
  } from './harness-templates.js';
22
+ import { routeWarningForStatusline } from './route-scan.js';
22
23
 
23
24
  const require = createRequire(import.meta.url);
24
25
  function readHarnessState() {
@@ -276,6 +277,82 @@ export function harnessPromote(ruleText, { root = findProjectRoot(), scope = 'pr
276
277
  return { path: rmPath, root, scope };
277
278
  }
278
279
 
280
+ /**
281
+ * harness pull — copy the user's GLOBAL ratchet rules (~/.claude/ratchet.md)
282
+ * into the current project's .claude/ratchet.md, on demand. Opt-in by design:
283
+ * install/init never auto-injects rules; this is the explicit "땡겨오기" verb.
284
+ *
285
+ * Rules are deduped by text (ignoring the leading YYYY-MM-DD stamp) so
286
+ * repeated pulls are idempotent. With `includeBlock`, the global CLAUDE.md
287
+ * harness block (including any user customizations) is also copied into the
288
+ * project CLAUDE.md — replacing the project's block if one exists.
289
+ *
290
+ * Returns { root, added, skippedRules, wrote, skipped }.
291
+ */
292
+ export function harnessPull({ root = findProjectRoot(), includeBlock = false } = {}) {
293
+ const result = { root, added: [], skippedRules: 0, wrote: [], skipped: [] };
294
+ const stripDate = (t) => t.replace(/^\d{4}-\d{2}-\d{2}:\s*/, '').trim();
295
+
296
+ // 1) Ratchet rules: global → project, dedup by rule text.
297
+ const globalRules = harnessListRules({ scope: 'global' }).rules;
298
+ const projPath = ratchetMdPath(root);
299
+ let content = existsSync(projPath)
300
+ ? readFileSync(projPath, 'utf8')
301
+ : harnessRatchetMdInitial();
302
+ const have = new Set(
303
+ harnessListRules({ root, scope: 'project' }).rules.map((r) => stripDate(r.text)),
304
+ );
305
+ for (const g of globalRules) {
306
+ const key = stripDate(g.text);
307
+ if (have.has(key)) {
308
+ result.skippedRules += 1;
309
+ continue;
310
+ }
311
+ content = appendRatchetRule(content, key);
312
+ have.add(key);
313
+ result.added.push(key);
314
+ }
315
+ if (result.added.length) {
316
+ mkdirSync(dirname(projPath), { recursive: true });
317
+ writeFileSync(projPath, content);
318
+ result.wrote.push(projPath);
319
+ }
320
+
321
+ // 2) Harness block (opt-in): copy the global block as-is so user edits to
322
+ // the global sections travel with it.
323
+ if (includeBlock) {
324
+ const gPath = globalClaudeMdPath();
325
+ const blockRe = new RegExp(
326
+ `${escapeRe(HARNESS_BLOCK_BEGIN)}[\\s\\S]*?${escapeRe(HARNESS_BLOCK_END)}\\n?`,
327
+ 'm',
328
+ );
329
+ const gContent = existsSync(gPath) ? readFileSync(gPath, 'utf8') : '';
330
+ const m = gContent.match(blockRe);
331
+ if (!m) {
332
+ result.skipped.push(`${gPath} (no global harness block to pull)`);
333
+ } else {
334
+ const block = m[0];
335
+ const pPath = claudeMdPath(root);
336
+ if (existsSync(pPath)) {
337
+ const pc = readFileSync(pPath, 'utf8');
338
+ if (pc.includes(HARNESS_BLOCK_BEGIN)) {
339
+ writeFileSync(pPath, pc.replace(blockRe, block));
340
+ result.wrote.push(`${pPath} (harness block replaced with global copy)`);
341
+ } else {
342
+ const sep = pc.endsWith('\n') ? '\n' : '\n\n';
343
+ writeFileSync(pPath, pc + sep + block);
344
+ result.wrote.push(`${pPath} (harness block appended from global)`);
345
+ }
346
+ } else {
347
+ writeFileSync(pPath, block);
348
+ result.wrote.push(pPath);
349
+ }
350
+ }
351
+ }
352
+
353
+ return result;
354
+ }
355
+
279
356
  /**
280
357
  * harness list — return numbered ratchet rules from .claude/ratchet.md.
281
358
  * Numbering is 1-based and matches `harness rm <N>`.
@@ -362,6 +439,14 @@ export function harnessStatusForStatusline(cfg, { root } = {}) {
362
439
  else if (state.pevSkip) warning = 'PEV-skip';
363
440
  }
364
441
  }
442
+ // Lowest precedence: route-scan delegation candidate (`route? R<N>`).
443
+ // Session-quality warnings above always win — routing is an optimization
444
+ // nudge, not a correctness signal. Cheap: one small cached-JSON read.
445
+ if (!warning) {
446
+ try {
447
+ warning = routeWarningForStatusline(projectRoot);
448
+ } catch { /* scan cache unreadable — stay silent */ }
449
+ }
365
450
  return { ...status, warning };
366
451
  }
367
452
 
package/src/history.js CHANGED
@@ -69,7 +69,8 @@ function saveState(state) {
69
69
  function chipKo(chip) {
70
70
  if (!chip) return chip;
71
71
  const map = {
72
- '⚠ 1M ON': '⚠ 1M 컨텍스트 활성',
72
+ '⚠ Ctx 200k+': '⚠ 컨텍스트 200k 초과',
73
+ '⚠ 1M ON': '⚠ 1M 컨텍스트 활성', // legacy (pre-v2.18)
73
74
  '⚠ Cache miss': '⚠ 캐시 미스',
74
75
  '⚠ Rebuild churn': '⚠ 캐시 재빌드 빈발',
75
76
  '⚠ Input spike': '⚠ 입력 급증',
@@ -92,6 +93,9 @@ function chipKo(chip) {
92
93
  */
93
94
  function detailKo(detail) {
94
95
  if (!detail) return detail;
96
+ const m0 = detail.match(/^Single-request context exceeded 200k \(max (\d+)k tokens\)$/);
97
+ if (m0) return `단일 요청 컨텍스트 200k 초과 (최대 ${m0[1]}k 토큰)`;
98
+ // legacy detail shape (pre-v2.18)
95
99
  const m1 = detail.match(/^Context auto-promoted to 1M \(max single-request (\d+)k tokens\)$/);
96
100
  if (m1) return `1M 컨텍스트 자동 활성 (단일 요청 최대 ${m1[1]}k 토큰)`;
97
101
  const m2 = detail.match(/^session ([^:]+): (.+)$/);
package/src/installer.js CHANGED
@@ -202,10 +202,47 @@ export function installStatusline({ force = false } = {}) {
202
202
  return { path: file, action: 'updated', reason: 'replaced previous statusLine' };
203
203
  }
204
204
 
205
+ // Registers the SessionStart hook that surfaces route-scan delegation
206
+ // candidates as session context (startup + /clear). Idempotent: skips when a
207
+ // claude-token-saver route-scan hook is already present; never touches other
208
+ // hooks the user configured.
209
+ const ROUTE_SCAN_HOOK_COMMAND = 'claude-token-saver route-scan --hook';
210
+
211
+ export function installSessionStartHook() {
212
+ const dir = claudeUserDir();
213
+ const file = join(dir, 'settings.json');
214
+ mkdirSync(dir, { recursive: true });
215
+
216
+ let settings = {};
217
+ if (existsSync(file)) {
218
+ try {
219
+ settings = JSON.parse(readFileSync(file, 'utf8'));
220
+ } catch (e) {
221
+ return { path: file, action: 'skipped', reason: `unreadable JSON (${e.message})` };
222
+ }
223
+ }
224
+
225
+ settings.hooks = settings.hooks || {};
226
+ const list = Array.isArray(settings.hooks.SessionStart) ? settings.hooks.SessionStart : [];
227
+ const already = list.some((m) =>
228
+ Array.isArray(m?.hooks) && m.hooks.some((h) => typeof h?.command === 'string' && h.command.includes('route-scan --hook')),
229
+ );
230
+ if (already) return { path: file, action: 'exists' };
231
+
232
+ list.push({
233
+ matcher: 'startup|clear',
234
+ hooks: [{ type: 'command', command: ROUTE_SCAN_HOOK_COMMAND, timeout: 10 }],
235
+ });
236
+ settings.hooks.SessionStart = list;
237
+ writeFileSync(file, JSON.stringify(settings, null, 2) + '\n');
238
+ return { path: file, action: 'created' };
239
+ }
240
+
205
241
  export function installAll({ force = false } = {}) {
206
242
  return {
207
243
  skill: installSkill({ force }),
208
244
  statusline: installStatusline({ force }),
245
+ sessionStartHook: installSessionStartHook(),
209
246
  legacy: removeLegacyCommand(),
210
247
  };
211
248
  }