claude-token-saver 2.16.0 → 2.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/advice.js CHANGED
@@ -39,25 +39,33 @@ function toggleShortcut() {
39
39
  */
40
40
  export const ISSUE_MESSAGES = {
41
41
  LARGE_INPUT_PER_REQUEST: {
42
- title: 'Per-request input tokens are unusually large (1M context suspected)',
43
- titleKo: '요청당 입력 토큰이 비정상적으로 큼 (1M 컨텍스트 의심)',
42
+ title: 'Per-request input tokens are unusually large (context past 200k)',
43
+ titleKo: '요청당 입력 토큰이 비정상적으로 큼 (컨텍스트 200k 초과)',
44
+ // Since Opus 4.7 there is NO long-context price premium — 1M is standard
45
+ // rate and the default window on current models. The cost driver is the
46
+ // token volume itself: a 500k-token context re-reads ~500k tokens every
47
+ // turn (cache-read billed) and burns the subscription 5H/7D windows
48
+ // several times faster. That's what this warning is about.
44
49
  explain:
45
- 'Since Opus 4.7, 1M context is priced at the standard rate, and Max plans auto-promote ' +
46
- 'sessions to 1M. Once context goes past 200k, long-context pricing kicks in and cache reuse drops.',
50
+ 'Current models default to a 1M window with no long-context premium — but the token ' +
51
+ 'volume itself is the cost: every turn re-reads the whole context (billed as cache reads) ' +
52
+ 'and drains the 5H/7D rate-limit windows several times faster. Cache reuse also drops.',
47
53
  explainKo:
48
- 'Opus 4.7부터 1M 컨텍스트가 표준 요금이 되었고, Max 플랜은 세션을 자동으로 1M으로 승격합니다. ' +
49
- '200k를 넘는 순간 장기 컨텍스트 요금이 적용되며 캐시 재사용률도 떨어집니다.',
54
+ '현재 모델은 1M 윈도가 기본이고 장기 컨텍스트 프리미엄도 없습니다 — 하지만 토큰량 자체가 비용입니다. ' +
55
+ '매 턴 컨텍스트 전체를 다시 읽고(캐시 읽기 과금) 5H/7D 한도도 몇 배 빠르게 소모됩니다. ' +
56
+ '캐시 재사용률도 떨어집니다.',
50
57
  actions: () => [
51
58
  {
52
- label: 'Disable 1M context (env var)',
53
- labelKo: '1M 컨텍스트 끄기 (환경변수)',
54
- commands: disable1mEnvSnippet(),
55
- },
56
- {
57
- label: 'In-session toggle',
58
- labelKo: '세션 내 즉시 토글',
59
- commands: [`Press ${toggleShortcut()} to toggle on/off instantly`],
60
- commandsKo: [`${toggleShortcut()} 누르면 즉시 on/off 토글`],
59
+ label: 'Compact or clear when context grows (first lever)',
60
+ labelKo: '컨텍스트가 커지면 /compact 또는 /clear (1순위)',
61
+ commands: [
62
+ '/compact — summarize the session history in place',
63
+ '/clear — drop history entirely at a clean task boundary',
64
+ ],
65
+ commandsKo: [
66
+ '/compact — 세션 히스토리를 그 자리에서 요약',
67
+ '/clear — 작업 분기점에서 히스토리를 통째로 비우기',
68
+ ],
61
69
  },
62
70
  {
63
71
  label: 'Cap extended-thinking budget (⚠ check /effort first)',
@@ -90,16 +98,14 @@ export const ISSUE_MESSAGES = {
90
98
  ],
91
99
  },
92
100
  {
93
- label: 'Compact or clear when context grows',
94
- labelKo: '컨텍스트가 커지면 /compact 또는 /clear',
95
- commands: [
96
- '/compact — summarize the session history in place',
97
- '/clear — drop history entirely at a clean task boundary',
98
- ],
99
- commandsKo: [
100
- '/compact — 세션 히스토리를 그 자리에서 요약',
101
- '/clear — 작업 분기점에서 히스토리를 통째로 비우기',
102
- ],
101
+ label: 'Cap the window at 200k if you never need more (optional)',
102
+ labelKo: '더 큰 윈도가 필요 없으면 200k로 제한 (선택)',
103
+ commands: disable1mEnvSnippet().concat([
104
+ `Or press ${toggleShortcut()} to toggle in-session`,
105
+ ]),
106
+ commandsKo: disable1mEnvSnippet().concat([
107
+ `또는 ${toggleShortcut()}로 세션 내 즉시 토글`,
108
+ ]),
103
109
  },
104
110
  {
105
111
  label: '⚠ Known bug #31640',
@@ -337,11 +343,13 @@ export const ISSUE_MESSAGES = {
337
343
  title: 'Context window is approaching the limit',
338
344
  titleKo: '컨텍스트 창이 한계에 근접',
339
345
  explain:
340
- 'As context grows toward 200k (or 1M), each turn becomes more expensive and cache ' +
341
- 'reuse efficiency drops. Past 200k, long-context pricing applies.',
346
+ 'As context grows, each turn becomes more expensive (the whole context is re-read every ' +
347
+ 'turn) and cache reuse efficiency drops. There is no price premium past 200k on current ' +
348
+ 'models — the token volume itself is the cost, and it drains the 5H/7D caps faster.',
342
349
  explainKo:
343
- '컨텍스트가 200k(또는 1M)에 가까워질수록 매 턴 비용이 높아지고 캐시 재사용 효율이 떨어집니다. ' +
344
- '200k를 넘으면 장기 컨텍스트 요금이 적용됩니다.',
350
+ '컨텍스트가 커질수록 매 턴 비용이 높아지고(전체 컨텍스트를 매 턴 다시 읽음) 캐시 재사용 효율이 ' +
351
+ '떨어집니다. 현재 모델은 200k 초과 프리미엄이 없습니다 — 토큰량 자체가 비용이며 5H/7D 한도도 ' +
352
+ '더 빠르게 소모됩니다.',
345
353
  actions: () => [
346
354
  {
347
355
  label: '/compact at the next natural break',
@@ -435,8 +443,8 @@ export const ISSUE_MESSAGES = {
435
443
  */
436
444
  export const ISSUE_TIPS = {
437
445
  LARGE_INPUT_PER_REQUEST: {
438
- en: 'Check `/effort` — `xhigh` is the #1 cap killer; switch to `/effort medium` (or `low`); disable 1M context; `/compact` when context grows',
439
- ko: '`/effort` 확인 — `xhigh`가 캡 소진 1순위 원인, `medium`(또는 `low`)으로 복귀; 1M 컨텍스트 끄기; 컨텍스트 커지면 `/compact`',
446
+ en: '`/compact` or `/clear` — big contexts re-bill every turn and burn the 5H/7D caps; check `/effort` (`xhigh` is the #1 cap killer)',
447
+ ko: '`/compact` 또는 `/clear` — 큰 컨텍스트는 매 턴 재과금되고 5H/7D 한도를 태움; `/effort` 확인 (`xhigh`가 캡 소진 1순위)',
440
448
  },
441
449
  LOW_HIT_RATE: {
442
450
  en: 'Continue with `claude --continue`; keep CLAUDE.md trim — every line ships every turn',
@@ -463,8 +471,8 @@ export const ISSUE_TIPS = {
463
471
  ko: '지금 바로 아무 프롬프트나 보내 TTL 타이머 리셋; 또는 자리 비우기 전 `/compact` (Pro = 5분, Max = 1시간)',
464
472
  },
465
473
  CONTEXT_NEAR_LIMIT: {
466
- en: '`/compact` before hitting the limit; `/clear` + re-attach only needed files; disable 1M: `export CLAUDE_MODEL_CONTEXT=200000`',
467
- ko: '한계 도달 전 `/compact`; `/clear` 후 필요한 파일만 재첨부; 1M 끄기: `export CLAUDE_MODEL_CONTEXT=200000`',
474
+ en: '`/compact` before hitting the limit; `/clear` + re-attach only needed files',
475
+ ko: '한계 도달 전 `/compact`; `/clear` 후 필요한 파일만 재첨부',
468
476
  },
469
477
  };
470
478
 
@@ -474,6 +482,9 @@ export const ISSUE_TIPS = {
474
482
  * need a fallback so history.js can still surface the right tip.
475
483
  */
476
484
  export const CHIP_TO_CODES = {
485
+ '⚠ Ctx 200k+': ['LARGE_INPUT_PER_REQUEST'],
486
+ // Legacy chip name (pre-v2.18) — kept so `last`/`history` can still resolve
487
+ // codes from history files written by older versions.
477
488
  '⚠ 1M ON': ['LARGE_INPUT_PER_REQUEST'],
478
489
  '⚠ Cache miss': ['LOW_HIT_RATE'],
479
490
  '⚠ Input spike': ['LARGE_INPUT_PER_REQUEST'],
@@ -502,7 +513,11 @@ export const CAP_TIPS = {
502
513
  * shown to a global audience.
503
514
  */
504
515
  export function chipForIssues(issues, contextWindow) {
505
- if (contextWindow?.size === '1M') return '⚠ 1M ON';
516
+ // Fires on *actual usage* (a real request carried >210k input tokens), not
517
+ // on the model merely supporting 1M — current models are all 1M by default
518
+ // with no price premium, so "1M ON" stopped being a meaningful alarm. The
519
+ // meaningful signal is "your context genuinely exceeded 200k".
520
+ if (contextWindow?.size === '1M') return '⚠ Ctx 200k+';
506
521
  const codes = issues.map((i) => i.code);
507
522
  if (codes.includes('LARGE_INPUT_PER_REQUEST')) return '⚠ Input spike';
508
523
  if (codes.includes('BUCKET_5M_DOMINANT')) return '⚠ 5m TTL';
package/src/cost.js CHANGED
@@ -8,7 +8,17 @@
8
8
  */
9
9
 
10
10
  const PRICING = {
11
- // Opus 4.5+ (new pricing tier — includes 4.5, 4.6, 4.7, and future)
11
+ // Fable 5 / Mythos 5 — premium tier above Opus ($10/$50). Cache write
12
+ // rates follow the standard multipliers (1.25x input for 5m, 2x for 1h),
13
+ // cache read is 0.1x input.
14
+ 'claude-fable-5': {
15
+ input: 10.0,
16
+ cacheWrite5m: 12.5,
17
+ cacheWrite1h: 20.0,
18
+ cacheRead: 1.0,
19
+ output: 50.0,
20
+ },
21
+ // Opus 4.5+ (new pricing tier — includes 4.5, 4.6, 4.7, 4.8, and future)
12
22
  'claude-opus-new': {
13
23
  input: 5.0,
14
24
  cacheWrite5m: 6.25,
@@ -66,6 +76,11 @@ function detectPricingTier(model) {
66
76
  if (!model) return 'claude-sonnet';
67
77
  const m = model.toLowerCase();
68
78
 
79
+ // Fable 5 / Mythos 5 — must be checked before the generic fallback:
80
+ // without this, 'claude-fable-5' fell through to the Sonnet tier and
81
+ // under-estimated costs ~3x ($3/$15 vs the real $10/$50).
82
+ if (m.includes('fable') || m.includes('mythos')) return 'claude-fable-5';
83
+
69
84
  if (m.includes('opus')) {
70
85
  // Opus 4.5, 4.6, 4.7, and future 5+ use the new reduced pricing.
71
86
  if (/opus[-_.]?4[-_.]?[5-9]\b/.test(m)) return 'claude-opus-new';
package/src/demo.js CHANGED
@@ -26,6 +26,7 @@ const SCENARIOS = [
26
26
  savings: 2123,
27
27
  elapsedSec: 30,
28
28
  contextSize: '200k',
29
+ ctxUsedPct: 34,
29
30
  spikeChip: null,
30
31
  caps: HEALTHY_CAPS,
31
32
  },
@@ -98,14 +99,15 @@ const SCENARIOS = [
98
99
  },
99
100
  {
100
101
  name: 'ctx-1m',
101
- label: '⚠ 1M context auto-on',
102
+ label: '⚠ Context past 200k',
102
103
  data: {
103
104
  hitRate: 0.78,
104
105
  pct1h: 0.92,
105
106
  savings: 1340,
106
107
  elapsedSec: 30,
107
108
  contextSize: '1M',
108
- spikeChip: '⚠ 1M ON',
109
+ ctxUsedPct: 28, // 28% of 1M ≈ 280k actually in context
110
+ spikeChip: '⚠ Ctx 200k+',
109
111
  caps: HEALTHY_CAPS,
110
112
  },
111
113
  },
@@ -271,7 +273,7 @@ export function buildTableDemoData(options = {}) {
271
273
  spikeReport: { spikes, baseline: { p95: 940_000 } },
272
274
  contextWindow: { size: '1M', maxContext: 280_000 },
273
275
  lastActivity: Date.now() - 60 * 1000,
274
- spikeChip: '⚠ 1M ON',
276
+ spikeChip: '⚠ Ctx 200k+',
275
277
  };
276
278
  }
277
279
 
@@ -301,7 +303,7 @@ export function buildScenarioData(scenarioName, options) {
301
303
  if (!scenario) return null;
302
304
  }
303
305
 
304
- const { hitRate, pct1h, pct5m, savings, elapsedSec, contextSize, spikeChip, caps } = scenario.data;
306
+ const { hitRate, pct1h, pct5m, savings, elapsedSec, contextSize, ctxUsedPct, spikeChip, caps } = scenario.data;
305
307
  return {
306
308
  summary: { hitRate },
307
309
  ttl: { pct1h, pct5m: pct5m ?? (1 - pct1h) },
@@ -314,6 +316,11 @@ export function buildScenarioData(scenarioName, options) {
314
316
  },
315
317
  lastActivity: Date.now() - elapsedSec * 1000,
316
318
  contextWindow: { size: contextSize },
319
+ // Live fill level (`📦 68%`) — scenarios that set ctxUsedPct exercise the
320
+ // stdin-driven segment; the rest fall back to the size-based chip.
321
+ ctxLive: ctxUsedPct != null
322
+ ? { usedPct: ctxUsedPct, size: contextSize === '1M' ? 1_000_000 : 200_000 }
323
+ : undefined,
317
324
  spikeChip,
318
325
  caps: buildCapsShape(caps),
319
326
  model: DEFAULT_MODEL,
@@ -214,7 +214,7 @@ export function formatNoSession({ caps = null, model = null, windowLabel = '' }
214
214
  * @param {string[]|null} [opts.segments] - whitelist of segments to render. Names: cap-warn, spike, harness, model, hit, ttl, saved, ctx, period, plus per-window keys (`five_hour`, `seven_day`, …). `5h`/`7d` are kept as aliases for back-compat. Null/undefined = all.
215
215
  */
216
216
  export function formatReport(data, { color = true, verbose = false, timer = true, mode = 'text', segments = null } = {}) {
217
- const { summary, ttl, cost, options, lastActivity, contextWindow, spikeChip, caps, model } = data;
217
+ const { summary, ttl, cost, options, lastActivity, contextWindow, ctxLive, spikeChip, caps, model } = data;
218
218
  const { hitRate } = summary;
219
219
 
220
220
  // Hit rate → color signal
@@ -321,15 +321,36 @@ export function formatReport(data, { color = true, verbose = false, timer = true
321
321
  }
322
322
  }
323
323
 
324
- // Context window chip (e.g. "📦 1M" or "📦 200k"). 1M gets a warning color
325
- // because it's the expensive default on Max plans after Opus 4.7.
324
+ // Context chip. Two data sources, best first:
325
+ //
326
+ // 1. Live fill level from Claude Code's stdin (`context_window.used_percentage`)
327
+ // — the current session's actual usage, refreshed every render. Rendered
328
+ // as `📦 68%` and colored by fill (green <70, yellow 70–89, red 90+),
329
+ // matching the cap-segment tone scale.
330
+ // 2. Fallback (table view / older Claude Code): transcript-inferred window
331
+ // size. Note the semantics: `size === '1M'` means a real request already
332
+ // carried >210k input tokens — actual heavy usage, not just the model
333
+ // supporting 1M. Current models are all 1M by default with no price
334
+ // premium, so this renders yellow ("your context is genuinely big"),
335
+ // not red ("expensive mode on") like it used to.
326
336
  let ctxSeg = null;
327
- if (contextWindow && contextWindow.size && contextWindow.size !== 'unknown') {
337
+ if (ctxLive && Number.isFinite(ctxLive.usedPct)) {
338
+ const pct = Math.max(0, Math.round(ctxLive.usedPct));
339
+ const tone = pct >= 90 ? RED : pct >= 70 ? YELLOW : GREEN;
340
+ const sizeLabel = ctxLive.size
341
+ ? (ctxLive.size >= 900_000 ? '1M' : `${Math.round(ctxLive.size / 1000)}k`)
342
+ : null;
343
+ const longLabel = sizeLabel ? `${pct}% of ${sizeLabel}` : `${pct}%`;
344
+ if (isIcon && verbose) {
345
+ ctxSeg = `${c(tone)}📦 Ctx ${longLabel}${c(RESET)}`;
346
+ } else if (isIcon) {
347
+ ctxSeg = `${c(tone)}📦 ${pct}%${c(RESET)}`;
348
+ } else {
349
+ ctxSeg = `${c(tone)}Ctx ${longLabel}${c(RESET)}`;
350
+ }
351
+ } else if (contextWindow && contextWindow.size && contextWindow.size !== 'unknown') {
328
352
  const label = contextWindow.size === '1M' ? '1M' : '200k';
329
- const ctxColor = contextWindow.size === '1M' ? RED : GREEN;
330
- // `Ctx` (not the full word `Context`) across every mode — the icon already
331
- // tells the eye what the chip is, and the short form fits the same cadence
332
- // as `Hit`/`Saved` peers when we eventually shorten those too.
353
+ const ctxColor = contextWindow.size === '1M' ? YELLOW : GREEN;
333
354
  if (isIcon && verbose) {
334
355
  ctxSeg = `${c(ctxColor)}📦 Ctx ${label}${c(RESET)}`;
335
356
  } else if (isIcon) {
@@ -86,7 +86,7 @@ function renderSpikeSection(spikes, contextWindow) {
86
86
  lines.push(r(` ${'─'.repeat(50)}`));
87
87
  if (contextWindow && contextWindow.size === '1M') {
88
88
  lines.push(
89
- r(` Context mode: 1M (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`),
89
+ r(` Context usage exceeded 200k (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`),
90
90
  );
91
91
  lines.push('');
92
92
  }
@@ -197,7 +197,7 @@ export function formatReport({ summary: sum, trend, ttl, anomalies, cost, option
197
197
  if (contextWindow && contextWindow.size !== 'unknown') {
198
198
  const note =
199
199
  contextWindow.size === '1M'
200
- ? '⚠ 1M context active (Opus 4.7+ Max default). Disable with CLAUDE_CODE_DISABLE_1M_CONTEXT=1'
200
+ ? '⚠ Context exceeded 200k in recent requests — big contexts re-bill every turn and drain the 5H/7D caps. Use /compact or /clear.'
201
201
  : '✓ 200k context (standard)';
202
202
  lines.push(` Context window: ${contextWindow.size} ${note}`);
203
203
  lines.push(` (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`);
package/src/history.js CHANGED
@@ -69,7 +69,8 @@ function saveState(state) {
69
69
  function chipKo(chip) {
70
70
  if (!chip) return chip;
71
71
  const map = {
72
- '⚠ 1M ON': '⚠ 1M 컨텍스트 활성',
72
+ '⚠ Ctx 200k+': '⚠ 컨텍스트 200k 초과',
73
+ '⚠ 1M ON': '⚠ 1M 컨텍스트 활성', // legacy (pre-v2.18)
73
74
  '⚠ Cache miss': '⚠ 캐시 미스',
74
75
  '⚠ Rebuild churn': '⚠ 캐시 재빌드 빈발',
75
76
  '⚠ Input spike': '⚠ 입력 급증',
@@ -92,6 +93,9 @@ function chipKo(chip) {
92
93
  */
93
94
  function detailKo(detail) {
94
95
  if (!detail) return detail;
96
+ const m0 = detail.match(/^Single-request context exceeded 200k \(max (\d+)k tokens\)$/);
97
+ if (m0) return `단일 요청 컨텍스트 200k 초과 (최대 ${m0[1]}k 토큰)`;
98
+ // legacy detail shape (pre-v2.18)
95
99
  const m1 = detail.match(/^Context auto-promoted to 1M \(max single-request (\d+)k tokens\)$/);
96
100
  if (m1) return `1M 컨텍스트 자동 활성 (단일 요청 최대 ${m1[1]}k 토큰)`;
97
101
  const m2 = detail.match(/^session ([^:]+): (.+)$/);