claude-token-saver 2.16.0 → 2.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.en.md +131 -193
- package/README.md +123 -186
- package/bin/cli.js +21 -1
- package/package.json +1 -1
- package/src/advice.js +49 -34
- package/src/cost.js +16 -1
- package/src/demo.js +11 -4
- package/src/formatters/statusline.js +29 -8
- package/src/formatters/table.js +2 -2
- package/src/history.js +5 -1
package/src/advice.js
CHANGED
|
@@ -39,25 +39,33 @@ function toggleShortcut() {
|
|
|
39
39
|
*/
|
|
40
40
|
export const ISSUE_MESSAGES = {
|
|
41
41
|
LARGE_INPUT_PER_REQUEST: {
|
|
42
|
-
title: 'Per-request input tokens are unusually large (
|
|
43
|
-
titleKo: '요청당 입력 토큰이 비정상적으로 큼 (
|
|
42
|
+
title: 'Per-request input tokens are unusually large (context past 200k)',
|
|
43
|
+
titleKo: '요청당 입력 토큰이 비정상적으로 큼 (컨텍스트 200k 초과)',
|
|
44
|
+
// Since Opus 4.7 there is NO long-context price premium — 1M is standard
|
|
45
|
+
// rate and the default window on current models. The cost driver is the
|
|
46
|
+
// token volume itself: a 500k-token context re-reads ~500k tokens every
|
|
47
|
+
// turn (cache-read billed) and burns the subscription 5H/7D windows
|
|
48
|
+
// several times faster. That's what this warning is about.
|
|
44
49
|
explain:
|
|
45
|
-
'
|
|
46
|
-
'
|
|
50
|
+
'Current models default to a 1M window with no long-context premium — but the token ' +
|
|
51
|
+
'volume itself is the cost: every turn re-reads the whole context (billed as cache reads) ' +
|
|
52
|
+
'and drains the 5H/7D rate-limit windows several times faster. Cache reuse also drops.',
|
|
47
53
|
explainKo:
|
|
48
|
-
'
|
|
49
|
-
'
|
|
54
|
+
'현재 모델은 1M 윈도가 기본이고 장기 컨텍스트 프리미엄도 없습니다 — 하지만 토큰량 자체가 비용입니다. ' +
|
|
55
|
+
'매 턴 컨텍스트 전체를 다시 읽고(캐시 읽기 과금) 5H/7D 한도도 몇 배 빠르게 소모됩니다. ' +
|
|
56
|
+
'캐시 재사용률도 떨어집니다.',
|
|
50
57
|
actions: () => [
|
|
51
58
|
{
|
|
52
|
-
label: '
|
|
53
|
-
labelKo: '
|
|
54
|
-
commands:
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
59
|
+
label: 'Compact or clear when context grows (first lever)',
|
|
60
|
+
labelKo: '컨텍스트가 커지면 /compact 또는 /clear (1순위)',
|
|
61
|
+
commands: [
|
|
62
|
+
'/compact — summarize the session history in place',
|
|
63
|
+
'/clear — drop history entirely at a clean task boundary',
|
|
64
|
+
],
|
|
65
|
+
commandsKo: [
|
|
66
|
+
'/compact — 세션 히스토리를 그 자리에서 요약',
|
|
67
|
+
'/clear — 작업 분기점에서 히스토리를 통째로 비우기',
|
|
68
|
+
],
|
|
61
69
|
},
|
|
62
70
|
{
|
|
63
71
|
label: 'Cap extended-thinking budget (⚠ check /effort first)',
|
|
@@ -90,16 +98,14 @@ export const ISSUE_MESSAGES = {
|
|
|
90
98
|
],
|
|
91
99
|
},
|
|
92
100
|
{
|
|
93
|
-
label: '
|
|
94
|
-
labelKo: '
|
|
95
|
-
commands: [
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
'/clear — 작업 분기점에서 히스토리를 통째로 비우기',
|
|
102
|
-
],
|
|
101
|
+
label: 'Cap the window at 200k if you never need more (optional)',
|
|
102
|
+
labelKo: '더 큰 윈도가 필요 없으면 200k로 제한 (선택)',
|
|
103
|
+
commands: disable1mEnvSnippet().concat([
|
|
104
|
+
`Or press ${toggleShortcut()} to toggle in-session`,
|
|
105
|
+
]),
|
|
106
|
+
commandsKo: disable1mEnvSnippet().concat([
|
|
107
|
+
`또는 ${toggleShortcut()}로 세션 내 즉시 토글`,
|
|
108
|
+
]),
|
|
103
109
|
},
|
|
104
110
|
{
|
|
105
111
|
label: '⚠ Known bug #31640',
|
|
@@ -337,11 +343,13 @@ export const ISSUE_MESSAGES = {
|
|
|
337
343
|
title: 'Context window is approaching the limit',
|
|
338
344
|
titleKo: '컨텍스트 창이 한계에 근접',
|
|
339
345
|
explain:
|
|
340
|
-
'As context grows
|
|
341
|
-
'reuse efficiency drops.
|
|
346
|
+
'As context grows, each turn becomes more expensive (the whole context is re-read every ' +
|
|
347
|
+
'turn) and cache reuse efficiency drops. There is no price premium past 200k on current ' +
|
|
348
|
+
'models — the token volume itself is the cost, and it drains the 5H/7D caps faster.',
|
|
342
349
|
explainKo:
|
|
343
|
-
'컨텍스트가
|
|
344
|
-
'200k
|
|
350
|
+
'컨텍스트가 커질수록 매 턴 비용이 높아지고(전체 컨텍스트를 매 턴 다시 읽음) 캐시 재사용 효율이 ' +
|
|
351
|
+
'떨어집니다. 현재 모델은 200k 초과 프리미엄이 없습니다 — 토큰량 자체가 비용이며 5H/7D 한도도 ' +
|
|
352
|
+
'더 빠르게 소모됩니다.',
|
|
345
353
|
actions: () => [
|
|
346
354
|
{
|
|
347
355
|
label: '/compact at the next natural break',
|
|
@@ -435,8 +443,8 @@ export const ISSUE_MESSAGES = {
|
|
|
435
443
|
*/
|
|
436
444
|
export const ISSUE_TIPS = {
|
|
437
445
|
LARGE_INPUT_PER_REQUEST: {
|
|
438
|
-
en: '
|
|
439
|
-
ko: '`/
|
|
446
|
+
en: '`/compact` or `/clear` — big contexts re-bill every turn and burn the 5H/7D caps; check `/effort` (`xhigh` is the #1 cap killer)',
|
|
447
|
+
ko: '`/compact` 또는 `/clear` — 큰 컨텍스트는 매 턴 재과금되고 5H/7D 한도를 태움; `/effort` 확인 (`xhigh`가 캡 소진 1순위)',
|
|
440
448
|
},
|
|
441
449
|
LOW_HIT_RATE: {
|
|
442
450
|
en: 'Continue with `claude --continue`; keep CLAUDE.md trim — every line ships every turn',
|
|
@@ -463,8 +471,8 @@ export const ISSUE_TIPS = {
|
|
|
463
471
|
ko: '지금 바로 아무 프롬프트나 보내 TTL 타이머 리셋; 또는 자리 비우기 전 `/compact` (Pro = 5분, Max = 1시간)',
|
|
464
472
|
},
|
|
465
473
|
CONTEXT_NEAR_LIMIT: {
|
|
466
|
-
en: '`/compact` before hitting the limit; `/clear` + re-attach only needed files
|
|
467
|
-
ko: '한계 도달 전 `/compact`; `/clear` 후 필요한 파일만
|
|
474
|
+
en: '`/compact` before hitting the limit; `/clear` + re-attach only needed files',
|
|
475
|
+
ko: '한계 도달 전 `/compact`; `/clear` 후 필요한 파일만 재첨부',
|
|
468
476
|
},
|
|
469
477
|
};
|
|
470
478
|
|
|
@@ -474,6 +482,9 @@ export const ISSUE_TIPS = {
|
|
|
474
482
|
* need a fallback so history.js can still surface the right tip.
|
|
475
483
|
*/
|
|
476
484
|
export const CHIP_TO_CODES = {
|
|
485
|
+
'⚠ Ctx 200k+': ['LARGE_INPUT_PER_REQUEST'],
|
|
486
|
+
// Legacy chip name (pre-v2.18) — kept so `last`/`history` can still resolve
|
|
487
|
+
// codes from history files written by older versions.
|
|
477
488
|
'⚠ 1M ON': ['LARGE_INPUT_PER_REQUEST'],
|
|
478
489
|
'⚠ Cache miss': ['LOW_HIT_RATE'],
|
|
479
490
|
'⚠ Input spike': ['LARGE_INPUT_PER_REQUEST'],
|
|
@@ -502,7 +513,11 @@ export const CAP_TIPS = {
|
|
|
502
513
|
* shown to a global audience.
|
|
503
514
|
*/
|
|
504
515
|
export function chipForIssues(issues, contextWindow) {
|
|
505
|
-
|
|
516
|
+
// Fires on *actual usage* (a real request carried >210k input tokens), not
|
|
517
|
+
// on the model merely supporting 1M — current models are all 1M by default
|
|
518
|
+
// with no price premium, so "1M ON" stopped being a meaningful alarm. The
|
|
519
|
+
// meaningful signal is "your context genuinely exceeded 200k".
|
|
520
|
+
if (contextWindow?.size === '1M') return '⚠ Ctx 200k+';
|
|
506
521
|
const codes = issues.map((i) => i.code);
|
|
507
522
|
if (codes.includes('LARGE_INPUT_PER_REQUEST')) return '⚠ Input spike';
|
|
508
523
|
if (codes.includes('BUCKET_5M_DOMINANT')) return '⚠ 5m TTL';
|
package/src/cost.js
CHANGED
|
@@ -8,7 +8,17 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
const PRICING = {
|
|
11
|
-
//
|
|
11
|
+
// Fable 5 / Mythos 5 — premium tier above Opus ($10/$50). Cache write
|
|
12
|
+
// rates follow the standard multipliers (1.25x input for 5m, 2x for 1h),
|
|
13
|
+
// cache read is 0.1x input.
|
|
14
|
+
'claude-fable-5': {
|
|
15
|
+
input: 10.0,
|
|
16
|
+
cacheWrite5m: 12.5,
|
|
17
|
+
cacheWrite1h: 20.0,
|
|
18
|
+
cacheRead: 1.0,
|
|
19
|
+
output: 50.0,
|
|
20
|
+
},
|
|
21
|
+
// Opus 4.5+ (new pricing tier — includes 4.5, 4.6, 4.7, 4.8, and future)
|
|
12
22
|
'claude-opus-new': {
|
|
13
23
|
input: 5.0,
|
|
14
24
|
cacheWrite5m: 6.25,
|
|
@@ -66,6 +76,11 @@ function detectPricingTier(model) {
|
|
|
66
76
|
if (!model) return 'claude-sonnet';
|
|
67
77
|
const m = model.toLowerCase();
|
|
68
78
|
|
|
79
|
+
// Fable 5 / Mythos 5 — must be checked before the generic fallback:
|
|
80
|
+
// without this, 'claude-fable-5' fell through to the Sonnet tier and
|
|
81
|
+
// under-estimated costs ~3x ($3/$15 vs the real $10/$50).
|
|
82
|
+
if (m.includes('fable') || m.includes('mythos')) return 'claude-fable-5';
|
|
83
|
+
|
|
69
84
|
if (m.includes('opus')) {
|
|
70
85
|
// Opus 4.5, 4.6, 4.7, and future 5+ use the new reduced pricing.
|
|
71
86
|
if (/opus[-_.]?4[-_.]?[5-9]\b/.test(m)) return 'claude-opus-new';
|
package/src/demo.js
CHANGED
|
@@ -26,6 +26,7 @@ const SCENARIOS = [
|
|
|
26
26
|
savings: 2123,
|
|
27
27
|
elapsedSec: 30,
|
|
28
28
|
contextSize: '200k',
|
|
29
|
+
ctxUsedPct: 34,
|
|
29
30
|
spikeChip: null,
|
|
30
31
|
caps: HEALTHY_CAPS,
|
|
31
32
|
},
|
|
@@ -98,14 +99,15 @@ const SCENARIOS = [
|
|
|
98
99
|
},
|
|
99
100
|
{
|
|
100
101
|
name: 'ctx-1m',
|
|
101
|
-
label: '⚠
|
|
102
|
+
label: '⚠ Context past 200k',
|
|
102
103
|
data: {
|
|
103
104
|
hitRate: 0.78,
|
|
104
105
|
pct1h: 0.92,
|
|
105
106
|
savings: 1340,
|
|
106
107
|
elapsedSec: 30,
|
|
107
108
|
contextSize: '1M',
|
|
108
|
-
|
|
109
|
+
ctxUsedPct: 28, // 28% of 1M ≈ 280k actually in context
|
|
110
|
+
spikeChip: '⚠ Ctx 200k+',
|
|
109
111
|
caps: HEALTHY_CAPS,
|
|
110
112
|
},
|
|
111
113
|
},
|
|
@@ -271,7 +273,7 @@ export function buildTableDemoData(options = {}) {
|
|
|
271
273
|
spikeReport: { spikes, baseline: { p95: 940_000 } },
|
|
272
274
|
contextWindow: { size: '1M', maxContext: 280_000 },
|
|
273
275
|
lastActivity: Date.now() - 60 * 1000,
|
|
274
|
-
spikeChip: '⚠
|
|
276
|
+
spikeChip: '⚠ Ctx 200k+',
|
|
275
277
|
};
|
|
276
278
|
}
|
|
277
279
|
|
|
@@ -301,7 +303,7 @@ export function buildScenarioData(scenarioName, options) {
|
|
|
301
303
|
if (!scenario) return null;
|
|
302
304
|
}
|
|
303
305
|
|
|
304
|
-
const { hitRate, pct1h, pct5m, savings, elapsedSec, contextSize, spikeChip, caps } = scenario.data;
|
|
306
|
+
const { hitRate, pct1h, pct5m, savings, elapsedSec, contextSize, ctxUsedPct, spikeChip, caps } = scenario.data;
|
|
305
307
|
return {
|
|
306
308
|
summary: { hitRate },
|
|
307
309
|
ttl: { pct1h, pct5m: pct5m ?? (1 - pct1h) },
|
|
@@ -314,6 +316,11 @@ export function buildScenarioData(scenarioName, options) {
|
|
|
314
316
|
},
|
|
315
317
|
lastActivity: Date.now() - elapsedSec * 1000,
|
|
316
318
|
contextWindow: { size: contextSize },
|
|
319
|
+
// Live fill level (`📦 68%`) — scenarios that set ctxUsedPct exercise the
|
|
320
|
+
// stdin-driven segment; the rest fall back to the size-based chip.
|
|
321
|
+
ctxLive: ctxUsedPct != null
|
|
322
|
+
? { usedPct: ctxUsedPct, size: contextSize === '1M' ? 1_000_000 : 200_000 }
|
|
323
|
+
: undefined,
|
|
317
324
|
spikeChip,
|
|
318
325
|
caps: buildCapsShape(caps),
|
|
319
326
|
model: DEFAULT_MODEL,
|
|
@@ -214,7 +214,7 @@ export function formatNoSession({ caps = null, model = null, windowLabel = '' }
|
|
|
214
214
|
* @param {string[]|null} [opts.segments] - whitelist of segments to render. Names: cap-warn, spike, harness, model, hit, ttl, saved, ctx, period, plus per-window keys (`five_hour`, `seven_day`, …). `5h`/`7d` are kept as aliases for back-compat. Null/undefined = all.
|
|
215
215
|
*/
|
|
216
216
|
export function formatReport(data, { color = true, verbose = false, timer = true, mode = 'text', segments = null } = {}) {
|
|
217
|
-
const { summary, ttl, cost, options, lastActivity, contextWindow, spikeChip, caps, model } = data;
|
|
217
|
+
const { summary, ttl, cost, options, lastActivity, contextWindow, ctxLive, spikeChip, caps, model } = data;
|
|
218
218
|
const { hitRate } = summary;
|
|
219
219
|
|
|
220
220
|
// Hit rate → color signal
|
|
@@ -321,15 +321,36 @@ export function formatReport(data, { color = true, verbose = false, timer = true
|
|
|
321
321
|
}
|
|
322
322
|
}
|
|
323
323
|
|
|
324
|
-
// Context
|
|
325
|
-
//
|
|
324
|
+
// Context chip. Two data sources, best first:
|
|
325
|
+
//
|
|
326
|
+
// 1. Live fill level from Claude Code's stdin (`context_window.used_percentage`)
|
|
327
|
+
// — the current session's actual usage, refreshed every render. Rendered
|
|
328
|
+
// as `📦 68%` and colored by fill (green <70, yellow 70–89, red 90+),
|
|
329
|
+
// matching the cap-segment tone scale.
|
|
330
|
+
// 2. Fallback (table view / older Claude Code): transcript-inferred window
|
|
331
|
+
// size. Note the semantics: `size === '1M'` means a real request already
|
|
332
|
+
// carried >210k input tokens — actual heavy usage, not just the model
|
|
333
|
+
// supporting 1M. Current models are all 1M by default with no price
|
|
334
|
+
// premium, so this renders yellow ("your context is genuinely big"),
|
|
335
|
+
// not red ("expensive mode on") like it used to.
|
|
326
336
|
let ctxSeg = null;
|
|
327
|
-
if (
|
|
337
|
+
if (ctxLive && Number.isFinite(ctxLive.usedPct)) {
|
|
338
|
+
const pct = Math.max(0, Math.round(ctxLive.usedPct));
|
|
339
|
+
const tone = pct >= 90 ? RED : pct >= 70 ? YELLOW : GREEN;
|
|
340
|
+
const sizeLabel = ctxLive.size
|
|
341
|
+
? (ctxLive.size >= 900_000 ? '1M' : `${Math.round(ctxLive.size / 1000)}k`)
|
|
342
|
+
: null;
|
|
343
|
+
const longLabel = sizeLabel ? `${pct}% of ${sizeLabel}` : `${pct}%`;
|
|
344
|
+
if (isIcon && verbose) {
|
|
345
|
+
ctxSeg = `${c(tone)}📦 Ctx ${longLabel}${c(RESET)}`;
|
|
346
|
+
} else if (isIcon) {
|
|
347
|
+
ctxSeg = `${c(tone)}📦 ${pct}%${c(RESET)}`;
|
|
348
|
+
} else {
|
|
349
|
+
ctxSeg = `${c(tone)}Ctx ${longLabel}${c(RESET)}`;
|
|
350
|
+
}
|
|
351
|
+
} else if (contextWindow && contextWindow.size && contextWindow.size !== 'unknown') {
|
|
328
352
|
const label = contextWindow.size === '1M' ? '1M' : '200k';
|
|
329
|
-
const ctxColor = contextWindow.size === '1M' ?
|
|
330
|
-
// `Ctx` (not the full word `Context`) across every mode — the icon already
|
|
331
|
-
// tells the eye what the chip is, and the short form fits the same cadence
|
|
332
|
-
// as `Hit`/`Saved` peers when we eventually shorten those too.
|
|
353
|
+
const ctxColor = contextWindow.size === '1M' ? YELLOW : GREEN;
|
|
333
354
|
if (isIcon && verbose) {
|
|
334
355
|
ctxSeg = `${c(ctxColor)}📦 Ctx ${label}${c(RESET)}`;
|
|
335
356
|
} else if (isIcon) {
|
package/src/formatters/table.js
CHANGED
|
@@ -86,7 +86,7 @@ function renderSpikeSection(spikes, contextWindow) {
|
|
|
86
86
|
lines.push(r(` ${'─'.repeat(50)}`));
|
|
87
87
|
if (contextWindow && contextWindow.size === '1M') {
|
|
88
88
|
lines.push(
|
|
89
|
-
r(` Context
|
|
89
|
+
r(` Context usage exceeded 200k (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`),
|
|
90
90
|
);
|
|
91
91
|
lines.push('');
|
|
92
92
|
}
|
|
@@ -197,7 +197,7 @@ export function formatReport({ summary: sum, trend, ttl, anomalies, cost, option
|
|
|
197
197
|
if (contextWindow && contextWindow.size !== 'unknown') {
|
|
198
198
|
const note =
|
|
199
199
|
contextWindow.size === '1M'
|
|
200
|
-
? '⚠
|
|
200
|
+
? '⚠ Context exceeded 200k in recent requests — big contexts re-bill every turn and drain the 5H/7D caps. Use /compact or /clear.'
|
|
201
201
|
: '✓ 200k context (standard)';
|
|
202
202
|
lines.push(` Context window: ${contextWindow.size} ${note}`);
|
|
203
203
|
lines.push(` (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`);
|
package/src/history.js
CHANGED
|
@@ -69,7 +69,8 @@ function saveState(state) {
|
|
|
69
69
|
function chipKo(chip) {
|
|
70
70
|
if (!chip) return chip;
|
|
71
71
|
const map = {
|
|
72
|
-
'⚠
|
|
72
|
+
'⚠ Ctx 200k+': '⚠ 컨텍스트 200k 초과',
|
|
73
|
+
'⚠ 1M ON': '⚠ 1M 컨텍스트 활성', // legacy (pre-v2.18)
|
|
73
74
|
'⚠ Cache miss': '⚠ 캐시 미스',
|
|
74
75
|
'⚠ Rebuild churn': '⚠ 캐시 재빌드 빈발',
|
|
75
76
|
'⚠ Input spike': '⚠ 입력 급증',
|
|
@@ -92,6 +93,9 @@ function chipKo(chip) {
|
|
|
92
93
|
*/
|
|
93
94
|
function detailKo(detail) {
|
|
94
95
|
if (!detail) return detail;
|
|
96
|
+
const m0 = detail.match(/^Single-request context exceeded 200k \(max (\d+)k tokens\)$/);
|
|
97
|
+
if (m0) return `단일 요청 컨텍스트 200k 초과 (최대 ${m0[1]}k 토큰)`;
|
|
98
|
+
// legacy detail shape (pre-v2.18)
|
|
95
99
|
const m1 = detail.match(/^Context auto-promoted to 1M \(max single-request (\d+)k tokens\)$/);
|
|
96
100
|
if (m1) return `1M 컨텍스트 자동 활성 (단일 요청 최대 ${m1[1]}k 토큰)`;
|
|
97
101
|
const m2 = detail.match(/^session ([^:]+): (.+)$/);
|