claude-token-saver 2.17.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.en.md +57 -3
- package/README.md +57 -3
- package/bin/cli.js +254 -4
- package/package.json +1 -1
- package/src/advice.js +49 -34
- package/src/demo.js +11 -4
- package/src/formatters/statusline.js +29 -8
- package/src/formatters/table.js +2 -2
- package/src/frugon-export.js +211 -0
- package/src/harness.js +85 -0
- package/src/history.js +5 -1
- package/src/installer.js +37 -0
- package/src/route-scan.js +286 -0
package/src/advice.js
CHANGED
|
@@ -39,25 +39,33 @@ function toggleShortcut() {
|
|
|
39
39
|
*/
|
|
40
40
|
export const ISSUE_MESSAGES = {
|
|
41
41
|
LARGE_INPUT_PER_REQUEST: {
|
|
42
|
-
title: 'Per-request input tokens are unusually large (
|
|
43
|
-
titleKo: '요청당 입력 토큰이 비정상적으로 큼 (
|
|
42
|
+
title: 'Per-request input tokens are unusually large (context past 200k)',
|
|
43
|
+
titleKo: '요청당 입력 토큰이 비정상적으로 큼 (컨텍스트 200k 초과)',
|
|
44
|
+
// Since Opus 4.7 there is NO long-context price premium — 1M is standard
|
|
45
|
+
// rate and the default window on current models. The cost driver is the
|
|
46
|
+
// token volume itself: a 500k-token context re-reads ~500k tokens every
|
|
47
|
+
// turn (cache-read billed) and burns the subscription 5H/7D windows
|
|
48
|
+
// several times faster. That's what this warning is about.
|
|
44
49
|
explain:
|
|
45
|
-
'
|
|
46
|
-
'
|
|
50
|
+
'Current models default to a 1M window with no long-context premium — but the token ' +
|
|
51
|
+
'volume itself is the cost: every turn re-reads the whole context (billed as cache reads) ' +
|
|
52
|
+
'and drains the 5H/7D rate-limit windows several times faster. Cache reuse also drops.',
|
|
47
53
|
explainKo:
|
|
48
|
-
'
|
|
49
|
-
'
|
|
54
|
+
'현재 모델은 1M 윈도가 기본이고 장기 컨텍스트 프리미엄도 없습니다 — 하지만 토큰량 자체가 비용입니다. ' +
|
|
55
|
+
'매 턴 컨텍스트 전체를 다시 읽고(캐시 읽기 과금) 5H/7D 한도도 몇 배 빠르게 소모됩니다. ' +
|
|
56
|
+
'캐시 재사용률도 떨어집니다.',
|
|
50
57
|
actions: () => [
|
|
51
58
|
{
|
|
52
|
-
label: '
|
|
53
|
-
labelKo: '
|
|
54
|
-
commands:
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
59
|
+
label: 'Compact or clear when context grows (first lever)',
|
|
60
|
+
labelKo: '컨텍스트가 커지면 /compact 또는 /clear (1순위)',
|
|
61
|
+
commands: [
|
|
62
|
+
'/compact — summarize the session history in place',
|
|
63
|
+
'/clear — drop history entirely at a clean task boundary',
|
|
64
|
+
],
|
|
65
|
+
commandsKo: [
|
|
66
|
+
'/compact — 세션 히스토리를 그 자리에서 요약',
|
|
67
|
+
'/clear — 작업 분기점에서 히스토리를 통째로 비우기',
|
|
68
|
+
],
|
|
61
69
|
},
|
|
62
70
|
{
|
|
63
71
|
label: 'Cap extended-thinking budget (⚠ check /effort first)',
|
|
@@ -90,16 +98,14 @@ export const ISSUE_MESSAGES = {
|
|
|
90
98
|
],
|
|
91
99
|
},
|
|
92
100
|
{
|
|
93
|
-
label: '
|
|
94
|
-
labelKo: '
|
|
95
|
-
commands: [
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
'/clear — 작업 분기점에서 히스토리를 통째로 비우기',
|
|
102
|
-
],
|
|
101
|
+
label: 'Cap the window at 200k if you never need more (optional)',
|
|
102
|
+
labelKo: '더 큰 윈도가 필요 없으면 200k로 제한 (선택)',
|
|
103
|
+
commands: disable1mEnvSnippet().concat([
|
|
104
|
+
`Or press ${toggleShortcut()} to toggle in-session`,
|
|
105
|
+
]),
|
|
106
|
+
commandsKo: disable1mEnvSnippet().concat([
|
|
107
|
+
`또는 ${toggleShortcut()}로 세션 내 즉시 토글`,
|
|
108
|
+
]),
|
|
103
109
|
},
|
|
104
110
|
{
|
|
105
111
|
label: '⚠ Known bug #31640',
|
|
@@ -337,11 +343,13 @@ export const ISSUE_MESSAGES = {
|
|
|
337
343
|
title: 'Context window is approaching the limit',
|
|
338
344
|
titleKo: '컨텍스트 창이 한계에 근접',
|
|
339
345
|
explain:
|
|
340
|
-
'As context grows
|
|
341
|
-
'reuse efficiency drops.
|
|
346
|
+
'As context grows, each turn becomes more expensive (the whole context is re-read every ' +
|
|
347
|
+
'turn) and cache reuse efficiency drops. There is no price premium past 200k on current ' +
|
|
348
|
+
'models — the token volume itself is the cost, and it drains the 5H/7D caps faster.',
|
|
342
349
|
explainKo:
|
|
343
|
-
'컨텍스트가
|
|
344
|
-
'200k
|
|
350
|
+
'컨텍스트가 커질수록 매 턴 비용이 높아지고(전체 컨텍스트를 매 턴 다시 읽음) 캐시 재사용 효율이 ' +
|
|
351
|
+
'떨어집니다. 현재 모델은 200k 초과 프리미엄이 없습니다 — 토큰량 자체가 비용이며 5H/7D 한도도 ' +
|
|
352
|
+
'더 빠르게 소모됩니다.',
|
|
345
353
|
actions: () => [
|
|
346
354
|
{
|
|
347
355
|
label: '/compact at the next natural break',
|
|
@@ -435,8 +443,8 @@ export const ISSUE_MESSAGES = {
|
|
|
435
443
|
*/
|
|
436
444
|
export const ISSUE_TIPS = {
|
|
437
445
|
LARGE_INPUT_PER_REQUEST: {
|
|
438
|
-
en: '
|
|
439
|
-
ko: '`/
|
|
446
|
+
en: '`/compact` or `/clear` — big contexts re-bill every turn and burn the 5H/7D caps; check `/effort` (`xhigh` is the #1 cap killer)',
|
|
447
|
+
ko: '`/compact` 또는 `/clear` — 큰 컨텍스트는 매 턴 재과금되고 5H/7D 한도를 태움; `/effort` 확인 (`xhigh`가 캡 소진 1순위)',
|
|
440
448
|
},
|
|
441
449
|
LOW_HIT_RATE: {
|
|
442
450
|
en: 'Continue with `claude --continue`; keep CLAUDE.md trim — every line ships every turn',
|
|
@@ -463,8 +471,8 @@ export const ISSUE_TIPS = {
|
|
|
463
471
|
ko: '지금 바로 아무 프롬프트나 보내 TTL 타이머 리셋; 또는 자리 비우기 전 `/compact` (Pro = 5분, Max = 1시간)',
|
|
464
472
|
},
|
|
465
473
|
CONTEXT_NEAR_LIMIT: {
|
|
466
|
-
en: '`/compact` before hitting the limit; `/clear` + re-attach only needed files
|
|
467
|
-
ko: '한계 도달 전 `/compact`; `/clear` 후 필요한 파일만
|
|
474
|
+
en: '`/compact` before hitting the limit; `/clear` + re-attach only needed files',
|
|
475
|
+
ko: '한계 도달 전 `/compact`; `/clear` 후 필요한 파일만 재첨부',
|
|
468
476
|
},
|
|
469
477
|
};
|
|
470
478
|
|
|
@@ -474,6 +482,9 @@ export const ISSUE_TIPS = {
|
|
|
474
482
|
* need a fallback so history.js can still surface the right tip.
|
|
475
483
|
*/
|
|
476
484
|
export const CHIP_TO_CODES = {
|
|
485
|
+
'⚠ Ctx 200k+': ['LARGE_INPUT_PER_REQUEST'],
|
|
486
|
+
// Legacy chip name (pre-v2.18) — kept so `last`/`history` can still resolve
|
|
487
|
+
// codes from history files written by older versions.
|
|
477
488
|
'⚠ 1M ON': ['LARGE_INPUT_PER_REQUEST'],
|
|
478
489
|
'⚠ Cache miss': ['LOW_HIT_RATE'],
|
|
479
490
|
'⚠ Input spike': ['LARGE_INPUT_PER_REQUEST'],
|
|
@@ -502,7 +513,11 @@ export const CAP_TIPS = {
|
|
|
502
513
|
* shown to a global audience.
|
|
503
514
|
*/
|
|
504
515
|
export function chipForIssues(issues, contextWindow) {
|
|
505
|
-
|
|
516
|
+
// Fires on *actual usage* (a real request carried >210k input tokens), not
|
|
517
|
+
// on the model merely supporting 1M — current models are all 1M by default
|
|
518
|
+
// with no price premium, so "1M ON" stopped being a meaningful alarm. The
|
|
519
|
+
// meaningful signal is "your context genuinely exceeded 200k".
|
|
520
|
+
if (contextWindow?.size === '1M') return '⚠ Ctx 200k+';
|
|
506
521
|
const codes = issues.map((i) => i.code);
|
|
507
522
|
if (codes.includes('LARGE_INPUT_PER_REQUEST')) return '⚠ Input spike';
|
|
508
523
|
if (codes.includes('BUCKET_5M_DOMINANT')) return '⚠ 5m TTL';
|
package/src/demo.js
CHANGED
|
@@ -26,6 +26,7 @@ const SCENARIOS = [
|
|
|
26
26
|
savings: 2123,
|
|
27
27
|
elapsedSec: 30,
|
|
28
28
|
contextSize: '200k',
|
|
29
|
+
ctxUsedPct: 34,
|
|
29
30
|
spikeChip: null,
|
|
30
31
|
caps: HEALTHY_CAPS,
|
|
31
32
|
},
|
|
@@ -98,14 +99,15 @@ const SCENARIOS = [
|
|
|
98
99
|
},
|
|
99
100
|
{
|
|
100
101
|
name: 'ctx-1m',
|
|
101
|
-
label: '⚠
|
|
102
|
+
label: '⚠ Context past 200k',
|
|
102
103
|
data: {
|
|
103
104
|
hitRate: 0.78,
|
|
104
105
|
pct1h: 0.92,
|
|
105
106
|
savings: 1340,
|
|
106
107
|
elapsedSec: 30,
|
|
107
108
|
contextSize: '1M',
|
|
108
|
-
|
|
109
|
+
ctxUsedPct: 28, // 28% of 1M ≈ 280k actually in context
|
|
110
|
+
spikeChip: '⚠ Ctx 200k+',
|
|
109
111
|
caps: HEALTHY_CAPS,
|
|
110
112
|
},
|
|
111
113
|
},
|
|
@@ -271,7 +273,7 @@ export function buildTableDemoData(options = {}) {
|
|
|
271
273
|
spikeReport: { spikes, baseline: { p95: 940_000 } },
|
|
272
274
|
contextWindow: { size: '1M', maxContext: 280_000 },
|
|
273
275
|
lastActivity: Date.now() - 60 * 1000,
|
|
274
|
-
spikeChip: '⚠
|
|
276
|
+
spikeChip: '⚠ Ctx 200k+',
|
|
275
277
|
};
|
|
276
278
|
}
|
|
277
279
|
|
|
@@ -301,7 +303,7 @@ export function buildScenarioData(scenarioName, options) {
|
|
|
301
303
|
if (!scenario) return null;
|
|
302
304
|
}
|
|
303
305
|
|
|
304
|
-
const { hitRate, pct1h, pct5m, savings, elapsedSec, contextSize, spikeChip, caps } = scenario.data;
|
|
306
|
+
const { hitRate, pct1h, pct5m, savings, elapsedSec, contextSize, ctxUsedPct, spikeChip, caps } = scenario.data;
|
|
305
307
|
return {
|
|
306
308
|
summary: { hitRate },
|
|
307
309
|
ttl: { pct1h, pct5m: pct5m ?? (1 - pct1h) },
|
|
@@ -314,6 +316,11 @@ export function buildScenarioData(scenarioName, options) {
|
|
|
314
316
|
},
|
|
315
317
|
lastActivity: Date.now() - elapsedSec * 1000,
|
|
316
318
|
contextWindow: { size: contextSize },
|
|
319
|
+
// Live fill level (`📦 68%`) — scenarios that set ctxUsedPct exercise the
|
|
320
|
+
// stdin-driven segment; the rest fall back to the size-based chip.
|
|
321
|
+
ctxLive: ctxUsedPct != null
|
|
322
|
+
? { usedPct: ctxUsedPct, size: contextSize === '1M' ? 1_000_000 : 200_000 }
|
|
323
|
+
: undefined,
|
|
317
324
|
spikeChip,
|
|
318
325
|
caps: buildCapsShape(caps),
|
|
319
326
|
model: DEFAULT_MODEL,
|
|
@@ -214,7 +214,7 @@ export function formatNoSession({ caps = null, model = null, windowLabel = '' }
|
|
|
214
214
|
* @param {string[]|null} [opts.segments] - whitelist of segments to render. Names: cap-warn, spike, harness, model, hit, ttl, saved, ctx, period, plus per-window keys (`five_hour`, `seven_day`, …). `5h`/`7d` are kept as aliases for back-compat. Null/undefined = all.
|
|
215
215
|
*/
|
|
216
216
|
export function formatReport(data, { color = true, verbose = false, timer = true, mode = 'text', segments = null } = {}) {
|
|
217
|
-
const { summary, ttl, cost, options, lastActivity, contextWindow, spikeChip, caps, model } = data;
|
|
217
|
+
const { summary, ttl, cost, options, lastActivity, contextWindow, ctxLive, spikeChip, caps, model } = data;
|
|
218
218
|
const { hitRate } = summary;
|
|
219
219
|
|
|
220
220
|
// Hit rate → color signal
|
|
@@ -321,15 +321,36 @@ export function formatReport(data, { color = true, verbose = false, timer = true
|
|
|
321
321
|
}
|
|
322
322
|
}
|
|
323
323
|
|
|
324
|
-
// Context
|
|
325
|
-
//
|
|
324
|
+
// Context chip. Two data sources, best first:
|
|
325
|
+
//
|
|
326
|
+
// 1. Live fill level from Claude Code's stdin (`context_window.used_percentage`)
|
|
327
|
+
// — the current session's actual usage, refreshed every render. Rendered
|
|
328
|
+
// as `📦 68%` and colored by fill (green <70, yellow 70–89, red 90+),
|
|
329
|
+
// matching the cap-segment tone scale.
|
|
330
|
+
// 2. Fallback (table view / older Claude Code): transcript-inferred window
|
|
331
|
+
// size. Note the semantics: `size === '1M'` means a real request already
|
|
332
|
+
// carried >210k input tokens — actual heavy usage, not just the model
|
|
333
|
+
// supporting 1M. Current models are all 1M by default with no price
|
|
334
|
+
// premium, so this renders yellow ("your context is genuinely big"),
|
|
335
|
+
// not red ("expensive mode on") like it used to.
|
|
326
336
|
let ctxSeg = null;
|
|
327
|
-
if (
|
|
337
|
+
if (ctxLive && Number.isFinite(ctxLive.usedPct)) {
|
|
338
|
+
const pct = Math.max(0, Math.round(ctxLive.usedPct));
|
|
339
|
+
const tone = pct >= 90 ? RED : pct >= 70 ? YELLOW : GREEN;
|
|
340
|
+
const sizeLabel = ctxLive.size
|
|
341
|
+
? (ctxLive.size >= 900_000 ? '1M' : `${Math.round(ctxLive.size / 1000)}k`)
|
|
342
|
+
: null;
|
|
343
|
+
const longLabel = sizeLabel ? `${pct}% of ${sizeLabel}` : `${pct}%`;
|
|
344
|
+
if (isIcon && verbose) {
|
|
345
|
+
ctxSeg = `${c(tone)}📦 Ctx ${longLabel}${c(RESET)}`;
|
|
346
|
+
} else if (isIcon) {
|
|
347
|
+
ctxSeg = `${c(tone)}📦 ${pct}%${c(RESET)}`;
|
|
348
|
+
} else {
|
|
349
|
+
ctxSeg = `${c(tone)}Ctx ${longLabel}${c(RESET)}`;
|
|
350
|
+
}
|
|
351
|
+
} else if (contextWindow && contextWindow.size && contextWindow.size !== 'unknown') {
|
|
328
352
|
const label = contextWindow.size === '1M' ? '1M' : '200k';
|
|
329
|
-
const ctxColor = contextWindow.size === '1M' ?
|
|
330
|
-
// `Ctx` (not the full word `Context`) across every mode — the icon already
|
|
331
|
-
// tells the eye what the chip is, and the short form fits the same cadence
|
|
332
|
-
// as `Hit`/`Saved` peers when we eventually shorten those too.
|
|
353
|
+
const ctxColor = contextWindow.size === '1M' ? YELLOW : GREEN;
|
|
333
354
|
if (isIcon && verbose) {
|
|
334
355
|
ctxSeg = `${c(ctxColor)}📦 Ctx ${label}${c(RESET)}`;
|
|
335
356
|
} else if (isIcon) {
|
package/src/formatters/table.js
CHANGED
|
@@ -86,7 +86,7 @@ function renderSpikeSection(spikes, contextWindow) {
|
|
|
86
86
|
lines.push(r(` ${'─'.repeat(50)}`));
|
|
87
87
|
if (contextWindow && contextWindow.size === '1M') {
|
|
88
88
|
lines.push(
|
|
89
|
-
r(` Context
|
|
89
|
+
r(` Context usage exceeded 200k (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`),
|
|
90
90
|
);
|
|
91
91
|
lines.push('');
|
|
92
92
|
}
|
|
@@ -197,7 +197,7 @@ export function formatReport({ summary: sum, trend, ttl, anomalies, cost, option
|
|
|
197
197
|
if (contextWindow && contextWindow.size !== 'unknown') {
|
|
198
198
|
const note =
|
|
199
199
|
contextWindow.size === '1M'
|
|
200
|
-
? '⚠
|
|
200
|
+
? '⚠ Context exceeded 200k in recent requests — big contexts re-bill every turn and drain the 5H/7D caps. Use /compact or /clear.'
|
|
201
201
|
: '✓ 200k context (standard)';
|
|
202
202
|
lines.push(` Context window: ${contextWindow.size} ${note}`);
|
|
203
203
|
lines.push(` (max recent single-request input ${formatContextSize(contextWindow.maxContext)} tokens)`);
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* frugon export — convert Claude Code session transcripts into the
|
|
3
|
+
* OpenAI-compatible JSONL log format frugon analyzes.
|
|
4
|
+
* (frugon: local LLM cost analyzer — github.com/Rodiun/frugon)
|
|
5
|
+
*
|
|
6
|
+
* One output line per API call:
|
|
7
|
+
* {
|
|
8
|
+
* "model": "claude-opus-4-8",
|
|
9
|
+
* "timestamp": "2026-07-12T02:11:05.123Z",
|
|
10
|
+
* "usage": { "prompt_tokens": 1234, "completion_tokens": 56 },
|
|
11
|
+
* "request": { "messages": [ ...stubs..., { "role": "user", "content": "<last user prompt>" } ] },
|
|
12
|
+
* "response": { "choices": [ { "message": { "role": "assistant", "content": "<reply>" } } ] }
|
|
13
|
+
* }
|
|
14
|
+
*
|
|
15
|
+
* Design notes (kept in sync with frugon 0.2.x internals):
|
|
16
|
+
* - frugon prefers the usage block for token counts, so message content is
|
|
17
|
+
* never re-tokenized — stubs with empty content are safe.
|
|
18
|
+
* - frugon's easy/hard difficulty score reads prompt_tokens, completion_tokens
|
|
19
|
+
* and conversation depth (len(messages) - 1, saturating at 6 turns). We emit
|
|
20
|
+
* up to MAX_STUB_MESSAGES role-alternating stubs so depth survives the
|
|
21
|
+
* export without duplicating the whole conversation into every record.
|
|
22
|
+
* - frugon has no notion of prompt caching: every prompt token is priced at
|
|
23
|
+
* the base input rate. Claude Code sessions are cache-read heavy (~90%+),
|
|
24
|
+
* so raw totals would overstate spend ~10x. By default we fold Anthropic's
|
|
25
|
+
* cache multipliers (5m write 1.25x, 1h write 2x, read 0.1x) into an
|
|
26
|
+
* "effective" prompt_tokens so frugon's dollar figures match reality.
|
|
27
|
+
* Pass cacheWeighted: false for raw physical token counts.
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import { createReadStream, createWriteStream } from 'node:fs';
|
|
31
|
+
import { createInterface } from 'node:readline';
|
|
32
|
+
import { discoverSessionFiles } from './parser.js';
|
|
33
|
+
|
|
34
|
+
// Depth cap: frugon's turn signal saturates at 6 turns (len(messages)-1 >= 6),
|
|
35
|
+
// so 7 messages carry the maximum-depth signal at minimum size.
|
|
36
|
+
const MAX_STUB_MESSAGES = 7;
|
|
37
|
+
|
|
38
|
+
// Anthropic cache multipliers relative to the base input rate — uniform
|
|
39
|
+
// across model tiers (see src/cost.js PRICING).
|
|
40
|
+
const CACHE_WEIGHTS = { write5m: 1.25, write1h: 2, read: 0.1 };
|
|
41
|
+
|
|
42
|
+
/** Strip context-window suffixes like "[1m]" so frugon's pricing table matches. */
|
|
43
|
+
export function normalizeModelId(model) {
|
|
44
|
+
return String(model || 'unknown').replace(/\[[^\]]*\]$/, '');
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Effective prompt tokens: what the call *costs* expressed in base-rate
|
|
49
|
+
* input tokens, so frugon (which prices all prompt tokens at the input rate)
|
|
50
|
+
* reproduces the real cache-discounted spend.
|
|
51
|
+
*/
|
|
52
|
+
export function effectivePromptTokens(r) {
|
|
53
|
+
const tracked = (r.ephemeral5mTokens || 0) + (r.ephemeral1hTokens || 0);
|
|
54
|
+
const untracked = Math.max(0, (r.cacheCreationTokens || 0) - tracked);
|
|
55
|
+
return Math.round(
|
|
56
|
+
(r.inputTokens || 0) +
|
|
57
|
+
((r.ephemeral5mTokens || 0) + untracked) * CACHE_WEIGHTS.write5m +
|
|
58
|
+
(r.ephemeral1hTokens || 0) * CACHE_WEIGHTS.write1h +
|
|
59
|
+
(r.cacheReadTokens || 0) * CACHE_WEIGHTS.read,
|
|
60
|
+
);
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** Raw physical prompt tokens (input + cache writes + cache reads). */
|
|
64
|
+
export function rawPromptTokens(r) {
|
|
65
|
+
return (r.inputTokens || 0) + (r.cacheCreationTokens || 0) + (r.cacheReadTokens || 0);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Extract plain text from a Claude transcript message content field. */
|
|
69
|
+
function contentText(content) {
|
|
70
|
+
if (typeof content === 'string') return content;
|
|
71
|
+
if (!Array.isArray(content)) return '';
|
|
72
|
+
return content
|
|
73
|
+
.filter((b) => b && b.type === 'text' && typeof b.text === 'string')
|
|
74
|
+
.map((b) => b.text)
|
|
75
|
+
.join('\n');
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Parse one session transcript into frugon records.
|
|
80
|
+
* Deduplicates by requestId (last-write-wins, matching parser.js) while
|
|
81
|
+
* tracking the conversation depth and last user prompt at each call.
|
|
82
|
+
*/
|
|
83
|
+
export async function collectSessionRecords(filePath, { cacheWeighted = true, includeContent = true } = {}) {
|
|
84
|
+
const records = new Map();
|
|
85
|
+
let depth = 0;
|
|
86
|
+
let lastUserText = '';
|
|
87
|
+
let lastCwd = '';
|
|
88
|
+
|
|
89
|
+
const rl = createInterface({
|
|
90
|
+
input: createReadStream(filePath, { encoding: 'utf8' }),
|
|
91
|
+
crlfDelay: Infinity,
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
for await (const line of rl) {
|
|
95
|
+
let entry;
|
|
96
|
+
try {
|
|
97
|
+
entry = JSON.parse(line);
|
|
98
|
+
} catch {
|
|
99
|
+
continue;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
const msg = entry.message;
|
|
103
|
+
if (typeof entry.cwd === 'string' && entry.cwd) lastCwd = entry.cwd;
|
|
104
|
+
if (entry.type === 'user' && msg) {
|
|
105
|
+
depth += 1;
|
|
106
|
+
const text = contentText(msg.content);
|
|
107
|
+
if (text) lastUserText = text;
|
|
108
|
+
continue;
|
|
109
|
+
}
|
|
110
|
+
if (entry.type !== 'assistant' || !msg) continue;
|
|
111
|
+
depth += 1;
|
|
112
|
+
|
|
113
|
+
if (!msg.usage || !msg.id) continue;
|
|
114
|
+
// "<synthetic>" is Claude Code's placeholder for locally-generated
|
|
115
|
+
// entries (e.g. error stubs) — no real API call, nothing to price.
|
|
116
|
+
if (msg.model === '<synthetic>') continue;
|
|
117
|
+
const usage = msg.usage;
|
|
118
|
+
const reqId = entry.requestId || msg.id;
|
|
119
|
+
const r = {
|
|
120
|
+
inputTokens: usage.input_tokens || 0,
|
|
121
|
+
cacheCreationTokens: usage.cache_creation_input_tokens || 0,
|
|
122
|
+
cacheReadTokens: usage.cache_read_input_tokens || 0,
|
|
123
|
+
ephemeral5mTokens: usage.cache_creation?.ephemeral_5m_input_tokens || 0,
|
|
124
|
+
ephemeral1hTokens: usage.cache_creation?.ephemeral_1h_input_tokens || 0,
|
|
125
|
+
outputTokens: usage.output_tokens || 0,
|
|
126
|
+
};
|
|
127
|
+
|
|
128
|
+
records.set(reqId, {
|
|
129
|
+
model: normalizeModelId(msg.model),
|
|
130
|
+
timestamp: entry.timestamp || null,
|
|
131
|
+
prompt_tokens: cacheWeighted ? effectivePromptTokens(r) : rawPromptTokens(r),
|
|
132
|
+
completion_tokens: r.outputTokens,
|
|
133
|
+
depth,
|
|
134
|
+
userText: includeContent ? lastUserText : '',
|
|
135
|
+
assistantText: includeContent ? contentText(msg.content) : '',
|
|
136
|
+
cwd: lastCwd,
|
|
137
|
+
});
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
return [...records.values()];
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** Build the frugon JSONL object for one collected record. */
|
|
144
|
+
export function toFrugonRecord(rec) {
|
|
145
|
+
const msgCount = Math.max(1, Math.min(rec.depth, MAX_STUB_MESSAGES));
|
|
146
|
+
const messages = [];
|
|
147
|
+
for (let i = 0; i < msgCount - 1; i++) {
|
|
148
|
+
messages.push({ role: i % 2 === 0 ? 'user' : 'assistant', content: '' });
|
|
149
|
+
}
|
|
150
|
+
messages.push({ role: 'user', content: rec.userText || '' });
|
|
151
|
+
|
|
152
|
+
const out = {
|
|
153
|
+
model: rec.model,
|
|
154
|
+
request: { messages },
|
|
155
|
+
response: {
|
|
156
|
+
choices: [{ message: { role: 'assistant', content: rec.assistantText || '' } }],
|
|
157
|
+
},
|
|
158
|
+
usage: {
|
|
159
|
+
prompt_tokens: rec.prompt_tokens,
|
|
160
|
+
completion_tokens: rec.completion_tokens,
|
|
161
|
+
},
|
|
162
|
+
};
|
|
163
|
+
if (rec.timestamp) out.timestamp = rec.timestamp;
|
|
164
|
+
return out;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Export Claude Code transcripts to a frugon-compatible JSONL file.
|
|
169
|
+
*
|
|
170
|
+
* @param {object} options
|
|
171
|
+
* days lookback window (default 30)
|
|
172
|
+
* projectFilter substring match on the project dir name
|
|
173
|
+
* outPath output JSONL path
|
|
174
|
+
* cacheWeighted fold cache pricing into prompt_tokens (default true)
|
|
175
|
+
* includeContent include user prompt / assistant reply text (default true)
|
|
176
|
+
* @returns {Promise<{records:number, sessions:number, models:Object, outPath:string}>}
|
|
177
|
+
*/
|
|
178
|
+
export async function exportFrugonLogs({
|
|
179
|
+
days = 30,
|
|
180
|
+
projectFilter,
|
|
181
|
+
outPath,
|
|
182
|
+
cacheWeighted = true,
|
|
183
|
+
includeContent = true,
|
|
184
|
+
} = {}) {
|
|
185
|
+
const files = await discoverSessionFiles({ days, projectFilter });
|
|
186
|
+
const models = {};
|
|
187
|
+
let recordCount = 0;
|
|
188
|
+
let sessionCount = 0;
|
|
189
|
+
|
|
190
|
+
const stream = createWriteStream(outPath, { encoding: 'utf8' });
|
|
191
|
+
for (const f of files) {
|
|
192
|
+
let recs;
|
|
193
|
+
try {
|
|
194
|
+
recs = await collectSessionRecords(f.path, { cacheWeighted, includeContent });
|
|
195
|
+
} catch {
|
|
196
|
+
continue;
|
|
197
|
+
}
|
|
198
|
+
if (recs.length === 0) continue;
|
|
199
|
+
sessionCount += 1;
|
|
200
|
+
for (const rec of recs) {
|
|
201
|
+
stream.write(JSON.stringify(toFrugonRecord(rec)) + '\n');
|
|
202
|
+
models[rec.model] = (models[rec.model] || 0) + 1;
|
|
203
|
+
recordCount += 1;
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
await new Promise((resolve, reject) => {
|
|
207
|
+
stream.end((err) => (err ? reject(err) : resolve()));
|
|
208
|
+
});
|
|
209
|
+
|
|
210
|
+
return { records: recordCount, sessions: sessionCount, models, outPath };
|
|
211
|
+
}
|
package/src/harness.js
CHANGED
|
@@ -19,6 +19,7 @@ import {
|
|
|
19
19
|
harnessRatchetMdInitial,
|
|
20
20
|
appendRatchetRule,
|
|
21
21
|
} from './harness-templates.js';
|
|
22
|
+
import { routeWarningForStatusline } from './route-scan.js';
|
|
22
23
|
|
|
23
24
|
const require = createRequire(import.meta.url);
|
|
24
25
|
function readHarnessState() {
|
|
@@ -276,6 +277,82 @@ export function harnessPromote(ruleText, { root = findProjectRoot(), scope = 'pr
|
|
|
276
277
|
return { path: rmPath, root, scope };
|
|
277
278
|
}
|
|
278
279
|
|
|
280
|
+
/**
|
|
281
|
+
* harness pull — copy the user's GLOBAL ratchet rules (~/.claude/ratchet.md)
|
|
282
|
+
* into the current project's .claude/ratchet.md, on demand. Opt-in by design:
|
|
283
|
+
* install/init never auto-injects rules; this is the explicit "땡겨오기" verb.
|
|
284
|
+
*
|
|
285
|
+
* Rules are deduped by text (ignoring the leading YYYY-MM-DD stamp) so
|
|
286
|
+
* repeated pulls are idempotent. With `includeBlock`, the global CLAUDE.md
|
|
287
|
+
* harness block (including any user customizations) is also copied into the
|
|
288
|
+
* project CLAUDE.md — replacing the project's block if one exists.
|
|
289
|
+
*
|
|
290
|
+
* Returns { root, added, skippedRules, wrote, skipped }.
|
|
291
|
+
*/
|
|
292
|
+
export function harnessPull({ root = findProjectRoot(), includeBlock = false } = {}) {
|
|
293
|
+
const result = { root, added: [], skippedRules: 0, wrote: [], skipped: [] };
|
|
294
|
+
const stripDate = (t) => t.replace(/^\d{4}-\d{2}-\d{2}:\s*/, '').trim();
|
|
295
|
+
|
|
296
|
+
// 1) Ratchet rules: global → project, dedup by rule text.
|
|
297
|
+
const globalRules = harnessListRules({ scope: 'global' }).rules;
|
|
298
|
+
const projPath = ratchetMdPath(root);
|
|
299
|
+
let content = existsSync(projPath)
|
|
300
|
+
? readFileSync(projPath, 'utf8')
|
|
301
|
+
: harnessRatchetMdInitial();
|
|
302
|
+
const have = new Set(
|
|
303
|
+
harnessListRules({ root, scope: 'project' }).rules.map((r) => stripDate(r.text)),
|
|
304
|
+
);
|
|
305
|
+
for (const g of globalRules) {
|
|
306
|
+
const key = stripDate(g.text);
|
|
307
|
+
if (have.has(key)) {
|
|
308
|
+
result.skippedRules += 1;
|
|
309
|
+
continue;
|
|
310
|
+
}
|
|
311
|
+
content = appendRatchetRule(content, key);
|
|
312
|
+
have.add(key);
|
|
313
|
+
result.added.push(key);
|
|
314
|
+
}
|
|
315
|
+
if (result.added.length) {
|
|
316
|
+
mkdirSync(dirname(projPath), { recursive: true });
|
|
317
|
+
writeFileSync(projPath, content);
|
|
318
|
+
result.wrote.push(projPath);
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// 2) Harness block (opt-in): copy the global block as-is so user edits to
|
|
322
|
+
// the global sections travel with it.
|
|
323
|
+
if (includeBlock) {
|
|
324
|
+
const gPath = globalClaudeMdPath();
|
|
325
|
+
const blockRe = new RegExp(
|
|
326
|
+
`${escapeRe(HARNESS_BLOCK_BEGIN)}[\\s\\S]*?${escapeRe(HARNESS_BLOCK_END)}\\n?`,
|
|
327
|
+
'm',
|
|
328
|
+
);
|
|
329
|
+
const gContent = existsSync(gPath) ? readFileSync(gPath, 'utf8') : '';
|
|
330
|
+
const m = gContent.match(blockRe);
|
|
331
|
+
if (!m) {
|
|
332
|
+
result.skipped.push(`${gPath} (no global harness block to pull)`);
|
|
333
|
+
} else {
|
|
334
|
+
const block = m[0];
|
|
335
|
+
const pPath = claudeMdPath(root);
|
|
336
|
+
if (existsSync(pPath)) {
|
|
337
|
+
const pc = readFileSync(pPath, 'utf8');
|
|
338
|
+
if (pc.includes(HARNESS_BLOCK_BEGIN)) {
|
|
339
|
+
writeFileSync(pPath, pc.replace(blockRe, block));
|
|
340
|
+
result.wrote.push(`${pPath} (harness block replaced with global copy)`);
|
|
341
|
+
} else {
|
|
342
|
+
const sep = pc.endsWith('\n') ? '\n' : '\n\n';
|
|
343
|
+
writeFileSync(pPath, pc + sep + block);
|
|
344
|
+
result.wrote.push(`${pPath} (harness block appended from global)`);
|
|
345
|
+
}
|
|
346
|
+
} else {
|
|
347
|
+
writeFileSync(pPath, block);
|
|
348
|
+
result.wrote.push(pPath);
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
return result;
|
|
354
|
+
}
|
|
355
|
+
|
|
279
356
|
/**
|
|
280
357
|
* harness list — return numbered ratchet rules from .claude/ratchet.md.
|
|
281
358
|
* Numbering is 1-based and matches `harness rm <N>`.
|
|
@@ -362,6 +439,14 @@ export function harnessStatusForStatusline(cfg, { root } = {}) {
|
|
|
362
439
|
else if (state.pevSkip) warning = 'PEV-skip';
|
|
363
440
|
}
|
|
364
441
|
}
|
|
442
|
+
// Lowest precedence: route-scan delegation candidate (`route? R<N>`).
|
|
443
|
+
// Session-quality warnings above always win — routing is an optimization
|
|
444
|
+
// nudge, not a correctness signal. Cheap: one small cached-JSON read.
|
|
445
|
+
if (!warning) {
|
|
446
|
+
try {
|
|
447
|
+
warning = routeWarningForStatusline(projectRoot);
|
|
448
|
+
} catch { /* scan cache unreadable — stay silent */ }
|
|
449
|
+
}
|
|
365
450
|
return { ...status, warning };
|
|
366
451
|
}
|
|
367
452
|
|
package/src/history.js
CHANGED
|
@@ -69,7 +69,8 @@ function saveState(state) {
|
|
|
69
69
|
function chipKo(chip) {
|
|
70
70
|
if (!chip) return chip;
|
|
71
71
|
const map = {
|
|
72
|
-
'⚠
|
|
72
|
+
'⚠ Ctx 200k+': '⚠ 컨텍스트 200k 초과',
|
|
73
|
+
'⚠ 1M ON': '⚠ 1M 컨텍스트 활성', // legacy (pre-v2.18)
|
|
73
74
|
'⚠ Cache miss': '⚠ 캐시 미스',
|
|
74
75
|
'⚠ Rebuild churn': '⚠ 캐시 재빌드 빈발',
|
|
75
76
|
'⚠ Input spike': '⚠ 입력 급증',
|
|
@@ -92,6 +93,9 @@ function chipKo(chip) {
|
|
|
92
93
|
*/
|
|
93
94
|
function detailKo(detail) {
|
|
94
95
|
if (!detail) return detail;
|
|
96
|
+
const m0 = detail.match(/^Single-request context exceeded 200k \(max (\d+)k tokens\)$/);
|
|
97
|
+
if (m0) return `단일 요청 컨텍스트 200k 초과 (최대 ${m0[1]}k 토큰)`;
|
|
98
|
+
// legacy detail shape (pre-v2.18)
|
|
95
99
|
const m1 = detail.match(/^Context auto-promoted to 1M \(max single-request (\d+)k tokens\)$/);
|
|
96
100
|
if (m1) return `1M 컨텍스트 자동 활성 (단일 요청 최대 ${m1[1]}k 토큰)`;
|
|
97
101
|
const m2 = detail.match(/^session ([^:]+): (.+)$/);
|
package/src/installer.js
CHANGED
|
@@ -202,10 +202,47 @@ export function installStatusline({ force = false } = {}) {
|
|
|
202
202
|
return { path: file, action: 'updated', reason: 'replaced previous statusLine' };
|
|
203
203
|
}
|
|
204
204
|
|
|
205
|
+
// Registers the SessionStart hook that surfaces route-scan delegation
|
|
206
|
+
// candidates as session context (startup + /clear). Idempotent: skips when a
|
|
207
|
+
// claude-token-saver route-scan hook is already present; never touches other
|
|
208
|
+
// hooks the user configured.
|
|
209
|
+
const ROUTE_SCAN_HOOK_COMMAND = 'claude-token-saver route-scan --hook';
|
|
210
|
+
|
|
211
|
+
export function installSessionStartHook() {
|
|
212
|
+
const dir = claudeUserDir();
|
|
213
|
+
const file = join(dir, 'settings.json');
|
|
214
|
+
mkdirSync(dir, { recursive: true });
|
|
215
|
+
|
|
216
|
+
let settings = {};
|
|
217
|
+
if (existsSync(file)) {
|
|
218
|
+
try {
|
|
219
|
+
settings = JSON.parse(readFileSync(file, 'utf8'));
|
|
220
|
+
} catch (e) {
|
|
221
|
+
return { path: file, action: 'skipped', reason: `unreadable JSON (${e.message})` };
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
settings.hooks = settings.hooks || {};
|
|
226
|
+
const list = Array.isArray(settings.hooks.SessionStart) ? settings.hooks.SessionStart : [];
|
|
227
|
+
const already = list.some((m) =>
|
|
228
|
+
Array.isArray(m?.hooks) && m.hooks.some((h) => typeof h?.command === 'string' && h.command.includes('route-scan --hook')),
|
|
229
|
+
);
|
|
230
|
+
if (already) return { path: file, action: 'exists' };
|
|
231
|
+
|
|
232
|
+
list.push({
|
|
233
|
+
matcher: 'startup|clear',
|
|
234
|
+
hooks: [{ type: 'command', command: ROUTE_SCAN_HOOK_COMMAND, timeout: 10 }],
|
|
235
|
+
});
|
|
236
|
+
settings.hooks.SessionStart = list;
|
|
237
|
+
writeFileSync(file, JSON.stringify(settings, null, 2) + '\n');
|
|
238
|
+
return { path: file, action: 'created' };
|
|
239
|
+
}
|
|
240
|
+
|
|
205
241
|
export function installAll({ force = false } = {}) {
|
|
206
242
|
return {
|
|
207
243
|
skill: installSkill({ force }),
|
|
208
244
|
statusline: installStatusline({ force }),
|
|
245
|
+
sessionStartHook: installSessionStartHook(),
|
|
209
246
|
legacy: removeLegacyCommand(),
|
|
210
247
|
};
|
|
211
248
|
}
|