context-doctor 0.17.0 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -15
- package/dist/accuracy.d.ts +6 -4
- package/dist/accuracy.js +9 -7
- package/dist/calibration.d.ts +8 -0
- package/dist/calibration.js +17 -3
- package/dist/cli.js +13 -4
- package/dist/hook.js +7 -4
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/mcp.js +5 -5
- package/dist/optimize.d.ts +5 -0
- package/dist/optimize.js +32 -22
- package/dist/parse.d.ts +2 -0
- package/dist/parse.js +2 -1
- package/dist/preferences.d.ts +1 -1
- package/dist/preferences.js +1 -1
- package/dist/profile.js +9 -2
- package/dist/proxy.d.ts +15 -0
- package/dist/proxy.js +45 -8
- package/dist/sketch.d.ts +7 -3
- package/dist/sketch.js +51 -30
- package/dist/tokenizer-measure.d.ts +42 -0
- package/dist/tokenizer-measure.js +172 -0
- package/dist/tokens.d.ts +31 -1
- package/dist/tokens.js +39 -5
- package/package.json +1 -1
package/dist/profile.js
CHANGED
|
@@ -146,11 +146,18 @@ function filesReadBy(toolName, toolCallText) {
|
|
|
146
146
|
return [...paths];
|
|
147
147
|
}
|
|
148
148
|
export function profileConversation(conv, model) {
|
|
149
|
+
// An explicit model wins; otherwise the request's own model field decides
|
|
150
|
+
// which tokenizer ratios apply.
|
|
151
|
+
model = model ?? conv.model;
|
|
152
|
+
// With no model at all, Anthropic's request format still says whose
|
|
153
|
+
// tokenizer counts it. Only the ratios use this; pricing and the window
|
|
154
|
+
// stay unknown rather than guessed.
|
|
155
|
+
const tokenizerModel = model ?? (conv.sourceFormat === "anthropic" ? "claude" : undefined);
|
|
149
156
|
// Learned from the user's own exact counts, if they ever fetched any.
|
|
150
|
-
const calibration = calibrationFor(
|
|
157
|
+
const calibration = calibrationFor(tokenizerModel);
|
|
151
158
|
const perMessage = conv.messages.map((m) => ({
|
|
152
159
|
msg: m,
|
|
153
|
-
tokens: Math.round(estimateTokens(m.text) * calibration.factor) + MESSAGE_OVERHEAD_TOKENS,
|
|
160
|
+
tokens: Math.round(estimateTokens(m.text, tokenizerModel) * calibration.factor) + MESSAGE_OVERHEAD_TOKENS,
|
|
154
161
|
}));
|
|
155
162
|
const totalTokens = perMessage.reduce((sum, p) => sum + p.tokens, 0);
|
|
156
163
|
const categories = {
|
package/dist/proxy.d.ts
CHANGED
|
@@ -23,9 +23,24 @@ export interface ProxyOptions extends OptimizeOptions {
|
|
|
23
23
|
* the user explicitly opts in (e.g. --host 0.0.0.0 inside a container).
|
|
24
24
|
*/
|
|
25
25
|
host?: string;
|
|
26
|
+
/**
|
|
27
|
+
* When set, every request except /health must arrive under the path prefix
|
|
28
|
+
* `/t/<token>/`, which is stripped before routing. This is what makes the
|
|
29
|
+
* proxy safe to put on a public URL (a tunnel) for apps whose servers call
|
|
30
|
+
* the base URL, such as Cursor with your own OpenAI key: those apps can set a
|
|
31
|
+
* URL but not a header, so the secret rides in the path. Compared with
|
|
32
|
+
* constant time; a wrong or missing prefix gets 401 and no upstream call.
|
|
33
|
+
*/
|
|
34
|
+
token?: string;
|
|
26
35
|
anthropicUpstream?: string;
|
|
27
36
|
openaiUpstream?: string;
|
|
28
37
|
}
|
|
38
|
+
/**
|
|
39
|
+
* Remove a leading `/t/<token>` from a request path, or return undefined when
|
|
40
|
+
* the prefix is absent or the token differs. The comparison is constant time
|
|
41
|
+
* so the token cannot be guessed a character at a time.
|
|
42
|
+
*/
|
|
43
|
+
export declare function stripToken(url: string, token: string): string | undefined;
|
|
29
44
|
export interface ProxyStats {
|
|
30
45
|
startedAt: string;
|
|
31
46
|
requests: number;
|
package/dist/proxy.js
CHANGED
|
@@ -12,13 +12,31 @@
|
|
|
12
12
|
* Streaming responses are piped through unchanged.
|
|
13
13
|
*/
|
|
14
14
|
import http from "node:http";
|
|
15
|
+
import { timingSafeEqual } from "node:crypto";
|
|
15
16
|
import { optimizeConversation } from "./optimize.js";
|
|
16
|
-
import { formatTokens } from "./tokens.js";
|
|
17
|
+
import { formatTokens, CHARS_PER_TOKEN, providerFor } from "./tokens.js";
|
|
17
18
|
import { formatUsd, inputCostUsd, pricingFor } from "./pricing.js";
|
|
18
19
|
import { recordLedger } from "./ledger.js";
|
|
19
20
|
/** Connection-level headers that must not be forwarded. */
|
|
20
21
|
const SKIP_REQUEST_HEADERS = new Set(["host", "content-length", "connection", "transfer-encoding", "accept-encoding", "expect"]);
|
|
21
22
|
const SKIP_RESPONSE_HEADERS = new Set(["content-length", "content-encoding", "transfer-encoding", "connection"]);
|
|
23
|
+
/**
|
|
24
|
+
* Remove a leading `/t/<token>` from a request path, or return undefined when
|
|
25
|
+
* the prefix is absent or the token differs. The comparison is constant time
|
|
26
|
+
* so the token cannot be guessed a character at a time.
|
|
27
|
+
*/
|
|
28
|
+
export function stripToken(url, token) {
|
|
29
|
+
const prefix = "/t/";
|
|
30
|
+
if (!url.startsWith(prefix))
|
|
31
|
+
return undefined;
|
|
32
|
+
const end = url.indexOf("/", prefix.length);
|
|
33
|
+
const candidate = end === -1 ? url.slice(prefix.length) : url.slice(prefix.length, end);
|
|
34
|
+
const a = Buffer.from(candidate), b = Buffer.from(token);
|
|
35
|
+
if (a.length !== b.length || !timingSafeEqual(a, b))
|
|
36
|
+
return undefined;
|
|
37
|
+
const rest = end === -1 ? "/" : url.slice(end);
|
|
38
|
+
return rest;
|
|
39
|
+
}
|
|
22
40
|
function upstreamFor(url, opts) {
|
|
23
41
|
if (url.startsWith("/v1/messages"))
|
|
24
42
|
return opts.anthropicUpstream ?? "https://api.anthropic.com";
|
|
@@ -75,13 +93,23 @@ export function startProxy(opts = {}) {
|
|
|
75
93
|
console.error(`[context-doctor] cache advisor: ${msg}`);
|
|
76
94
|
};
|
|
77
95
|
const server = http.createServer(async (req, res) => {
|
|
78
|
-
|
|
96
|
+
let url = req.url ?? "/";
|
|
79
97
|
try {
|
|
80
98
|
if (url === "/health") {
|
|
81
99
|
res.setHeader("content-type", "application/json");
|
|
82
100
|
res.end(JSON.stringify({ ok: true, service: "context-doctor-proxy" }));
|
|
83
101
|
return;
|
|
84
102
|
}
|
|
103
|
+
if (opts.token) {
|
|
104
|
+
const stripped = stripToken(url, opts.token);
|
|
105
|
+
if (stripped === undefined) {
|
|
106
|
+
res.statusCode = 401;
|
|
107
|
+
res.setHeader("content-type", "application/json");
|
|
108
|
+
res.end(JSON.stringify({ error: "context-doctor proxy: this proxy requires its token in the path: /t/<token>/v1/..." }));
|
|
109
|
+
return;
|
|
110
|
+
}
|
|
111
|
+
url = stripped;
|
|
112
|
+
}
|
|
85
113
|
if (url === "/stats") {
|
|
86
114
|
res.setHeader("content-type", "application/json");
|
|
87
115
|
res.end(JSON.stringify({ ...stats, estUsdSaved: Number(stats.estUsdSaved.toFixed(4)) }, null, 2));
|
|
@@ -127,7 +155,9 @@ export function startProxy(opts = {}) {
|
|
|
127
155
|
if (url.startsWith("/v1/messages") && requestModel) {
|
|
128
156
|
const stablePrefix = JSON.stringify(parsedBody.tools ?? null) + JSON.stringify(parsedBody.system ?? null);
|
|
129
157
|
const hasBreakpoint = body.includes("cache_control");
|
|
130
|
-
|
|
158
|
+
const stablePrefixTokens = Math.round(stablePrefix.length / CHARS_PER_TOKEN[providerFor(requestModel)].code);
|
|
159
|
+
// Anthropic will not cache a prefix under ~1024 tokens; below that the advice is useless.
|
|
160
|
+
if (stablePrefixTokens >= 1024 && !hasBreakpoint) {
|
|
131
161
|
// Say WHERE, not just that. A breakpoint caches everything up
|
|
132
162
|
// to and including the block it sits on, so it belongs on the
|
|
133
163
|
// LAST stable block: the final tool definition if there are
|
|
@@ -135,7 +165,7 @@ export function startProxy(opts = {}) {
|
|
|
135
165
|
const where = Array.isArray(parsedBody.tools) && parsedBody.tools.length > 0
|
|
136
166
|
? `the last entry in "tools" (tools come before system in the cached prefix)`
|
|
137
167
|
: `the last block of "system"`;
|
|
138
|
-
advise(`~${
|
|
168
|
+
advise(`~${stablePrefixTokens}+ tokens of stable system/tools on ${requestModel} without cache_control. ` +
|
|
139
169
|
`Add {"cache_control":{"type":"ephemeral"}} to ${where}; everything before it then bills at ~10% on every call`);
|
|
140
170
|
}
|
|
141
171
|
const fp = fnv1a(stablePrefix);
|
|
@@ -160,7 +190,7 @@ export function startProxy(opts = {}) {
|
|
|
160
190
|
stable++;
|
|
161
191
|
if (stable >= 2) {
|
|
162
192
|
const stableChars = msgs.slice(0, stable).reduce((n, m) => n + JSON.stringify(m).length, 0);
|
|
163
|
-
const stableTokens = Math.round(stableChars /
|
|
193
|
+
const stableTokens = Math.round(stableChars / CHARS_PER_TOKEN[providerFor(requestModel)].code);
|
|
164
194
|
// Anthropic will not cache a prefix under ~1024 tokens (2048 on Haiku).
|
|
165
195
|
if (stableTokens >= 1024) {
|
|
166
196
|
advise(`messages #0-#${stable - 1} (~${stableTokens} tokens) were identical to the previous ${requestModel} request and carry no cache_control. ` +
|
|
@@ -278,11 +308,18 @@ export function startProxy(opts = {}) {
|
|
|
278
308
|
server.on("close", checkpoint);
|
|
279
309
|
const host = opts.host ?? "127.0.0.1";
|
|
280
310
|
server.listen(port, host, () => {
|
|
311
|
+
// Print the token as <token>, never the value: this log is what people paste into bug reports.
|
|
312
|
+
const prefix = opts.token ? "/t/<token>" : "";
|
|
281
313
|
console.error(`context-doctor proxy listening on http://${host}:${port}`);
|
|
282
|
-
console.error(` Anthropic apps/SDKs: export ANTHROPIC_BASE_URL=http://localhost:${port}`);
|
|
283
|
-
console.error(` OpenAI apps/SDKs: export OPENAI_BASE_URL=http://localhost:${port}/v1`);
|
|
314
|
+
console.error(` Anthropic apps/SDKs: export ANTHROPIC_BASE_URL=http://localhost:${port}${prefix}`);
|
|
315
|
+
console.error(` OpenAI apps/SDKs: export OPENAI_BASE_URL=http://localhost:${port}${prefix}/v1`);
|
|
284
316
|
console.error(` Every request's context is optimized in flight; savings are logged here.`);
|
|
285
|
-
console.error(` Cumulative savings: http://localhost:${port}/stats`);
|
|
317
|
+
console.error(` Cumulative savings: http://localhost:${port}${prefix}/stats`);
|
|
318
|
+
if (opts.token) {
|
|
319
|
+
console.error(` Token required: every path except /health must start with /t/<token>/.`);
|
|
320
|
+
console.error(` Cursor with your own OpenAI key: expose this port on HTTPS (a tunnel), then Settings > Models > OpenAI API Key >`);
|
|
321
|
+
console.error(` "Override OpenAI Base URL" = https://<your-host>/t/<token>/v1. Cursor's servers call that URL, so 127.0.0.1 will not work there.`);
|
|
322
|
+
}
|
|
286
323
|
});
|
|
287
324
|
return server;
|
|
288
325
|
}
|
package/dist/sketch.d.ts
CHANGED
|
@@ -6,8 +6,10 @@
|
|
|
6
6
|
* JSON means re-typing 50k+ tokens as a tool argument. No model does that, and
|
|
7
7
|
* it would double the context it is meant to measure. A sketch is ~100 output
|
|
8
8
|
* tokens: turn count plus the handful of blocks that matter (pastes, tool
|
|
9
|
-
* results, images, repeats).
|
|
10
|
-
*
|
|
9
|
+
* results, images, repeats). Measured error: -20% to +9% on the conversation
|
|
10
|
+
* total from turn count alone, ±25% on a code block sized by lines, ±15% on
|
|
11
|
+
* one sized by chars. Coarse, stated, and enough to find what to drop; it turns
|
|
12
|
+
* "call profile_context" from an impossible instruction into a cheap one.
|
|
11
13
|
*/
|
|
12
14
|
export type SketchKind = "paste" | "code" | "tool_result" | "image" | "base64" | "text";
|
|
13
15
|
export interface SketchBlock {
|
|
@@ -52,7 +54,9 @@ export interface SketchProfile {
|
|
|
52
54
|
perTurnUsd?: number;
|
|
53
55
|
perTurnCachedUsd?: number;
|
|
54
56
|
}
|
|
55
|
-
export declare function blockTokens(b: SketchBlock): number;
|
|
57
|
+
export declare function blockTokens(b: SketchBlock, model?: string): number;
|
|
58
|
+
/** Tokens for one plain exchange under this model's tokenizer. */
|
|
59
|
+
export declare function exchangeTokens(model?: string): number;
|
|
56
60
|
export declare function profileSketch(sketch: ConversationSketch): SketchProfile;
|
|
57
61
|
export declare function renderSketchProfile(p: SketchProfile): string;
|
|
58
62
|
/** Profile a sketch, log it to the ledger like a hook check, and render. */
|
package/dist/sketch.js
CHANGED
|
@@ -6,51 +6,72 @@
|
|
|
6
6
|
* JSON means re-typing 50k+ tokens as a tool argument. No model does that, and
|
|
7
7
|
* it would double the context it is meant to measure. A sketch is ~100 output
|
|
8
8
|
* tokens: turn count plus the handful of blocks that matter (pastes, tool
|
|
9
|
-
* results, images, repeats).
|
|
10
|
-
*
|
|
9
|
+
* results, images, repeats). Measured error: -20% to +9% on the conversation
|
|
10
|
+
* total from turn count alone, ±25% on a code block sized by lines, ±15% on
|
|
11
|
+
* one sized by chars. Coarse, stated, and enough to find what to drop; it turns
|
|
12
|
+
* "call profile_context" from an impossible instruction into a cheap one.
|
|
11
13
|
*/
|
|
12
|
-
import { contextWindowFor } from "./tokens.js";
|
|
13
|
-
import { formatTokens } from "./tokens.js";
|
|
14
|
+
import { CHARS_PER_TOKEN, contextWindowFor, formatTokens, providerFor } from "./tokens.js";
|
|
14
15
|
import { formatUsd, inputCostUsd, pricingFor } from "./pricing.js";
|
|
15
16
|
import { recordLedger } from "./ledger.js";
|
|
16
|
-
//
|
|
17
|
-
//
|
|
18
|
-
//
|
|
19
|
-
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
|
|
23
|
-
|
|
17
|
+
// Every size here is in CHARACTERS, measured on real data, and converted to
|
|
18
|
+
// tokens with the model's own ratio (tokens.ts), so a Claude chat and a GPT
|
|
19
|
+
// chat of the same text get different, correct counts.
|
|
20
|
+
//
|
|
21
|
+
// A plain exchange: the user's message plus the assistant's reply. Median over
|
|
22
|
+
// 1,283 exchanges in 59 Claude Code sessions (p25 1,177, p75 3,008). Sizing a
|
|
23
|
+
// 30+ exchange chat from its turn count alone landed within -20% to +9% of the
|
|
24
|
+
// real visible total (p10 to p90, 12 sessions): individual turns vary a lot,
|
|
25
|
+
// long chats average it out.
|
|
26
|
+
const CHARS_PER_EXCHANGE = 2060;
|
|
27
|
+
// Mean chars per non-empty line. Code: median over 719 source files (a line
|
|
28
|
+
// count for code is within about ±25%). Logs and tool output: median over
|
|
29
|
+
// 2,095 tool results, but they range 38 to 100 chars a line, so the schema
|
|
30
|
+
// asks for chars or tokens on those, and lines are the fallback.
|
|
31
|
+
const CHARS_PER_LINE = {
|
|
32
|
+
code: 42, tool_result: 56, paste: 56, text: 80, base64: 76, image: 0,
|
|
24
33
|
};
|
|
25
|
-
|
|
26
|
-
const
|
|
27
|
-
// When a block carries no size at all.
|
|
28
|
-
const
|
|
29
|
-
|
|
34
|
+
// Prose, including the space after each word (6.3 measured on assistant text).
|
|
35
|
+
const CHARS_PER_WORD = 6.3;
|
|
36
|
+
// When a block carries no size at all.
|
|
37
|
+
const DEFAULT_CHARS = {
|
|
38
|
+
paste: 2400, code: 2400, tool_result: 2400, text: 900, base64: 12000, image: 0,
|
|
30
39
|
};
|
|
40
|
+
// Images are billed by pixels, not bytes: ~1,600 tokens for a full-size
|
|
41
|
+
// screenshot on Claude, fewer on smaller images and on GPT. One figure is enough
|
|
42
|
+
// for a sketch.
|
|
43
|
+
const IMAGE_TOKENS = 1500;
|
|
31
44
|
const LARGE_BLOCK_TOKENS = 2000;
|
|
32
45
|
const LONG_HISTORY_TURNS = 30;
|
|
33
46
|
const HANDOFF_SUMMARY_TOKENS = 300;
|
|
34
47
|
const RECENT_TURNS_KEPT = 6;
|
|
35
|
-
|
|
48
|
+
/** Code, tool output and base64 tokenize at the code ratio; pastes and text at the prose ratio. */
|
|
49
|
+
function ratioFor(kind, model) {
|
|
50
|
+
const r = CHARS_PER_TOKEN[providerFor(model)];
|
|
51
|
+
return kind === "code" || kind === "tool_result" || kind === "base64" ? r.code : r.prose;
|
|
52
|
+
}
|
|
53
|
+
export function blockTokens(b, model) {
|
|
36
54
|
if (b.kind === "image")
|
|
37
|
-
return b.approx_tokens
|
|
55
|
+
return b.approx_tokens && b.approx_tokens > 0 ? Math.round(b.approx_tokens) : IMAGE_TOKENS;
|
|
38
56
|
if (b.approx_tokens && b.approx_tokens > 0)
|
|
39
57
|
return Math.round(b.approx_tokens);
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
58
|
+
const chars = b.approx_chars && b.approx_chars > 0 ? b.approx_chars
|
|
59
|
+
: b.approx_lines && b.approx_lines > 0 ? b.approx_lines * CHARS_PER_LINE[b.kind]
|
|
60
|
+
: b.approx_words && b.approx_words > 0 ? b.approx_words * CHARS_PER_WORD
|
|
61
|
+
: DEFAULT_CHARS[b.kind];
|
|
62
|
+
return Math.round(chars / ratioFor(b.kind, model));
|
|
63
|
+
}
|
|
64
|
+
/** Tokens for one plain exchange under this model's tokenizer. */
|
|
65
|
+
export function exchangeTokens(model) {
|
|
66
|
+
return Math.round(CHARS_PER_EXCHANGE / CHARS_PER_TOKEN[providerFor(model)].prose);
|
|
47
67
|
}
|
|
48
68
|
export function profileSketch(sketch) {
|
|
49
69
|
const turns = Math.max(0, Math.floor(sketch.turns || 0));
|
|
50
70
|
const blocks = Array.isArray(sketch.blocks) ? sketch.blocks : [];
|
|
51
|
-
const
|
|
71
|
+
const perExchange = exchangeTokens(sketch.model);
|
|
72
|
+
const baselineTokens = turns * perExchange;
|
|
52
73
|
// A repeated block costs its size every time it appears.
|
|
53
|
-
const sized = blocks.map((b) => ({ block: b, tokens: blockTokens(b), copies: Math.max(1, Math.floor(b.repeated ?? 1)) }));
|
|
74
|
+
const sized = blocks.map((b) => ({ block: b, tokens: blockTokens(b, sketch.model), copies: Math.max(1, Math.floor(b.repeated ?? 1)) }));
|
|
54
75
|
const blockTotal = sized.reduce((n, s) => n + s.tokens * s.copies, 0);
|
|
55
76
|
const totalTokens = baselineTokens + blockTotal;
|
|
56
77
|
const findings = [];
|
|
@@ -101,7 +122,7 @@ export function profileSketch(sketch) {
|
|
|
101
122
|
});
|
|
102
123
|
}
|
|
103
124
|
if (turns >= LONG_HISTORY_TURNS) {
|
|
104
|
-
const afterHandoff = RECENT_TURNS_KEPT *
|
|
125
|
+
const afterHandoff = RECENT_TURNS_KEPT * perExchange + HANDOFF_SUMMARY_TOKENS;
|
|
105
126
|
findings.push({
|
|
106
127
|
id: "long_history",
|
|
107
128
|
severity: totalTokens > 100_000 ? "high" : "warn",
|
|
@@ -140,7 +161,7 @@ export function renderSketchProfile(p) {
|
|
|
140
161
|
const lines = [];
|
|
141
162
|
const window = p.usagePct !== undefined ? ` (~${p.usagePct}% of ${formatTokens(p.contextWindow)})` : "";
|
|
142
163
|
lines.push(`Context estimate from sketch: ~${formatTokens(p.totalTokens)} tokens${window}, ${p.turns} turns.`);
|
|
143
|
-
lines.push(` ${formatTokens(p.baselineTokens)} plain conversation + ${formatTokens(p.blockTokens)} in pastes, tool output and images. Estimate, ±
|
|
164
|
+
lines.push(` ${formatTokens(p.baselineTokens)} plain conversation + ${formatTokens(p.blockTokens)} in pastes, tool output and images. Estimate, usually within ±20%.`);
|
|
144
165
|
if (p.perTurnUsd !== undefined) {
|
|
145
166
|
lines.push(` Re-read on every turn: ${formatUsd(p.perTurnUsd)} at list price, ${formatUsd(p.perTurnCachedUsd)} when cached. On a subscription this is what spends the usage limit.`);
|
|
146
167
|
}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Measure a model's real chars-per-token from Claude Code transcripts, with no
|
|
3
|
+
* API key and no tokenizer: the API's own counts are already in the file.
|
|
4
|
+
*
|
|
5
|
+
* Two independent measurements, so one can check the other:
|
|
6
|
+
*
|
|
7
|
+
* - PROSE. An assistant reply with no thinking block is billed as exactly
|
|
8
|
+
* `output_tokens`, and all of it is visible text in the transcript. Visible
|
|
9
|
+
* chars / output_tokens is the tokenizer's ratio on the model's own prose.
|
|
10
|
+
*
|
|
11
|
+
* - BLOCKS (code, tool output, pastes). Between two consecutive API calls in
|
|
12
|
+
* one session the prompt is the old prompt plus what was appended: the
|
|
13
|
+
* previous reply (all of `output_tokens`, thinking included, since a tool
|
|
14
|
+
* loop re-sends it) and the new user-side content. When that content is a
|
|
15
|
+
* single large block, (growth − previous output_tokens) is its exact size.
|
|
16
|
+
*
|
|
17
|
+
* The injected reminders the harness adds make the block figure slightly
|
|
18
|
+
* pessimistic (a few dozen tokens on blocks of thousands), which is why only
|
|
19
|
+
* blocks over 6k chars count.
|
|
20
|
+
*/
|
|
21
|
+
export interface RatioStats {
|
|
22
|
+
samples: number;
|
|
23
|
+
median: number;
|
|
24
|
+
p10: number;
|
|
25
|
+
p90: number;
|
|
26
|
+
}
|
|
27
|
+
export interface ModelRatios {
|
|
28
|
+
model: string;
|
|
29
|
+
prose?: RatioStats;
|
|
30
|
+
blocks?: RatioStats;
|
|
31
|
+
/** What the estimator uses for this model: prose, code. */
|
|
32
|
+
assumed: {
|
|
33
|
+
prose: number;
|
|
34
|
+
code: number;
|
|
35
|
+
};
|
|
36
|
+
}
|
|
37
|
+
export interface TokenizerReport {
|
|
38
|
+
sessionsScanned: number;
|
|
39
|
+
models: ModelRatios[];
|
|
40
|
+
}
|
|
41
|
+
export declare function measureTokenizer(limit?: number, paths?: string[]): TokenizerReport;
|
|
42
|
+
export declare function renderTokenizer(report: TokenizerReport): string;
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Measure a model's real chars-per-token from Claude Code transcripts, with no
|
|
3
|
+
* API key and no tokenizer: the API's own counts are already in the file.
|
|
4
|
+
*
|
|
5
|
+
* Two independent measurements, so one can check the other:
|
|
6
|
+
*
|
|
7
|
+
* - PROSE. An assistant reply with no thinking block is billed as exactly
|
|
8
|
+
* `output_tokens`, and all of it is visible text in the transcript. Visible
|
|
9
|
+
* chars / output_tokens is the tokenizer's ratio on the model's own prose.
|
|
10
|
+
*
|
|
11
|
+
* - BLOCKS (code, tool output, pastes). Between two consecutive API calls in
|
|
12
|
+
* one session the prompt is the old prompt plus what was appended: the
|
|
13
|
+
* previous reply (all of `output_tokens`, thinking included, since a tool
|
|
14
|
+
* loop re-sends it) and the new user-side content. When that content is a
|
|
15
|
+
* single large block, (growth − previous output_tokens) is its exact size.
|
|
16
|
+
*
|
|
17
|
+
* The injected reminders the harness adds make the block figure slightly
|
|
18
|
+
* pessimistic (a few dozen tokens on blocks of thousands), which is why only
|
|
19
|
+
* blocks over 6k chars count.
|
|
20
|
+
*/
|
|
21
|
+
import { forEachLine, listSessions } from "./session.js";
|
|
22
|
+
import { CHARS_PER_TOKEN, providerFor } from "./tokens.js";
|
|
23
|
+
const MIN_PROSE_CHARS = 1500;
|
|
24
|
+
const MIN_BLOCK_CHARS = 6000;
|
|
25
|
+
function stats(values) {
|
|
26
|
+
if (values.length === 0)
|
|
27
|
+
return undefined;
|
|
28
|
+
const v = [...values].sort((a, b) => a - b);
|
|
29
|
+
const at = (q) => v[Math.min(v.length - 1, Math.floor(q * (v.length - 1)))];
|
|
30
|
+
const mid = Math.floor(v.length / 2);
|
|
31
|
+
const median = v.length % 2 ? v[mid] : (v[mid - 1] + v[mid]) / 2;
|
|
32
|
+
return { samples: v.length, median, p10: at(0.1), p90: at(0.9) };
|
|
33
|
+
}
|
|
34
|
+
function promptTotal(u) {
|
|
35
|
+
const n = (k) => (typeof u[k] === "number" ? u[k] : 0);
|
|
36
|
+
return n("input_tokens") + n("cache_read_input_tokens") + n("cache_creation_input_tokens");
|
|
37
|
+
}
|
|
38
|
+
function userText(content) {
|
|
39
|
+
if (typeof content === "string")
|
|
40
|
+
return content;
|
|
41
|
+
if (!Array.isArray(content))
|
|
42
|
+
return "";
|
|
43
|
+
let out = "";
|
|
44
|
+
for (const p of content) {
|
|
45
|
+
if (p?.type === "text" && typeof p.text === "string")
|
|
46
|
+
out += p.text;
|
|
47
|
+
else if (p?.type === "tool_result") {
|
|
48
|
+
const c = p.content;
|
|
49
|
+
if (typeof c === "string")
|
|
50
|
+
out += c;
|
|
51
|
+
else if (Array.isArray(c))
|
|
52
|
+
for (const x of c)
|
|
53
|
+
if (typeof x?.text === "string")
|
|
54
|
+
out += x.text;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
return out;
|
|
58
|
+
}
|
|
59
|
+
/** Scan one transcript into per-request records. Never throws on bad lines. */
|
|
60
|
+
function readRequests(path) {
|
|
61
|
+
const reqs = [];
|
|
62
|
+
let cur;
|
|
63
|
+
forEachLine(path, (line) => {
|
|
64
|
+
let e;
|
|
65
|
+
try {
|
|
66
|
+
e = JSON.parse(line);
|
|
67
|
+
}
|
|
68
|
+
catch {
|
|
69
|
+
return;
|
|
70
|
+
}
|
|
71
|
+
if (!e || e.isSidechain)
|
|
72
|
+
return;
|
|
73
|
+
if (e.type === "assistant" && e.message && typeof e.message === "object") {
|
|
74
|
+
const m = e.message;
|
|
75
|
+
const u = (m.usage ?? {});
|
|
76
|
+
if (!cur || cur.id !== m.id) {
|
|
77
|
+
cur = { id: String(m.id), model: m.model, prompt: promptTotal(u), output: 0, types: new Set(), textChars: 0, after: [] };
|
|
78
|
+
reqs.push(cur);
|
|
79
|
+
}
|
|
80
|
+
if (typeof u.output_tokens === "number")
|
|
81
|
+
cur.output = Math.max(cur.output, u.output_tokens);
|
|
82
|
+
for (const p of Array.isArray(m.content) ? m.content : []) {
|
|
83
|
+
cur.types.add(String(p?.type));
|
|
84
|
+
if (p?.type === "text" && typeof p.text === "string")
|
|
85
|
+
cur.textChars += p.text.length;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
else if (e.type === "user" && cur) {
|
|
89
|
+
const meta = Boolean(e.isMeta || e.isCompactSummary);
|
|
90
|
+
const text = userText(e.message?.content);
|
|
91
|
+
cur.after.push({ chars: text.length, meta });
|
|
92
|
+
}
|
|
93
|
+
});
|
|
94
|
+
return reqs;
|
|
95
|
+
}
|
|
96
|
+
export function measureTokenizer(limit = 60, paths) {
|
|
97
|
+
const prose = new Map();
|
|
98
|
+
const blocks = new Map();
|
|
99
|
+
const push = (m, k, v) => (m.get(k) ?? m.set(k, []).get(k)).push(v);
|
|
100
|
+
let scanned = 0;
|
|
101
|
+
const targets = paths ?? listSessions(limit).map((s) => s.path);
|
|
102
|
+
for (const path of targets) {
|
|
103
|
+
let reqs;
|
|
104
|
+
try {
|
|
105
|
+
reqs = readRequests(path);
|
|
106
|
+
}
|
|
107
|
+
catch {
|
|
108
|
+
continue;
|
|
109
|
+
}
|
|
110
|
+
if (reqs.length === 0)
|
|
111
|
+
continue;
|
|
112
|
+
scanned++;
|
|
113
|
+
for (const r of reqs) {
|
|
114
|
+
if (!r.model || r.model.startsWith("<"))
|
|
115
|
+
continue;
|
|
116
|
+
if (r.types.size === 1 && r.types.has("text") && r.textChars >= MIN_PROSE_CHARS && r.output > 0) {
|
|
117
|
+
push(prose, r.model, r.textChars / r.output);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
for (let i = 0; i + 1 < reqs.length; i++) {
|
|
121
|
+
const a = reqs[i], b = reqs[i + 1];
|
|
122
|
+
if (!a.model || a.model !== b.model)
|
|
123
|
+
continue;
|
|
124
|
+
const real = a.after.filter((x) => !x.meta);
|
|
125
|
+
if (a.after.some((x) => x.meta) || real.length !== 1 || real[0].chars < MIN_BLOCK_CHARS)
|
|
126
|
+
continue;
|
|
127
|
+
const exact = b.prompt - a.prompt - a.output;
|
|
128
|
+
if (exact <= 0)
|
|
129
|
+
continue; // a compaction or cache reset, not growth
|
|
130
|
+
push(blocks, a.model, real[0].chars / exact);
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
const models = [...new Set([...prose.keys(), ...blocks.keys()])]
|
|
134
|
+
.map((model) => ({
|
|
135
|
+
model,
|
|
136
|
+
prose: stats(prose.get(model) ?? []),
|
|
137
|
+
blocks: stats(blocks.get(model) ?? []),
|
|
138
|
+
assumed: CHARS_PER_TOKEN[providerFor(model)],
|
|
139
|
+
}))
|
|
140
|
+
.sort((x, y) => ((y.prose?.samples ?? 0) + (y.blocks?.samples ?? 0)) - ((x.prose?.samples ?? 0) + (x.blocks?.samples ?? 0)));
|
|
141
|
+
return { sessionsScanned: scanned, models };
|
|
142
|
+
}
|
|
143
|
+
export function renderTokenizer(report) {
|
|
144
|
+
const lines = [];
|
|
145
|
+
lines.push("Tokenizer check — chars per token, measured from the API's own counts");
|
|
146
|
+
lines.push("─".repeat(56));
|
|
147
|
+
if (report.models.length === 0) {
|
|
148
|
+
lines.push("No usable samples (needs Claude Code sessions with usage recorded).");
|
|
149
|
+
return lines.join("\n");
|
|
150
|
+
}
|
|
151
|
+
const f = (n) => n.toFixed(2);
|
|
152
|
+
const err = (assumed, real) => {
|
|
153
|
+
// Estimated tokens / real tokens − 1, from chars/token on each side.
|
|
154
|
+
const pct = Math.round((real / assumed - 1) * 100);
|
|
155
|
+
return pct === 0 ? "exact" : `estimates ${pct > 0 ? "+" : ""}${pct}%`;
|
|
156
|
+
};
|
|
157
|
+
const row = (label, st, unit, assumed) => ` ${label.padEnd(7)} ${f(st.median)} ${`(p10 ${f(st.p10)}, p90 ${f(st.p90)}, ${st.samples} ${unit})`.padEnd(34)} ` +
|
|
158
|
+
`estimator ${f(assumed)} → ${err(assumed, st.median)}`;
|
|
159
|
+
for (const m of report.models) {
|
|
160
|
+
lines.push(m.model);
|
|
161
|
+
if (m.prose) {
|
|
162
|
+
lines.push(row("prose", m.prose, "replies", m.assumed.prose));
|
|
163
|
+
}
|
|
164
|
+
if (m.blocks) {
|
|
165
|
+
lines.push(row("blocks", m.blocks, "blocks", m.assumed.code));
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
lines.push("");
|
|
169
|
+
lines.push("\"estimates +x%\" means the estimator reports x% more tokens than the API bills for");
|
|
170
|
+
lines.push("the same text (negative: fewer). Blocks are mostly code and tool output.");
|
|
171
|
+
return lines.join("\n");
|
|
172
|
+
}
|
package/dist/tokens.d.ts
CHANGED
|
@@ -11,7 +11,37 @@
|
|
|
11
11
|
export type Provider = "anthropic" | "openai" | "google" | "generic";
|
|
12
12
|
export declare function contextWindowFor(model?: string): number | undefined;
|
|
13
13
|
export declare function providerFor(model?: string): Provider;
|
|
14
|
-
|
|
14
|
+
/**
|
|
15
|
+
* Characters per token, by provider and by content type.
|
|
16
|
+
*
|
|
17
|
+
* Anthropic: measured 2026-09 against the API's own counts in 59 Claude Code
|
|
18
|
+
* sessions (Opus 4.7 to 5, Fable 5.x), two independent ways that agree.
|
|
19
|
+
* Prose: 504 assistant replies with no thinking block, visible text divided by
|
|
20
|
+
* the exact output_tokens: median 2.75 (p10 2.4, p90 3.0). Code and tool
|
|
21
|
+
* output: 474 single appended blocks over 6k chars, sized by the exact growth
|
|
22
|
+
* of the billed prompt between consecutive calls: median 2.4 (p10 2.1, p90 2.8).
|
|
23
|
+
* The ratios this tool used before (4.0 / 3.2) undercounted current Claude
|
|
24
|
+
* models by about 1.45x on prose and 1.33x on code.
|
|
25
|
+
*
|
|
26
|
+
* OpenAI, Google and unknown models keep 4.0 / 3.2, the usual figures for
|
|
27
|
+
* o200k-class tokenizers on English and code. They are not re-measured here:
|
|
28
|
+
* Codex rollouts truncate tool output before the model sees it, so the same
|
|
29
|
+
* delta method does not isolate a block. `analyze --exact` calibrates any
|
|
30
|
+
* provider from its own tokenizer on your machine.
|
|
31
|
+
*/
|
|
32
|
+
export declare const CHARS_PER_TOKEN: Record<Provider, {
|
|
33
|
+
prose: number;
|
|
34
|
+
code: number;
|
|
35
|
+
}>;
|
|
36
|
+
/** True when text is dense with code/JSON symbols and tokenizes more finely. */
|
|
37
|
+
export declare function isCodeLike(text: string): boolean;
|
|
38
|
+
/** Chars per token to use for this text under this model's tokenizer. */
|
|
39
|
+
export declare function charsPerTokenFor(text: string, model?: string): number;
|
|
40
|
+
/**
|
|
41
|
+
* Estimated tokens for `text`. Pass the model when you know it: Claude's
|
|
42
|
+
* tokenizer produces ~40% more tokens than the provider-neutral default.
|
|
43
|
+
*/
|
|
44
|
+
export declare function estimateTokens(text: string, model?: string): number;
|
|
15
45
|
/** Per-message structural overhead (role markers, delimiters) is roughly constant. */
|
|
16
46
|
export declare const MESSAGE_OVERHEAD_TOKENS = 4;
|
|
17
47
|
export declare function formatTokens(n: number): string;
|
package/dist/tokens.js
CHANGED
|
@@ -49,17 +49,51 @@ function symbolDensity(text) {
|
|
|
49
49
|
const symbols = text.match(/[{}[\]()<>;:=_\/\\|"'`#$%&*+^~-]/g);
|
|
50
50
|
return (symbols?.length ?? 0) / text.length;
|
|
51
51
|
}
|
|
52
|
-
|
|
52
|
+
/**
|
|
53
|
+
* Characters per token, by provider and by content type.
|
|
54
|
+
*
|
|
55
|
+
* Anthropic: measured 2026-09 against the API's own counts in 59 Claude Code
|
|
56
|
+
* sessions (Opus 4.7 to 5, Fable 5.x), two independent ways that agree.
|
|
57
|
+
* Prose: 504 assistant replies with no thinking block, visible text divided by
|
|
58
|
+
* the exact output_tokens: median 2.75 (p10 2.4, p90 3.0). Code and tool
|
|
59
|
+
* output: 474 single appended blocks over 6k chars, sized by the exact growth
|
|
60
|
+
* of the billed prompt between consecutive calls: median 2.4 (p10 2.1, p90 2.8).
|
|
61
|
+
* The ratios this tool used before (4.0 / 3.2) undercounted current Claude
|
|
62
|
+
* models by about 1.45x on prose and 1.33x on code.
|
|
63
|
+
*
|
|
64
|
+
* OpenAI, Google and unknown models keep 4.0 / 3.2, the usual figures for
|
|
65
|
+
* o200k-class tokenizers on English and code. They are not re-measured here:
|
|
66
|
+
* Codex rollouts truncate tool output before the model sees it, so the same
|
|
67
|
+
* delta method does not isolate a block. `analyze --exact` calibrates any
|
|
68
|
+
* provider from its own tokenizer on your machine.
|
|
69
|
+
*/
|
|
70
|
+
export const CHARS_PER_TOKEN = {
|
|
71
|
+
anthropic: { prose: 2.75, code: 2.4 },
|
|
72
|
+
openai: { prose: 4.0, code: 3.2 },
|
|
73
|
+
google: { prose: 4.0, code: 3.2 },
|
|
74
|
+
generic: { prose: 4.0, code: 3.2 },
|
|
75
|
+
};
|
|
76
|
+
/** True when text is dense with code/JSON symbols and tokenizes more finely. */
|
|
77
|
+
export function isCodeLike(text) {
|
|
78
|
+
return symbolDensity(text) > 0.08;
|
|
79
|
+
}
|
|
80
|
+
/** Chars per token to use for this text under this model's tokenizer. */
|
|
81
|
+
export function charsPerTokenFor(text, model) {
|
|
82
|
+
const ratios = CHARS_PER_TOKEN[providerFor(model)];
|
|
83
|
+
return isCodeLike(text) ? ratios.code : ratios.prose;
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* Estimated tokens for `text`. Pass the model when you know it: Claude's
|
|
87
|
+
* tokenizer produces ~40% more tokens than the provider-neutral default.
|
|
88
|
+
*/
|
|
89
|
+
export function estimateTokens(text, model) {
|
|
53
90
|
// Public API: callers outside this package pass whatever they have, and a
|
|
54
91
|
// TypeError from a token estimator is never the useful answer.
|
|
55
92
|
if (typeof text !== "string")
|
|
56
93
|
text = String(text ?? "");
|
|
57
94
|
if (!text)
|
|
58
95
|
return 0;
|
|
59
|
-
|
|
60
|
-
const density = symbolDensity(text);
|
|
61
|
-
const charsPerToken = density > 0.08 ? 3.2 : 4.0;
|
|
62
|
-
return Math.ceil(text.length / charsPerToken);
|
|
96
|
+
return Math.ceil(text.length / charsPerTokenFor(text, model));
|
|
63
97
|
}
|
|
64
98
|
/** Per-message structural overhead (role markers, delimiters) is roughly constant. */
|
|
65
99
|
export const MESSAGE_OVERHEAD_TOKENS = 4;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "context-doctor",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.19.0",
|
|
4
4
|
"description": "Profile and optimize LLM context windows. See what's eating your tokens and fix it — works with Claude Code, Claude Desktop, Cursor, Codex (OpenAI), and any MCP-capable AI app.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude",
|