context-doctor 0.17.0 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/profile.js CHANGED
@@ -146,11 +146,18 @@ function filesReadBy(toolName, toolCallText) {
146
146
  return [...paths];
147
147
  }
148
148
  export function profileConversation(conv, model) {
149
+ // An explicit model wins; otherwise the request's own model field decides
150
+ // which tokenizer ratios apply.
151
+ model = model ?? conv.model;
152
+ // With no model at all, Anthropic's request format still says whose
153
+ // tokenizer counts it. Only the ratios use this; pricing and the window
154
+ // stay unknown rather than guessed.
155
+ const tokenizerModel = model ?? (conv.sourceFormat === "anthropic" ? "claude" : undefined);
149
156
  // Learned from the user's own exact counts, if they ever fetched any.
150
- const calibration = calibrationFor(model);
157
+ const calibration = calibrationFor(tokenizerModel);
151
158
  const perMessage = conv.messages.map((m) => ({
152
159
  msg: m,
153
- tokens: Math.round(estimateTokens(m.text) * calibration.factor) + MESSAGE_OVERHEAD_TOKENS,
160
+ tokens: Math.round(estimateTokens(m.text, tokenizerModel) * calibration.factor) + MESSAGE_OVERHEAD_TOKENS,
154
161
  }));
155
162
  const totalTokens = perMessage.reduce((sum, p) => sum + p.tokens, 0);
156
163
  const categories = {
package/dist/proxy.d.ts CHANGED
@@ -23,9 +23,24 @@ export interface ProxyOptions extends OptimizeOptions {
23
23
  * the user explicitly opts in (e.g. --host 0.0.0.0 inside a container).
24
24
  */
25
25
  host?: string;
26
+ /**
27
+ * When set, every request except /health must arrive under the path prefix
28
+ * `/t/<token>/`, which is stripped before routing. This is what makes the
29
+ * proxy safe to put on a public URL (a tunnel) for apps whose servers call
30
+ * the base URL, such as Cursor with your own OpenAI key: those apps can set a
31
+ * URL but not a header, so the secret rides in the path. Compared with
32
+ * constant time; a wrong or missing prefix gets 401 and no upstream call.
33
+ */
34
+ token?: string;
26
35
  anthropicUpstream?: string;
27
36
  openaiUpstream?: string;
28
37
  }
38
+ /**
39
+ * Remove a leading `/t/<token>` from a request path, or return undefined when
40
+ * the prefix is absent or the token differs. The comparison is constant time
41
+ * so the token cannot be guessed a character at a time.
42
+ */
43
+ export declare function stripToken(url: string, token: string): string | undefined;
29
44
  export interface ProxyStats {
30
45
  startedAt: string;
31
46
  requests: number;
package/dist/proxy.js CHANGED
@@ -12,13 +12,31 @@
12
12
  * Streaming responses are piped through unchanged.
13
13
  */
14
14
  import http from "node:http";
15
+ import { timingSafeEqual } from "node:crypto";
15
16
  import { optimizeConversation } from "./optimize.js";
16
- import { formatTokens } from "./tokens.js";
17
+ import { formatTokens, CHARS_PER_TOKEN, providerFor } from "./tokens.js";
17
18
  import { formatUsd, inputCostUsd, pricingFor } from "./pricing.js";
18
19
  import { recordLedger } from "./ledger.js";
19
20
  /** Connection-level headers that must not be forwarded. */
20
21
  const SKIP_REQUEST_HEADERS = new Set(["host", "content-length", "connection", "transfer-encoding", "accept-encoding", "expect"]);
21
22
  const SKIP_RESPONSE_HEADERS = new Set(["content-length", "content-encoding", "transfer-encoding", "connection"]);
23
+ /**
24
+ * Remove a leading `/t/<token>` from a request path, or return undefined when
25
+ * the prefix is absent or the token differs. The comparison is constant time
26
+ * so the token cannot be guessed a character at a time.
27
+ */
28
+ export function stripToken(url, token) {
29
+ const prefix = "/t/";
30
+ if (!url.startsWith(prefix))
31
+ return undefined;
32
+ const end = url.indexOf("/", prefix.length);
33
+ const candidate = end === -1 ? url.slice(prefix.length) : url.slice(prefix.length, end);
34
+ const a = Buffer.from(candidate), b = Buffer.from(token);
35
+ if (a.length !== b.length || !timingSafeEqual(a, b))
36
+ return undefined;
37
+ const rest = end === -1 ? "/" : url.slice(end);
38
+ return rest;
39
+ }
22
40
  function upstreamFor(url, opts) {
23
41
  if (url.startsWith("/v1/messages"))
24
42
  return opts.anthropicUpstream ?? "https://api.anthropic.com";
@@ -75,13 +93,23 @@ export function startProxy(opts = {}) {
75
93
  console.error(`[context-doctor] cache advisor: ${msg}`);
76
94
  };
77
95
  const server = http.createServer(async (req, res) => {
78
- const url = req.url ?? "/";
96
+ let url = req.url ?? "/";
79
97
  try {
80
98
  if (url === "/health") {
81
99
  res.setHeader("content-type", "application/json");
82
100
  res.end(JSON.stringify({ ok: true, service: "context-doctor-proxy" }));
83
101
  return;
84
102
  }
103
+ if (opts.token) {
104
+ const stripped = stripToken(url, opts.token);
105
+ if (stripped === undefined) {
106
+ res.statusCode = 401;
107
+ res.setHeader("content-type", "application/json");
108
+ res.end(JSON.stringify({ error: "context-doctor proxy: this proxy requires its token in the path: /t/<token>/v1/..." }));
109
+ return;
110
+ }
111
+ url = stripped;
112
+ }
85
113
  if (url === "/stats") {
86
114
  res.setHeader("content-type", "application/json");
87
115
  res.end(JSON.stringify({ ...stats, estUsdSaved: Number(stats.estUsdSaved.toFixed(4)) }, null, 2));
@@ -127,7 +155,9 @@ export function startProxy(opts = {}) {
127
155
  if (url.startsWith("/v1/messages") && requestModel) {
128
156
  const stablePrefix = JSON.stringify(parsedBody.tools ?? null) + JSON.stringify(parsedBody.system ?? null);
129
157
  const hasBreakpoint = body.includes("cache_control");
130
- if (stablePrefix.length > 4000 && !hasBreakpoint) {
158
+ const stablePrefixTokens = Math.round(stablePrefix.length / CHARS_PER_TOKEN[providerFor(requestModel)].code);
159
+ // Anthropic will not cache a prefix under ~1024 tokens; below that the advice is useless.
160
+ if (stablePrefixTokens >= 1024 && !hasBreakpoint) {
131
161
  // Say WHERE, not just that. A breakpoint caches everything up
132
162
  // to and including the block it sits on, so it belongs on the
133
163
  // LAST stable block: the final tool definition if there are
@@ -135,7 +165,7 @@ export function startProxy(opts = {}) {
135
165
  const where = Array.isArray(parsedBody.tools) && parsedBody.tools.length > 0
136
166
  ? `the last entry in "tools" (tools come before system in the cached prefix)`
137
167
  : `the last block of "system"`;
138
- advise(`~${Math.round(stablePrefix.length / 4)}+ tokens of stable system/tools on ${requestModel} without cache_control. ` +
168
+ advise(`~${stablePrefixTokens}+ tokens of stable system/tools on ${requestModel} without cache_control. ` +
139
169
  `Add {"cache_control":{"type":"ephemeral"}} to ${where}; everything before it then bills at ~10% on every call`);
140
170
  }
141
171
  const fp = fnv1a(stablePrefix);
@@ -160,7 +190,7 @@ export function startProxy(opts = {}) {
160
190
  stable++;
161
191
  if (stable >= 2) {
162
192
  const stableChars = msgs.slice(0, stable).reduce((n, m) => n + JSON.stringify(m).length, 0);
163
- const stableTokens = Math.round(stableChars / 4);
193
+ const stableTokens = Math.round(stableChars / CHARS_PER_TOKEN[providerFor(requestModel)].code);
164
194
  // Anthropic will not cache a prefix under ~1024 tokens (2048 on Haiku).
165
195
  if (stableTokens >= 1024) {
166
196
  advise(`messages #0-#${stable - 1} (~${stableTokens} tokens) were identical to the previous ${requestModel} request and carry no cache_control. ` +
@@ -278,11 +308,18 @@ export function startProxy(opts = {}) {
278
308
  server.on("close", checkpoint);
279
309
  const host = opts.host ?? "127.0.0.1";
280
310
  server.listen(port, host, () => {
311
+ // Print the token as <token>, never the value: this log is what people paste into bug reports.
312
+ const prefix = opts.token ? "/t/<token>" : "";
281
313
  console.error(`context-doctor proxy listening on http://${host}:${port}`);
282
- console.error(` Anthropic apps/SDKs: export ANTHROPIC_BASE_URL=http://localhost:${port}`);
283
- console.error(` OpenAI apps/SDKs: export OPENAI_BASE_URL=http://localhost:${port}/v1`);
314
+ console.error(` Anthropic apps/SDKs: export ANTHROPIC_BASE_URL=http://localhost:${port}${prefix}`);
315
+ console.error(` OpenAI apps/SDKs: export OPENAI_BASE_URL=http://localhost:${port}${prefix}/v1`);
284
316
  console.error(` Every request's context is optimized in flight; savings are logged here.`);
285
- console.error(` Cumulative savings: http://localhost:${port}/stats`);
317
+ console.error(` Cumulative savings: http://localhost:${port}${prefix}/stats`);
318
+ if (opts.token) {
319
+ console.error(` Token required: every path except /health must start with /t/<token>/.`);
320
+ console.error(` Cursor with your own OpenAI key: expose this port on HTTPS (a tunnel), then Settings > Models > OpenAI API Key >`);
321
+ console.error(` "Override OpenAI Base URL" = https://<your-host>/t/<token>/v1. Cursor's servers call that URL, so 127.0.0.1 will not work there.`);
322
+ }
286
323
  });
287
324
  return server;
288
325
  }
package/dist/sketch.d.ts CHANGED
@@ -6,8 +6,10 @@
6
6
  * JSON means re-typing 50k+ tokens as a tool argument. No model does that, and
7
7
  * it would double the context it is meant to measure. A sketch is ~100 output
8
8
  * tokens: turn count plus the handful of blocks that matter (pastes, tool
9
- * results, images, repeats). The estimate is coarse (±30%) and says so, but it
10
- * turns "call profile_context" from an impossible instruction into a cheap one.
9
+ * results, images, repeats). Measured error: -20% to +9% on the conversation
10
+ * total from turn count alone, ±25% on a code block sized by lines, ±15% on
11
+ * one sized by chars. Coarse, stated, and enough to find what to drop; it turns
12
+ * "call profile_context" from an impossible instruction into a cheap one.
11
13
  */
12
14
  export type SketchKind = "paste" | "code" | "tool_result" | "image" | "base64" | "text";
13
15
  export interface SketchBlock {
@@ -52,7 +54,9 @@ export interface SketchProfile {
52
54
  perTurnUsd?: number;
53
55
  perTurnCachedUsd?: number;
54
56
  }
55
- export declare function blockTokens(b: SketchBlock): number;
57
+ export declare function blockTokens(b: SketchBlock, model?: string): number;
58
+ /** Tokens for one plain exchange under this model's tokenizer. */
59
+ export declare function exchangeTokens(model?: string): number;
56
60
  export declare function profileSketch(sketch: ConversationSketch): SketchProfile;
57
61
  export declare function renderSketchProfile(p: SketchProfile): string;
58
62
  /** Profile a sketch, log it to the ledger like a hook check, and render. */
package/dist/sketch.js CHANGED
@@ -6,51 +6,72 @@
6
6
  * JSON means re-typing 50k+ tokens as a tool argument. No model does that, and
7
7
  * it would double the context it is meant to measure. A sketch is ~100 output
8
8
  * tokens: turn count plus the handful of blocks that matter (pastes, tool
9
- * results, images, repeats). The estimate is coarse (±30%) and says so, but it
10
- * turns "call profile_context" from an impossible instruction into a cheap one.
9
+ * results, images, repeats). Measured error: -20% to +9% on the conversation
10
+ * total from turn count alone, ±25% on a code block sized by lines, ±15% on
11
+ * one sized by chars. Coarse, stated, and enough to find what to drop; it turns
12
+ * "call profile_context" from an impossible instruction into a cheap one.
11
13
  */
12
- import { contextWindowFor } from "./tokens.js";
13
- import { formatTokens } from "./tokens.js";
14
+ import { CHARS_PER_TOKEN, contextWindowFor, formatTokens, providerFor } from "./tokens.js";
14
15
  import { formatUsd, inputCostUsd, pricingFor } from "./pricing.js";
15
16
  import { recordLedger } from "./ledger.js";
16
- // A plain chat turn without attachments: a short user message and a normal
17
- // assistant reply. Measured across Claude Code transcripts the median user turn
18
- // is ~120 tokens and the median assistant turn ~450; chat apps run similar.
19
- const BASELINE_TOKENS_PER_TURN = 570;
20
- // Tokens per unit when the model reports size in lines/words/chars. Code and
21
- // tool output are denser per line than prose; words are ~1.35 tokens each.
22
- const TOKENS_PER_LINE = {
23
- code: 12, tool_result: 12, paste: 14, text: 14, base64: 40, image: 0,
17
+ // Every size here is in CHARACTERS, measured on real data, and converted to
18
+ // tokens with the model's own ratio (tokens.ts), so a Claude chat and a GPT
19
+ // chat of the same text get different, correct counts.
20
+ //
21
+ // A plain exchange: the user's message plus the assistant's reply. Median over
22
+ // 1,283 exchanges in 59 Claude Code sessions (p25 1,177, p75 3,008). Sizing a
23
+ // 30+ exchange chat from its turn count alone landed within -20% to +9% of the
24
+ // real visible total (p10 to p90, 12 sessions): individual turns vary a lot,
25
+ // long chats average it out.
26
+ const CHARS_PER_EXCHANGE = 2060;
27
+ // Mean chars per non-empty line. Code: median over 719 source files (a line
28
+ // count for code is within about ±25%). Logs and tool output: median over
29
+ // 2,095 tool results, but they range 38 to 100 chars a line, so the schema
30
+ // asks for chars or tokens on those, and lines are the fallback.
31
+ const CHARS_PER_LINE = {
32
+ code: 42, tool_result: 56, paste: 56, text: 80, base64: 76, image: 0,
24
33
  };
25
- const TOKENS_PER_WORD = 1.35;
26
- const CHARS_PER_TOKEN = 4;
27
- // When a block carries no size at all. Images are billed at a near-fixed rate.
28
- const DEFAULT_TOKENS = {
29
- image: 1500, paste: 800, code: 800, tool_result: 800, base64: 4000, text: 300,
34
+ // Prose, including the space after each word (6.3 measured on assistant text).
35
+ const CHARS_PER_WORD = 6.3;
36
+ // When a block carries no size at all.
37
+ const DEFAULT_CHARS = {
38
+ paste: 2400, code: 2400, tool_result: 2400, text: 900, base64: 12000, image: 0,
30
39
  };
40
+ // Images are billed by pixels, not bytes: ~1,600 tokens for a full-size
41
+ // screenshot on Claude, fewer on smaller images and on GPT. One figure is enough
42
+ // for a sketch.
43
+ const IMAGE_TOKENS = 1500;
31
44
  const LARGE_BLOCK_TOKENS = 2000;
32
45
  const LONG_HISTORY_TURNS = 30;
33
46
  const HANDOFF_SUMMARY_TOKENS = 300;
34
47
  const RECENT_TURNS_KEPT = 6;
35
- export function blockTokens(b) {
48
+ /** Code, tool output and base64 tokenize at the code ratio; pastes and text at the prose ratio. */
49
+ function ratioFor(kind, model) {
50
+ const r = CHARS_PER_TOKEN[providerFor(model)];
51
+ return kind === "code" || kind === "tool_result" || kind === "base64" ? r.code : r.prose;
52
+ }
53
+ export function blockTokens(b, model) {
36
54
  if (b.kind === "image")
37
- return b.approx_tokens ?? DEFAULT_TOKENS.image;
55
+ return b.approx_tokens && b.approx_tokens > 0 ? Math.round(b.approx_tokens) : IMAGE_TOKENS;
38
56
  if (b.approx_tokens && b.approx_tokens > 0)
39
57
  return Math.round(b.approx_tokens);
40
- if (b.approx_lines && b.approx_lines > 0)
41
- return Math.round(b.approx_lines * TOKENS_PER_LINE[b.kind]);
42
- if (b.approx_words && b.approx_words > 0)
43
- return Math.round(b.approx_words * TOKENS_PER_WORD);
44
- if (b.approx_chars && b.approx_chars > 0)
45
- return Math.round(b.approx_chars / CHARS_PER_TOKEN);
46
- return DEFAULT_TOKENS[b.kind];
58
+ const chars = b.approx_chars && b.approx_chars > 0 ? b.approx_chars
59
+ : b.approx_lines && b.approx_lines > 0 ? b.approx_lines * CHARS_PER_LINE[b.kind]
60
+ : b.approx_words && b.approx_words > 0 ? b.approx_words * CHARS_PER_WORD
61
+ : DEFAULT_CHARS[b.kind];
62
+ return Math.round(chars / ratioFor(b.kind, model));
63
+ }
64
+ /** Tokens for one plain exchange under this model's tokenizer. */
65
+ export function exchangeTokens(model) {
66
+ return Math.round(CHARS_PER_EXCHANGE / CHARS_PER_TOKEN[providerFor(model)].prose);
47
67
  }
48
68
  export function profileSketch(sketch) {
49
69
  const turns = Math.max(0, Math.floor(sketch.turns || 0));
50
70
  const blocks = Array.isArray(sketch.blocks) ? sketch.blocks : [];
51
- const baselineTokens = turns * BASELINE_TOKENS_PER_TURN;
71
+ const perExchange = exchangeTokens(sketch.model);
72
+ const baselineTokens = turns * perExchange;
52
73
  // A repeated block costs its size every time it appears.
53
- const sized = blocks.map((b) => ({ block: b, tokens: blockTokens(b), copies: Math.max(1, Math.floor(b.repeated ?? 1)) }));
74
+ const sized = blocks.map((b) => ({ block: b, tokens: blockTokens(b, sketch.model), copies: Math.max(1, Math.floor(b.repeated ?? 1)) }));
54
75
  const blockTotal = sized.reduce((n, s) => n + s.tokens * s.copies, 0);
55
76
  const totalTokens = baselineTokens + blockTotal;
56
77
  const findings = [];
@@ -101,7 +122,7 @@ export function profileSketch(sketch) {
101
122
  });
102
123
  }
103
124
  if (turns >= LONG_HISTORY_TURNS) {
104
- const afterHandoff = RECENT_TURNS_KEPT * BASELINE_TOKENS_PER_TURN + HANDOFF_SUMMARY_TOKENS;
125
+ const afterHandoff = RECENT_TURNS_KEPT * perExchange + HANDOFF_SUMMARY_TOKENS;
105
126
  findings.push({
106
127
  id: "long_history",
107
128
  severity: totalTokens > 100_000 ? "high" : "warn",
@@ -140,7 +161,7 @@ export function renderSketchProfile(p) {
140
161
  const lines = [];
141
162
  const window = p.usagePct !== undefined ? ` (~${p.usagePct}% of ${formatTokens(p.contextWindow)})` : "";
142
163
  lines.push(`Context estimate from sketch: ~${formatTokens(p.totalTokens)} tokens${window}, ${p.turns} turns.`);
143
- lines.push(` ${formatTokens(p.baselineTokens)} plain conversation + ${formatTokens(p.blockTokens)} in pastes, tool output and images. Estimate, ±30%.`);
164
+ lines.push(` ${formatTokens(p.baselineTokens)} plain conversation + ${formatTokens(p.blockTokens)} in pastes, tool output and images. Estimate, usually within ±20%.`);
144
165
  if (p.perTurnUsd !== undefined) {
145
166
  lines.push(` Re-read on every turn: ${formatUsd(p.perTurnUsd)} at list price, ${formatUsd(p.perTurnCachedUsd)} when cached. On a subscription this is what spends the usage limit.`);
146
167
  }
@@ -0,0 +1,42 @@
1
+ /**
2
+ * Measure a model's real chars-per-token from Claude Code transcripts, with no
3
+ * API key and no tokenizer: the API's own counts are already in the file.
4
+ *
5
+ * Two independent measurements, so one can check the other:
6
+ *
7
+ * - PROSE. An assistant reply with no thinking block is billed as exactly
8
+ * `output_tokens`, and all of it is visible text in the transcript. Visible
9
+ * chars / output_tokens is the tokenizer's ratio on the model's own prose.
10
+ *
11
+ * - BLOCKS (code, tool output, pastes). Between two consecutive API calls in
12
+ * one session the prompt is the old prompt plus what was appended: the
13
+ * previous reply (all of `output_tokens`, thinking included, since a tool
14
+ * loop re-sends it) and the new user-side content. When that content is a
15
+ * single large block, (growth − previous output_tokens) is its exact size.
16
+ *
17
+ * The injected reminders the harness adds make the block figure slightly
18
+ * pessimistic (a few dozen tokens on blocks of thousands), which is why only
19
+ * blocks over 6k chars count.
20
+ */
21
+ export interface RatioStats {
22
+ samples: number;
23
+ median: number;
24
+ p10: number;
25
+ p90: number;
26
+ }
27
+ export interface ModelRatios {
28
+ model: string;
29
+ prose?: RatioStats;
30
+ blocks?: RatioStats;
31
+ /** What the estimator uses for this model: prose, code. */
32
+ assumed: {
33
+ prose: number;
34
+ code: number;
35
+ };
36
+ }
37
+ export interface TokenizerReport {
38
+ sessionsScanned: number;
39
+ models: ModelRatios[];
40
+ }
41
+ export declare function measureTokenizer(limit?: number, paths?: string[]): TokenizerReport;
42
+ export declare function renderTokenizer(report: TokenizerReport): string;
@@ -0,0 +1,172 @@
1
+ /**
2
+ * Measure a model's real chars-per-token from Claude Code transcripts, with no
3
+ * API key and no tokenizer: the API's own counts are already in the file.
4
+ *
5
+ * Two independent measurements, so one can check the other:
6
+ *
7
+ * - PROSE. An assistant reply with no thinking block is billed as exactly
8
+ * `output_tokens`, and all of it is visible text in the transcript. Visible
9
+ * chars / output_tokens is the tokenizer's ratio on the model's own prose.
10
+ *
11
+ * - BLOCKS (code, tool output, pastes). Between two consecutive API calls in
12
+ * one session the prompt is the old prompt plus what was appended: the
13
+ * previous reply (all of `output_tokens`, thinking included, since a tool
14
+ * loop re-sends it) and the new user-side content. When that content is a
15
+ * single large block, (growth − previous output_tokens) is its exact size.
16
+ *
17
+ * The injected reminders the harness adds make the block figure slightly
18
+ * pessimistic (a few dozen tokens on blocks of thousands), which is why only
19
+ * blocks over 6k chars count.
20
+ */
21
+ import { forEachLine, listSessions } from "./session.js";
22
+ import { CHARS_PER_TOKEN, providerFor } from "./tokens.js";
23
+ const MIN_PROSE_CHARS = 1500;
24
+ const MIN_BLOCK_CHARS = 6000;
25
+ function stats(values) {
26
+ if (values.length === 0)
27
+ return undefined;
28
+ const v = [...values].sort((a, b) => a - b);
29
+ const at = (q) => v[Math.min(v.length - 1, Math.floor(q * (v.length - 1)))];
30
+ const mid = Math.floor(v.length / 2);
31
+ const median = v.length % 2 ? v[mid] : (v[mid - 1] + v[mid]) / 2;
32
+ return { samples: v.length, median, p10: at(0.1), p90: at(0.9) };
33
+ }
34
+ function promptTotal(u) {
35
+ const n = (k) => (typeof u[k] === "number" ? u[k] : 0);
36
+ return n("input_tokens") + n("cache_read_input_tokens") + n("cache_creation_input_tokens");
37
+ }
38
+ function userText(content) {
39
+ if (typeof content === "string")
40
+ return content;
41
+ if (!Array.isArray(content))
42
+ return "";
43
+ let out = "";
44
+ for (const p of content) {
45
+ if (p?.type === "text" && typeof p.text === "string")
46
+ out += p.text;
47
+ else if (p?.type === "tool_result") {
48
+ const c = p.content;
49
+ if (typeof c === "string")
50
+ out += c;
51
+ else if (Array.isArray(c))
52
+ for (const x of c)
53
+ if (typeof x?.text === "string")
54
+ out += x.text;
55
+ }
56
+ }
57
+ return out;
58
+ }
59
+ /** Scan one transcript into per-request records. Never throws on bad lines. */
60
+ function readRequests(path) {
61
+ const reqs = [];
62
+ let cur;
63
+ forEachLine(path, (line) => {
64
+ let e;
65
+ try {
66
+ e = JSON.parse(line);
67
+ }
68
+ catch {
69
+ return;
70
+ }
71
+ if (!e || e.isSidechain)
72
+ return;
73
+ if (e.type === "assistant" && e.message && typeof e.message === "object") {
74
+ const m = e.message;
75
+ const u = (m.usage ?? {});
76
+ if (!cur || cur.id !== m.id) {
77
+ cur = { id: String(m.id), model: m.model, prompt: promptTotal(u), output: 0, types: new Set(), textChars: 0, after: [] };
78
+ reqs.push(cur);
79
+ }
80
+ if (typeof u.output_tokens === "number")
81
+ cur.output = Math.max(cur.output, u.output_tokens);
82
+ for (const p of Array.isArray(m.content) ? m.content : []) {
83
+ cur.types.add(String(p?.type));
84
+ if (p?.type === "text" && typeof p.text === "string")
85
+ cur.textChars += p.text.length;
86
+ }
87
+ }
88
+ else if (e.type === "user" && cur) {
89
+ const meta = Boolean(e.isMeta || e.isCompactSummary);
90
+ const text = userText(e.message?.content);
91
+ cur.after.push({ chars: text.length, meta });
92
+ }
93
+ });
94
+ return reqs;
95
+ }
96
+ export function measureTokenizer(limit = 60, paths) {
97
+ const prose = new Map();
98
+ const blocks = new Map();
99
+ const push = (m, k, v) => (m.get(k) ?? m.set(k, []).get(k)).push(v);
100
+ let scanned = 0;
101
+ const targets = paths ?? listSessions(limit).map((s) => s.path);
102
+ for (const path of targets) {
103
+ let reqs;
104
+ try {
105
+ reqs = readRequests(path);
106
+ }
107
+ catch {
108
+ continue;
109
+ }
110
+ if (reqs.length === 0)
111
+ continue;
112
+ scanned++;
113
+ for (const r of reqs) {
114
+ if (!r.model || r.model.startsWith("<"))
115
+ continue;
116
+ if (r.types.size === 1 && r.types.has("text") && r.textChars >= MIN_PROSE_CHARS && r.output > 0) {
117
+ push(prose, r.model, r.textChars / r.output);
118
+ }
119
+ }
120
+ for (let i = 0; i + 1 < reqs.length; i++) {
121
+ const a = reqs[i], b = reqs[i + 1];
122
+ if (!a.model || a.model !== b.model)
123
+ continue;
124
+ const real = a.after.filter((x) => !x.meta);
125
+ if (a.after.some((x) => x.meta) || real.length !== 1 || real[0].chars < MIN_BLOCK_CHARS)
126
+ continue;
127
+ const exact = b.prompt - a.prompt - a.output;
128
+ if (exact <= 0)
129
+ continue; // a compaction or cache reset, not growth
130
+ push(blocks, a.model, real[0].chars / exact);
131
+ }
132
+ }
133
+ const models = [...new Set([...prose.keys(), ...blocks.keys()])]
134
+ .map((model) => ({
135
+ model,
136
+ prose: stats(prose.get(model) ?? []),
137
+ blocks: stats(blocks.get(model) ?? []),
138
+ assumed: CHARS_PER_TOKEN[providerFor(model)],
139
+ }))
140
+ .sort((x, y) => ((y.prose?.samples ?? 0) + (y.blocks?.samples ?? 0)) - ((x.prose?.samples ?? 0) + (x.blocks?.samples ?? 0)));
141
+ return { sessionsScanned: scanned, models };
142
+ }
143
+ export function renderTokenizer(report) {
144
+ const lines = [];
145
+ lines.push("Tokenizer check — chars per token, measured from the API's own counts");
146
+ lines.push("─".repeat(56));
147
+ if (report.models.length === 0) {
148
+ lines.push("No usable samples (needs Claude Code sessions with usage recorded).");
149
+ return lines.join("\n");
150
+ }
151
+ const f = (n) => n.toFixed(2);
152
+ const err = (assumed, real) => {
153
+ // Estimated tokens / real tokens − 1, from chars/token on each side.
154
+ const pct = Math.round((real / assumed - 1) * 100);
155
+ return pct === 0 ? "exact" : `estimates ${pct > 0 ? "+" : ""}${pct}%`;
156
+ };
157
+ const row = (label, st, unit, assumed) => ` ${label.padEnd(7)} ${f(st.median)} ${`(p10 ${f(st.p10)}, p90 ${f(st.p90)}, ${st.samples} ${unit})`.padEnd(34)} ` +
158
+ `estimator ${f(assumed)} → ${err(assumed, st.median)}`;
159
+ for (const m of report.models) {
160
+ lines.push(m.model);
161
+ if (m.prose) {
162
+ lines.push(row("prose", m.prose, "replies", m.assumed.prose));
163
+ }
164
+ if (m.blocks) {
165
+ lines.push(row("blocks", m.blocks, "blocks", m.assumed.code));
166
+ }
167
+ }
168
+ lines.push("");
169
+ lines.push("\"estimates +x%\" means the estimator reports x% more tokens than the API bills for");
170
+ lines.push("the same text (negative: fewer). Blocks are mostly code and tool output.");
171
+ return lines.join("\n");
172
+ }
package/dist/tokens.d.ts CHANGED
@@ -11,7 +11,37 @@
11
11
  export type Provider = "anthropic" | "openai" | "google" | "generic";
12
12
  export declare function contextWindowFor(model?: string): number | undefined;
13
13
  export declare function providerFor(model?: string): Provider;
14
- export declare function estimateTokens(text: string): number;
14
+ /**
15
+ * Characters per token, by provider and by content type.
16
+ *
17
+ * Anthropic: measured 2026-09 against the API's own counts in 59 Claude Code
18
+ * sessions (Opus 4.7 to 5, Fable 5.x), two independent ways that agree.
19
+ * Prose: 504 assistant replies with no thinking block, visible text divided by
20
+ * the exact output_tokens: median 2.75 (p10 2.4, p90 3.0). Code and tool
21
+ * output: 474 single appended blocks over 6k chars, sized by the exact growth
22
+ * of the billed prompt between consecutive calls: median 2.4 (p10 2.1, p90 2.8).
23
+ * The ratios this tool used before (4.0 / 3.2) undercounted current Claude
24
+ * models by about 1.45x on prose and 1.33x on code.
25
+ *
26
+ * OpenAI, Google and unknown models keep 4.0 / 3.2, the usual figures for
27
+ * o200k-class tokenizers on English and code. They are not re-measured here:
28
+ * Codex rollouts truncate tool output before the model sees it, so the same
29
+ * delta method does not isolate a block. `analyze --exact` calibrates any
30
+ * provider from its own tokenizer on your machine.
31
+ */
32
+ export declare const CHARS_PER_TOKEN: Record<Provider, {
33
+ prose: number;
34
+ code: number;
35
+ }>;
36
+ /** True when text is dense with code/JSON symbols and tokenizes more finely. */
37
+ export declare function isCodeLike(text: string): boolean;
38
+ /** Chars per token to use for this text under this model's tokenizer. */
39
+ export declare function charsPerTokenFor(text: string, model?: string): number;
40
+ /**
41
+ * Estimated tokens for `text`. Pass the model when you know it: Claude's
42
+ * tokenizer produces ~40% more tokens than the provider-neutral default.
43
+ */
44
+ export declare function estimateTokens(text: string, model?: string): number;
15
45
  /** Per-message structural overhead (role markers, delimiters) is roughly constant. */
16
46
  export declare const MESSAGE_OVERHEAD_TOKENS = 4;
17
47
  export declare function formatTokens(n: number): string;
package/dist/tokens.js CHANGED
@@ -49,17 +49,51 @@ function symbolDensity(text) {
49
49
  const symbols = text.match(/[{}[\]()<>;:=_\/\\|"'`#$%&*+^~-]/g);
50
50
  return (symbols?.length ?? 0) / text.length;
51
51
  }
52
- export function estimateTokens(text) {
52
+ /**
53
+ * Characters per token, by provider and by content type.
54
+ *
55
+ * Anthropic: measured 2026-09 against the API's own counts in 59 Claude Code
56
+ * sessions (Opus 4.7 to 5, Fable 5.x), two independent ways that agree.
57
+ * Prose: 504 assistant replies with no thinking block, visible text divided by
58
+ * the exact output_tokens: median 2.75 (p10 2.4, p90 3.0). Code and tool
59
+ * output: 474 single appended blocks over 6k chars, sized by the exact growth
60
+ * of the billed prompt between consecutive calls: median 2.4 (p10 2.1, p90 2.8).
61
+ * The ratios this tool used before (4.0 / 3.2) undercounted current Claude
62
+ * models by about 1.45x on prose and 1.33x on code.
63
+ *
64
+ * OpenAI, Google and unknown models keep 4.0 / 3.2, the usual figures for
65
+ * o200k-class tokenizers on English and code. They are not re-measured here:
66
+ * Codex rollouts truncate tool output before the model sees it, so the same
67
+ * delta method does not isolate a block. `analyze --exact` calibrates any
68
+ * provider from its own tokenizer on your machine.
69
+ */
70
+ export const CHARS_PER_TOKEN = {
71
+ anthropic: { prose: 2.75, code: 2.4 },
72
+ openai: { prose: 4.0, code: 3.2 },
73
+ google: { prose: 4.0, code: 3.2 },
74
+ generic: { prose: 4.0, code: 3.2 },
75
+ };
76
+ /** True when text is dense with code/JSON symbols and tokenizes more finely. */
77
+ export function isCodeLike(text) {
78
+ return symbolDensity(text) > 0.08;
79
+ }
80
+ /** Chars per token to use for this text under this model's tokenizer. */
81
+ export function charsPerTokenFor(text, model) {
82
+ const ratios = CHARS_PER_TOKEN[providerFor(model)];
83
+ return isCodeLike(text) ? ratios.code : ratios.prose;
84
+ }
85
+ /**
86
+ * Estimated tokens for `text`. Pass the model when you know it: Claude's
87
+ * tokenizer produces ~40% more tokens than the provider-neutral default.
88
+ */
89
+ export function estimateTokens(text, model) {
53
90
  // Public API: callers outside this package pass whatever they have, and a
54
91
  // TypeError from a token estimator is never the useful answer.
55
92
  if (typeof text !== "string")
56
93
  text = String(text ?? "");
57
94
  if (!text)
58
95
  return 0;
59
- // Denser tokenization for code/JSON-like content, lighter for plain prose.
60
- const density = symbolDensity(text);
61
- const charsPerToken = density > 0.08 ? 3.2 : 4.0;
62
- return Math.ceil(text.length / charsPerToken);
96
+ return Math.ceil(text.length / charsPerTokenFor(text, model));
63
97
  }
64
98
  /** Per-message structural overhead (role markers, delimiters) is roughly constant. */
65
99
  export const MESSAGE_OVERHEAD_TOKENS = 4;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "context-doctor",
3
- "version": "0.17.0",
3
+ "version": "0.19.0",
4
4
  "description": "Profile and optimize LLM context windows. See what's eating your tokens and fix it — works with Claude Code, Claude Desktop, Cursor, Codex (OpenAI), and any MCP-capable AI app.",
5
5
  "keywords": [
6
6
  "claude",