@oh-my-pi/pi-agent-core 17.3.8 → 17.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,8 +3,8 @@
3
3
  */
4
4
 
5
5
  import type { ToolResultMessage } from "@oh-my-pi/pi-ai";
6
+ import type { Tokenizer } from "../tokenizer";
6
7
  import type { AgentMessage, AgentToolCall } from "../types";
7
- import { estimateTokens } from "./compaction";
8
8
  import type { SessionEntry, SessionMessageEntry } from "./entries";
9
9
  import { invalidateMessageCache } from "./message-cache";
10
10
  import {
@@ -140,13 +140,13 @@ function estimatePrunedSavings(tokens: number, notice: string): number {
140
140
  * (cacheWrite premium) if that entry is mutated in place. Used to keep prune
141
141
  * mutations inside the cheap-to-recache tail.
142
142
  */
143
- function computeMessageSuffixTokens(entries: readonly SessionEntry[]): number[] {
143
+ function computeMessageSuffixTokens(entries: readonly SessionEntry[], tokenizer: Tokenizer): number[] {
144
144
  const suffix = new Array<number>(entries.length);
145
145
  let accumulated = 0;
146
146
  for (let i = entries.length - 1; i >= 0; i--) {
147
147
  suffix[i] = accumulated;
148
148
  const entry = entries[i];
149
- if (entry.type === "message") accumulated += estimateTokens(entry.message as AgentMessage);
149
+ if (entry.type === "message") accumulated += tokenizer.countMessage(entry.message as AgentMessage);
150
150
  }
151
151
  return suffix;
152
152
  }
@@ -181,6 +181,7 @@ interface SupersedeCandidate {
181
181
  */
182
182
  function collectSupersededResults(
183
183
  entries: readonly SessionEntry[],
184
+ tokenizer: Tokenizer,
184
185
  toolCallsById: ReadonlyMap<string, AgentToolCall>,
185
186
  supersedeKey: SupersedeKeyFn,
186
187
  protectedTools: readonly ProtectedToolMatcher[],
@@ -204,7 +205,7 @@ function collectSupersededResults(
204
205
  entry: entry as SessionMessageEntry,
205
206
  message,
206
207
  index: i,
207
- tokens: estimateTokens(message as AgentMessage),
208
+ tokens: tokenizer.countMessage(message as AgentMessage),
208
209
  notice: SUPERSEDED_NOTICE,
209
210
  });
210
211
  }
@@ -219,6 +220,7 @@ function collectSupersededResults(
219
220
  */
220
221
  function collectUselessResults(
221
222
  entries: readonly SessionEntry[],
223
+ tokenizer: Tokenizer,
222
224
  toolCallsById: ReadonlyMap<string, AgentToolCall>,
223
225
  protectedTools: readonly ProtectedToolMatcher[],
224
226
  exclude: ReadonlySet<ToolResultMessage>,
@@ -230,7 +232,7 @@ function collectUselessResults(
230
232
  if (message?.useless !== true || message.prunedAt !== undefined || message.isError === true) continue;
231
233
  if (exclude.has(message)) continue;
232
234
  if (isProtectedToolResult(message, toolCallsById.get(message.toolCallId), protectedTools)) continue;
233
- const tokens = estimateTokens(message as AgentMessage);
235
+ const tokens = tokenizer.countMessage(message as AgentMessage);
234
236
  if (estimatePrunedSavings(tokens, USELESS_NOTICE) <= 0) continue;
235
237
  candidates.push({ entry: entry as SessionMessageEntry, message, index: i, tokens, notice: USELESS_NOTICE });
236
238
  }
@@ -246,14 +248,18 @@ function collectUselessResults(
246
248
  * the provider cache is cold anyway (then all still-sent candidates flush).
247
249
  * Never mutates entries before `keepBoundaryId` (summarized away — not sent).
248
250
  */
249
- export function pruneSupersededToolResults(entries: SessionEntry[], config: SupersedePruneConfig): PruneResult {
251
+ export function pruneSupersededToolResults(
252
+ entries: SessionEntry[],
253
+ tokenizer: Tokenizer,
254
+ config: SupersedePruneConfig,
255
+ ): PruneResult {
250
256
  const toolCallsById = collectToolCallsById(entries);
251
257
  const candidates = config.supersedeKey
252
- ? collectSupersededResults(entries, toolCallsById, config.supersedeKey, config.protectedTools)
258
+ ? collectSupersededResults(entries, tokenizer, toolCallsById, config.supersedeKey, config.protectedTools)
253
259
  : [];
254
260
  if (config.pruneUseless) {
255
261
  const exclude = new Set(candidates.map(candidate => candidate.message));
256
- candidates.push(...collectUselessResults(entries, toolCallsById, config.protectedTools, exclude));
262
+ candidates.push(...collectUselessResults(entries, tokenizer, toolCallsById, config.protectedTools, exclude));
257
263
  candidates.sort((a, b) => a.index - b.index);
258
264
  }
259
265
  if (candidates.length === 0) return { prunedCount: 0, tokensSaved: 0 };
@@ -284,7 +290,7 @@ export function pruneSupersededToolResults(entries: SessionEntry[], config: Supe
284
290
  // Mutating a candidate re-writes its suffix in the warm cache, so prune only
285
291
  // when that suffix is small (cheap-to-recache tail) and the candidate sits
286
292
  // at/after the compaction boundary.
287
- const suffixTokens = computeMessageSuffixTokens(entries);
293
+ const suffixTokens = computeMessageSuffixTokens(entries, tokenizer);
288
294
  toPrune = candidates.filter(
289
295
  candidate => candidate.index >= boundaryIndex && suffixTokens[candidate.index] <= suffixTokenLimit,
290
296
  );
@@ -302,7 +308,11 @@ export function pruneSupersededToolResults(entries: SessionEntry[], config: Supe
302
308
  return { prunedCount: toPrune.length, tokensSaved };
303
309
  }
304
310
 
305
- export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig = DEFAULT_PRUNE_CONFIG): PruneResult {
311
+ export function pruneToolOutputs(
312
+ entries: SessionEntry[],
313
+ tokenizer: Tokenizer,
314
+ config: PruneConfig = DEFAULT_PRUNE_CONFIG,
315
+ ): PruneResult {
306
316
  let accumulatedTokens = 0;
307
317
  let tokensSaved = 0;
308
318
  let prunedCount = 0;
@@ -311,7 +321,7 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
311
321
  const toolCallsById = collectToolCallsById(entries);
312
322
  const supersededMessages = config.supersedeKey
313
323
  ? new Set(
314
- collectSupersededResults(entries, toolCallsById, config.supersedeKey, config.protectedTools).map(
324
+ collectSupersededResults(entries, tokenizer, toolCallsById, config.supersedeKey, config.protectedTools).map(
315
325
  candidate => candidate.message,
316
326
  ),
317
327
  )
@@ -321,6 +331,7 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
321
331
  ? new Set(
322
332
  collectUselessResults(
323
333
  entries,
334
+ tokenizer,
324
335
  toolCallsById,
325
336
  config.protectedTools,
326
337
  supersededMessages ?? new Set(),
@@ -331,14 +342,15 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
331
342
  const boundaryIndex = resolveBoundaryIndex(entries, config.keepBoundaryId);
332
343
  const cacheWarmSuffixTokens = config.cacheWarmSuffixTokens;
333
344
  // All-message suffix per index, only when the cache guard is armed.
334
- const messageSuffix = cacheWarmSuffixTokens === undefined ? undefined : computeMessageSuffixTokens(entries);
345
+ const messageSuffix =
346
+ cacheWarmSuffixTokens === undefined ? undefined : computeMessageSuffixTokens(entries, tokenizer);
335
347
 
336
348
  for (let i = entries.length - 1; i >= 0; i--) {
337
349
  const entry = entries[i];
338
350
  const message = getToolResultMessage(entry);
339
351
  if (!message) continue;
340
352
 
341
- const tokens = estimateTokens(message as AgentMessage);
353
+ const tokens = tokenizer.countMessage(message as AgentMessage);
342
354
  const isProtected = isProtectedToolResult(message, toolCallsById.get(message.toolCallId), config.protectedTools);
343
355
 
344
356
  if (message.prunedAt !== undefined) {
@@ -11,9 +11,8 @@
11
11
  */
12
12
 
13
13
  import type { TextContent, ToolResultMessage } from "@oh-my-pi/pi-ai";
14
- import { countTokens } from "../tokenizer";
14
+ import type { Tokenizer } from "../tokenizer";
15
15
  import type { AgentMessage } from "../types";
16
- import { estimateTokens } from "./compaction";
17
16
  import type { CustomMessageEntry, SessionEntry, SessionMessageEntry } from "./entries";
18
17
  import { invalidateMessageCache } from "./message-cache";
19
18
  import {
@@ -123,15 +122,15 @@ function toolResultText(message: ToolResultMessage): string {
123
122
  }
124
123
 
125
124
  /** Estimate the token contribution of an entry for the protect-recent window. */
126
- function entryTokens(entry: SessionEntry): number {
125
+ function entryTokens(entry: SessionEntry, tokenizer: Tokenizer): number {
127
126
  if (entry.type === "message") {
128
- return estimateTokens(entry.message);
127
+ return tokenizer.countMessage(entry.message);
129
128
  }
130
129
  if (entry.type === "custom_message") {
131
130
  const content = entry.content;
132
- if (typeof content === "string") return content.length === 0 ? 0 : countTokens(content);
131
+ if (typeof content === "string") return content.length === 0 ? 0 : tokenizer.countTokens(content);
133
132
  const fragments = content.filter((block): block is TextContent => block.type === "text").map(block => block.text);
134
- return fragments.length === 0 ? 0 : countTokens(fragments);
133
+ return fragments.length === 0 ? 0 : tokenizer.countTokens(fragments);
135
134
  }
136
135
  return 0;
137
136
  }
@@ -222,6 +221,7 @@ function pushBlockRegions(
222
221
  entry: SessionMessageEntry | CustomMessageEntry,
223
222
  blockIndex: number,
224
223
  text: string,
224
+ tokenizer: Tokenizer,
225
225
  config: ShakeConfig,
226
226
  label: string,
227
227
  out: ShakeRegion[],
@@ -229,7 +229,7 @@ function pushBlockRegions(
229
229
  for (const range of scanTextForBlockRanges(text)) {
230
230
  const slice = text.slice(range.start, range.end);
231
231
  if (slice.length === 0) continue;
232
- const tokens = countTokens(slice);
232
+ const tokens = tokenizer.countTokens(slice);
233
233
  if (tokens < config.fenceMinTokens) continue;
234
234
  out.push({
235
235
  kind: "block",
@@ -246,6 +246,7 @@ function pushBlockRegions(
246
246
 
247
247
  function collectBlockRegions(
248
248
  entry: SessionMessageEntry | CustomMessageEntry,
249
+ tokenizer: Tokenizer,
249
250
  config: ShakeConfig,
250
251
  out: ShakeRegion[],
251
252
  ): void {
@@ -254,34 +255,35 @@ function collectBlockRegions(
254
255
  if (message.role === "assistant") {
255
256
  for (let bi = 0; bi < message.content.length; bi++) {
256
257
  const block = message.content[bi];
257
- if (block.type === "text") pushBlockRegions(entry, bi, block.text, config, "assistant", out);
258
+ if (block.type === "text") pushBlockRegions(entry, bi, block.text, tokenizer, config, "assistant", out);
258
259
  }
259
260
  return;
260
261
  }
261
262
  if (message.role === "user" || message.role === "developer") {
262
- scanContentBlocks(entry, message.content, config, message.role, out);
263
+ scanContentBlocks(entry, message.content, tokenizer, config, message.role, out);
263
264
  }
264
265
  return;
265
266
  }
266
267
  // custom_message
267
- scanContentBlocks(entry, entry.content, config, entry.customType, out);
268
+ scanContentBlocks(entry, entry.content, tokenizer, config, entry.customType, out);
268
269
  }
269
270
 
270
271
  function scanContentBlocks(
271
272
  entry: SessionMessageEntry | CustomMessageEntry,
272
273
  content: string | Array<{ type: string; text?: string }>,
274
+ tokenizer: Tokenizer,
273
275
  config: ShakeConfig,
274
276
  label: string,
275
277
  out: ShakeRegion[],
276
278
  ): void {
277
279
  if (typeof content === "string") {
278
- pushBlockRegions(entry, -1, content, config, label, out);
280
+ pushBlockRegions(entry, -1, content, tokenizer, config, label, out);
279
281
  return;
280
282
  }
281
283
  for (let bi = 0; bi < content.length; bi++) {
282
284
  const block = content[bi];
283
285
  if (block.type === "text" && typeof block.text === "string") {
284
- pushBlockRegions(entry, bi, block.text, config, label, out);
286
+ pushBlockRegions(entry, bi, block.text, tokenizer, config, label, out);
285
287
  }
286
288
  }
287
289
  }
@@ -300,7 +302,7 @@ function scanContentBlocks(
300
302
  * and regions never span a message boundary. When the combined estimated
301
303
  * savings is below `minSavings`, returns `[]` (no-op).
302
304
  */
303
- export function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig): ShakeRegion[] {
305
+ export function collectShakeRegions(entries: SessionEntry[], tokenizer: Tokenizer, config: ShakeConfig): ShakeRegion[] {
304
306
  const n = entries.length;
305
307
  if (n === 0) return [];
306
308
 
@@ -309,7 +311,7 @@ export function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig
309
311
  let acc = 0;
310
312
  for (let i = n - 1; i >= 0; i--) {
311
313
  accumulatedAfter[i] = acc;
312
- acc += entryTokens(entries[i]);
314
+ acc += entryTokens(entries[i], tokenizer);
313
315
  }
314
316
 
315
317
  const toolCallsById = collectToolCallsById(entries);
@@ -342,7 +344,7 @@ export function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig
342
344
  regions.push({
343
345
  kind: "toolResult",
344
346
  entry: entry as SessionMessageEntry,
345
- tokens: estimateTokens(toolResult as AgentMessage),
347
+ tokens: tokenizer.countMessage(toolResult as AgentMessage),
346
348
  originalText: text,
347
349
  label: toolResult.toolName,
348
350
  });
@@ -350,7 +352,7 @@ export function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig
350
352
  }
351
353
 
352
354
  if (entry.type === "message" || entry.type === "custom_message") {
353
- collectBlockRegions(entry as SessionMessageEntry | CustomMessageEntry, config, regions);
355
+ collectBlockRegions(entry as SessionMessageEntry | CustomMessageEntry, tokenizer, config, regions);
354
356
  }
355
357
  }
356
358
 
@@ -0,0 +1,111 @@
1
+ /**
2
+ * Provider-anchored transcript token accounting.
3
+ *
4
+ * Local tokenization is the expensive way to answer "how big is this
5
+ * conversation?" — and usually the wrong one, because the provider already
6
+ * answered it. Every settled assistant turn carries `usage` covering the exact
7
+ * prompt it was sent: the system prompt, the tool schemas, and every message up
8
+ * to and including itself. The only genuinely unaccounted-for text is the tail
9
+ * appended *after* that turn.
10
+ *
11
+ * These helpers locate the newest trustworthy usage report and tokenize only
12
+ * that tail, so a long session pays counting proportional to one turn instead
13
+ * of to the whole transcript, every turn.
14
+ *
15
+ * Trust rules for an anchor (mirroring the provider contract):
16
+ * - Assistant role only — nothing else carries `usage`.
17
+ * - Not `aborted` / `error`: those turns report partial or zero usage.
18
+ * - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
19
+ */
20
+
21
+ import type { AssistantMessage } from "@oh-my-pi/pi-ai";
22
+ import type { MessageCountOptions, Tokenizer } from "../tokenizer";
23
+ import type { AgentMessage } from "../types";
24
+ import { calculateContextTokens, hasContextTokenUsage } from "./compaction";
25
+
26
+ /** A provider usage report that accounts for a prefix of the transcript. */
27
+ export interface TranscriptUsageAnchor {
28
+ /** Index in the scanned array; messages at or before it are provider-accounted. */
29
+ index: number;
30
+ /** The anchoring assistant turn. */
31
+ message: AssistantMessage;
32
+ /** Conversation tokens the provider reported for that prompt. */
33
+ tokens: number;
34
+ }
35
+
36
+ /**
37
+ * Whether this message's provider usage may anchor transcript accounting.
38
+ *
39
+ * The single home for the trust rules — every anchor scan MUST route through
40
+ * it so a stale-usage rule can never drift between the transcript walkers and
41
+ * the session-entry walkers.
42
+ */
43
+ export function isTranscriptUsageAnchor(message: AgentMessage): message is AssistantMessage {
44
+ if (message.role !== "assistant") return false;
45
+ const assistant = message as AssistantMessage;
46
+ if (assistant.stopReason === "aborted" || assistant.stopReason === "error") return false;
47
+ return assistant.usage !== undefined && hasContextTokenUsage(assistant.usage);
48
+ }
49
+
50
+ /**
51
+ * Newest assistant turn in `messages[fromIndex..]` whose usage can anchor the
52
+ * transcript, or `undefined` when none qualifies (fresh context, or every
53
+ * recent turn aborted/errored).
54
+ *
55
+ * `fromIndex` excludes turns whose usage is stale — anything a compaction
56
+ * summarized away describes a prompt that is no longer sent.
57
+ */
58
+ export function findTranscriptUsageAnchor(
59
+ messages: readonly AgentMessage[],
60
+ fromIndex = 0,
61
+ ): TranscriptUsageAnchor | undefined {
62
+ for (let index = messages.length - 1; index >= fromIndex; index--) {
63
+ const message = messages[index];
64
+ if (!isTranscriptUsageAnchor(message)) continue;
65
+ return { index, message, tokens: calculateContextTokens(message.usage) };
66
+ }
67
+ return undefined;
68
+ }
69
+
70
+ /** Options for {@link estimateTranscriptTokens}. */
71
+ export interface TranscriptTokenOptions {
72
+ /**
73
+ * Gates the anchor search only: usage at or before this index is stale (a
74
+ * compaction rewrote the prompt it describes) and must not anchor. Content
75
+ * accounting is governed separately by {@link countFromIndex}.
76
+ */
77
+ anchorFromIndex?: number;
78
+ /**
79
+ * First message whose content is counted locally when no anchor is found.
80
+ * Defaults to 0 (count the whole transcript), which is what a floor
81
+ * estimate wants; pass the compaction boundary to skip summarized-away
82
+ * messages entirely.
83
+ */
84
+ countFromIndex?: number;
85
+ /** Forwarded to {@link Tokenizer.countMessage} for every locally counted message. */
86
+ excludeEncryptedReasoning?: boolean;
87
+ }
88
+
89
+ /**
90
+ * Conversation tokens for `messages`: the provider's own report for everything
91
+ * it already covers, plus a local count of only the unaccounted-for tail.
92
+ *
93
+ * An anchored result already includes the non-message prefix (system prompt +
94
+ * tool schemas) because the provider charged it; an unanchored result is a
95
+ * message-only sum. Callers that add non-message tokens on top MUST branch on
96
+ * {@link findTranscriptUsageAnchor} rather than assuming one shape.
97
+ */
98
+ export function estimateTranscriptTokens(
99
+ messages: readonly AgentMessage[],
100
+ tokenizer: Tokenizer,
101
+ options?: TranscriptTokenOptions,
102
+ ): number {
103
+ const estimateOptions: MessageCountOptions | undefined =
104
+ options?.excludeEncryptedReasoning === true ? { excludeEncryptedReasoning: true } : undefined;
105
+ const anchor = findTranscriptUsageAnchor(messages, options?.anchorFromIndex ?? 0);
106
+ let total = anchor?.tokens ?? 0;
107
+ for (let index = anchor ? anchor.index + 1 : (options?.countFromIndex ?? 0); index < messages.length; index++) {
108
+ total += tokenizer.countMessage(messages[index], estimateOptions);
109
+ }
110
+ return total;
111
+ }
package/src/tokenizer.ts CHANGED
@@ -1,27 +1,282 @@
1
- import { countTokens as countTokensNat } from "@oh-my-pi/pi-natives";
1
+ import type { Model } from "@oh-my-pi/pi-ai";
2
+ import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
3
+ import { countTokens as countTokensNat, Encoding } from "@oh-my-pi/pi-natives";
4
+ import { stringifyJson } from "@oh-my-pi/pi-utils";
5
+ import * as snapcompact from "@oh-my-pi/snapcompact";
6
+ import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
7
+ import type { AgentMessage } from "./types";
2
8
 
3
- const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && Bun.env.NODE_ENV !== "test";
9
+ const testEnv = Bun.env.NODE_ENV === "test";
10
+ const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && !testEnv;
4
11
 
5
- function estimateTokens(text: string) {
12
+ const NATIVE_ENCODING: Record<ModelTokenizer, Encoding> = {
13
+ "claude-v3": Encoding.ClaudeV3,
14
+ "claude-v47": Encoding.ClaudeV47,
15
+ "claude-v5": Encoding.ClaudeV5,
16
+ "claude-v5-sonnet": Encoding.ClaudeV5Sonnet,
17
+ qwen3: Encoding.Qwen3,
18
+ "deepseek-v3": Encoding.DeepSeekV3,
19
+ "kimi-k2": Encoding.KimiK2,
20
+ glm5: Encoding.Glm5,
21
+ };
22
+
23
+ /** Maps the catalog-resolved tokenizer family to its native implementation. */
24
+ export function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null {
25
+ return model?.tokenizer ? NATIVE_ENCODING[model.tokenizer] : null;
26
+ }
27
+
28
+ /**
29
+ * `strict` always pays for an exact native count (the catalog-resolved
30
+ * tokenizer when known, o200k_base otherwise). `approximate` and
31
+ * `upperbound` prefer the same exact count for known tokenizer families or
32
+ * when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
33
+ * `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
34
+ * undercounts).
35
+ */
36
+ export type TokenCountMode = "strict" | "approximate" | "upperbound";
37
+
38
+ /** Options for {@link Tokenizer.countMessage} / {@link Tokenizer.countMessages}. */
39
+ export interface MessageCountOptions {
40
+ /**
41
+ * Drop opaque provider reasoning payloads (`thinkingSignature`,
42
+ * `redactedThinking`, native server-tool blocks) from the estimate. Those
43
+ * are billed by the provider on replay, so the default counts them — but
44
+ * their *local* byte size can diverge wildly from what the provider
45
+ * charges, so the compaction floor (which only needs the reliably-countable,
46
+ * on-wire-compressible content) excludes them to avoid false triggers on
47
+ * thinking-heavy turns.
48
+ */
49
+ excludeEncryptedReasoning?: boolean;
50
+ }
51
+
52
+ function byteEstimate(text: string): number {
6
53
  return (Buffer.byteLength(text, "utf-8") + 3) >> 2;
7
54
  }
8
55
 
9
- export function countTokens(text: string | string[]): number {
10
- if (accurate) {
11
- return countTokensNat(text);
12
- } else if (Array.isArray(text)) {
13
- return text.reduce((sum, t) => sum + estimateTokens(t), 0);
14
- } else {
15
- return estimateTokens(text);
16
- }
56
+ function byteLength(text: string): number {
57
+ return Buffer.byteLength(text, "utf-8");
17
58
  }
18
59
 
19
- export function countTokensConservatively(text: string | string[]): number {
20
- if (accurate) {
21
- return countTokensNat(text);
22
- } else if (Array.isArray(text)) {
23
- return text.reduce((sum, value) => sum + Buffer.byteLength(value, "utf-8"), 0);
24
- } else {
25
- return Buffer.byteLength(text, "utf-8");
60
+ function sumFragments(text: string | string[], perFragment: (t: string) => number): number {
61
+ return Array.isArray(text) ? text.reduce((sum, t) => sum + perFragment(t), 0) : perFragment(text);
62
+ }
63
+
64
+ /** Verdict from {@link Tokenizer.checkTokenBudget}. */
65
+ export interface TokenBudgetCheck {
66
+ /** Whether the text fits the budget. */
67
+ fits: boolean;
68
+ /**
69
+ * Token count behind the verdict: the exact native count when `exact` is
70
+ * set, otherwise the cheap byte upper bound (which already fit, so it is
71
+ * only an over-estimate of a count known to be under budget).
72
+ */
73
+ tokens: number;
74
+ /** Whether the exact tokenizer had to run because the cheap bound busted. */
75
+ exact: boolean;
76
+ }
77
+
78
+ /**
79
+ * Image content has no tokenizer representation; charge a fixed estimate
80
+ * matching what providers typically bill for inline images.
81
+ */
82
+ const IMAGE_TOKEN_ESTIMATE = 1200;
83
+
84
+ /**
85
+ * Memoized estimates for one message under this tokenizer's encoding, split by
86
+ * the {@link MessageCountOptions.excludeEncryptedReasoning} option so the two
87
+ * variants never collide. `version` snapshots {@link messageEstimateVersion} at
88
+ * write time; an owner mutation (prune/shake/strip-images) bumps the version,
89
+ * which invalidates the entry in every live Tokenizer at once.
90
+ */
91
+ interface MessageEstimate {
92
+ version: number;
93
+ default?: number;
94
+ floored?: number;
95
+ }
96
+
97
+ /**
98
+ * Model-aware local token counter. Immutable: the catalog-resolved encoding
99
+ * is fixed at construction, so a cached count can never straddle two
100
+ * encodings. An `Agent` owns one for its active model (swapping the instance
101
+ * when the model's encoding changes); one-shot flows construct their own for
102
+ * the model that will be billed. Known tokenizer families use exact native
103
+ * counts; unknown models keep the fast byte estimate (or o200k when
104
+ * `PI_TOKENIZER_ACCURATE=1`).
105
+ */
106
+ export class Tokenizer {
107
+ readonly #encoding: Encoding | null;
108
+
109
+ /**
110
+ * Per-message estimate memo. Keyed by message identity, deliberately not a
111
+ * symbol-tagged property: callers spread messages to derive throwaway
112
+ * variants for counting (`estimateBranchSummaryTokens` does
113
+ * `countMessage({ ...message, content: truncated })`), and a property-borne
114
+ * cache would ride along the spread. Identity keying gives clones a fresh
115
+ * count.
116
+ */
117
+ #estimates = new WeakMap<AgentMessage, MessageEstimate>();
118
+
119
+ constructor(model?: Pick<Model, "tokenizer"> | null) {
120
+ this.#encoding = tokenizerEncodingForModel(model);
121
+ }
122
+
123
+ get encoding(): Encoding | null {
124
+ return this.#encoding;
125
+ }
126
+
127
+ countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
128
+ if (mode === "strict") return countTokensNat(text, this.#encoding);
129
+ if (!testEnv && this.#encoding !== null) return countTokensNat(text, this.#encoding);
130
+ if (accurate) return countTokensNat(text);
131
+ return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
132
+ }
133
+
134
+ /**
135
+ * Cheap-first budget probe — the way to ask "does this fit in `budget`
136
+ * tokens?" without tokenizing the world.
137
+ *
138
+ * Byte length is a hard upper bound on token count (every token consumes at
139
+ * least one input byte), so text whose raw bytes already fit the budget
140
+ * cannot possibly exceed it — that verdict is returned without tokenizing at
141
+ * all. Only text that busts the bound is ambiguous, and only that case pays
142
+ * for the exact count. Since the bound overshoots ~4x on ordinary prose, the
143
+ * common "comfortably under budget" answer is free.
144
+ */
145
+ checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
146
+ const bound = sumFragments(text, byteLength);
147
+ if (bound <= budget) return { fits: true, tokens: bound, exact: false };
148
+ const tokens = this.countTokens(text, "strict");
149
+ return { fits: tokens <= budget, tokens, exact: true };
150
+ }
151
+
152
+ /**
153
+ * Token estimate for one message under this tokenizer's encoding.
154
+ *
155
+ * Settled historical messages are counted once and reused until an owner
156
+ * (prune/shake/strip-images) calls `invalidateMessageCache`; streaming
157
+ * assistants bypass the memo entirely (see the message-cache settle-gate
158
+ * invariant). Image blocks charge a fixed per-image estimate.
159
+ */
160
+ countMessage(message: AgentMessage, options?: MessageCountOptions): number {
161
+ const floored = options?.excludeEncryptedReasoning === true;
162
+ if (!isEstimateCacheable(message)) return this.#measureMessage(message, floored);
163
+ const version = messageEstimateVersion(message);
164
+ let entry = this.#estimates.get(message);
165
+ if (entry === undefined || entry.version !== version) {
166
+ entry = { version };
167
+ this.#estimates.set(message, entry);
168
+ }
169
+ const cached = floored ? entry.floored : entry.default;
170
+ if (cached !== undefined) return cached;
171
+ const result = this.#measureMessage(message, floored);
172
+ if (floored) entry.floored = result;
173
+ else entry.default = result;
174
+ return result;
175
+ }
176
+
177
+ /** Sum of {@link countMessage} over `messages`. */
178
+ countMessages(messages: readonly AgentMessage[], options?: MessageCountOptions): number {
179
+ let total = 0;
180
+ for (const message of messages) total += this.countMessage(message, options);
181
+ return total;
182
+ }
183
+
184
+ #measureMessage(message: AgentMessage, excludeEncryptedReasoning: boolean): number {
185
+ const fragments: string[] = [];
186
+ let extra = 0;
187
+ // Declaration-merged app roles (the coding-agent's bashExecution) are
188
+ // invisible to this package's union, so the discriminant is read as data.
189
+ const role: string = message.role;
190
+ if (role === "bashExecution") {
191
+ if ("command" in message && typeof message.command === "string") fragments.push(message.command);
192
+ if ("output" in message && typeof message.output === "string") fragments.push(message.output);
193
+ return fragments.length === 0 ? 0 : this.countTokens(fragments);
194
+ }
195
+
196
+ switch (message.role) {
197
+ case "user": {
198
+ const content: string | Array<{ type: string; text?: string }> = message.content;
199
+ if (typeof content === "string") {
200
+ fragments.push(content);
201
+ } else if (Array.isArray(content)) {
202
+ for (const block of content) {
203
+ if (block.type === "text" && block.text) {
204
+ fragments.push(block.text);
205
+ }
206
+ }
207
+ }
208
+ break;
209
+ }
210
+ case "assistant": {
211
+ for (const block of message.content) {
212
+ if (block.type === "text") {
213
+ fragments.push(block.text);
214
+ } else if (block.type === "thinking") {
215
+ fragments.push(block.thinking);
216
+ // Providers charge for the opaque signature/reasoning payload that
217
+ // rides alongside the thinking text (OpenAI Responses encrypted
218
+ // reasoning items, Anthropic signed thinking blocks, etc.). Without
219
+ // counting it, this estimator can read ~half of the provider-reported
220
+ // usage on thinking-heavy turns — see #2275 for the resulting
221
+ // compaction-trigger / post-check metric divergence. The compaction
222
+ // floor excludes it (its local byte size diverges from provider billing).
223
+ if (block.thinkingSignature && !excludeEncryptedReasoning) {
224
+ fragments.push(block.thinkingSignature);
225
+ }
226
+ } else if (block.type === "toolCall") {
227
+ fragments.push(block.name);
228
+ fragments.push(stringifyJson(block.arguments) ?? "null");
229
+ } else if (block.type === "redactedThinking") {
230
+ // Encrypted reasoning blob the provider still bills for on replay;
231
+ // excluded from the compaction floor for the same reason as above.
232
+ if (!excludeEncryptedReasoning) fragments.push(block.data);
233
+ } else if (block.type === "anthropicServerTool") {
234
+ // Native Anthropic server-tool call/result replayed verbatim on the
235
+ // wire (server_tool_use input and opaque result content). The provider
236
+ // still bills for it on same-provider replay; excluded from the
237
+ // compaction floor like other encrypted reasoning because its local
238
+ // byte size diverges from provider billing.
239
+ if (!excludeEncryptedReasoning) fragments.push(stringifyJson(block.block) ?? "null");
240
+ }
241
+ }
242
+ break;
243
+ }
244
+ case "hookMessage":
245
+ case "toolResult": {
246
+ if (typeof message.content === "string") {
247
+ fragments.push(message.content);
248
+ } else {
249
+ for (const block of message.content) {
250
+ if (block.type === "text" && block.text) {
251
+ fragments.push(block.text);
252
+ } else if (block.type === "image") {
253
+ extra += IMAGE_TOKEN_ESTIMATE;
254
+ }
255
+ }
256
+ }
257
+ break;
258
+ }
259
+ case "branchSummary":
260
+ case "compactionSummary": {
261
+ fragments.push(message.summary);
262
+ if (message.role === "compactionSummary") {
263
+ if (message.blocks) {
264
+ for (const block of message.blocks) {
265
+ if (block.type === "text") fragments.push(block.text);
266
+ else extra += snapcompact.FRAME_TOKEN_ESTIMATE;
267
+ }
268
+ } else if (message.images) {
269
+ // Snapcompact frames render at ≥1568px; providers bill the downscaled cap.
270
+ extra += message.images.length * snapcompact.FRAME_TOKEN_ESTIMATE;
271
+ }
272
+ }
273
+ break;
274
+ }
275
+ default:
276
+ return 0;
277
+ }
278
+
279
+ if (fragments.length === 0) return extra;
280
+ return extra + this.countTokens(fragments);
26
281
  }
27
282
  }