@oh-my-pi/pi-agent-core 18.4.0 → 18.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,29 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.4.2] - 2026-09-28
6
+
7
+ ### Added
8
+
9
+ - Added tool_execution_end events that fire as each tool call settles for live UI updates
10
+
11
+ ### Changed
12
+
13
+ - Emitted tool result messages in the order of tool calls, preserving call order regardless of completion order
14
+ - Reduced repeated token-counting work with a bounded, model-scoped cache of exact text and short-message fragment counts.
15
+
16
+ ### Fixed
17
+
18
+ - Fixed an issue where streaming tool call arguments could be incorrectly modified in-place
19
+
20
+ ## [18.4.1] - 2026-09-28
21
+
22
+ ### Fixed
23
+
24
+ - Fixed fitted output caps overshooting the context window by a few tokens on strict Chat Completions hosts (e.g. llama.cpp), causing 400s.
25
+ - Fixed native remote compaction sending requests already estimated past the model's context window (e.g. after re-expanding history behind another provider's native boundary); it now fails fast so the next configured compaction method runs ([#13502](https://github.com/can1357/oh-my-pi/issues/13502))
26
+ - Fixed V2 remote compaction retrying a standalone stream `error` event three times and reporting it as `stream closed before response.completed`; the upstream status, code, and message (e.g. `context_too_large`) are now surfaced ([#13502](https://github.com/can1357/oh-my-pi/issues/13502))
27
+
5
28
  ## [18.4.0] - 2026-09-28
6
29
 
7
30
  ### Changed
@@ -33,6 +33,8 @@ export interface TrimRemoteCompactionInputResult {
33
33
  rewrittenOutputs: number;
34
34
  estimatedTokensBefore: number;
35
35
  estimatedTokensAfter: number;
36
+ /** Whether `input` fits the model window; false means it must not be sent. */
37
+ fits: boolean;
36
38
  }
37
39
  /**
38
40
  * Preserve the full native transcript unless trailing tool outputs alone push a
@@ -41,6 +43,15 @@ export interface TrimRemoteCompactionInputResult {
41
43
  * matching Codex's recovery path for oversized tool turns.
42
44
  */
43
45
  export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, tokenizer: Tokenizer, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
46
+ /**
47
+ * Refuse a native compaction request whose prepared input cannot fit the model
48
+ * window, before any network I/O. Re-expanded history behind an unreadable
49
+ * native boundary can exceed the window even when live context does not.
50
+ *
51
+ * @throws Error flagged `ContextOverflow` when `trimmed.fits` is false, so
52
+ * compaction callers skip retries and advance to the next method.
53
+ */
54
+ export declare function assertRemoteCompactionInputFits(trimmed: TrimRemoteCompactionInputResult, model: Model): void;
44
55
  export type OpenAiRemoteCompactionItem = {
45
56
  type: "compaction" | "compaction_summary";
46
57
  encrypted_content?: string;
@@ -2,6 +2,16 @@ import type { Context, Model } from "@oh-my-pi/pi-ai";
2
2
  import type { Tokenizer } from "./tokenizer.js";
3
3
  /** Smallest output cap {@link fitOutputTokensToContextWindow} will request. */
4
4
  export declare const MIN_FITTED_OUTPUT_TOKENS = 1024;
5
+ /**
6
+ * Absolute headway subtracted from the remaining room (unconditionally, on anchored and
7
+ * fully-local counts alike). Proportional padding covers
8
+ * tokenizer drift that scales with the prompt, but a host can still count a few tokens more
9
+ * than any local estimate can see (chat-template framing, reasoning wrappers). Measured
10
+ * against a 262,144-token host: fitted caps landed +11 to +42 tokens over and were rejected
11
+ * with 400s. 64 tokens covers the observed drift with margin to spare; the
12
+ * {@link MIN_FITTED_OUTPUT_TOKENS} floor still applies.
13
+ */
14
+ export declare const OUTPUT_FIT_HEADWAY_TOKENS = 64;
5
15
  /**
6
16
  * Output cap for a request, so prompt plus output stays inside the model's
7
17
  * context window.
@@ -26,7 +36,8 @@ export declare const MIN_FITTED_OUTPUT_TOKENS = 1024;
26
36
  * OpenRouter-hosted model with no caller cap: the transport omits the catalog
27
37
  * default there so each upstream self-caps, and a fitted value would turn into
28
38
  * an explicit cap that filters upstreams). Otherwise returns
29
- * the remaining room (never below {@link MIN_FITTED_OUTPUT_TOKENS}); a
39
+ * the remaining room minus {@link OUTPUT_FIT_HEADWAY_TOKENS} (never below
40
+ * {@link MIN_FITTED_OUTPUT_TOKENS}); a
30
41
  * prompt that fills the whole window still overflows and is left to the
31
42
  * caller's compaction. Near a full window the floor means a turn can stop on
32
43
  * `length` instead of failing with a 400.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "18.4.0",
4
+ "version": "18.4.2",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": {
@@ -38,16 +38,16 @@
38
38
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
39
39
  },
40
40
  "dependencies": {
41
- "@oh-my-pi/pi-ai": "18.4.0",
42
- "@oh-my-pi/pi-catalog": "18.4.0",
43
- "@oh-my-pi/pi-natives": "18.4.0",
44
- "@oh-my-pi/pi-utils": "18.4.0",
45
- "@oh-my-pi/pi-wire": "18.4.0",
46
- "@oh-my-pi/snapcompact": "18.4.0",
41
+ "@oh-my-pi/pi-ai": "18.4.2",
42
+ "@oh-my-pi/pi-catalog": "18.4.2",
43
+ "@oh-my-pi/pi-natives": "18.4.2",
44
+ "@oh-my-pi/pi-utils": "18.4.2",
45
+ "@oh-my-pi/pi-wire": "18.4.2",
46
+ "@oh-my-pi/snapcompact": "18.4.2",
47
47
  "@opentelemetry/api": "^1.9.1"
48
48
  },
49
49
  "devDependencies": {
50
- "@oh-my-pi/omptype": "18.4.0",
50
+ "@oh-my-pi/omptype": "18.4.2",
51
51
  "@opentelemetry/context-async-hooks": "^2.9.0",
52
52
  "@opentelemetry/sdk-trace-base": "^2.9.0",
53
53
  "@types/bun": "^1.3.14"
package/src/agent-loop.ts CHANGED
@@ -51,7 +51,7 @@ import {
51
51
  recoverHarmonyToolCall,
52
52
  signalListLabel,
53
53
  } from "@oh-my-pi/pi-ai/utils/harmony-leak";
54
- import { logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
54
+ import { cloneJsonTree, logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
55
55
  import { INTENT_FIELD } from "@oh-my-pi/pi-wire";
56
56
  import { LiveSteeringChannel } from "./live-steering";
57
57
  import { agentPauseGate } from "./pause";
@@ -377,13 +377,17 @@ function snapshotAssistantContentBlock(block: AssistantContentBlock): AssistantC
377
377
  case "redactedThinking":
378
378
  return { ...block };
379
379
  case "anthropicServerTool":
380
- return { ...block, block: structuredCloneJSON(block.block) };
380
+ return { ...block, block: cloneJsonTree(block.block) };
381
381
  case "fallback":
382
382
  return { ...block, from: { ...block.from }, to: { ...block.to } };
383
383
  case "toolCall": {
384
384
  const snap = {
385
385
  ...block,
386
- arguments: structuredCloneJSON(block.arguments),
386
+ // Providers mutate streaming arguments in place (owned-stream, GLM)
387
+ // as well as replacing them, so containers are always copied; the
388
+ // strings inside are immutable and shared, keeping the per-delta
389
+ // cost independent of the argument payload size.
390
+ arguments: cloneJsonTree(block.arguments),
387
391
  providerMetadata: snapshotToolCallProviderMetadata(block.providerMetadata),
388
392
  };
389
393
  // Object spread copies enumerable symbols in Bun, but the Cursor
@@ -2859,6 +2863,13 @@ async function prepareToolCallDispatch(
2859
2863
  if (toolCall.type !== "toolCall") continue;
2860
2864
  if ((toolCall as CursorExecResolvedCarrier)[kCursorExecResolved] === true) continue;
2861
2865
  const tool = resolveToolForCall(context.tools, toolCall, resolveFallbackTool);
2866
+ // A host fallback accepts aliases (`xd://recall`, a mis-separated MCP
2867
+ // name) that providers reject when replayed as a function-call name.
2868
+ // Record the call under the resolved tool's canonical name so history,
2869
+ // persistence, and replay agree; custom-wire calls keep their wire name.
2870
+ if (tool && toolCall.name !== tool.name && toolCall.name !== tool.customWireName) {
2871
+ toolCall.name = tool.name;
2872
+ }
2862
2873
  const entry: PreparedToolCall = { tool, args: toolCall.arguments as Record<string, unknown> };
2863
2874
  prepared.set(toolCall.id, entry);
2864
2875
  let argsForExecution = toolCall.arguments as Record<string, unknown>;
@@ -3020,6 +3031,11 @@ async function speculativeFinalCalls(
3020
3031
  /**
3021
3032
  * Execute tool calls from an assistant message. Returns model-visible context
3022
3033
  * only after every result has settled, preserving assistant call order.
3034
+ *
3035
+ * `tool_execution_end` fires as each call settles so live UI updates promptly;
3036
+ * result `message_start`/`message_end` events (which append to agent state and
3037
+ * the persisted session) are held until every earlier call has a result, so
3038
+ * history always pairs results in call order regardless of completion order.
3023
3039
  */
3024
3040
  async function executeToolCalls(
3025
3041
  currentContext: AgentContext,
@@ -3206,6 +3222,18 @@ async function executeToolCalls(
3206
3222
  await checkAsideInterrupts();
3207
3223
  };
3208
3224
 
3225
+ // Index of the first record whose result message has not been emitted yet.
3226
+ let nextResultIndex = 0;
3227
+ const flushResultMessages = (): void => {
3228
+ for (; nextResultIndex < records.length; nextResultIndex++) {
3229
+ const message = records[nextResultIndex].toolResultMessage;
3230
+ if (!message) return;
3231
+ emittedToolResults.push(message);
3232
+ stream.push({ type: "message_start", message });
3233
+ stream.push({ type: "message_end", message });
3234
+ }
3235
+ };
3236
+
3209
3237
  const emitToolResult = (record: (typeof records)[number], result: AgentToolResult<any>, isError: boolean): void => {
3210
3238
  if (record.resultEmitted) return;
3211
3239
  const { toolCall } = record;
@@ -3241,10 +3269,7 @@ async function executeToolCalls(
3241
3269
  record.isError = isError;
3242
3270
  record.toolResultMessage = toolResultMessage;
3243
3271
  record.resultEmitted = true;
3244
- emittedToolResults.push(toolResultMessage);
3245
-
3246
- stream.push({ type: "message_start", message: toolResultMessage });
3247
- stream.push({ type: "message_end", message: toolResultMessage });
3272
+ flushResultMessages();
3248
3273
  };
3249
3274
 
3250
3275
  const runTool = async (record: (typeof records)[number], index: number): Promise<void> => {
@@ -597,6 +597,14 @@ function handleCompactionV2Event(
597
597
  if (type === "response.failed" || type === "response.incomplete") {
598
598
  throw new Error(formatCompactionV2Failure(event, type));
599
599
  }
600
+
601
+ // A standalone `error` event terminates the stream. Keep its status so a
602
+ // deterministic 4xx (e.g. context_too_large) is not retried as a dropped stream.
603
+ if (type === "error") {
604
+ const message = formatCompactionV2Failure(event, type);
605
+ const status = numberField(event, "status");
606
+ throw status === undefined ? new Error(message) : new AIError.ProviderHttpError(message, status);
607
+ }
600
608
  }
601
609
 
602
610
  function parseCompactionV2Usage(event: Record<string, unknown>): CompactionV2Usage | undefined {
@@ -629,8 +637,9 @@ function formatCompactionV2Failure(event: Record<string, unknown>, type: string)
629
637
  : response && isRecord(response.error)
630
638
  ? response.error
631
639
  : undefined;
632
- const message = error ? stringField(error, "message") : undefined;
633
- const code = error ? (stringField(error, "code") ?? stringField(error, "type")) : undefined;
640
+ // Responses `error` events carry code/message at the top level.
641
+ const message = stringField(error ?? event, "message");
642
+ const code = error ? (stringField(error, "code") ?? stringField(error, "type")) : stringField(event, "code");
634
643
  return `V2 compaction stream ${type}${code ? ` (${code})` : ""}${message ? `: ${message}` : ""}`;
635
644
  }
636
645
 
@@ -69,6 +69,7 @@ import {
69
69
  defaultConvertToLlm,
70
70
  } from "./messages";
71
71
  import {
72
+ assertRemoteCompactionInputFits,
72
73
  buildOpenAiNativeHistory,
73
74
  getPreservedOpenAiRemoteCompactionData,
74
75
  isOpenAiRemoteCompactionApi,
@@ -1748,6 +1749,7 @@ export async function compact(
1748
1749
  contextWindow: model.contextWindow,
1749
1750
  });
1750
1751
  }
1752
+ assertRemoteCompactionInputFits(trimmed, model);
1751
1753
  const requestOptions = {
1752
1754
  sessionId: summaryOptions.sessionId,
1753
1755
  promptCacheKey: summaryOptions.promptCacheKey,
@@ -15,7 +15,7 @@
15
15
  * with `{ summary, shortSummary? }`.
16
16
  */
17
17
 
18
- import { ProviderHttpError } from "@oh-my-pi/pi-ai/error";
18
+ import { attach, create, Flag, ProviderHttpError } from "@oh-my-pi/pi-ai/error";
19
19
  import { getCodexAttestationHeader } from "@oh-my-pi/pi-ai/providers/openai-codex-attestation";
20
20
  import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-compaction";
21
21
  import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer";
@@ -122,6 +122,8 @@ export interface TrimRemoteCompactionInputResult {
122
122
  rewrittenOutputs: number;
123
123
  estimatedTokensBefore: number;
124
124
  estimatedTokensAfter: number;
125
+ /** Whether `input` fits the model window; false means it must not be sent. */
126
+ fits: boolean;
125
127
  }
126
128
 
127
129
  /** Verdict for one remote-compaction request measured against the model window. */
@@ -199,6 +201,7 @@ export function trimRemoteCompactionInputToContextWindow(
199
201
  rewrittenOutputs: 0,
200
202
  estimatedTokensBefore: before.tokens,
201
203
  estimatedTokensAfter: before.tokens,
204
+ fits: true,
202
205
  };
203
206
  }
204
207
 
@@ -222,6 +225,7 @@ export function trimRemoteCompactionInputToContextWindow(
222
225
  rewrittenOutputs: 0,
223
226
  estimatedTokensBefore: before.tokens,
224
227
  estimatedTokensAfter: before.tokens,
228
+ fits: false,
225
229
  };
226
230
  }
227
231
 
@@ -230,9 +234,29 @@ export function trimRemoteCompactionInputToContextWindow(
230
234
  rewrittenOutputs,
231
235
  estimatedTokensBefore: before.tokens,
232
236
  estimatedTokensAfter: after.tokens,
237
+ fits: true,
233
238
  };
234
239
  }
235
240
 
241
+ /**
242
+ * Refuse a native compaction request whose prepared input cannot fit the model
243
+ * window, before any network I/O. Re-expanded history behind an unreadable
244
+ * native boundary can exceed the window even when live context does not.
245
+ *
246
+ * @throws Error flagged `ContextOverflow` when `trimmed.fits` is false, so
247
+ * compaction callers skip retries and advance to the next method.
248
+ */
249
+ export function assertRemoteCompactionInputFits(trimmed: TrimRemoteCompactionInputResult, model: Model): void {
250
+ if (trimmed.fits) return;
251
+ throw attach(
252
+ new Error(
253
+ `Remote compaction input exceeds the context window of ${model.provider}/${model.id}: ` +
254
+ `estimated ${trimmed.estimatedTokensAfter} tokens > ${model.contextWindow}`,
255
+ ),
256
+ create(Flag.ContextOverflow),
257
+ );
258
+ }
259
+
236
260
  /** Race the caller's signal against the request timeout; `timeoutMs <= 0` disables the watchdog. */
237
261
  function withRequestTimeout(signal: AbortSignal | undefined, timeoutMs: number): AbortSignal | undefined {
238
262
  if (timeoutMs <= 0) return signal;
@@ -794,6 +818,7 @@ export async function requestOpenAiRemoteCompaction(
794
818
  contextWindow: model.contextWindow,
795
819
  });
796
820
  }
821
+ assertRemoteCompactionInputFits(trimmed, model);
797
822
  const request: OpenAiRemoteCompactionRequest = {
798
823
  model: requestModel,
799
824
  // Preserve the native transcript. Only oversized trailing tool outputs are
@@ -17,6 +17,17 @@ export const MIN_FITTED_OUTPUT_TOKENS = 1024;
17
17
  */
18
18
  const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
19
19
 
20
+ /**
21
+ * Absolute headway subtracted from the remaining room (unconditionally, on anchored and
22
+ * fully-local counts alike). Proportional padding covers
23
+ * tokenizer drift that scales with the prompt, but a host can still count a few tokens more
24
+ * than any local estimate can see (chat-template framing, reasoning wrappers). Measured
25
+ * against a 262,144-token host: fitted caps landed +11 to +42 tokens over and were rejected
26
+ * with 400s. 64 tokens covers the observed drift with margin to spare; the
27
+ * {@link MIN_FITTED_OUTPUT_TOKENS} floor still applies.
28
+ */
29
+ export const OUTPUT_FIT_HEADWAY_TOKENS = 64;
30
+
20
31
  /**
21
32
  * Output cap for a request, so prompt plus output stays inside the model's
22
33
  * context window.
@@ -41,7 +52,8 @@ const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
41
52
  * OpenRouter-hosted model with no caller cap: the transport omits the catalog
42
53
  * default there so each upstream self-caps, and a fitted value would turn into
43
54
  * an explicit cap that filters upstreams). Otherwise returns
44
- * the remaining room (never below {@link MIN_FITTED_OUTPUT_TOKENS}); a
55
+ * the remaining room minus {@link OUTPUT_FIT_HEADWAY_TOKENS} (never below
56
+ * {@link MIN_FITTED_OUTPUT_TOKENS}); a
45
57
  * prompt that fills the whole window still overflows and is left to the
46
58
  * caller's compaction. Near a full window the floor means a turn can stop on
47
59
  * `length` instead of failing with a 400.
@@ -67,7 +79,7 @@ export function fitOutputTokensToContextWindow(
67
79
  if (!requested || !contextWindow || contextWindow <= 0) return maxTokens;
68
80
  if (stopsOutputAtContextWindow(model)) return maxTokens;
69
81
 
70
- const room = contextWindow - countPromptTokens(context, tokenizer);
82
+ const room = contextWindow - countPromptTokens(context, tokenizer) - OUTPUT_FIT_HEADWAY_TOKENS;
71
83
  if (room >= requested) return maxTokens;
72
84
  return Math.max(MIN_FITTED_OUTPUT_TOKENS, room);
73
85
  }
package/src/tokenizer.ts CHANGED
@@ -1,7 +1,8 @@
1
1
  import type { Model } from "@oh-my-pi/pi-ai";
2
2
  import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
3
3
  import * as natives from "@oh-my-pi/pi-natives";
4
- import { stringifyJson } from "@oh-my-pi/pi-utils";
4
+ import { materializeString, stringifyJson } from "@oh-my-pi/pi-utils";
5
+ import { LRUCache } from "@oh-my-pi/pi-utils/lru";
5
6
  import * as snapcompact from "@oh-my-pi/snapcompact";
6
7
  import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
7
8
  import type { AgentMessage } from "./types";
@@ -67,13 +68,43 @@ interface NativeTokenCount {
67
68
  exact: boolean;
68
69
  }
69
70
 
71
+ // Growing streamed text and large tool results must not evict the reusable
72
+ // short fragments. Account for UTF-16 key storage plus a per-entry allowance.
73
+ const NATIVE_CACHE_MAX_LENGTH = 16 * 1024;
74
+
75
+ function countNativeFragment(
76
+ text: string,
77
+ encoding: natives.Encoding | null | undefined,
78
+ counts: LRUCache<string, number>,
79
+ ): number {
80
+ if (text.length > NATIVE_CACHE_MAX_LENGTH) return natives.countTokens(text, encoding);
81
+ const cached = counts.get(text);
82
+ if (cached !== undefined) return cached;
83
+ const tokens = natives.countTokens(text, encoding);
84
+ // Detach sliced strings so a small key cannot retain a much larger source.
85
+ counts.set(materializeString(text), tokens);
86
+ return tokens;
87
+ }
88
+
70
89
  function countTokensNat(
71
90
  text: string | string[],
72
91
  encoding: natives.Encoding | null | undefined,
73
92
  mode: TokenCountMode,
93
+ counts: LRUCache<string, number>,
74
94
  ): NativeTokenCount {
75
95
  try {
76
- return { tokens: natives.countTokens(text, encoding), exact: true };
96
+ let tokens: number;
97
+ if (typeof text === "string") {
98
+ tokens = countNativeFragment(text, encoding, counts);
99
+ } else if (text.length > 0 && text.length < 16) {
100
+ // The native API sums independent fragments, not their concatenation.
101
+ // Keep its parallel batch path for arrays of 16 or more fragments.
102
+ tokens = 0;
103
+ for (const fragment of text) tokens += countNativeFragment(fragment, encoding, counts);
104
+ } else {
105
+ tokens = natives.countTokens(text, encoding);
106
+ }
107
+ return { tokens, exact: true };
77
108
  } catch (error) {
78
109
  if (
79
110
  !(error instanceof Error) ||
@@ -131,6 +162,13 @@ interface MessageEstimate {
131
162
  export class Tokenizer {
132
163
  readonly #encoding: natives.Encoding | null;
133
164
 
165
+ /** Exact counts only; byte fallbacks remain mode-dependent and uncached. */
166
+ readonly #nativeCounts = new LRUCache<string, number>({
167
+ max: 256,
168
+ maxSize: 512 * 1024,
169
+ sizeCalculation: (_tokens, text) => text.length * 2 + 64,
170
+ });
171
+
134
172
  /**
135
173
  * Per-message estimate memo. Keyed by message identity, deliberately not a
136
174
  * symbol-tagged property: callers spread messages to derive throwaway
@@ -150,9 +188,10 @@ export class Tokenizer {
150
188
  }
151
189
 
152
190
  countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
153
- if (mode === "strict") return countTokensNat(text, this.#encoding, mode).tokens;
154
- if (!testEnv && this.#encoding !== null) return countTokensNat(text, this.#encoding, mode).tokens;
155
- if (accurate) return countTokensNat(text, undefined, mode).tokens;
191
+ if (mode === "strict") return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
192
+ if (!testEnv && this.#encoding !== null)
193
+ return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
194
+ if (accurate) return countTokensNat(text, undefined, mode, this.#nativeCounts).tokens;
156
195
  return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
157
196
  }
158
197
 
@@ -170,7 +209,7 @@ export class Tokenizer {
170
209
  checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
171
210
  const bound = sumFragments(text, byteLength);
172
211
  if (bound <= budget) return { fits: true, tokens: bound, exact: false };
173
- const result = countTokensNat(text, this.#encoding, "strict");
212
+ const result = countTokensNat(text, this.#encoding, "strict", this.#nativeCounts);
174
213
  return { fits: result.tokens <= budget, tokens: result.tokens, exact: result.exact };
175
214
  }
176
215