@oh-my-pi/pi-agent-core 18.4.0 → 18.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/dist/types/compaction/openai.d.ts +11 -0
- package/dist/types/output-budget.d.ts +12 -1
- package/package.json +8 -8
- package/src/agent-loop.ts +32 -7
- package/src/compaction/compaction-v2-streaming.ts +11 -2
- package/src/compaction/compaction.ts +2 -0
- package/src/compaction/openai.ts +26 -1
- package/src/output-budget.ts +14 -2
- package/src/tokenizer.ts +45 -6
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,29 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.4.2] - 2026-09-28
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Added tool_execution_end events that fire as each tool call settles for live UI updates
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
|
|
13
|
+
- Emitted tool result messages in the order of tool calls, preserving call order regardless of completion order
|
|
14
|
+
- Reduced repeated token-counting work with a bounded, model-scoped cache of exact text and short-message fragment counts.
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- Fixed an issue where streaming tool call arguments could be incorrectly modified in-place
|
|
19
|
+
|
|
20
|
+
## [18.4.1] - 2026-09-28
|
|
21
|
+
|
|
22
|
+
### Fixed
|
|
23
|
+
|
|
24
|
+
- Fixed fitted output caps overshooting the context window by a few tokens on strict Chat Completions hosts (e.g. llama.cpp), causing 400s.
|
|
25
|
+
- Fixed native remote compaction sending requests already estimated past the model's context window (e.g. after re-expanding history behind another provider's native boundary); it now fails fast so the next configured compaction method runs ([#13502](https://github.com/can1357/oh-my-pi/issues/13502))
|
|
26
|
+
- Fixed V2 remote compaction retrying a standalone stream `error` event three times and reporting it as `stream closed before response.completed`; the upstream status, code, and message (e.g. `context_too_large`) are now surfaced ([#13502](https://github.com/can1357/oh-my-pi/issues/13502))
|
|
27
|
+
|
|
5
28
|
## [18.4.0] - 2026-09-28
|
|
6
29
|
|
|
7
30
|
### Changed
|
|
@@ -33,6 +33,8 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
33
33
|
rewrittenOutputs: number;
|
|
34
34
|
estimatedTokensBefore: number;
|
|
35
35
|
estimatedTokensAfter: number;
|
|
36
|
+
/** Whether `input` fits the model window; false means it must not be sent. */
|
|
37
|
+
fits: boolean;
|
|
36
38
|
}
|
|
37
39
|
/**
|
|
38
40
|
* Preserve the full native transcript unless trailing tool outputs alone push a
|
|
@@ -41,6 +43,15 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
41
43
|
* matching Codex's recovery path for oversized tool turns.
|
|
42
44
|
*/
|
|
43
45
|
export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, tokenizer: Tokenizer, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
|
|
46
|
+
/**
|
|
47
|
+
* Refuse a native compaction request whose prepared input cannot fit the model
|
|
48
|
+
* window, before any network I/O. Re-expanded history behind an unreadable
|
|
49
|
+
* native boundary can exceed the window even when live context does not.
|
|
50
|
+
*
|
|
51
|
+
* @throws Error flagged `ContextOverflow` when `trimmed.fits` is false, so
|
|
52
|
+
* compaction callers skip retries and advance to the next method.
|
|
53
|
+
*/
|
|
54
|
+
export declare function assertRemoteCompactionInputFits(trimmed: TrimRemoteCompactionInputResult, model: Model): void;
|
|
44
55
|
export type OpenAiRemoteCompactionItem = {
|
|
45
56
|
type: "compaction" | "compaction_summary";
|
|
46
57
|
encrypted_content?: string;
|
|
@@ -2,6 +2,16 @@ import type { Context, Model } from "@oh-my-pi/pi-ai";
|
|
|
2
2
|
import type { Tokenizer } from "./tokenizer.js";
|
|
3
3
|
/** Smallest output cap {@link fitOutputTokensToContextWindow} will request. */
|
|
4
4
|
export declare const MIN_FITTED_OUTPUT_TOKENS = 1024;
|
|
5
|
+
/**
|
|
6
|
+
* Absolute headway subtracted from the remaining room (unconditionally, on anchored and
|
|
7
|
+
* fully-local counts alike). Proportional padding covers
|
|
8
|
+
* tokenizer drift that scales with the prompt, but a host can still count a few tokens more
|
|
9
|
+
* than any local estimate can see (chat-template framing, reasoning wrappers). Measured
|
|
10
|
+
* against a 262,144-token host: fitted caps landed +11 to +42 tokens over and were rejected
|
|
11
|
+
* with 400s. 64 tokens covers the observed drift with margin to spare; the
|
|
12
|
+
* {@link MIN_FITTED_OUTPUT_TOKENS} floor still applies.
|
|
13
|
+
*/
|
|
14
|
+
export declare const OUTPUT_FIT_HEADWAY_TOKENS = 64;
|
|
5
15
|
/**
|
|
6
16
|
* Output cap for a request, so prompt plus output stays inside the model's
|
|
7
17
|
* context window.
|
|
@@ -26,7 +36,8 @@ export declare const MIN_FITTED_OUTPUT_TOKENS = 1024;
|
|
|
26
36
|
* OpenRouter-hosted model with no caller cap: the transport omits the catalog
|
|
27
37
|
* default there so each upstream self-caps, and a fitted value would turn into
|
|
28
38
|
* an explicit cap that filters upstreams). Otherwise returns
|
|
29
|
-
* the remaining room
|
|
39
|
+
* the remaining room minus {@link OUTPUT_FIT_HEADWAY_TOKENS} (never below
|
|
40
|
+
* {@link MIN_FITTED_OUTPUT_TOKENS}); a
|
|
30
41
|
* prompt that fills the whole window still overflows and is left to the
|
|
31
42
|
* caller's compaction. Near a full window the floor means a turn can stop on
|
|
32
43
|
* `length` instead of failing with a 400.
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-agent-core",
|
|
4
|
-
"version": "18.4.
|
|
4
|
+
"version": "18.4.2",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": {
|
|
@@ -38,16 +38,16 @@
|
|
|
38
38
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
39
39
|
},
|
|
40
40
|
"dependencies": {
|
|
41
|
-
"@oh-my-pi/pi-ai": "18.4.
|
|
42
|
-
"@oh-my-pi/pi-catalog": "18.4.
|
|
43
|
-
"@oh-my-pi/pi-natives": "18.4.
|
|
44
|
-
"@oh-my-pi/pi-utils": "18.4.
|
|
45
|
-
"@oh-my-pi/pi-wire": "18.4.
|
|
46
|
-
"@oh-my-pi/snapcompact": "18.4.
|
|
41
|
+
"@oh-my-pi/pi-ai": "18.4.2",
|
|
42
|
+
"@oh-my-pi/pi-catalog": "18.4.2",
|
|
43
|
+
"@oh-my-pi/pi-natives": "18.4.2",
|
|
44
|
+
"@oh-my-pi/pi-utils": "18.4.2",
|
|
45
|
+
"@oh-my-pi/pi-wire": "18.4.2",
|
|
46
|
+
"@oh-my-pi/snapcompact": "18.4.2",
|
|
47
47
|
"@opentelemetry/api": "^1.9.1"
|
|
48
48
|
},
|
|
49
49
|
"devDependencies": {
|
|
50
|
-
"@oh-my-pi/omptype": "18.4.
|
|
50
|
+
"@oh-my-pi/omptype": "18.4.2",
|
|
51
51
|
"@opentelemetry/context-async-hooks": "^2.9.0",
|
|
52
52
|
"@opentelemetry/sdk-trace-base": "^2.9.0",
|
|
53
53
|
"@types/bun": "^1.3.14"
|
package/src/agent-loop.ts
CHANGED
|
@@ -51,7 +51,7 @@ import {
|
|
|
51
51
|
recoverHarmonyToolCall,
|
|
52
52
|
signalListLabel,
|
|
53
53
|
} from "@oh-my-pi/pi-ai/utils/harmony-leak";
|
|
54
|
-
import { logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
|
54
|
+
import { cloneJsonTree, logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
|
55
55
|
import { INTENT_FIELD } from "@oh-my-pi/pi-wire";
|
|
56
56
|
import { LiveSteeringChannel } from "./live-steering";
|
|
57
57
|
import { agentPauseGate } from "./pause";
|
|
@@ -377,13 +377,17 @@ function snapshotAssistantContentBlock(block: AssistantContentBlock): AssistantC
|
|
|
377
377
|
case "redactedThinking":
|
|
378
378
|
return { ...block };
|
|
379
379
|
case "anthropicServerTool":
|
|
380
|
-
return { ...block, block:
|
|
380
|
+
return { ...block, block: cloneJsonTree(block.block) };
|
|
381
381
|
case "fallback":
|
|
382
382
|
return { ...block, from: { ...block.from }, to: { ...block.to } };
|
|
383
383
|
case "toolCall": {
|
|
384
384
|
const snap = {
|
|
385
385
|
...block,
|
|
386
|
-
arguments
|
|
386
|
+
// Providers mutate streaming arguments in place (owned-stream, GLM)
|
|
387
|
+
// as well as replacing them, so containers are always copied; the
|
|
388
|
+
// strings inside are immutable and shared, keeping the per-delta
|
|
389
|
+
// cost independent of the argument payload size.
|
|
390
|
+
arguments: cloneJsonTree(block.arguments),
|
|
387
391
|
providerMetadata: snapshotToolCallProviderMetadata(block.providerMetadata),
|
|
388
392
|
};
|
|
389
393
|
// Object spread copies enumerable symbols in Bun, but the Cursor
|
|
@@ -2859,6 +2863,13 @@ async function prepareToolCallDispatch(
|
|
|
2859
2863
|
if (toolCall.type !== "toolCall") continue;
|
|
2860
2864
|
if ((toolCall as CursorExecResolvedCarrier)[kCursorExecResolved] === true) continue;
|
|
2861
2865
|
const tool = resolveToolForCall(context.tools, toolCall, resolveFallbackTool);
|
|
2866
|
+
// A host fallback accepts aliases (`xd://recall`, a mis-separated MCP
|
|
2867
|
+
// name) that providers reject when replayed as a function-call name.
|
|
2868
|
+
// Record the call under the resolved tool's canonical name so history,
|
|
2869
|
+
// persistence, and replay agree; custom-wire calls keep their wire name.
|
|
2870
|
+
if (tool && toolCall.name !== tool.name && toolCall.name !== tool.customWireName) {
|
|
2871
|
+
toolCall.name = tool.name;
|
|
2872
|
+
}
|
|
2862
2873
|
const entry: PreparedToolCall = { tool, args: toolCall.arguments as Record<string, unknown> };
|
|
2863
2874
|
prepared.set(toolCall.id, entry);
|
|
2864
2875
|
let argsForExecution = toolCall.arguments as Record<string, unknown>;
|
|
@@ -3020,6 +3031,11 @@ async function speculativeFinalCalls(
|
|
|
3020
3031
|
/**
|
|
3021
3032
|
* Execute tool calls from an assistant message. Returns model-visible context
|
|
3022
3033
|
* only after every result has settled, preserving assistant call order.
|
|
3034
|
+
*
|
|
3035
|
+
* `tool_execution_end` fires as each call settles so live UI updates promptly;
|
|
3036
|
+
* result `message_start`/`message_end` events (which append to agent state and
|
|
3037
|
+
* the persisted session) are held until every earlier call has a result, so
|
|
3038
|
+
* history always pairs results in call order regardless of completion order.
|
|
3023
3039
|
*/
|
|
3024
3040
|
async function executeToolCalls(
|
|
3025
3041
|
currentContext: AgentContext,
|
|
@@ -3206,6 +3222,18 @@ async function executeToolCalls(
|
|
|
3206
3222
|
await checkAsideInterrupts();
|
|
3207
3223
|
};
|
|
3208
3224
|
|
|
3225
|
+
// Index of the first record whose result message has not been emitted yet.
|
|
3226
|
+
let nextResultIndex = 0;
|
|
3227
|
+
const flushResultMessages = (): void => {
|
|
3228
|
+
for (; nextResultIndex < records.length; nextResultIndex++) {
|
|
3229
|
+
const message = records[nextResultIndex].toolResultMessage;
|
|
3230
|
+
if (!message) return;
|
|
3231
|
+
emittedToolResults.push(message);
|
|
3232
|
+
stream.push({ type: "message_start", message });
|
|
3233
|
+
stream.push({ type: "message_end", message });
|
|
3234
|
+
}
|
|
3235
|
+
};
|
|
3236
|
+
|
|
3209
3237
|
const emitToolResult = (record: (typeof records)[number], result: AgentToolResult<any>, isError: boolean): void => {
|
|
3210
3238
|
if (record.resultEmitted) return;
|
|
3211
3239
|
const { toolCall } = record;
|
|
@@ -3241,10 +3269,7 @@ async function executeToolCalls(
|
|
|
3241
3269
|
record.isError = isError;
|
|
3242
3270
|
record.toolResultMessage = toolResultMessage;
|
|
3243
3271
|
record.resultEmitted = true;
|
|
3244
|
-
|
|
3245
|
-
|
|
3246
|
-
stream.push({ type: "message_start", message: toolResultMessage });
|
|
3247
|
-
stream.push({ type: "message_end", message: toolResultMessage });
|
|
3272
|
+
flushResultMessages();
|
|
3248
3273
|
};
|
|
3249
3274
|
|
|
3250
3275
|
const runTool = async (record: (typeof records)[number], index: number): Promise<void> => {
|
|
@@ -597,6 +597,14 @@ function handleCompactionV2Event(
|
|
|
597
597
|
if (type === "response.failed" || type === "response.incomplete") {
|
|
598
598
|
throw new Error(formatCompactionV2Failure(event, type));
|
|
599
599
|
}
|
|
600
|
+
|
|
601
|
+
// A standalone `error` event terminates the stream. Keep its status so a
|
|
602
|
+
// deterministic 4xx (e.g. context_too_large) is not retried as a dropped stream.
|
|
603
|
+
if (type === "error") {
|
|
604
|
+
const message = formatCompactionV2Failure(event, type);
|
|
605
|
+
const status = numberField(event, "status");
|
|
606
|
+
throw status === undefined ? new Error(message) : new AIError.ProviderHttpError(message, status);
|
|
607
|
+
}
|
|
600
608
|
}
|
|
601
609
|
|
|
602
610
|
function parseCompactionV2Usage(event: Record<string, unknown>): CompactionV2Usage | undefined {
|
|
@@ -629,8 +637,9 @@ function formatCompactionV2Failure(event: Record<string, unknown>, type: string)
|
|
|
629
637
|
: response && isRecord(response.error)
|
|
630
638
|
? response.error
|
|
631
639
|
: undefined;
|
|
632
|
-
|
|
633
|
-
const
|
|
640
|
+
// Responses `error` events carry code/message at the top level.
|
|
641
|
+
const message = stringField(error ?? event, "message");
|
|
642
|
+
const code = error ? (stringField(error, "code") ?? stringField(error, "type")) : stringField(event, "code");
|
|
634
643
|
return `V2 compaction stream ${type}${code ? ` (${code})` : ""}${message ? `: ${message}` : ""}`;
|
|
635
644
|
}
|
|
636
645
|
|
|
@@ -69,6 +69,7 @@ import {
|
|
|
69
69
|
defaultConvertToLlm,
|
|
70
70
|
} from "./messages";
|
|
71
71
|
import {
|
|
72
|
+
assertRemoteCompactionInputFits,
|
|
72
73
|
buildOpenAiNativeHistory,
|
|
73
74
|
getPreservedOpenAiRemoteCompactionData,
|
|
74
75
|
isOpenAiRemoteCompactionApi,
|
|
@@ -1748,6 +1749,7 @@ export async function compact(
|
|
|
1748
1749
|
contextWindow: model.contextWindow,
|
|
1749
1750
|
});
|
|
1750
1751
|
}
|
|
1752
|
+
assertRemoteCompactionInputFits(trimmed, model);
|
|
1751
1753
|
const requestOptions = {
|
|
1752
1754
|
sessionId: summaryOptions.sessionId,
|
|
1753
1755
|
promptCacheKey: summaryOptions.promptCacheKey,
|
package/src/compaction/openai.ts
CHANGED
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* with `{ summary, shortSummary? }`.
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
|
-
import { ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
|
18
|
+
import { attach, create, Flag, ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
|
19
19
|
import { getCodexAttestationHeader } from "@oh-my-pi/pi-ai/providers/openai-codex-attestation";
|
|
20
20
|
import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-compaction";
|
|
21
21
|
import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer";
|
|
@@ -122,6 +122,8 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
122
122
|
rewrittenOutputs: number;
|
|
123
123
|
estimatedTokensBefore: number;
|
|
124
124
|
estimatedTokensAfter: number;
|
|
125
|
+
/** Whether `input` fits the model window; false means it must not be sent. */
|
|
126
|
+
fits: boolean;
|
|
125
127
|
}
|
|
126
128
|
|
|
127
129
|
/** Verdict for one remote-compaction request measured against the model window. */
|
|
@@ -199,6 +201,7 @@ export function trimRemoteCompactionInputToContextWindow(
|
|
|
199
201
|
rewrittenOutputs: 0,
|
|
200
202
|
estimatedTokensBefore: before.tokens,
|
|
201
203
|
estimatedTokensAfter: before.tokens,
|
|
204
|
+
fits: true,
|
|
202
205
|
};
|
|
203
206
|
}
|
|
204
207
|
|
|
@@ -222,6 +225,7 @@ export function trimRemoteCompactionInputToContextWindow(
|
|
|
222
225
|
rewrittenOutputs: 0,
|
|
223
226
|
estimatedTokensBefore: before.tokens,
|
|
224
227
|
estimatedTokensAfter: before.tokens,
|
|
228
|
+
fits: false,
|
|
225
229
|
};
|
|
226
230
|
}
|
|
227
231
|
|
|
@@ -230,9 +234,29 @@ export function trimRemoteCompactionInputToContextWindow(
|
|
|
230
234
|
rewrittenOutputs,
|
|
231
235
|
estimatedTokensBefore: before.tokens,
|
|
232
236
|
estimatedTokensAfter: after.tokens,
|
|
237
|
+
fits: true,
|
|
233
238
|
};
|
|
234
239
|
}
|
|
235
240
|
|
|
241
|
+
/**
|
|
242
|
+
* Refuse a native compaction request whose prepared input cannot fit the model
|
|
243
|
+
* window, before any network I/O. Re-expanded history behind an unreadable
|
|
244
|
+
* native boundary can exceed the window even when live context does not.
|
|
245
|
+
*
|
|
246
|
+
* @throws Error flagged `ContextOverflow` when `trimmed.fits` is false, so
|
|
247
|
+
* compaction callers skip retries and advance to the next method.
|
|
248
|
+
*/
|
|
249
|
+
export function assertRemoteCompactionInputFits(trimmed: TrimRemoteCompactionInputResult, model: Model): void {
|
|
250
|
+
if (trimmed.fits) return;
|
|
251
|
+
throw attach(
|
|
252
|
+
new Error(
|
|
253
|
+
`Remote compaction input exceeds the context window of ${model.provider}/${model.id}: ` +
|
|
254
|
+
`estimated ${trimmed.estimatedTokensAfter} tokens > ${model.contextWindow}`,
|
|
255
|
+
),
|
|
256
|
+
create(Flag.ContextOverflow),
|
|
257
|
+
);
|
|
258
|
+
}
|
|
259
|
+
|
|
236
260
|
/** Race the caller's signal against the request timeout; `timeoutMs <= 0` disables the watchdog. */
|
|
237
261
|
function withRequestTimeout(signal: AbortSignal | undefined, timeoutMs: number): AbortSignal | undefined {
|
|
238
262
|
if (timeoutMs <= 0) return signal;
|
|
@@ -794,6 +818,7 @@ export async function requestOpenAiRemoteCompaction(
|
|
|
794
818
|
contextWindow: model.contextWindow,
|
|
795
819
|
});
|
|
796
820
|
}
|
|
821
|
+
assertRemoteCompactionInputFits(trimmed, model);
|
|
797
822
|
const request: OpenAiRemoteCompactionRequest = {
|
|
798
823
|
model: requestModel,
|
|
799
824
|
// Preserve the native transcript. Only oversized trailing tool outputs are
|
package/src/output-budget.ts
CHANGED
|
@@ -17,6 +17,17 @@ export const MIN_FITTED_OUTPUT_TOKENS = 1024;
|
|
|
17
17
|
*/
|
|
18
18
|
const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
|
|
19
19
|
|
|
20
|
+
/**
|
|
21
|
+
* Absolute headway subtracted from the remaining room (unconditionally, on anchored and
|
|
22
|
+
* fully-local counts alike). Proportional padding covers
|
|
23
|
+
* tokenizer drift that scales with the prompt, but a host can still count a few tokens more
|
|
24
|
+
* than any local estimate can see (chat-template framing, reasoning wrappers). Measured
|
|
25
|
+
* against a 262,144-token host: fitted caps landed +11 to +42 tokens over and were rejected
|
|
26
|
+
* with 400s. 64 tokens covers the observed drift with margin to spare; the
|
|
27
|
+
* {@link MIN_FITTED_OUTPUT_TOKENS} floor still applies.
|
|
28
|
+
*/
|
|
29
|
+
export const OUTPUT_FIT_HEADWAY_TOKENS = 64;
|
|
30
|
+
|
|
20
31
|
/**
|
|
21
32
|
* Output cap for a request, so prompt plus output stays inside the model's
|
|
22
33
|
* context window.
|
|
@@ -41,7 +52,8 @@ const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
|
|
|
41
52
|
* OpenRouter-hosted model with no caller cap: the transport omits the catalog
|
|
42
53
|
* default there so each upstream self-caps, and a fitted value would turn into
|
|
43
54
|
* an explicit cap that filters upstreams). Otherwise returns
|
|
44
|
-
* the remaining room
|
|
55
|
+
* the remaining room minus {@link OUTPUT_FIT_HEADWAY_TOKENS} (never below
|
|
56
|
+
* {@link MIN_FITTED_OUTPUT_TOKENS}); a
|
|
45
57
|
* prompt that fills the whole window still overflows and is left to the
|
|
46
58
|
* caller's compaction. Near a full window the floor means a turn can stop on
|
|
47
59
|
* `length` instead of failing with a 400.
|
|
@@ -67,7 +79,7 @@ export function fitOutputTokensToContextWindow(
|
|
|
67
79
|
if (!requested || !contextWindow || contextWindow <= 0) return maxTokens;
|
|
68
80
|
if (stopsOutputAtContextWindow(model)) return maxTokens;
|
|
69
81
|
|
|
70
|
-
const room = contextWindow - countPromptTokens(context, tokenizer);
|
|
82
|
+
const room = contextWindow - countPromptTokens(context, tokenizer) - OUTPUT_FIT_HEADWAY_TOKENS;
|
|
71
83
|
if (room >= requested) return maxTokens;
|
|
72
84
|
return Math.max(MIN_FITTED_OUTPUT_TOKENS, room);
|
|
73
85
|
}
|
package/src/tokenizer.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import type { Model } from "@oh-my-pi/pi-ai";
|
|
2
2
|
import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
|
|
3
3
|
import * as natives from "@oh-my-pi/pi-natives";
|
|
4
|
-
import { stringifyJson } from "@oh-my-pi/pi-utils";
|
|
4
|
+
import { materializeString, stringifyJson } from "@oh-my-pi/pi-utils";
|
|
5
|
+
import { LRUCache } from "@oh-my-pi/pi-utils/lru";
|
|
5
6
|
import * as snapcompact from "@oh-my-pi/snapcompact";
|
|
6
7
|
import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
|
|
7
8
|
import type { AgentMessage } from "./types";
|
|
@@ -67,13 +68,43 @@ interface NativeTokenCount {
|
|
|
67
68
|
exact: boolean;
|
|
68
69
|
}
|
|
69
70
|
|
|
71
|
+
// Growing streamed text and large tool results must not evict the reusable
|
|
72
|
+
// short fragments. Account for UTF-16 key storage plus a per-entry allowance.
|
|
73
|
+
const NATIVE_CACHE_MAX_LENGTH = 16 * 1024;
|
|
74
|
+
|
|
75
|
+
function countNativeFragment(
|
|
76
|
+
text: string,
|
|
77
|
+
encoding: natives.Encoding | null | undefined,
|
|
78
|
+
counts: LRUCache<string, number>,
|
|
79
|
+
): number {
|
|
80
|
+
if (text.length > NATIVE_CACHE_MAX_LENGTH) return natives.countTokens(text, encoding);
|
|
81
|
+
const cached = counts.get(text);
|
|
82
|
+
if (cached !== undefined) return cached;
|
|
83
|
+
const tokens = natives.countTokens(text, encoding);
|
|
84
|
+
// Detach sliced strings so a small key cannot retain a much larger source.
|
|
85
|
+
counts.set(materializeString(text), tokens);
|
|
86
|
+
return tokens;
|
|
87
|
+
}
|
|
88
|
+
|
|
70
89
|
function countTokensNat(
|
|
71
90
|
text: string | string[],
|
|
72
91
|
encoding: natives.Encoding | null | undefined,
|
|
73
92
|
mode: TokenCountMode,
|
|
93
|
+
counts: LRUCache<string, number>,
|
|
74
94
|
): NativeTokenCount {
|
|
75
95
|
try {
|
|
76
|
-
|
|
96
|
+
let tokens: number;
|
|
97
|
+
if (typeof text === "string") {
|
|
98
|
+
tokens = countNativeFragment(text, encoding, counts);
|
|
99
|
+
} else if (text.length > 0 && text.length < 16) {
|
|
100
|
+
// The native API sums independent fragments, not their concatenation.
|
|
101
|
+
// Keep its parallel batch path for arrays of 16 or more fragments.
|
|
102
|
+
tokens = 0;
|
|
103
|
+
for (const fragment of text) tokens += countNativeFragment(fragment, encoding, counts);
|
|
104
|
+
} else {
|
|
105
|
+
tokens = natives.countTokens(text, encoding);
|
|
106
|
+
}
|
|
107
|
+
return { tokens, exact: true };
|
|
77
108
|
} catch (error) {
|
|
78
109
|
if (
|
|
79
110
|
!(error instanceof Error) ||
|
|
@@ -131,6 +162,13 @@ interface MessageEstimate {
|
|
|
131
162
|
export class Tokenizer {
|
|
132
163
|
readonly #encoding: natives.Encoding | null;
|
|
133
164
|
|
|
165
|
+
/** Exact counts only; byte fallbacks remain mode-dependent and uncached. */
|
|
166
|
+
readonly #nativeCounts = new LRUCache<string, number>({
|
|
167
|
+
max: 256,
|
|
168
|
+
maxSize: 512 * 1024,
|
|
169
|
+
sizeCalculation: (_tokens, text) => text.length * 2 + 64,
|
|
170
|
+
});
|
|
171
|
+
|
|
134
172
|
/**
|
|
135
173
|
* Per-message estimate memo. Keyed by message identity, deliberately not a
|
|
136
174
|
* symbol-tagged property: callers spread messages to derive throwaway
|
|
@@ -150,9 +188,10 @@ export class Tokenizer {
|
|
|
150
188
|
}
|
|
151
189
|
|
|
152
190
|
countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
|
|
153
|
-
if (mode === "strict") return countTokensNat(text, this.#encoding, mode).tokens;
|
|
154
|
-
if (!testEnv && this.#encoding !== null)
|
|
155
|
-
|
|
191
|
+
if (mode === "strict") return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
|
|
192
|
+
if (!testEnv && this.#encoding !== null)
|
|
193
|
+
return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
|
|
194
|
+
if (accurate) return countTokensNat(text, undefined, mode, this.#nativeCounts).tokens;
|
|
156
195
|
return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
|
|
157
196
|
}
|
|
158
197
|
|
|
@@ -170,7 +209,7 @@ export class Tokenizer {
|
|
|
170
209
|
checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
|
|
171
210
|
const bound = sumFragments(text, byteLength);
|
|
172
211
|
if (bound <= budget) return { fits: true, tokens: bound, exact: false };
|
|
173
|
-
const result = countTokensNat(text, this.#encoding, "strict");
|
|
212
|
+
const result = countTokensNat(text, this.#encoding, "strict", this.#nativeCounts);
|
|
174
213
|
return { fits: result.tokens <= budget, tokens: result.tokens, exact: result.exact };
|
|
175
214
|
}
|
|
176
215
|
|