@oh-my-pi/pi-agent-core 18.4.1 → 18.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/dist/types/agent.d.ts +4 -0
- package/dist/types/speculative-execution.d.ts +2 -1
- package/dist/types/tool-context.d.ts +3 -2
- package/dist/types/types.d.ts +25 -1
- package/package.json +8 -8
- package/src/agent-loop.ts +31 -15
- package/src/agent.ts +6 -0
- package/src/compaction/openai.ts +8 -3
- package/src/speculative-execution.ts +12 -0
- package/src/tokenizer.ts +45 -6
- package/src/tool-context.ts +9 -3
- package/src/types.ts +27 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,33 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.4.3] - 2026-09-28
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Added `transformAssistantMessagePreservesToolCalls`, letting stream speculation and direct speculative candidates run under a `transformAssistantMessage` that never rewrites streamed tool calls
|
|
10
|
+
- Added `authorizeLaunch` to the speculative execution host and coordinator so tool stream sessions can start host-approved effectful work (e.g. subagents) before their call dispatches
|
|
11
|
+
|
|
12
|
+
### Fixed
|
|
13
|
+
|
|
14
|
+
- Fixed auto-compaction with the `remote` method failing on long Codex/OpenAI sessions with "Remote compaction input exceeds the context window" ([#13611](https://github.com/can1357/oh-my-pi/issues/13611))
|
|
15
|
+
- Fixed passive tool-call context being repeated when several calls in one batch returned the same text; identical per-call context is now delivered once, at its first position ([#13633](https://github.com/can1357/oh-my-pi/pull/13633) by [@andrebrait](https://github.com/andrebrait))
|
|
16
|
+
|
|
17
|
+
## [18.4.2] - 2026-09-28
|
|
18
|
+
|
|
19
|
+
### Added
|
|
20
|
+
|
|
21
|
+
- Added tool_execution_end events that fire as each tool call settles for live UI updates
|
|
22
|
+
|
|
23
|
+
### Changed
|
|
24
|
+
|
|
25
|
+
- Emitted tool result messages in the order of tool calls, preserving call order regardless of completion order
|
|
26
|
+
- Reduced repeated token-counting work with a bounded, model-scoped cache of exact text and short-message fragment counts.
|
|
27
|
+
|
|
28
|
+
### Fixed
|
|
29
|
+
|
|
30
|
+
- Fixed an issue where streaming tool call arguments could be incorrectly modified in-place
|
|
31
|
+
|
|
5
32
|
## [18.4.1] - 2026-09-28
|
|
6
33
|
|
|
7
34
|
### Fixed
|
package/dist/types/agent.d.ts
CHANGED
|
@@ -221,6 +221,8 @@ export interface AgentOptions {
|
|
|
221
221
|
* tool-call arguments). See {@link AgentLoopConfig.transformAssistantMessage}.
|
|
222
222
|
*/
|
|
223
223
|
transformAssistantMessage?: AgentLoopConfig["transformAssistantMessage"];
|
|
224
|
+
/** See {@link AgentLoopConfig.transformAssistantMessagePreservesToolCalls}. */
|
|
225
|
+
transformAssistantMessagePreservesToolCalls?: boolean;
|
|
224
226
|
/**
|
|
225
227
|
* Opt-in OpenTelemetry instrumentation. Passing `{}` enables the loop's
|
|
226
228
|
* GenAI-semantic-convention spans using the global tracer provider. See
|
|
@@ -257,6 +259,8 @@ export declare class Agent {
|
|
|
257
259
|
* UI emission, and tool dispatch. Reassign at any time to swap the implementation.
|
|
258
260
|
*/
|
|
259
261
|
transformAssistantMessage?: AgentLoopConfig["transformAssistantMessage"];
|
|
262
|
+
/** Declares {@link transformAssistantMessage} never rewrites streamed tool calls; reassign alongside it. */
|
|
263
|
+
transformAssistantMessagePreservesToolCalls?: boolean;
|
|
260
264
|
/**
|
|
261
265
|
* Hook that peeks whether interrupting IRC asides are queued for the next boundary.
|
|
262
266
|
*/
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { type AssistantMessage } from "@oh-my-pi/pi-ai";
|
|
2
|
-
import type { AgentContext, AgentLoopConfig, AgentTool, AgentToolCall, AgentToolResult, SpeculativeChildDefinition, SpeculativeChildHandle, SpeculativeToolExecutionConfig, ToolSpeculationEffect, ToolSpeculationStreamSession } from "./types.js";
|
|
2
|
+
import type { AgentContext, AgentLoopConfig, AgentTool, AgentToolCall, AgentToolResult, SpeculativeAuthorization, SpeculativeChildDefinition, SpeculativeChildHandle, SpeculativeLaunchContext, SpeculativeToolExecutionConfig, ToolSpeculationEffect, ToolSpeculationStreamSession } from "./types.js";
|
|
3
3
|
export type SpeculativeRawOutcome = {
|
|
4
4
|
result: AgentToolResult<unknown>;
|
|
5
5
|
isError: boolean;
|
|
@@ -55,6 +55,7 @@ export declare class SpeculativeOperationCoordinator {
|
|
|
55
55
|
*/
|
|
56
56
|
directExecutionArgsFor(toolCallId: string, rawArgs: Readonly<Record<string, unknown>>): Record<string, unknown> | undefined;
|
|
57
57
|
reconcileFinalCalls(calls: ReadonlyMap<string, AgentToolCall>): Promise<void>;
|
|
58
|
+
authorizeLaunch(context: SpeculativeLaunchContext): Promise<SpeculativeAuthorization>;
|
|
58
59
|
discardChildren(parentToolCallId: string, reason: string): Promise<void>;
|
|
59
60
|
claim(tool: AgentTool | undefined, toolCall: AgentToolCall, args: Record<string, unknown>): Promise<SpeculativeRawOutcome | undefined>;
|
|
60
61
|
admitFinalized(context: AgentContext, toolCall: AgentToolCall, loopConfig: AgentLoopConfig, signal: AbortSignal | undefined): void;
|
|
@@ -20,8 +20,9 @@ export type ToolResultWithAdditionalContext = ToolResultMessage & {
|
|
|
20
20
|
*/
|
|
21
21
|
export declare function isNonBlankContext(value: unknown): value is string;
|
|
22
22
|
/**
|
|
23
|
-
* Join passive context values in order, dropping blanks
|
|
24
|
-
*
|
|
23
|
+
* Join passive context values in order, dropping blanks and repeats of an
|
|
24
|
+
* earlier value (compared without surrounding whitespace; the first original
|
|
25
|
+
* is kept). Returns undefined when nothing remains.
|
|
25
26
|
*/
|
|
26
27
|
export declare function joinAdditionalContext(values: Iterable<string | undefined>): string | undefined;
|
|
27
28
|
/**
|
package/dist/types/types.d.ts
CHANGED
|
@@ -515,6 +515,13 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
|
|
515
515
|
* the turn).
|
|
516
516
|
*/
|
|
517
517
|
transformAssistantMessage?: (message: AssistantMessage, signal?: AbortSignal) => Promise<void> | void;
|
|
518
|
+
/**
|
|
519
|
+
* Declares that {@link transformAssistantMessage} never rewrites or removes a
|
|
520
|
+
* tool call the model streamed (it may edit text or append new calls). Stream
|
|
521
|
+
* speculation sessions and direct speculative candidates plan from streamed
|
|
522
|
+
* calls, so they stay disabled under a transform unless this is set.
|
|
523
|
+
*/
|
|
524
|
+
transformAssistantMessagePreservesToolCalls?: boolean;
|
|
518
525
|
/**
|
|
519
526
|
* Called after a tool finishes executing, before `tool_execution_end` and the
|
|
520
527
|
* tool-result message are emitted.
|
|
@@ -659,8 +666,22 @@ export interface SpeculativeExecutionHost {
|
|
|
659
666
|
validate?(context: SpeculativeCommitContext): boolean | Promise<boolean>;
|
|
660
667
|
commit?(context: SpeculativeCommitContext, commitDefault: () => Promise<AgentToolResult<unknown>>): Promise<SpeculativeCommitDecision>;
|
|
661
668
|
discard?(context: SpeculativeDiscardContext): void | Promise<void>;
|
|
669
|
+
/**
|
|
670
|
+
* Authorize a stream session to start effectful work (e.g. subagents) from
|
|
671
|
+
* partially streamed arguments. The session owns that work and must abort it
|
|
672
|
+
* when the finalized call is invalid, blocked, or changed. Hosts without this
|
|
673
|
+
* hook deny every launch.
|
|
674
|
+
*/
|
|
675
|
+
authorizeLaunch?(context: SpeculativeLaunchContext): SpeculativeAuthorization | Promise<SpeculativeAuthorization>;
|
|
662
676
|
close?(reason: string): void | Promise<void>;
|
|
663
677
|
}
|
|
678
|
+
/** Effectful work a tool-owned stream session asks to start before its outer call dispatches. */
|
|
679
|
+
export interface SpeculativeLaunchContext {
|
|
680
|
+
tool: SpeculativeToolReference;
|
|
681
|
+
toolCall: AgentToolCall;
|
|
682
|
+
/** Arguments the launch was planned from: the streamed prefix of the outer call. */
|
|
683
|
+
args: Readonly<Record<string, unknown>>;
|
|
684
|
+
}
|
|
664
685
|
export interface ToolSpeculationStreamContext {
|
|
665
686
|
readonly coordinator: SpeculativeOperationSink;
|
|
666
687
|
readonly parentToolCallId: string;
|
|
@@ -703,6 +724,8 @@ export interface ToolSpeculationStreamSession {
|
|
|
703
724
|
export interface SpeculativeOperationSink {
|
|
704
725
|
readonly maxInFlight: number;
|
|
705
726
|
admit(definition: SpeculativeChildDefinition): Promise<SpeculativeChildHandle | undefined>;
|
|
727
|
+
/** Host-gated permission for effectful stream work; see {@link SpeculativeExecutionHost.authorizeLaunch}. */
|
|
728
|
+
authorizeLaunch?(context: SpeculativeLaunchContext): Promise<SpeculativeAuthorization>;
|
|
706
729
|
discardChildren?(parentToolCallId: string, reason: string): void | Promise<void>;
|
|
707
730
|
close(reason: string): void | Promise<void>;
|
|
708
731
|
}
|
|
@@ -755,7 +778,8 @@ export interface SpeculativeToolExecutionConfig {
|
|
|
755
778
|
* ignored when `block` is true.
|
|
756
779
|
*
|
|
757
780
|
* Set `additionalContext` to attach passive model-visible context to this call.
|
|
758
|
-
* Non-empty values from a tool batch are injected in assistant tool-call order
|
|
781
|
+
* Non-empty values from a tool batch are injected in assistant tool-call order,
|
|
782
|
+
* a value identical to an earlier one in the batch only once,
|
|
759
783
|
* after every result settles and before the next provider request. It is
|
|
760
784
|
* dropped when the call is blocked or skipped, or when its final result is an
|
|
761
785
|
* error (including an approval denial raised by the tool's own gate). Within a
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-agent-core",
|
|
4
|
-
"version": "18.4.
|
|
4
|
+
"version": "18.4.3",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": {
|
|
@@ -38,16 +38,16 @@
|
|
|
38
38
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
39
39
|
},
|
|
40
40
|
"dependencies": {
|
|
41
|
-
"@oh-my-pi/pi-ai": "18.4.
|
|
42
|
-
"@oh-my-pi/pi-catalog": "18.4.
|
|
43
|
-
"@oh-my-pi/pi-natives": "18.4.
|
|
44
|
-
"@oh-my-pi/pi-utils": "18.4.
|
|
45
|
-
"@oh-my-pi/pi-wire": "18.4.
|
|
46
|
-
"@oh-my-pi/snapcompact": "18.4.
|
|
41
|
+
"@oh-my-pi/pi-ai": "18.4.3",
|
|
42
|
+
"@oh-my-pi/pi-catalog": "18.4.3",
|
|
43
|
+
"@oh-my-pi/pi-natives": "18.4.3",
|
|
44
|
+
"@oh-my-pi/pi-utils": "18.4.3",
|
|
45
|
+
"@oh-my-pi/pi-wire": "18.4.3",
|
|
46
|
+
"@oh-my-pi/snapcompact": "18.4.3",
|
|
47
47
|
"@opentelemetry/api": "^1.9.1"
|
|
48
48
|
},
|
|
49
49
|
"devDependencies": {
|
|
50
|
-
"@oh-my-pi/omptype": "18.4.
|
|
50
|
+
"@oh-my-pi/omptype": "18.4.3",
|
|
51
51
|
"@opentelemetry/context-async-hooks": "^2.9.0",
|
|
52
52
|
"@opentelemetry/sdk-trace-base": "^2.9.0",
|
|
53
53
|
"@types/bun": "^1.3.14"
|
package/src/agent-loop.ts
CHANGED
|
@@ -51,7 +51,7 @@ import {
|
|
|
51
51
|
recoverHarmonyToolCall,
|
|
52
52
|
signalListLabel,
|
|
53
53
|
} from "@oh-my-pi/pi-ai/utils/harmony-leak";
|
|
54
|
-
import { logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
|
54
|
+
import { cloneJsonTree, logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
|
55
55
|
import { INTENT_FIELD } from "@oh-my-pi/pi-wire";
|
|
56
56
|
import { LiveSteeringChannel } from "./live-steering";
|
|
57
57
|
import { agentPauseGate } from "./pause";
|
|
@@ -377,13 +377,17 @@ function snapshotAssistantContentBlock(block: AssistantContentBlock): AssistantC
|
|
|
377
377
|
case "redactedThinking":
|
|
378
378
|
return { ...block };
|
|
379
379
|
case "anthropicServerTool":
|
|
380
|
-
return { ...block, block:
|
|
380
|
+
return { ...block, block: cloneJsonTree(block.block) };
|
|
381
381
|
case "fallback":
|
|
382
382
|
return { ...block, from: { ...block.from }, to: { ...block.to } };
|
|
383
383
|
case "toolCall": {
|
|
384
384
|
const snap = {
|
|
385
385
|
...block,
|
|
386
|
-
arguments
|
|
386
|
+
// Providers mutate streaming arguments in place (owned-stream, GLM)
|
|
387
|
+
// as well as replacing them, so containers are always copied; the
|
|
388
|
+
// strings inside are immutable and shared, keeping the per-delta
|
|
389
|
+
// cost independent of the argument payload size.
|
|
390
|
+
arguments: cloneJsonTree(block.arguments),
|
|
387
391
|
providerMetadata: snapshotToolCallProviderMetadata(block.providerMetadata),
|
|
388
392
|
};
|
|
389
393
|
// Object spread copies enumerable symbols in Bun, but the Cursor
|
|
@@ -2094,6 +2098,8 @@ async function streamAssistantResponse(
|
|
|
2094
2098
|
signal: requestSignal,
|
|
2095
2099
|
})
|
|
2096
2100
|
: undefined;
|
|
2101
|
+
const speculationPlansFromStream =
|
|
2102
|
+
!config.transformAssistantMessage || config.transformAssistantMessagePreservesToolCalls === true;
|
|
2097
2103
|
|
|
2098
2104
|
let providerStreamSettled = false;
|
|
2099
2105
|
let speculationSettled = false;
|
|
@@ -2323,15 +2329,11 @@ async function streamAssistantResponse(
|
|
|
2323
2329
|
case "toolcall_delta":
|
|
2324
2330
|
case "toolcall_end":
|
|
2325
2331
|
if (partialMessage) {
|
|
2326
|
-
if (
|
|
2327
|
-
event.type === "toolcall_start" &&
|
|
2328
|
-
speculationCoordinator &&
|
|
2329
|
-
!config.transformAssistantMessage
|
|
2330
|
-
) {
|
|
2332
|
+
if (event.type === "toolcall_start" && speculationCoordinator && speculationPlansFromStream) {
|
|
2331
2333
|
// Stream sessions plan from pre-transform arguments, exactly like
|
|
2332
2334
|
// direct candidates (see admitFinalized below): with a transformer
|
|
2333
|
-
//
|
|
2334
|
-
// work started from the original would be phantom I/O.
|
|
2335
|
+
// that may rewrite calls, the authoritative call may differ, so any
|
|
2336
|
+
// speculative work started from the original would be phantom I/O.
|
|
2335
2337
|
speculationCoordinator.register(event.contentIndex);
|
|
2336
2338
|
const toolCall = event.partial.content[event.contentIndex];
|
|
2337
2339
|
if (toolCall?.type === "toolCall") {
|
|
@@ -2428,7 +2430,7 @@ async function streamAssistantResponse(
|
|
|
2428
2430
|
event.type === "toolcall_end" &&
|
|
2429
2431
|
speculationCoordinator &&
|
|
2430
2432
|
speculationConfig &&
|
|
2431
|
-
|
|
2433
|
+
speculationPlansFromStream
|
|
2432
2434
|
) {
|
|
2433
2435
|
speculationCoordinator.admitFinalized(context, event.toolCall, config, requestSignal);
|
|
2434
2436
|
}
|
|
@@ -3027,6 +3029,11 @@ async function speculativeFinalCalls(
|
|
|
3027
3029
|
/**
|
|
3028
3030
|
* Execute tool calls from an assistant message. Returns model-visible context
|
|
3029
3031
|
* only after every result has settled, preserving assistant call order.
|
|
3032
|
+
*
|
|
3033
|
+
* `tool_execution_end` fires as each call settles so live UI updates promptly;
|
|
3034
|
+
* result `message_start`/`message_end` events (which append to agent state and
|
|
3035
|
+
* the persisted session) are held until every earlier call has a result, so
|
|
3036
|
+
* history always pairs results in call order regardless of completion order.
|
|
3030
3037
|
*/
|
|
3031
3038
|
async function executeToolCalls(
|
|
3032
3039
|
currentContext: AgentContext,
|
|
@@ -3213,6 +3220,18 @@ async function executeToolCalls(
|
|
|
3213
3220
|
await checkAsideInterrupts();
|
|
3214
3221
|
};
|
|
3215
3222
|
|
|
3223
|
+
// Index of the first record whose result message has not been emitted yet.
|
|
3224
|
+
let nextResultIndex = 0;
|
|
3225
|
+
const flushResultMessages = (): void => {
|
|
3226
|
+
for (; nextResultIndex < records.length; nextResultIndex++) {
|
|
3227
|
+
const message = records[nextResultIndex].toolResultMessage;
|
|
3228
|
+
if (!message) return;
|
|
3229
|
+
emittedToolResults.push(message);
|
|
3230
|
+
stream.push({ type: "message_start", message });
|
|
3231
|
+
stream.push({ type: "message_end", message });
|
|
3232
|
+
}
|
|
3233
|
+
};
|
|
3234
|
+
|
|
3216
3235
|
const emitToolResult = (record: (typeof records)[number], result: AgentToolResult<any>, isError: boolean): void => {
|
|
3217
3236
|
if (record.resultEmitted) return;
|
|
3218
3237
|
const { toolCall } = record;
|
|
@@ -3248,10 +3267,7 @@ async function executeToolCalls(
|
|
|
3248
3267
|
record.isError = isError;
|
|
3249
3268
|
record.toolResultMessage = toolResultMessage;
|
|
3250
3269
|
record.resultEmitted = true;
|
|
3251
|
-
|
|
3252
|
-
|
|
3253
|
-
stream.push({ type: "message_start", message: toolResultMessage });
|
|
3254
|
-
stream.push({ type: "message_end", message: toolResultMessage });
|
|
3270
|
+
flushResultMessages();
|
|
3255
3271
|
};
|
|
3256
3272
|
|
|
3257
3273
|
const runTool = async (record: (typeof records)[number], index: number): Promise<void> => {
|
package/src/agent.ts
CHANGED
|
@@ -346,6 +346,8 @@ export interface AgentOptions {
|
|
|
346
346
|
* tool-call arguments). See {@link AgentLoopConfig.transformAssistantMessage}.
|
|
347
347
|
*/
|
|
348
348
|
transformAssistantMessage?: AgentLoopConfig["transformAssistantMessage"];
|
|
349
|
+
/** See {@link AgentLoopConfig.transformAssistantMessagePreservesToolCalls}. */
|
|
350
|
+
transformAssistantMessagePreservesToolCalls?: boolean;
|
|
349
351
|
|
|
350
352
|
/**
|
|
351
353
|
* Opt-in OpenTelemetry instrumentation. Passing `{}` enables the loop's
|
|
@@ -511,6 +513,8 @@ export class Agent {
|
|
|
511
513
|
* UI emission, and tool dispatch. Reassign at any time to swap the implementation.
|
|
512
514
|
*/
|
|
513
515
|
transformAssistantMessage?: AgentLoopConfig["transformAssistantMessage"];
|
|
516
|
+
/** Declares {@link transformAssistantMessage} never rewrites streamed tool calls; reassign alongside it. */
|
|
517
|
+
transformAssistantMessagePreservesToolCalls?: boolean;
|
|
514
518
|
/**
|
|
515
519
|
* Hook that peeks whether interrupting IRC asides are queued for the next boundary.
|
|
516
520
|
*/
|
|
@@ -576,6 +580,7 @@ export class Agent {
|
|
|
576
580
|
this.beforeToolCall = opts.beforeToolCall;
|
|
577
581
|
this.afterToolCall = opts.afterToolCall;
|
|
578
582
|
this.transformAssistantMessage = opts.transformAssistantMessage;
|
|
583
|
+
this.transformAssistantMessagePreservesToolCalls = opts.transformAssistantMessagePreservesToolCalls;
|
|
579
584
|
this.#telemetry = opts.telemetry;
|
|
580
585
|
this.#appendOnlyContext = opts.appendOnlyContext;
|
|
581
586
|
this.#transformProviderContext = opts.transformProviderContext;
|
|
@@ -1750,6 +1755,7 @@ export class Agent {
|
|
|
1750
1755
|
transformAssistantMessage: this.transformAssistantMessage
|
|
1751
1756
|
? (message, signal) => this.transformAssistantMessage?.(message, signal)
|
|
1752
1757
|
: undefined,
|
|
1758
|
+
transformAssistantMessagePreservesToolCalls: this.transformAssistantMessagePreservesToolCalls,
|
|
1753
1759
|
onAssistantMessageEvent: this.#onAssistantMessageEvent,
|
|
1754
1760
|
onHarmonyLeak: this.#onHarmonyLeak,
|
|
1755
1761
|
onTurnEnd: (messages, signal, context) => this.#onTurnEnd?.(messages, signal, context),
|
package/src/compaction/openai.ts
CHANGED
|
@@ -110,6 +110,10 @@ function normalizeRemoteCompactionEstimateValue(value: unknown): NormalizedEstim
|
|
|
110
110
|
const normalized: Record<string, unknown> = {};
|
|
111
111
|
let imageTokens = 0;
|
|
112
112
|
for (const [key, item] of Object.entries(record)) {
|
|
113
|
+
// Opaque encrypted reasoning/compaction state: its local base64 size far
|
|
114
|
+
// exceeds what the provider bills, so it stays out of the fit estimate
|
|
115
|
+
// (same policy as `MessageCountOptions.excludeEncryptedReasoning`).
|
|
116
|
+
if (key === "encrypted_content" && typeof item === "string") continue;
|
|
113
117
|
const result = normalizeRemoteCompactionEstimateValue(item);
|
|
114
118
|
normalized[key] = result.value;
|
|
115
119
|
imageTokens += result.imageTokens;
|
|
@@ -137,9 +141,10 @@ interface RemoteCompactionBudgetProbe {
|
|
|
137
141
|
/**
|
|
138
142
|
* Cheap-first sizing of a remote-compaction request. Images and the request
|
|
139
143
|
* frame are charged flat, so they come off the budget rather than through the
|
|
140
|
-
* tokenizer;
|
|
141
|
-
* {@link Tokenizer.checkTokenBudget}, which only
|
|
142
|
-
* the byte bound cannot already prove the request
|
|
144
|
+
* tokenizer; opaque `encrypted_content` payloads are excluded. The serialized
|
|
145
|
+
* transcript is then probed with {@link Tokenizer.checkTokenBudget}, which only
|
|
146
|
+
* pays for an exact count when the byte bound cannot already prove the request
|
|
147
|
+
* fits.
|
|
143
148
|
*/
|
|
144
149
|
function probeRemoteCompactionInputBudget(
|
|
145
150
|
input: Array<Record<string, unknown>>,
|
|
@@ -10,6 +10,7 @@ import type {
|
|
|
10
10
|
SpeculativeChildDefinition,
|
|
11
11
|
SpeculativeChildHandle,
|
|
12
12
|
SpeculativeCommitContext,
|
|
13
|
+
SpeculativeLaunchContext,
|
|
13
14
|
SpeculativeOperationContext,
|
|
14
15
|
SpeculativePhysicalOutcome,
|
|
15
16
|
SpeculativeResourceAccess,
|
|
@@ -520,6 +521,17 @@ export class SpeculativeOperationCoordinator {
|
|
|
520
521
|
}
|
|
521
522
|
}
|
|
522
523
|
|
|
524
|
+
async authorizeLaunch(context: SpeculativeLaunchContext): Promise<SpeculativeAuthorization> {
|
|
525
|
+
if (this.#closed) return { allowed: false, reason: "speculation coordinator is closed" };
|
|
526
|
+
const authorize = this.config.host?.authorizeLaunch;
|
|
527
|
+
if (!authorize) return { allowed: false, reason: "host does not authorize speculative launches" };
|
|
528
|
+
try {
|
|
529
|
+
return await authorize.call(this.config.host, context);
|
|
530
|
+
} catch {
|
|
531
|
+
return { allowed: false, reason: "host launch authorization failed" };
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
|
|
523
535
|
async discardChildren(parentToolCallId: string, reason: string): Promise<void> {
|
|
524
536
|
await this.#admission;
|
|
525
537
|
await Promise.all(
|
package/src/tokenizer.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import type { Model } from "@oh-my-pi/pi-ai";
|
|
2
2
|
import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
|
|
3
3
|
import * as natives from "@oh-my-pi/pi-natives";
|
|
4
|
-
import { stringifyJson } from "@oh-my-pi/pi-utils";
|
|
4
|
+
import { materializeString, stringifyJson } from "@oh-my-pi/pi-utils";
|
|
5
|
+
import { LRUCache } from "@oh-my-pi/pi-utils/lru";
|
|
5
6
|
import * as snapcompact from "@oh-my-pi/snapcompact";
|
|
6
7
|
import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
|
|
7
8
|
import type { AgentMessage } from "./types";
|
|
@@ -67,13 +68,43 @@ interface NativeTokenCount {
|
|
|
67
68
|
exact: boolean;
|
|
68
69
|
}
|
|
69
70
|
|
|
71
|
+
// Growing streamed text and large tool results must not evict the reusable
|
|
72
|
+
// short fragments. Account for UTF-16 key storage plus a per-entry allowance.
|
|
73
|
+
const NATIVE_CACHE_MAX_LENGTH = 16 * 1024;
|
|
74
|
+
|
|
75
|
+
function countNativeFragment(
|
|
76
|
+
text: string,
|
|
77
|
+
encoding: natives.Encoding | null | undefined,
|
|
78
|
+
counts: LRUCache<string, number>,
|
|
79
|
+
): number {
|
|
80
|
+
if (text.length > NATIVE_CACHE_MAX_LENGTH) return natives.countTokens(text, encoding);
|
|
81
|
+
const cached = counts.get(text);
|
|
82
|
+
if (cached !== undefined) return cached;
|
|
83
|
+
const tokens = natives.countTokens(text, encoding);
|
|
84
|
+
// Detach sliced strings so a small key cannot retain a much larger source.
|
|
85
|
+
counts.set(materializeString(text), tokens);
|
|
86
|
+
return tokens;
|
|
87
|
+
}
|
|
88
|
+
|
|
70
89
|
function countTokensNat(
|
|
71
90
|
text: string | string[],
|
|
72
91
|
encoding: natives.Encoding | null | undefined,
|
|
73
92
|
mode: TokenCountMode,
|
|
93
|
+
counts: LRUCache<string, number>,
|
|
74
94
|
): NativeTokenCount {
|
|
75
95
|
try {
|
|
76
|
-
|
|
96
|
+
let tokens: number;
|
|
97
|
+
if (typeof text === "string") {
|
|
98
|
+
tokens = countNativeFragment(text, encoding, counts);
|
|
99
|
+
} else if (text.length > 0 && text.length < 16) {
|
|
100
|
+
// The native API sums independent fragments, not their concatenation.
|
|
101
|
+
// Keep its parallel batch path for arrays of 16 or more fragments.
|
|
102
|
+
tokens = 0;
|
|
103
|
+
for (const fragment of text) tokens += countNativeFragment(fragment, encoding, counts);
|
|
104
|
+
} else {
|
|
105
|
+
tokens = natives.countTokens(text, encoding);
|
|
106
|
+
}
|
|
107
|
+
return { tokens, exact: true };
|
|
77
108
|
} catch (error) {
|
|
78
109
|
if (
|
|
79
110
|
!(error instanceof Error) ||
|
|
@@ -131,6 +162,13 @@ interface MessageEstimate {
|
|
|
131
162
|
export class Tokenizer {
|
|
132
163
|
readonly #encoding: natives.Encoding | null;
|
|
133
164
|
|
|
165
|
+
/** Exact counts only; byte fallbacks remain mode-dependent and uncached. */
|
|
166
|
+
readonly #nativeCounts = new LRUCache<string, number>({
|
|
167
|
+
max: 256,
|
|
168
|
+
maxSize: 512 * 1024,
|
|
169
|
+
sizeCalculation: (_tokens, text) => text.length * 2 + 64,
|
|
170
|
+
});
|
|
171
|
+
|
|
134
172
|
/**
|
|
135
173
|
* Per-message estimate memo. Keyed by message identity, deliberately not a
|
|
136
174
|
* symbol-tagged property: callers spread messages to derive throwaway
|
|
@@ -150,9 +188,10 @@ export class Tokenizer {
|
|
|
150
188
|
}
|
|
151
189
|
|
|
152
190
|
countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
|
|
153
|
-
if (mode === "strict") return countTokensNat(text, this.#encoding, mode).tokens;
|
|
154
|
-
if (!testEnv && this.#encoding !== null)
|
|
155
|
-
|
|
191
|
+
if (mode === "strict") return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
|
|
192
|
+
if (!testEnv && this.#encoding !== null)
|
|
193
|
+
return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
|
|
194
|
+
if (accurate) return countTokensNat(text, undefined, mode, this.#nativeCounts).tokens;
|
|
156
195
|
return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
|
|
157
196
|
}
|
|
158
197
|
|
|
@@ -170,7 +209,7 @@ export class Tokenizer {
|
|
|
170
209
|
checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
|
|
171
210
|
const bound = sumFragments(text, byteLength);
|
|
172
211
|
if (bound <= budget) return { fits: true, tokens: bound, exact: false };
|
|
173
|
-
const result = countTokensNat(text, this.#encoding, "strict");
|
|
212
|
+
const result = countTokensNat(text, this.#encoding, "strict", this.#nativeCounts);
|
|
174
213
|
return { fits: result.tokens <= budget, tokens: result.tokens, exact: result.exact };
|
|
175
214
|
}
|
|
176
215
|
|
package/src/tool-context.ts
CHANGED
|
@@ -24,13 +24,19 @@ export function isNonBlankContext(value: unknown): value is string {
|
|
|
24
24
|
}
|
|
25
25
|
|
|
26
26
|
/**
|
|
27
|
-
* Join passive context values in order, dropping blanks
|
|
28
|
-
*
|
|
27
|
+
* Join passive context values in order, dropping blanks and repeats of an
|
|
28
|
+
* earlier value (compared without surrounding whitespace; the first original
|
|
29
|
+
* is kept). Returns undefined when nothing remains.
|
|
29
30
|
*/
|
|
30
31
|
export function joinAdditionalContext(values: Iterable<string | undefined>): string | undefined {
|
|
32
|
+
const seen = new Set<string>();
|
|
31
33
|
const kept: string[] = [];
|
|
32
34
|
for (const value of values) {
|
|
33
|
-
if (isNonBlankContext(value))
|
|
35
|
+
if (!isNonBlankContext(value)) continue;
|
|
36
|
+
const key = value.trim();
|
|
37
|
+
if (seen.has(key)) continue;
|
|
38
|
+
seen.add(key);
|
|
39
|
+
kept.push(value);
|
|
34
40
|
}
|
|
35
41
|
return kept.length > 0 ? kept.join("\n\n") : undefined;
|
|
36
42
|
}
|
package/src/types.ts
CHANGED
|
@@ -595,6 +595,14 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
|
|
595
595
|
*/
|
|
596
596
|
transformAssistantMessage?: (message: AssistantMessage, signal?: AbortSignal) => Promise<void> | void;
|
|
597
597
|
|
|
598
|
+
/**
|
|
599
|
+
* Declares that {@link transformAssistantMessage} never rewrites or removes a
|
|
600
|
+
* tool call the model streamed (it may edit text or append new calls). Stream
|
|
601
|
+
* speculation sessions and direct speculative candidates plan from streamed
|
|
602
|
+
* calls, so they stay disabled under a transform unless this is set.
|
|
603
|
+
*/
|
|
604
|
+
transformAssistantMessagePreservesToolCalls?: boolean;
|
|
605
|
+
|
|
598
606
|
/**
|
|
599
607
|
* Called after a tool finishes executing, before `tool_execution_end` and the
|
|
600
608
|
* tool-result message are emitted.
|
|
@@ -740,9 +748,24 @@ export interface SpeculativeExecutionHost {
|
|
|
740
748
|
commitDefault: () => Promise<AgentToolResult<unknown>>,
|
|
741
749
|
): Promise<SpeculativeCommitDecision>;
|
|
742
750
|
discard?(context: SpeculativeDiscardContext): void | Promise<void>;
|
|
751
|
+
/**
|
|
752
|
+
* Authorize a stream session to start effectful work (e.g. subagents) from
|
|
753
|
+
* partially streamed arguments. The session owns that work and must abort it
|
|
754
|
+
* when the finalized call is invalid, blocked, or changed. Hosts without this
|
|
755
|
+
* hook deny every launch.
|
|
756
|
+
*/
|
|
757
|
+
authorizeLaunch?(context: SpeculativeLaunchContext): SpeculativeAuthorization | Promise<SpeculativeAuthorization>;
|
|
743
758
|
close?(reason: string): void | Promise<void>;
|
|
744
759
|
}
|
|
745
760
|
|
|
761
|
+
/** Effectful work a tool-owned stream session asks to start before its outer call dispatches. */
|
|
762
|
+
export interface SpeculativeLaunchContext {
|
|
763
|
+
tool: SpeculativeToolReference;
|
|
764
|
+
toolCall: AgentToolCall;
|
|
765
|
+
/** Arguments the launch was planned from: the streamed prefix of the outer call. */
|
|
766
|
+
args: Readonly<Record<string, unknown>>;
|
|
767
|
+
}
|
|
768
|
+
|
|
746
769
|
export interface ToolSpeculationStreamContext {
|
|
747
770
|
readonly coordinator: SpeculativeOperationSink;
|
|
748
771
|
readonly parentToolCallId: string;
|
|
@@ -789,6 +812,8 @@ export interface ToolSpeculationStreamSession {
|
|
|
789
812
|
export interface SpeculativeOperationSink {
|
|
790
813
|
readonly maxInFlight: number;
|
|
791
814
|
admit(definition: SpeculativeChildDefinition): Promise<SpeculativeChildHandle | undefined>;
|
|
815
|
+
/** Host-gated permission for effectful stream work; see {@link SpeculativeExecutionHost.authorizeLaunch}. */
|
|
816
|
+
authorizeLaunch?(context: SpeculativeLaunchContext): Promise<SpeculativeAuthorization>;
|
|
792
817
|
discardChildren?(parentToolCallId: string, reason: string): void | Promise<void>;
|
|
793
818
|
close(reason: string): void | Promise<void>;
|
|
794
819
|
}
|
|
@@ -850,7 +875,8 @@ export interface SpeculativeToolExecutionConfig {
|
|
|
850
875
|
* ignored when `block` is true.
|
|
851
876
|
*
|
|
852
877
|
* Set `additionalContext` to attach passive model-visible context to this call.
|
|
853
|
-
* Non-empty values from a tool batch are injected in assistant tool-call order
|
|
878
|
+
* Non-empty values from a tool batch are injected in assistant tool-call order,
|
|
879
|
+
* a value identical to an earlier one in the batch only once,
|
|
854
880
|
* after every result settles and before the next provider request. It is
|
|
855
881
|
* dropped when the call is blocked or skipped, or when its final result is an
|
|
856
882
|
* error (including an approval denial raised by the tool's own gate). Within a
|