@oh-my-pi/pi-agent-core 18.4.1 → 18.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,33 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.4.3] - 2026-09-28
6
+
7
+ ### Added
8
+
9
+ - Added `transformAssistantMessagePreservesToolCalls`, letting stream speculation and direct speculative candidates run under a `transformAssistantMessage` that never rewrites streamed tool calls
10
+ - Added `authorizeLaunch` to the speculative execution host and coordinator so tool stream sessions can start host-approved effectful work (e.g. subagents) before their call dispatches
11
+
12
+ ### Fixed
13
+
14
+ - Fixed auto-compaction with the `remote` method failing on long Codex/OpenAI sessions with "Remote compaction input exceeds the context window" ([#13611](https://github.com/can1357/oh-my-pi/issues/13611))
15
+ - Fixed passive tool-call context being repeated when several calls in one batch returned the same text; identical per-call context is now delivered once, at its first position ([#13633](https://github.com/can1357/oh-my-pi/pull/13633) by [@andrebrait](https://github.com/andrebrait))
16
+
17
+ ## [18.4.2] - 2026-09-28
18
+
19
+ ### Added
20
+
21
+ - Added tool_execution_end events that fire as each tool call settles for live UI updates
22
+
23
+ ### Changed
24
+
25
+ - Emitted tool result messages in the order of tool calls, preserving call order regardless of completion order
26
+ - Reduced repeated token-counting work with a bounded, model-scoped cache of exact text and short-message fragment counts.
27
+
28
+ ### Fixed
29
+
30
+ - Fixed an issue where streaming tool call arguments could be incorrectly modified in-place
31
+
5
32
  ## [18.4.1] - 2026-09-28
6
33
 
7
34
  ### Fixed
@@ -221,6 +221,8 @@ export interface AgentOptions {
221
221
  * tool-call arguments). See {@link AgentLoopConfig.transformAssistantMessage}.
222
222
  */
223
223
  transformAssistantMessage?: AgentLoopConfig["transformAssistantMessage"];
224
+ /** See {@link AgentLoopConfig.transformAssistantMessagePreservesToolCalls}. */
225
+ transformAssistantMessagePreservesToolCalls?: boolean;
224
226
  /**
225
227
  * Opt-in OpenTelemetry instrumentation. Passing `{}` enables the loop's
226
228
  * GenAI-semantic-convention spans using the global tracer provider. See
@@ -257,6 +259,8 @@ export declare class Agent {
257
259
  * UI emission, and tool dispatch. Reassign at any time to swap the implementation.
258
260
  */
259
261
  transformAssistantMessage?: AgentLoopConfig["transformAssistantMessage"];
262
+ /** Declares {@link transformAssistantMessage} never rewrites streamed tool calls; reassign alongside it. */
263
+ transformAssistantMessagePreservesToolCalls?: boolean;
260
264
  /**
261
265
  * Hook that peeks whether interrupting IRC asides are queued for the next boundary.
262
266
  */
@@ -1,5 +1,5 @@
1
1
  import { type AssistantMessage } from "@oh-my-pi/pi-ai";
2
- import type { AgentContext, AgentLoopConfig, AgentTool, AgentToolCall, AgentToolResult, SpeculativeChildDefinition, SpeculativeChildHandle, SpeculativeToolExecutionConfig, ToolSpeculationEffect, ToolSpeculationStreamSession } from "./types.js";
2
+ import type { AgentContext, AgentLoopConfig, AgentTool, AgentToolCall, AgentToolResult, SpeculativeAuthorization, SpeculativeChildDefinition, SpeculativeChildHandle, SpeculativeLaunchContext, SpeculativeToolExecutionConfig, ToolSpeculationEffect, ToolSpeculationStreamSession } from "./types.js";
3
3
  export type SpeculativeRawOutcome = {
4
4
  result: AgentToolResult<unknown>;
5
5
  isError: boolean;
@@ -55,6 +55,7 @@ export declare class SpeculativeOperationCoordinator {
55
55
  */
56
56
  directExecutionArgsFor(toolCallId: string, rawArgs: Readonly<Record<string, unknown>>): Record<string, unknown> | undefined;
57
57
  reconcileFinalCalls(calls: ReadonlyMap<string, AgentToolCall>): Promise<void>;
58
+ authorizeLaunch(context: SpeculativeLaunchContext): Promise<SpeculativeAuthorization>;
58
59
  discardChildren(parentToolCallId: string, reason: string): Promise<void>;
59
60
  claim(tool: AgentTool | undefined, toolCall: AgentToolCall, args: Record<string, unknown>): Promise<SpeculativeRawOutcome | undefined>;
60
61
  admitFinalized(context: AgentContext, toolCall: AgentToolCall, loopConfig: AgentLoopConfig, signal: AbortSignal | undefined): void;
@@ -20,8 +20,9 @@ export type ToolResultWithAdditionalContext = ToolResultMessage & {
20
20
  */
21
21
  export declare function isNonBlankContext(value: unknown): value is string;
22
22
  /**
23
- * Join passive context values in order, dropping blanks. Returns undefined
24
- * when nothing remains.
23
+ * Join passive context values in order, dropping blanks and repeats of an
24
+ * earlier value (compared without surrounding whitespace; the first original
25
+ * is kept). Returns undefined when nothing remains.
25
26
  */
26
27
  export declare function joinAdditionalContext(values: Iterable<string | undefined>): string | undefined;
27
28
  /**
@@ -515,6 +515,13 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
515
515
  * the turn).
516
516
  */
517
517
  transformAssistantMessage?: (message: AssistantMessage, signal?: AbortSignal) => Promise<void> | void;
518
+ /**
519
+ * Declares that {@link transformAssistantMessage} never rewrites or removes a
520
+ * tool call the model streamed (it may edit text or append new calls). Stream
521
+ * speculation sessions and direct speculative candidates plan from streamed
522
+ * calls, so they stay disabled under a transform unless this is set.
523
+ */
524
+ transformAssistantMessagePreservesToolCalls?: boolean;
518
525
  /**
519
526
  * Called after a tool finishes executing, before `tool_execution_end` and the
520
527
  * tool-result message are emitted.
@@ -659,8 +666,22 @@ export interface SpeculativeExecutionHost {
659
666
  validate?(context: SpeculativeCommitContext): boolean | Promise<boolean>;
660
667
  commit?(context: SpeculativeCommitContext, commitDefault: () => Promise<AgentToolResult<unknown>>): Promise<SpeculativeCommitDecision>;
661
668
  discard?(context: SpeculativeDiscardContext): void | Promise<void>;
669
+ /**
670
+ * Authorize a stream session to start effectful work (e.g. subagents) from
671
+ * partially streamed arguments. The session owns that work and must abort it
672
+ * when the finalized call is invalid, blocked, or changed. Hosts without this
673
+ * hook deny every launch.
674
+ */
675
+ authorizeLaunch?(context: SpeculativeLaunchContext): SpeculativeAuthorization | Promise<SpeculativeAuthorization>;
662
676
  close?(reason: string): void | Promise<void>;
663
677
  }
678
+ /** Effectful work a tool-owned stream session asks to start before its outer call dispatches. */
679
+ export interface SpeculativeLaunchContext {
680
+ tool: SpeculativeToolReference;
681
+ toolCall: AgentToolCall;
682
+ /** Arguments the launch was planned from: the streamed prefix of the outer call. */
683
+ args: Readonly<Record<string, unknown>>;
684
+ }
664
685
  export interface ToolSpeculationStreamContext {
665
686
  readonly coordinator: SpeculativeOperationSink;
666
687
  readonly parentToolCallId: string;
@@ -703,6 +724,8 @@ export interface ToolSpeculationStreamSession {
703
724
  export interface SpeculativeOperationSink {
704
725
  readonly maxInFlight: number;
705
726
  admit(definition: SpeculativeChildDefinition): Promise<SpeculativeChildHandle | undefined>;
727
+ /** Host-gated permission for effectful stream work; see {@link SpeculativeExecutionHost.authorizeLaunch}. */
728
+ authorizeLaunch?(context: SpeculativeLaunchContext): Promise<SpeculativeAuthorization>;
706
729
  discardChildren?(parentToolCallId: string, reason: string): void | Promise<void>;
707
730
  close(reason: string): void | Promise<void>;
708
731
  }
@@ -755,7 +778,8 @@ export interface SpeculativeToolExecutionConfig {
755
778
  * ignored when `block` is true.
756
779
  *
757
780
  * Set `additionalContext` to attach passive model-visible context to this call.
758
- * Non-empty values from a tool batch are injected in assistant tool-call order
781
+ * Non-empty values from a tool batch are injected in assistant tool-call order,
782
+ * a value identical to an earlier one in the batch only once,
759
783
  * after every result settles and before the next provider request. It is
760
784
  * dropped when the call is blocked or skipped, or when its final result is an
761
785
  * error (including an approval denial raised by the tool's own gate). Within a
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "18.4.1",
4
+ "version": "18.4.3",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": {
@@ -38,16 +38,16 @@
38
38
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
39
39
  },
40
40
  "dependencies": {
41
- "@oh-my-pi/pi-ai": "18.4.1",
42
- "@oh-my-pi/pi-catalog": "18.4.1",
43
- "@oh-my-pi/pi-natives": "18.4.1",
44
- "@oh-my-pi/pi-utils": "18.4.1",
45
- "@oh-my-pi/pi-wire": "18.4.1",
46
- "@oh-my-pi/snapcompact": "18.4.1",
41
+ "@oh-my-pi/pi-ai": "18.4.3",
42
+ "@oh-my-pi/pi-catalog": "18.4.3",
43
+ "@oh-my-pi/pi-natives": "18.4.3",
44
+ "@oh-my-pi/pi-utils": "18.4.3",
45
+ "@oh-my-pi/pi-wire": "18.4.3",
46
+ "@oh-my-pi/snapcompact": "18.4.3",
47
47
  "@opentelemetry/api": "^1.9.1"
48
48
  },
49
49
  "devDependencies": {
50
- "@oh-my-pi/omptype": "18.4.1",
50
+ "@oh-my-pi/omptype": "18.4.3",
51
51
  "@opentelemetry/context-async-hooks": "^2.9.0",
52
52
  "@opentelemetry/sdk-trace-base": "^2.9.0",
53
53
  "@types/bun": "^1.3.14"
package/src/agent-loop.ts CHANGED
@@ -51,7 +51,7 @@ import {
51
51
  recoverHarmonyToolCall,
52
52
  signalListLabel,
53
53
  } from "@oh-my-pi/pi-ai/utils/harmony-leak";
54
- import { logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
54
+ import { cloneJsonTree, logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
55
55
  import { INTENT_FIELD } from "@oh-my-pi/pi-wire";
56
56
  import { LiveSteeringChannel } from "./live-steering";
57
57
  import { agentPauseGate } from "./pause";
@@ -377,13 +377,17 @@ function snapshotAssistantContentBlock(block: AssistantContentBlock): AssistantC
377
377
  case "redactedThinking":
378
378
  return { ...block };
379
379
  case "anthropicServerTool":
380
- return { ...block, block: structuredCloneJSON(block.block) };
380
+ return { ...block, block: cloneJsonTree(block.block) };
381
381
  case "fallback":
382
382
  return { ...block, from: { ...block.from }, to: { ...block.to } };
383
383
  case "toolCall": {
384
384
  const snap = {
385
385
  ...block,
386
- arguments: structuredCloneJSON(block.arguments),
386
+ // Providers mutate streaming arguments in place (owned-stream, GLM)
387
+ // as well as replacing them, so containers are always copied; the
388
+ // strings inside are immutable and shared, keeping the per-delta
389
+ // cost independent of the argument payload size.
390
+ arguments: cloneJsonTree(block.arguments),
387
391
  providerMetadata: snapshotToolCallProviderMetadata(block.providerMetadata),
388
392
  };
389
393
  // Object spread copies enumerable symbols in Bun, but the Cursor
@@ -2094,6 +2098,8 @@ async function streamAssistantResponse(
2094
2098
  signal: requestSignal,
2095
2099
  })
2096
2100
  : undefined;
2101
+ const speculationPlansFromStream =
2102
+ !config.transformAssistantMessage || config.transformAssistantMessagePreservesToolCalls === true;
2097
2103
 
2098
2104
  let providerStreamSettled = false;
2099
2105
  let speculationSettled = false;
@@ -2323,15 +2329,11 @@ async function streamAssistantResponse(
2323
2329
  case "toolcall_delta":
2324
2330
  case "toolcall_end":
2325
2331
  if (partialMessage) {
2326
- if (
2327
- event.type === "toolcall_start" &&
2328
- speculationCoordinator &&
2329
- !config.transformAssistantMessage
2330
- ) {
2332
+ if (event.type === "toolcall_start" && speculationCoordinator && speculationPlansFromStream) {
2331
2333
  // Stream sessions plan from pre-transform arguments, exactly like
2332
2334
  // direct candidates (see admitFinalized below): with a transformer
2333
- // installed the authoritative call may differ, so any speculative
2334
- // work started from the original would be phantom I/O.
2335
+ // that may rewrite calls, the authoritative call may differ, so any
2336
+ // speculative work started from the original would be phantom I/O.
2335
2337
  speculationCoordinator.register(event.contentIndex);
2336
2338
  const toolCall = event.partial.content[event.contentIndex];
2337
2339
  if (toolCall?.type === "toolCall") {
@@ -2428,7 +2430,7 @@ async function streamAssistantResponse(
2428
2430
  event.type === "toolcall_end" &&
2429
2431
  speculationCoordinator &&
2430
2432
  speculationConfig &&
2431
- !config.transformAssistantMessage
2433
+ speculationPlansFromStream
2432
2434
  ) {
2433
2435
  speculationCoordinator.admitFinalized(context, event.toolCall, config, requestSignal);
2434
2436
  }
@@ -3027,6 +3029,11 @@ async function speculativeFinalCalls(
3027
3029
  /**
3028
3030
  * Execute tool calls from an assistant message. Returns model-visible context
3029
3031
  * only after every result has settled, preserving assistant call order.
3032
+ *
3033
+ * `tool_execution_end` fires as each call settles so live UI updates promptly;
3034
+ * result `message_start`/`message_end` events (which append to agent state and
3035
+ * the persisted session) are held until every earlier call has a result, so
3036
+ * history always pairs results in call order regardless of completion order.
3030
3037
  */
3031
3038
  async function executeToolCalls(
3032
3039
  currentContext: AgentContext,
@@ -3213,6 +3220,18 @@ async function executeToolCalls(
3213
3220
  await checkAsideInterrupts();
3214
3221
  };
3215
3222
 
3223
+ // Index of the first record whose result message has not been emitted yet.
3224
+ let nextResultIndex = 0;
3225
+ const flushResultMessages = (): void => {
3226
+ for (; nextResultIndex < records.length; nextResultIndex++) {
3227
+ const message = records[nextResultIndex].toolResultMessage;
3228
+ if (!message) return;
3229
+ emittedToolResults.push(message);
3230
+ stream.push({ type: "message_start", message });
3231
+ stream.push({ type: "message_end", message });
3232
+ }
3233
+ };
3234
+
3216
3235
  const emitToolResult = (record: (typeof records)[number], result: AgentToolResult<any>, isError: boolean): void => {
3217
3236
  if (record.resultEmitted) return;
3218
3237
  const { toolCall } = record;
@@ -3248,10 +3267,7 @@ async function executeToolCalls(
3248
3267
  record.isError = isError;
3249
3268
  record.toolResultMessage = toolResultMessage;
3250
3269
  record.resultEmitted = true;
3251
- emittedToolResults.push(toolResultMessage);
3252
-
3253
- stream.push({ type: "message_start", message: toolResultMessage });
3254
- stream.push({ type: "message_end", message: toolResultMessage });
3270
+ flushResultMessages();
3255
3271
  };
3256
3272
 
3257
3273
  const runTool = async (record: (typeof records)[number], index: number): Promise<void> => {
package/src/agent.ts CHANGED
@@ -346,6 +346,8 @@ export interface AgentOptions {
346
346
  * tool-call arguments). See {@link AgentLoopConfig.transformAssistantMessage}.
347
347
  */
348
348
  transformAssistantMessage?: AgentLoopConfig["transformAssistantMessage"];
349
+ /** See {@link AgentLoopConfig.transformAssistantMessagePreservesToolCalls}. */
350
+ transformAssistantMessagePreservesToolCalls?: boolean;
349
351
 
350
352
  /**
351
353
  * Opt-in OpenTelemetry instrumentation. Passing `{}` enables the loop's
@@ -511,6 +513,8 @@ export class Agent {
511
513
  * UI emission, and tool dispatch. Reassign at any time to swap the implementation.
512
514
  */
513
515
  transformAssistantMessage?: AgentLoopConfig["transformAssistantMessage"];
516
+ /** Declares {@link transformAssistantMessage} never rewrites streamed tool calls; reassign alongside it. */
517
+ transformAssistantMessagePreservesToolCalls?: boolean;
514
518
  /**
515
519
  * Hook that peeks whether interrupting IRC asides are queued for the next boundary.
516
520
  */
@@ -576,6 +580,7 @@ export class Agent {
576
580
  this.beforeToolCall = opts.beforeToolCall;
577
581
  this.afterToolCall = opts.afterToolCall;
578
582
  this.transformAssistantMessage = opts.transformAssistantMessage;
583
+ this.transformAssistantMessagePreservesToolCalls = opts.transformAssistantMessagePreservesToolCalls;
579
584
  this.#telemetry = opts.telemetry;
580
585
  this.#appendOnlyContext = opts.appendOnlyContext;
581
586
  this.#transformProviderContext = opts.transformProviderContext;
@@ -1750,6 +1755,7 @@ export class Agent {
1750
1755
  transformAssistantMessage: this.transformAssistantMessage
1751
1756
  ? (message, signal) => this.transformAssistantMessage?.(message, signal)
1752
1757
  : undefined,
1758
+ transformAssistantMessagePreservesToolCalls: this.transformAssistantMessagePreservesToolCalls,
1753
1759
  onAssistantMessageEvent: this.#onAssistantMessageEvent,
1754
1760
  onHarmonyLeak: this.#onHarmonyLeak,
1755
1761
  onTurnEnd: (messages, signal, context) => this.#onTurnEnd?.(messages, signal, context),
@@ -110,6 +110,10 @@ function normalizeRemoteCompactionEstimateValue(value: unknown): NormalizedEstim
110
110
  const normalized: Record<string, unknown> = {};
111
111
  let imageTokens = 0;
112
112
  for (const [key, item] of Object.entries(record)) {
113
+ // Opaque encrypted reasoning/compaction state: its local base64 size far
114
+ // exceeds what the provider bills, so it stays out of the fit estimate
115
+ // (same policy as `MessageCountOptions.excludeEncryptedReasoning`).
116
+ if (key === "encrypted_content" && typeof item === "string") continue;
113
117
  const result = normalizeRemoteCompactionEstimateValue(item);
114
118
  normalized[key] = result.value;
115
119
  imageTokens += result.imageTokens;
@@ -137,9 +141,10 @@ interface RemoteCompactionBudgetProbe {
137
141
  /**
138
142
  * Cheap-first sizing of a remote-compaction request. Images and the request
139
143
  * frame are charged flat, so they come off the budget rather than through the
140
- * tokenizer; the serialized transcript is then probed with
141
- * {@link Tokenizer.checkTokenBudget}, which only pays for an exact count when
142
- * the byte bound cannot already prove the request fits.
144
+ * tokenizer; opaque `encrypted_content` payloads are excluded. The serialized
145
+ * transcript is then probed with {@link Tokenizer.checkTokenBudget}, which only
146
+ * pays for an exact count when the byte bound cannot already prove the request
147
+ * fits.
143
148
  */
144
149
  function probeRemoteCompactionInputBudget(
145
150
  input: Array<Record<string, unknown>>,
@@ -10,6 +10,7 @@ import type {
10
10
  SpeculativeChildDefinition,
11
11
  SpeculativeChildHandle,
12
12
  SpeculativeCommitContext,
13
+ SpeculativeLaunchContext,
13
14
  SpeculativeOperationContext,
14
15
  SpeculativePhysicalOutcome,
15
16
  SpeculativeResourceAccess,
@@ -520,6 +521,17 @@ export class SpeculativeOperationCoordinator {
520
521
  }
521
522
  }
522
523
 
524
+ async authorizeLaunch(context: SpeculativeLaunchContext): Promise<SpeculativeAuthorization> {
525
+ if (this.#closed) return { allowed: false, reason: "speculation coordinator is closed" };
526
+ const authorize = this.config.host?.authorizeLaunch;
527
+ if (!authorize) return { allowed: false, reason: "host does not authorize speculative launches" };
528
+ try {
529
+ return await authorize.call(this.config.host, context);
530
+ } catch {
531
+ return { allowed: false, reason: "host launch authorization failed" };
532
+ }
533
+ }
534
+
523
535
  async discardChildren(parentToolCallId: string, reason: string): Promise<void> {
524
536
  await this.#admission;
525
537
  await Promise.all(
package/src/tokenizer.ts CHANGED
@@ -1,7 +1,8 @@
1
1
  import type { Model } from "@oh-my-pi/pi-ai";
2
2
  import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
3
3
  import * as natives from "@oh-my-pi/pi-natives";
4
- import { stringifyJson } from "@oh-my-pi/pi-utils";
4
+ import { materializeString, stringifyJson } from "@oh-my-pi/pi-utils";
5
+ import { LRUCache } from "@oh-my-pi/pi-utils/lru";
5
6
  import * as snapcompact from "@oh-my-pi/snapcompact";
6
7
  import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
7
8
  import type { AgentMessage } from "./types";
@@ -67,13 +68,43 @@ interface NativeTokenCount {
67
68
  exact: boolean;
68
69
  }
69
70
 
71
+ // Growing streamed text and large tool results must not evict the reusable
72
+ // short fragments. Account for UTF-16 key storage plus a per-entry allowance.
73
+ const NATIVE_CACHE_MAX_LENGTH = 16 * 1024;
74
+
75
+ function countNativeFragment(
76
+ text: string,
77
+ encoding: natives.Encoding | null | undefined,
78
+ counts: LRUCache<string, number>,
79
+ ): number {
80
+ if (text.length > NATIVE_CACHE_MAX_LENGTH) return natives.countTokens(text, encoding);
81
+ const cached = counts.get(text);
82
+ if (cached !== undefined) return cached;
83
+ const tokens = natives.countTokens(text, encoding);
84
+ // Detach sliced strings so a small key cannot retain a much larger source.
85
+ counts.set(materializeString(text), tokens);
86
+ return tokens;
87
+ }
88
+
70
89
  function countTokensNat(
71
90
  text: string | string[],
72
91
  encoding: natives.Encoding | null | undefined,
73
92
  mode: TokenCountMode,
93
+ counts: LRUCache<string, number>,
74
94
  ): NativeTokenCount {
75
95
  try {
76
- return { tokens: natives.countTokens(text, encoding), exact: true };
96
+ let tokens: number;
97
+ if (typeof text === "string") {
98
+ tokens = countNativeFragment(text, encoding, counts);
99
+ } else if (text.length > 0 && text.length < 16) {
100
+ // The native API sums independent fragments, not their concatenation.
101
+ // Keep its parallel batch path for arrays of 16 or more fragments.
102
+ tokens = 0;
103
+ for (const fragment of text) tokens += countNativeFragment(fragment, encoding, counts);
104
+ } else {
105
+ tokens = natives.countTokens(text, encoding);
106
+ }
107
+ return { tokens, exact: true };
77
108
  } catch (error) {
78
109
  if (
79
110
  !(error instanceof Error) ||
@@ -131,6 +162,13 @@ interface MessageEstimate {
131
162
  export class Tokenizer {
132
163
  readonly #encoding: natives.Encoding | null;
133
164
 
165
+ /** Exact counts only; byte fallbacks remain mode-dependent and uncached. */
166
+ readonly #nativeCounts = new LRUCache<string, number>({
167
+ max: 256,
168
+ maxSize: 512 * 1024,
169
+ sizeCalculation: (_tokens, text) => text.length * 2 + 64,
170
+ });
171
+
134
172
  /**
135
173
  * Per-message estimate memo. Keyed by message identity, deliberately not a
136
174
  * symbol-tagged property: callers spread messages to derive throwaway
@@ -150,9 +188,10 @@ export class Tokenizer {
150
188
  }
151
189
 
152
190
  countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
153
- if (mode === "strict") return countTokensNat(text, this.#encoding, mode).tokens;
154
- if (!testEnv && this.#encoding !== null) return countTokensNat(text, this.#encoding, mode).tokens;
155
- if (accurate) return countTokensNat(text, undefined, mode).tokens;
191
+ if (mode === "strict") return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
192
+ if (!testEnv && this.#encoding !== null)
193
+ return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
194
+ if (accurate) return countTokensNat(text, undefined, mode, this.#nativeCounts).tokens;
156
195
  return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
157
196
  }
158
197
 
@@ -170,7 +209,7 @@ export class Tokenizer {
170
209
  checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
171
210
  const bound = sumFragments(text, byteLength);
172
211
  if (bound <= budget) return { fits: true, tokens: bound, exact: false };
173
- const result = countTokensNat(text, this.#encoding, "strict");
212
+ const result = countTokensNat(text, this.#encoding, "strict", this.#nativeCounts);
174
213
  return { fits: result.tokens <= budget, tokens: result.tokens, exact: result.exact };
175
214
  }
176
215
 
@@ -24,13 +24,19 @@ export function isNonBlankContext(value: unknown): value is string {
24
24
  }
25
25
 
26
26
  /**
27
- * Join passive context values in order, dropping blanks. Returns undefined
28
- * when nothing remains.
27
+ * Join passive context values in order, dropping blanks and repeats of an
28
+ * earlier value (compared without surrounding whitespace; the first original
29
+ * is kept). Returns undefined when nothing remains.
29
30
  */
30
31
  export function joinAdditionalContext(values: Iterable<string | undefined>): string | undefined {
32
+ const seen = new Set<string>();
31
33
  const kept: string[] = [];
32
34
  for (const value of values) {
33
- if (isNonBlankContext(value)) kept.push(value);
35
+ if (!isNonBlankContext(value)) continue;
36
+ const key = value.trim();
37
+ if (seen.has(key)) continue;
38
+ seen.add(key);
39
+ kept.push(value);
34
40
  }
35
41
  return kept.length > 0 ? kept.join("\n\n") : undefined;
36
42
  }
package/src/types.ts CHANGED
@@ -595,6 +595,14 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
595
595
  */
596
596
  transformAssistantMessage?: (message: AssistantMessage, signal?: AbortSignal) => Promise<void> | void;
597
597
 
598
+ /**
599
+ * Declares that {@link transformAssistantMessage} never rewrites or removes a
600
+ * tool call the model streamed (it may edit text or append new calls). Stream
601
+ * speculation sessions and direct speculative candidates plan from streamed
602
+ * calls, so they stay disabled under a transform unless this is set.
603
+ */
604
+ transformAssistantMessagePreservesToolCalls?: boolean;
605
+
598
606
  /**
599
607
  * Called after a tool finishes executing, before `tool_execution_end` and the
600
608
  * tool-result message are emitted.
@@ -740,9 +748,24 @@ export interface SpeculativeExecutionHost {
740
748
  commitDefault: () => Promise<AgentToolResult<unknown>>,
741
749
  ): Promise<SpeculativeCommitDecision>;
742
750
  discard?(context: SpeculativeDiscardContext): void | Promise<void>;
751
+ /**
752
+ * Authorize a stream session to start effectful work (e.g. subagents) from
753
+ * partially streamed arguments. The session owns that work and must abort it
754
+ * when the finalized call is invalid, blocked, or changed. Hosts without this
755
+ * hook deny every launch.
756
+ */
757
+ authorizeLaunch?(context: SpeculativeLaunchContext): SpeculativeAuthorization | Promise<SpeculativeAuthorization>;
743
758
  close?(reason: string): void | Promise<void>;
744
759
  }
745
760
 
761
+ /** Effectful work a tool-owned stream session asks to start before its outer call dispatches. */
762
+ export interface SpeculativeLaunchContext {
763
+ tool: SpeculativeToolReference;
764
+ toolCall: AgentToolCall;
765
+ /** Arguments the launch was planned from: the streamed prefix of the outer call. */
766
+ args: Readonly<Record<string, unknown>>;
767
+ }
768
+
746
769
  export interface ToolSpeculationStreamContext {
747
770
  readonly coordinator: SpeculativeOperationSink;
748
771
  readonly parentToolCallId: string;
@@ -789,6 +812,8 @@ export interface ToolSpeculationStreamSession {
789
812
  export interface SpeculativeOperationSink {
790
813
  readonly maxInFlight: number;
791
814
  admit(definition: SpeculativeChildDefinition): Promise<SpeculativeChildHandle | undefined>;
815
+ /** Host-gated permission for effectful stream work; see {@link SpeculativeExecutionHost.authorizeLaunch}. */
816
+ authorizeLaunch?(context: SpeculativeLaunchContext): Promise<SpeculativeAuthorization>;
792
817
  discardChildren?(parentToolCallId: string, reason: string): void | Promise<void>;
793
818
  close(reason: string): void | Promise<void>;
794
819
  }
@@ -850,7 +875,8 @@ export interface SpeculativeToolExecutionConfig {
850
875
  * ignored when `block` is true.
851
876
  *
852
877
  * Set `additionalContext` to attach passive model-visible context to this call.
853
- * Non-empty values from a tool batch are injected in assistant tool-call order
878
+ * Non-empty values from a tool batch are injected in assistant tool-call order,
879
+ * a value identical to an earlier one in the batch only once,
854
880
  * after every result settles and before the next provider request. It is
855
881
  * dropped when the call is blocked or skipped, or when its final result is an
856
882
  * error (including an approval denial raised by the tool's own gate). Within a