@oh-my-pi/pi-agent-core 18.4.1 → 18.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,21 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.4.2] - 2026-09-28
6
+
7
+ ### Added
8
+
9
+ - Added tool_execution_end events that fire as each tool call settles for live UI updates
10
+
11
+ ### Changed
12
+
13
+ - Emitted tool result messages in the order of tool calls, preserving call order regardless of completion order
14
+ - Reduced repeated token-counting work with a bounded, model-scoped cache of exact text and short-message fragment counts.
15
+
16
+ ### Fixed
17
+
18
+ - Fixed an issue where streaming tool call arguments could be incorrectly modified in-place
19
+
5
20
  ## [18.4.1] - 2026-09-28
6
21
 
7
22
  ### Fixed
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "18.4.1",
4
+ "version": "18.4.2",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": {
@@ -38,16 +38,16 @@
38
38
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
39
39
  },
40
40
  "dependencies": {
41
- "@oh-my-pi/pi-ai": "18.4.1",
42
- "@oh-my-pi/pi-catalog": "18.4.1",
43
- "@oh-my-pi/pi-natives": "18.4.1",
44
- "@oh-my-pi/pi-utils": "18.4.1",
45
- "@oh-my-pi/pi-wire": "18.4.1",
46
- "@oh-my-pi/snapcompact": "18.4.1",
41
+ "@oh-my-pi/pi-ai": "18.4.2",
42
+ "@oh-my-pi/pi-catalog": "18.4.2",
43
+ "@oh-my-pi/pi-natives": "18.4.2",
44
+ "@oh-my-pi/pi-utils": "18.4.2",
45
+ "@oh-my-pi/pi-wire": "18.4.2",
46
+ "@oh-my-pi/snapcompact": "18.4.2",
47
47
  "@opentelemetry/api": "^1.9.1"
48
48
  },
49
49
  "devDependencies": {
50
- "@oh-my-pi/omptype": "18.4.1",
50
+ "@oh-my-pi/omptype": "18.4.2",
51
51
  "@opentelemetry/context-async-hooks": "^2.9.0",
52
52
  "@opentelemetry/sdk-trace-base": "^2.9.0",
53
53
  "@types/bun": "^1.3.14"
package/src/agent-loop.ts CHANGED
@@ -51,7 +51,7 @@ import {
51
51
  recoverHarmonyToolCall,
52
52
  signalListLabel,
53
53
  } from "@oh-my-pi/pi-ai/utils/harmony-leak";
54
- import { logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
54
+ import { cloneJsonTree, logger, sanitizeText, structuredCloneJSON } from "@oh-my-pi/pi-utils";
55
55
  import { INTENT_FIELD } from "@oh-my-pi/pi-wire";
56
56
  import { LiveSteeringChannel } from "./live-steering";
57
57
  import { agentPauseGate } from "./pause";
@@ -377,13 +377,17 @@ function snapshotAssistantContentBlock(block: AssistantContentBlock): AssistantC
377
377
  case "redactedThinking":
378
378
  return { ...block };
379
379
  case "anthropicServerTool":
380
- return { ...block, block: structuredCloneJSON(block.block) };
380
+ return { ...block, block: cloneJsonTree(block.block) };
381
381
  case "fallback":
382
382
  return { ...block, from: { ...block.from }, to: { ...block.to } };
383
383
  case "toolCall": {
384
384
  const snap = {
385
385
  ...block,
386
- arguments: structuredCloneJSON(block.arguments),
386
+ // Providers mutate streaming arguments in place (owned-stream, GLM)
387
+ // as well as replacing them, so containers are always copied; the
388
+ // strings inside are immutable and shared, keeping the per-delta
389
+ // cost independent of the argument payload size.
390
+ arguments: cloneJsonTree(block.arguments),
387
391
  providerMetadata: snapshotToolCallProviderMetadata(block.providerMetadata),
388
392
  };
389
393
  // Object spread copies enumerable symbols in Bun, but the Cursor
@@ -3027,6 +3031,11 @@ async function speculativeFinalCalls(
3027
3031
  /**
3028
3032
  * Execute tool calls from an assistant message. Returns model-visible context
3029
3033
  * only after every result has settled, preserving assistant call order.
3034
+ *
3035
+ * `tool_execution_end` fires as each call settles so live UI updates promptly;
3036
+ * result `message_start`/`message_end` events (which append to agent state and
3037
+ * the persisted session) are held until every earlier call has a result, so
3038
+ * history always pairs results in call order regardless of completion order.
3030
3039
  */
3031
3040
  async function executeToolCalls(
3032
3041
  currentContext: AgentContext,
@@ -3213,6 +3222,18 @@ async function executeToolCalls(
3213
3222
  await checkAsideInterrupts();
3214
3223
  };
3215
3224
 
3225
+ // Index of the first record whose result message has not been emitted yet.
3226
+ let nextResultIndex = 0;
3227
+ const flushResultMessages = (): void => {
3228
+ for (; nextResultIndex < records.length; nextResultIndex++) {
3229
+ const message = records[nextResultIndex].toolResultMessage;
3230
+ if (!message) return;
3231
+ emittedToolResults.push(message);
3232
+ stream.push({ type: "message_start", message });
3233
+ stream.push({ type: "message_end", message });
3234
+ }
3235
+ };
3236
+
3216
3237
  const emitToolResult = (record: (typeof records)[number], result: AgentToolResult<any>, isError: boolean): void => {
3217
3238
  if (record.resultEmitted) return;
3218
3239
  const { toolCall } = record;
@@ -3248,10 +3269,7 @@ async function executeToolCalls(
3248
3269
  record.isError = isError;
3249
3270
  record.toolResultMessage = toolResultMessage;
3250
3271
  record.resultEmitted = true;
3251
- emittedToolResults.push(toolResultMessage);
3252
-
3253
- stream.push({ type: "message_start", message: toolResultMessage });
3254
- stream.push({ type: "message_end", message: toolResultMessage });
3272
+ flushResultMessages();
3255
3273
  };
3256
3274
 
3257
3275
  const runTool = async (record: (typeof records)[number], index: number): Promise<void> => {
package/src/tokenizer.ts CHANGED
@@ -1,7 +1,8 @@
1
1
  import type { Model } from "@oh-my-pi/pi-ai";
2
2
  import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
3
3
  import * as natives from "@oh-my-pi/pi-natives";
4
- import { stringifyJson } from "@oh-my-pi/pi-utils";
4
+ import { materializeString, stringifyJson } from "@oh-my-pi/pi-utils";
5
+ import { LRUCache } from "@oh-my-pi/pi-utils/lru";
5
6
  import * as snapcompact from "@oh-my-pi/snapcompact";
6
7
  import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
7
8
  import type { AgentMessage } from "./types";
@@ -67,13 +68,43 @@ interface NativeTokenCount {
67
68
  exact: boolean;
68
69
  }
69
70
 
71
+ // Growing streamed text and large tool results must not evict the reusable
72
+ // short fragments. Account for UTF-16 key storage plus a per-entry allowance.
73
+ const NATIVE_CACHE_MAX_LENGTH = 16 * 1024;
74
+
75
+ function countNativeFragment(
76
+ text: string,
77
+ encoding: natives.Encoding | null | undefined,
78
+ counts: LRUCache<string, number>,
79
+ ): number {
80
+ if (text.length > NATIVE_CACHE_MAX_LENGTH) return natives.countTokens(text, encoding);
81
+ const cached = counts.get(text);
82
+ if (cached !== undefined) return cached;
83
+ const tokens = natives.countTokens(text, encoding);
84
+ // Detach sliced strings so a small key cannot retain a much larger source.
85
+ counts.set(materializeString(text), tokens);
86
+ return tokens;
87
+ }
88
+
70
89
  function countTokensNat(
71
90
  text: string | string[],
72
91
  encoding: natives.Encoding | null | undefined,
73
92
  mode: TokenCountMode,
93
+ counts: LRUCache<string, number>,
74
94
  ): NativeTokenCount {
75
95
  try {
76
- return { tokens: natives.countTokens(text, encoding), exact: true };
96
+ let tokens: number;
97
+ if (typeof text === "string") {
98
+ tokens = countNativeFragment(text, encoding, counts);
99
+ } else if (text.length > 0 && text.length < 16) {
100
+ // The native API sums independent fragments, not their concatenation.
101
+ // Keep its parallel batch path for arrays of 16 or more fragments.
102
+ tokens = 0;
103
+ for (const fragment of text) tokens += countNativeFragment(fragment, encoding, counts);
104
+ } else {
105
+ tokens = natives.countTokens(text, encoding);
106
+ }
107
+ return { tokens, exact: true };
77
108
  } catch (error) {
78
109
  if (
79
110
  !(error instanceof Error) ||
@@ -131,6 +162,13 @@ interface MessageEstimate {
131
162
  export class Tokenizer {
132
163
  readonly #encoding: natives.Encoding | null;
133
164
 
165
+ /** Exact counts only; byte fallbacks remain mode-dependent and uncached. */
166
+ readonly #nativeCounts = new LRUCache<string, number>({
167
+ max: 256,
168
+ maxSize: 512 * 1024,
169
+ sizeCalculation: (_tokens, text) => text.length * 2 + 64,
170
+ });
171
+
134
172
  /**
135
173
  * Per-message estimate memo. Keyed by message identity, deliberately not a
136
174
  * symbol-tagged property: callers spread messages to derive throwaway
@@ -150,9 +188,10 @@ export class Tokenizer {
150
188
  }
151
189
 
152
190
  countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
153
- if (mode === "strict") return countTokensNat(text, this.#encoding, mode).tokens;
154
- if (!testEnv && this.#encoding !== null) return countTokensNat(text, this.#encoding, mode).tokens;
155
- if (accurate) return countTokensNat(text, undefined, mode).tokens;
191
+ if (mode === "strict") return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
192
+ if (!testEnv && this.#encoding !== null)
193
+ return countTokensNat(text, this.#encoding, mode, this.#nativeCounts).tokens;
194
+ if (accurate) return countTokensNat(text, undefined, mode, this.#nativeCounts).tokens;
156
195
  return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
157
196
  }
158
197
 
@@ -170,7 +209,7 @@ export class Tokenizer {
170
209
  checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
171
210
  const bound = sumFragments(text, byteLength);
172
211
  if (bound <= budget) return { fits: true, tokens: bound, exact: false };
173
- const result = countTokensNat(text, this.#encoding, "strict");
212
+ const result = countTokensNat(text, this.#encoding, "strict", this.#nativeCounts);
174
213
  return { fits: result.tokens <= budget, tokens: result.tokens, exact: result.exact };
175
214
  }
176
215