@oh-my-pi/pi-agent-core 18.0.9 → 18.0.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,18 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.0.11] - 2026-08-29
6
+
7
+ ### Fixed
8
+
9
+ - Fixed agent startup and context compaction failures for models with unrecognized tokenizer encodings.
10
+
11
+ ## [18.0.10] - 2026-08-28
12
+
13
+ ### Added
14
+
15
+ - Added support for continuing interrupted agent runs with pending tool calls, allowing those calls to be retried before requesting the next model response.
16
+
5
17
  ## [18.0.9] - 2026-08-28
6
18
 
7
19
  ### Fixed
@@ -38,13 +38,26 @@ export declare function resolveOwnedDialectFromEnv(value: string | undefined): D
38
38
  * The prompt is added to the context and events are emitted for it.
39
39
  */
40
40
  export declare function agentLoop(prompts: AgentMessage[], context: AgentContext, config: AgentLoopConfig, signal?: AbortSignal, streamFn?: StreamFn): EventStream<AgentEvent, AgentMessage[]>;
41
+ /**
42
+ * Trailing assistant message whose runnable tool calls have no results yet.
43
+ *
44
+ * A harness that strips failed/aborted tool results in order to re-execute
45
+ * the calls (e.g. `AgentSession.retry`'s tool replay) leaves the transcript
46
+ * in this shape; {@link agentLoopContinue} resumes it by running those calls
47
+ * before the next model call. Cursor exec-resolved blocks are excluded — the
48
+ * provider already executed those server-side. `length`-truncated turns never
49
+ * qualify: their trailing call arguments may be incomplete.
50
+ */
51
+ export declare function unpairedToolCallTail(messages: readonly AgentMessage[]): AssistantMessage | undefined;
41
52
  /**
42
53
  * Continue an agent loop from the current context without adding a new message.
43
54
  * Used for retries - context already has user message or tool results.
44
55
  *
45
56
  * **Important:** The last message in context must convert to a `user` or `toolResult` message
46
- * via `convertToLlm`. If it doesn't, the LLM provider will reject the request.
47
- * This cannot be validated here since `convertToLlm` is only called once per turn.
57
+ * via `convertToLlm` — except for an assistant tail with unpaired runnable
58
+ * tool calls (see {@link unpairedToolCallTail}), which resumes by executing
59
+ * those calls first. Any other assistant tail is rejected here; other invalid
60
+ * tails cannot be validated since `convertToLlm` is only called once per turn.
48
61
  */
49
62
  export declare function agentLoopContinue(context: AgentContext, config: AgentLoopConfig, signal?: AbortSignal, streamFn?: StreamFn): EventStream<AgentEvent, AgentMessage[]>;
50
63
  /**
@@ -1,11 +1,12 @@
1
1
  import type { Model } from "@oh-my-pi/pi-ai";
2
- import { Encoding } from "@oh-my-pi/pi-natives";
2
+ import * as natives from "@oh-my-pi/pi-natives";
3
3
  import type { AgentMessage } from "./types.js";
4
4
  /** Maps the catalog-resolved tokenizer family to its native implementation. */
5
- export declare function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null;
5
+ export declare function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): natives.Encoding | null;
6
6
  /**
7
- * `strict` always pays for an exact native count (the catalog-resolved
8
- * tokenizer when known, o200k_base otherwise). `approximate` and
7
+ * `strict` always tries an exact native count (the catalog-resolved tokenizer
8
+ * when known, o200k_base otherwise), falling back to the byte upper bound when
9
+ * the loaded addon does not recognize that encoding. `approximate` and
9
10
  * `upperbound` prefer the same exact count for known tokenizer families or
10
11
  * when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
11
12
  * `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
@@ -31,11 +32,10 @@ export interface TokenBudgetCheck {
31
32
  fits: boolean;
32
33
  /**
33
34
  * Token count behind the verdict: the exact native count when `exact` is
34
- * set, otherwise the cheap byte upper bound (which already fit, so it is
35
- * only an over-estimate of a count known to be under budget).
35
+ * set, otherwise the conservative byte upper bound.
36
36
  */
37
37
  tokens: number;
38
- /** Whether the exact tokenizer had to run because the cheap bound busted. */
38
+ /** Whether `tokens` came from the exact native tokenizer. */
39
39
  exact: boolean;
40
40
  }
41
41
  /**
@@ -50,7 +50,7 @@ export interface TokenBudgetCheck {
50
50
  export declare class Tokenizer {
51
51
  #private;
52
52
  constructor(model?: Pick<Model, "tokenizer"> | null);
53
- get encoding(): Encoding | null;
53
+ get encoding(): natives.Encoding | null;
54
54
  countTokens(text: string | string[], mode?: TokenCountMode): number;
55
55
  /**
56
56
  * Cheap-first budget probe — the way to ask "does this fit in `budget`
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "18.0.9",
4
+ "version": "18.0.11",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -35,16 +35,16 @@
35
35
  "fmt": "biome format --write ."
36
36
  },
37
37
  "dependencies": {
38
- "@oh-my-pi/pi-ai": "18.0.9",
39
- "@oh-my-pi/pi-catalog": "18.0.9",
40
- "@oh-my-pi/pi-natives": "18.0.9",
41
- "@oh-my-pi/pi-utils": "18.0.9",
42
- "@oh-my-pi/pi-wire": "18.0.9",
43
- "@oh-my-pi/snapcompact": "18.0.9",
38
+ "@oh-my-pi/pi-ai": "18.0.11",
39
+ "@oh-my-pi/pi-catalog": "18.0.11",
40
+ "@oh-my-pi/pi-natives": "18.0.11",
41
+ "@oh-my-pi/pi-utils": "18.0.11",
42
+ "@oh-my-pi/pi-wire": "18.0.11",
43
+ "@oh-my-pi/snapcompact": "18.0.11",
44
44
  "@opentelemetry/api": "^1.9.1"
45
45
  },
46
46
  "devDependencies": {
47
- "@oh-my-pi/omptype": "18.0.9",
47
+ "@oh-my-pi/omptype": "18.0.11",
48
48
  "@opentelemetry/context-async-hooks": "^2.9.0",
49
49
  "@opentelemetry/sdk-trace-base": "^2.9.0",
50
50
  "@types/bun": "^1.3.14"
package/src/agent-loop.ts CHANGED
@@ -554,13 +554,36 @@ export function agentLoop(
554
554
  return stream;
555
555
  }
556
556
 
557
+ /**
558
+ * Trailing assistant message whose runnable tool calls have no results yet.
559
+ *
560
+ * A harness that strips failed/aborted tool results in order to re-execute
561
+ * the calls (e.g. `AgentSession.retry`'s tool replay) leaves the transcript
562
+ * in this shape; {@link agentLoopContinue} resumes it by running those calls
563
+ * before the next model call. Cursor exec-resolved blocks are excluded — the
564
+ * provider already executed those server-side. `length`-truncated turns never
565
+ * qualify: their trailing call arguments may be incomplete.
566
+ */
567
+ export function unpairedToolCallTail(messages: readonly AgentMessage[]): AssistantMessage | undefined {
568
+ const tail = messages[messages.length - 1];
569
+ if (tail?.role !== "assistant") return undefined;
570
+ // Mirrors the loop's `runnableStop` rule for fresh turns.
571
+ if (tail.stopReason !== "toolUse" && tail.stopReason !== "stop") return undefined;
572
+ const hasRunnable = tail.content.some(
573
+ c => c.type === "toolCall" && (c as CursorExecResolvedCarrier)[kCursorExecResolved] !== true,
574
+ );
575
+ return hasRunnable ? tail : undefined;
576
+ }
577
+
557
578
  /**
558
579
  * Continue an agent loop from the current context without adding a new message.
559
580
  * Used for retries - context already has user message or tool results.
560
581
  *
561
582
  * **Important:** The last message in context must convert to a `user` or `toolResult` message
562
- * via `convertToLlm`. If it doesn't, the LLM provider will reject the request.
563
- * This cannot be validated here since `convertToLlm` is only called once per turn.
583
+ * via `convertToLlm` — except for an assistant tail with unpaired runnable
584
+ * tool calls (see {@link unpairedToolCallTail}), which resumes by executing
585
+ * those calls first. Any other assistant tail is rejected here; other invalid
586
+ * tails cannot be validated since `convertToLlm` is only called once per turn.
564
587
  */
565
588
  export function agentLoopContinue(
566
589
  context: AgentContext,
@@ -572,7 +595,7 @@ export function agentLoopContinue(
572
595
  throw new Error("Cannot continue: no messages in context");
573
596
  }
574
597
 
575
- if (context.messages[context.messages.length - 1].role === "assistant") {
598
+ if (context.messages[context.messages.length - 1].role === "assistant" && !unpairedToolCallTail(context.messages)) {
576
599
  throw new Error("Cannot continue from message role: assistant");
577
600
  }
578
601
 
@@ -1061,6 +1084,43 @@ async function runLoopBody(
1061
1084
  let softSatisfies: SoftToolRequirement["satisfies"];
1062
1085
  let directiveResolvedForTurn = false;
1063
1086
  let turnOpen = false;
1087
+ // A trailing assistant message with unpaired tool calls: the harness
1088
+ // stripped the failed/aborted results so this run re-executes the calls
1089
+ // (AgentSession.retry's tool replay). Run them before the first model
1090
+ // call; the loop then continues normally on the fresh results. Queued
1091
+ // steering stays parked until after the batch — injecting a message
1092
+ // between the tool_use blocks and their results would break the
1093
+ // provider's pairing invariant.
1094
+ const resumeTail = unpairedToolCallTail(currentContext.messages);
1095
+ if (resumeTail) {
1096
+ stream.push({ type: "turn_start" });
1097
+ emitInputMessages(stream, messagesToEmit);
1098
+ messagesToEmit = [];
1099
+ turnOpen = true;
1100
+ const executionResult = await executeToolCalls(
1101
+ currentContext,
1102
+ resumeTail,
1103
+ signal,
1104
+ stream,
1105
+ config,
1106
+ telemetry,
1107
+ invokeAgentSpan,
1108
+ );
1109
+ for (const result of executionResult.toolResults) {
1110
+ currentContext.messages.push(result);
1111
+ newMessages.push(result);
1112
+ }
1113
+ await emitTurnEnd(stream, currentContext, resumeTail, executionResult.toolResults, config, signal, {
1114
+ willContinue: !isDeadlineExceeded(config.deadline),
1115
+ });
1116
+ turnOpen = false;
1117
+ // A tool hook may mark its completed result as terminal (e.g. subagent
1118
+ // yield) — same stop-before-next-model-call rule as the main loop.
1119
+ if (signal?.reason === TERMINAL_TOOL_RESULT_ABORT_REASON) {
1120
+ endAgentStream(stream, newMessages, telemetry, stepCounter.count);
1121
+ return;
1122
+ }
1123
+ }
1064
1124
 
1065
1125
  // Outer loop: continues when queued follow-up messages arrive after agent would stop
1066
1126
  while (true) {
package/src/agent.ts CHANGED
@@ -34,6 +34,7 @@ import {
34
34
  normalizeMessagesForProvider,
35
35
  normalizeTools,
36
36
  resolveOwnedDialectFromEnv,
37
+ unpairedToolCallTail,
37
38
  } from "./agent-loop";
38
39
  import type { AppendOnlyContextManager } from "./append-only-context";
39
40
  import { isProviderRefusalMessage } from "./replay-policy";
@@ -1249,6 +1250,15 @@ export class Agent {
1249
1250
  throw new Error("No messages to continue from");
1250
1251
  }
1251
1252
  if (messages[messages.length - 1].role === "assistant") {
1253
+ // A tail with unpaired runnable tool calls resumes by re-executing
1254
+ // them (see `unpairedToolCallTail` in agent-loop). This must win over
1255
+ // queued-message delivery: injecting a message between the tool_use
1256
+ // blocks and their results would break the provider's pairing
1257
+ // invariant. Queued messages drain inside the resumed loop instead.
1258
+ if (unpairedToolCallTail(messages)) {
1259
+ await this.#runLoop(undefined, undefined, signal, true);
1260
+ return;
1261
+ }
1252
1262
  const queuedSteering = await this.#dequeueSteeringMessagesAfterHooks(dequeueSignal);
1253
1263
  if (queuedSteering.length > 0) {
1254
1264
  await this.#runLoop(queuedSteering, { skipInitialSteeringPoll: true }, signal, true);
package/src/tokenizer.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import type { Model } from "@oh-my-pi/pi-ai";
2
2
  import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
3
- import { countTokens as countTokensNat, Encoding } from "@oh-my-pi/pi-natives";
3
+ import * as natives from "@oh-my-pi/pi-natives";
4
4
  import { stringifyJson } from "@oh-my-pi/pi-utils";
5
5
  import * as snapcompact from "@oh-my-pi/snapcompact";
6
6
  import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
@@ -9,25 +9,26 @@ import type { AgentMessage } from "./types";
9
9
  const testEnv = Bun.env.NODE_ENV === "test";
10
10
  const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && !testEnv;
11
11
 
12
- const NATIVE_ENCODING: Record<ModelTokenizer, Encoding> = {
13
- "claude-v3": Encoding.ClaudeV3,
14
- "claude-v47": Encoding.ClaudeV47,
15
- "claude-v5": Encoding.ClaudeV5,
16
- "claude-v5-sonnet": Encoding.ClaudeV5Sonnet,
17
- qwen3: Encoding.Qwen3,
18
- "deepseek-v3": Encoding.DeepSeekV3,
19
- "kimi-k2": Encoding.KimiK2,
20
- glm5: Encoding.Glm5,
12
+ const NATIVE_ENCODING: Record<ModelTokenizer, natives.Encoding> = {
13
+ "claude-v3": natives.Encoding.ClaudeV3,
14
+ "claude-v47": natives.Encoding.ClaudeV47,
15
+ "claude-v5": natives.Encoding.ClaudeV5,
16
+ "claude-v5-sonnet": natives.Encoding.ClaudeV5Sonnet,
17
+ qwen3: natives.Encoding.Qwen3,
18
+ "deepseek-v3": natives.Encoding.DeepSeekV3,
19
+ "kimi-k2": natives.Encoding.KimiK2,
20
+ glm5: natives.Encoding.Glm5,
21
21
  };
22
22
 
23
23
  /** Maps the catalog-resolved tokenizer family to its native implementation. */
24
- export function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null {
24
+ export function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): natives.Encoding | null {
25
25
  return model?.tokenizer ? NATIVE_ENCODING[model.tokenizer] : null;
26
26
  }
27
27
 
28
28
  /**
29
- * `strict` always pays for an exact native count (the catalog-resolved
30
- * tokenizer when known, o200k_base otherwise). `approximate` and
29
+ * `strict` always tries an exact native count (the catalog-resolved tokenizer
30
+ * when known, o200k_base otherwise), falling back to the byte upper bound when
31
+ * the loaded addon does not recognize that encoding. `approximate` and
31
32
  * `upperbound` prefer the same exact count for known tokenizer families or
32
33
  * when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
33
34
  * `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
@@ -61,17 +62,41 @@ function sumFragments(text: string | string[], perFragment: (t: string) => numbe
61
62
  return Array.isArray(text) ? text.reduce((sum, t) => sum + perFragment(t), 0) : perFragment(text);
62
63
  }
63
64
 
65
+ interface NativeTokenCount {
66
+ tokens: number;
67
+ exact: boolean;
68
+ }
69
+
70
+ function countTokensNat(
71
+ text: string | string[],
72
+ encoding: natives.Encoding | null | undefined,
73
+ mode: TokenCountMode,
74
+ ): NativeTokenCount {
75
+ try {
76
+ return { tokens: natives.countTokens(text, encoding), exact: true };
77
+ } catch (error) {
78
+ if (
79
+ !(error instanceof Error) ||
80
+ (!error.message.includes("does not match any variant of enum") &&
81
+ !error.message.includes("unknown enum variant"))
82
+ ) {
83
+ throw error;
84
+ }
85
+ const tokens = sumFragments(text, mode === "approximate" ? byteEstimate : byteLength);
86
+ return { tokens, exact: false };
87
+ }
88
+ }
89
+
64
90
  /** Verdict from {@link Tokenizer.checkTokenBudget}. */
65
91
  export interface TokenBudgetCheck {
66
92
  /** Whether the text fits the budget. */
67
93
  fits: boolean;
68
94
  /**
69
95
  * Token count behind the verdict: the exact native count when `exact` is
70
- * set, otherwise the cheap byte upper bound (which already fit, so it is
71
- * only an over-estimate of a count known to be under budget).
96
+ * set, otherwise the conservative byte upper bound.
72
97
  */
73
98
  tokens: number;
74
- /** Whether the exact tokenizer had to run because the cheap bound busted. */
99
+ /** Whether `tokens` came from the exact native tokenizer. */
75
100
  exact: boolean;
76
101
  }
77
102
 
@@ -104,7 +129,7 @@ interface MessageEstimate {
104
129
  * `PI_TOKENIZER_ACCURATE=1`).
105
130
  */
106
131
  export class Tokenizer {
107
- readonly #encoding: Encoding | null;
132
+ readonly #encoding: natives.Encoding | null;
108
133
 
109
134
  /**
110
135
  * Per-message estimate memo. Keyed by message identity, deliberately not a
@@ -120,14 +145,14 @@ export class Tokenizer {
120
145
  this.#encoding = tokenizerEncodingForModel(model);
121
146
  }
122
147
 
123
- get encoding(): Encoding | null {
148
+ get encoding(): natives.Encoding | null {
124
149
  return this.#encoding;
125
150
  }
126
151
 
127
152
  countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
128
- if (mode === "strict") return countTokensNat(text, this.#encoding);
129
- if (!testEnv && this.#encoding !== null) return countTokensNat(text, this.#encoding);
130
- if (accurate) return countTokensNat(text);
153
+ if (mode === "strict") return countTokensNat(text, this.#encoding, mode).tokens;
154
+ if (!testEnv && this.#encoding !== null) return countTokensNat(text, this.#encoding, mode).tokens;
155
+ if (accurate) return countTokensNat(text, undefined, mode).tokens;
131
156
  return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
132
157
  }
133
158
 
@@ -145,8 +170,8 @@ export class Tokenizer {
145
170
  checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
146
171
  const bound = sumFragments(text, byteLength);
147
172
  if (bound <= budget) return { fits: true, tokens: bound, exact: false };
148
- const tokens = this.countTokens(text, "strict");
149
- return { fits: tokens <= budget, tokens, exact: true };
173
+ const result = countTokensNat(text, this.#encoding, "strict");
174
+ return { fits: result.tokens <= budget, tokens: result.tokens, exact: result.exact };
150
175
  }
151
176
 
152
177
  /**