@oh-my-pi/pi-agent-core 18.4.0 → 18.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,14 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.4.1] - 2026-09-28
6
+
7
+ ### Fixed
8
+
9
+ - Fixed fitted output caps overshooting the context window by a few tokens on strict Chat Completions hosts (e.g. llama.cpp), causing 400s.
10
+ - Fixed native remote compaction sending requests already estimated past the model's context window (e.g. after re-expanding history behind another provider's native boundary); it now fails fast so the next configured compaction method runs ([#13502](https://github.com/can1357/oh-my-pi/issues/13502))
11
+ - Fixed V2 remote compaction retrying a standalone stream `error` event three times and reporting it as `stream closed before response.completed`; the upstream status, code, and message (e.g. `context_too_large`) are now surfaced ([#13502](https://github.com/can1357/oh-my-pi/issues/13502))
12
+
5
13
  ## [18.4.0] - 2026-09-28
6
14
 
7
15
  ### Changed
@@ -33,6 +33,8 @@ export interface TrimRemoteCompactionInputResult {
33
33
  rewrittenOutputs: number;
34
34
  estimatedTokensBefore: number;
35
35
  estimatedTokensAfter: number;
36
+ /** Whether `input` fits the model window; false means it must not be sent. */
37
+ fits: boolean;
36
38
  }
37
39
  /**
38
40
  * Preserve the full native transcript unless trailing tool outputs alone push a
@@ -41,6 +43,15 @@ export interface TrimRemoteCompactionInputResult {
41
43
  * matching Codex's recovery path for oversized tool turns.
42
44
  */
43
45
  export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, tokenizer: Tokenizer, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
46
+ /**
47
+ * Refuse a native compaction request whose prepared input cannot fit the model
48
+ * window, before any network I/O. Re-expanded history behind an unreadable
49
+ * native boundary can exceed the window even when live context does not.
50
+ *
51
+ * @throws Error flagged `ContextOverflow` when `trimmed.fits` is false, so
52
+ * compaction callers skip retries and advance to the next method.
53
+ */
54
+ export declare function assertRemoteCompactionInputFits(trimmed: TrimRemoteCompactionInputResult, model: Model): void;
44
55
  export type OpenAiRemoteCompactionItem = {
45
56
  type: "compaction" | "compaction_summary";
46
57
  encrypted_content?: string;
@@ -2,6 +2,16 @@ import type { Context, Model } from "@oh-my-pi/pi-ai";
2
2
  import type { Tokenizer } from "./tokenizer.js";
3
3
  /** Smallest output cap {@link fitOutputTokensToContextWindow} will request. */
4
4
  export declare const MIN_FITTED_OUTPUT_TOKENS = 1024;
5
+ /**
6
+ * Absolute headway subtracted from the remaining room (unconditionally, on anchored and
7
+ * fully-local counts alike). Proportional padding covers
8
+ * tokenizer drift that scales with the prompt, but a host can still count a few tokens more
9
+ * than any local estimate can see (chat-template framing, reasoning wrappers). Measured
10
+ * against a 262,144-token host: fitted caps landed +11 to +42 tokens over and were rejected
11
+ * with 400s. 64 tokens covers the observed drift with margin to spare; the
12
+ * {@link MIN_FITTED_OUTPUT_TOKENS} floor still applies.
13
+ */
14
+ export declare const OUTPUT_FIT_HEADWAY_TOKENS = 64;
5
15
  /**
6
16
  * Output cap for a request, so prompt plus output stays inside the model's
7
17
  * context window.
@@ -26,7 +36,8 @@ export declare const MIN_FITTED_OUTPUT_TOKENS = 1024;
26
36
  * OpenRouter-hosted model with no caller cap: the transport omits the catalog
27
37
  * default there so each upstream self-caps, and a fitted value would turn into
28
38
  * an explicit cap that filters upstreams). Otherwise returns
29
- * the remaining room (never below {@link MIN_FITTED_OUTPUT_TOKENS}); a
39
+ * the remaining room minus {@link OUTPUT_FIT_HEADWAY_TOKENS} (never below
40
+ * {@link MIN_FITTED_OUTPUT_TOKENS}); a
30
41
  * prompt that fills the whole window still overflows and is left to the
31
42
  * caller's compaction. Near a full window the floor means a turn can stop on
32
43
  * `length` instead of failing with a 400.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "18.4.0",
4
+ "version": "18.4.1",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": {
@@ -38,16 +38,16 @@
38
38
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
39
39
  },
40
40
  "dependencies": {
41
- "@oh-my-pi/pi-ai": "18.4.0",
42
- "@oh-my-pi/pi-catalog": "18.4.0",
43
- "@oh-my-pi/pi-natives": "18.4.0",
44
- "@oh-my-pi/pi-utils": "18.4.0",
45
- "@oh-my-pi/pi-wire": "18.4.0",
46
- "@oh-my-pi/snapcompact": "18.4.0",
41
+ "@oh-my-pi/pi-ai": "18.4.1",
42
+ "@oh-my-pi/pi-catalog": "18.4.1",
43
+ "@oh-my-pi/pi-natives": "18.4.1",
44
+ "@oh-my-pi/pi-utils": "18.4.1",
45
+ "@oh-my-pi/pi-wire": "18.4.1",
46
+ "@oh-my-pi/snapcompact": "18.4.1",
47
47
  "@opentelemetry/api": "^1.9.1"
48
48
  },
49
49
  "devDependencies": {
50
- "@oh-my-pi/omptype": "18.4.0",
50
+ "@oh-my-pi/omptype": "18.4.1",
51
51
  "@opentelemetry/context-async-hooks": "^2.9.0",
52
52
  "@opentelemetry/sdk-trace-base": "^2.9.0",
53
53
  "@types/bun": "^1.3.14"
package/src/agent-loop.ts CHANGED
@@ -2859,6 +2859,13 @@ async function prepareToolCallDispatch(
2859
2859
  if (toolCall.type !== "toolCall") continue;
2860
2860
  if ((toolCall as CursorExecResolvedCarrier)[kCursorExecResolved] === true) continue;
2861
2861
  const tool = resolveToolForCall(context.tools, toolCall, resolveFallbackTool);
2862
+ // A host fallback accepts aliases (`xd://recall`, a mis-separated MCP
2863
+ // name) that providers reject when replayed as a function-call name.
2864
+ // Record the call under the resolved tool's canonical name so history,
2865
+ // persistence, and replay agree; custom-wire calls keep their wire name.
2866
+ if (tool && toolCall.name !== tool.name && toolCall.name !== tool.customWireName) {
2867
+ toolCall.name = tool.name;
2868
+ }
2862
2869
  const entry: PreparedToolCall = { tool, args: toolCall.arguments as Record<string, unknown> };
2863
2870
  prepared.set(toolCall.id, entry);
2864
2871
  let argsForExecution = toolCall.arguments as Record<string, unknown>;
@@ -597,6 +597,14 @@ function handleCompactionV2Event(
597
597
  if (type === "response.failed" || type === "response.incomplete") {
598
598
  throw new Error(formatCompactionV2Failure(event, type));
599
599
  }
600
+
601
+ // A standalone `error` event terminates the stream. Keep its status so a
602
+ // deterministic 4xx (e.g. context_too_large) is not retried as a dropped stream.
603
+ if (type === "error") {
604
+ const message = formatCompactionV2Failure(event, type);
605
+ const status = numberField(event, "status");
606
+ throw status === undefined ? new Error(message) : new AIError.ProviderHttpError(message, status);
607
+ }
600
608
  }
601
609
 
602
610
  function parseCompactionV2Usage(event: Record<string, unknown>): CompactionV2Usage | undefined {
@@ -629,8 +637,9 @@ function formatCompactionV2Failure(event: Record<string, unknown>, type: string)
629
637
  : response && isRecord(response.error)
630
638
  ? response.error
631
639
  : undefined;
632
- const message = error ? stringField(error, "message") : undefined;
633
- const code = error ? (stringField(error, "code") ?? stringField(error, "type")) : undefined;
640
+ // Responses `error` events carry code/message at the top level.
641
+ const message = stringField(error ?? event, "message");
642
+ const code = error ? (stringField(error, "code") ?? stringField(error, "type")) : stringField(event, "code");
634
643
  return `V2 compaction stream ${type}${code ? ` (${code})` : ""}${message ? `: ${message}` : ""}`;
635
644
  }
636
645
 
@@ -69,6 +69,7 @@ import {
69
69
  defaultConvertToLlm,
70
70
  } from "./messages";
71
71
  import {
72
+ assertRemoteCompactionInputFits,
72
73
  buildOpenAiNativeHistory,
73
74
  getPreservedOpenAiRemoteCompactionData,
74
75
  isOpenAiRemoteCompactionApi,
@@ -1748,6 +1749,7 @@ export async function compact(
1748
1749
  contextWindow: model.contextWindow,
1749
1750
  });
1750
1751
  }
1752
+ assertRemoteCompactionInputFits(trimmed, model);
1751
1753
  const requestOptions = {
1752
1754
  sessionId: summaryOptions.sessionId,
1753
1755
  promptCacheKey: summaryOptions.promptCacheKey,
@@ -15,7 +15,7 @@
15
15
  * with `{ summary, shortSummary? }`.
16
16
  */
17
17
 
18
- import { ProviderHttpError } from "@oh-my-pi/pi-ai/error";
18
+ import { attach, create, Flag, ProviderHttpError } from "@oh-my-pi/pi-ai/error";
19
19
  import { getCodexAttestationHeader } from "@oh-my-pi/pi-ai/providers/openai-codex-attestation";
20
20
  import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-compaction";
21
21
  import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer";
@@ -122,6 +122,8 @@ export interface TrimRemoteCompactionInputResult {
122
122
  rewrittenOutputs: number;
123
123
  estimatedTokensBefore: number;
124
124
  estimatedTokensAfter: number;
125
+ /** Whether `input` fits the model window; false means it must not be sent. */
126
+ fits: boolean;
125
127
  }
126
128
 
127
129
  /** Verdict for one remote-compaction request measured against the model window. */
@@ -199,6 +201,7 @@ export function trimRemoteCompactionInputToContextWindow(
199
201
  rewrittenOutputs: 0,
200
202
  estimatedTokensBefore: before.tokens,
201
203
  estimatedTokensAfter: before.tokens,
204
+ fits: true,
202
205
  };
203
206
  }
204
207
 
@@ -222,6 +225,7 @@ export function trimRemoteCompactionInputToContextWindow(
222
225
  rewrittenOutputs: 0,
223
226
  estimatedTokensBefore: before.tokens,
224
227
  estimatedTokensAfter: before.tokens,
228
+ fits: false,
225
229
  };
226
230
  }
227
231
 
@@ -230,9 +234,29 @@ export function trimRemoteCompactionInputToContextWindow(
230
234
  rewrittenOutputs,
231
235
  estimatedTokensBefore: before.tokens,
232
236
  estimatedTokensAfter: after.tokens,
237
+ fits: true,
233
238
  };
234
239
  }
235
240
 
241
+ /**
242
+ * Refuse a native compaction request whose prepared input cannot fit the model
243
+ * window, before any network I/O. Re-expanded history behind an unreadable
244
+ * native boundary can exceed the window even when live context does not.
245
+ *
246
+ * @throws Error flagged `ContextOverflow` when `trimmed.fits` is false, so
247
+ * compaction callers skip retries and advance to the next method.
248
+ */
249
+ export function assertRemoteCompactionInputFits(trimmed: TrimRemoteCompactionInputResult, model: Model): void {
250
+ if (trimmed.fits) return;
251
+ throw attach(
252
+ new Error(
253
+ `Remote compaction input exceeds the context window of ${model.provider}/${model.id}: ` +
254
+ `estimated ${trimmed.estimatedTokensAfter} tokens > ${model.contextWindow}`,
255
+ ),
256
+ create(Flag.ContextOverflow),
257
+ );
258
+ }
259
+
236
260
  /** Race the caller's signal against the request timeout; `timeoutMs <= 0` disables the watchdog. */
237
261
  function withRequestTimeout(signal: AbortSignal | undefined, timeoutMs: number): AbortSignal | undefined {
238
262
  if (timeoutMs <= 0) return signal;
@@ -794,6 +818,7 @@ export async function requestOpenAiRemoteCompaction(
794
818
  contextWindow: model.contextWindow,
795
819
  });
796
820
  }
821
+ assertRemoteCompactionInputFits(trimmed, model);
797
822
  const request: OpenAiRemoteCompactionRequest = {
798
823
  model: requestModel,
799
824
  // Preserve the native transcript. Only oversized trailing tool outputs are
@@ -17,6 +17,17 @@ export const MIN_FITTED_OUTPUT_TOKENS = 1024;
17
17
  */
18
18
  const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
19
19
 
20
+ /**
21
+ * Absolute headway subtracted from the remaining room (unconditionally, on anchored and
22
+ * fully-local counts alike). Proportional padding covers
23
+ * tokenizer drift that scales with the prompt, but a host can still count a few tokens more
24
+ * than any local estimate can see (chat-template framing, reasoning wrappers). Measured
25
+ * against a 262,144-token host: fitted caps landed +11 to +42 tokens over and were rejected
26
+ * with 400s. 64 tokens covers the observed drift with margin to spare; the
27
+ * {@link MIN_FITTED_OUTPUT_TOKENS} floor still applies.
28
+ */
29
+ export const OUTPUT_FIT_HEADWAY_TOKENS = 64;
30
+
20
31
  /**
21
32
  * Output cap for a request, so prompt plus output stays inside the model's
22
33
  * context window.
@@ -41,7 +52,8 @@ const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
41
52
  * OpenRouter-hosted model with no caller cap: the transport omits the catalog
42
53
  * default there so each upstream self-caps, and a fitted value would turn into
43
54
  * an explicit cap that filters upstreams). Otherwise returns
44
- * the remaining room (never below {@link MIN_FITTED_OUTPUT_TOKENS}); a
55
+ * the remaining room minus {@link OUTPUT_FIT_HEADWAY_TOKENS} (never below
56
+ * {@link MIN_FITTED_OUTPUT_TOKENS}); a
45
57
  * prompt that fills the whole window still overflows and is left to the
46
58
  * caller's compaction. Near a full window the floor means a turn can stop on
47
59
  * `length` instead of failing with a 400.
@@ -67,7 +79,7 @@ export function fitOutputTokensToContextWindow(
67
79
  if (!requested || !contextWindow || contextWindow <= 0) return maxTokens;
68
80
  if (stopsOutputAtContextWindow(model)) return maxTokens;
69
81
 
70
- const room = contextWindow - countPromptTokens(context, tokenizer);
82
+ const room = contextWindow - countPromptTokens(context, tokenizer) - OUTPUT_FIT_HEADWAY_TOKENS;
71
83
  if (room >= requested) return maxTokens;
72
84
  return Math.max(MIN_FITTED_OUTPUT_TOKENS, room);
73
85
  }