@oh-my-pi/pi-ai 18.4.12 → 18.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,26 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.5.1] - 2026-10-03
6
+
7
+ ### Fixed
8
+
9
+ - Fixed DeepSeek and OpenAI Responses requests failing or entering retry loops when replayed tool calls contained repaired arguments, orphaned tool results, or missing reasoning context.
10
+ - Fixed requests to models that do not support sampling parameters from failing with HTTP 400 errors when accessed through non-native providers. Sampling parameters are now omitted for incompatible models, including requests made by chat judging, title generation, skill descriptions, and memory extraction.
11
+ - Fixed OpenRouter BYOK usage being reported as free; provider inference costs and applicable credits charges are now included in session and status-line cost reporting.
12
+ - Fixed Claude background and long-running bash commands losing their requested timeout and being terminated at the default deadline.
13
+ - Fixed cleared credential cooldowns being incorrectly restored by another concurrently running session.
14
+ - Fixed retryable Cursor provider errors, such as “Unable to reach the model provider,” from prematurely ending a turn.
15
+ - Fixed resumed OpenAI Responses sessions losing earlier plaintext reasoning, which could cause self-hosted Responses servers to reprocess the entire context after a restart.
16
+ - Fixed non-retryable HTTP 4xx responses being retried when their error messages contained transient-error terms such as “server_error,” “timeout,” or “overloaded.”
17
+ - Fixed Cursor models receiving earlier multi-step tool calls as if they occurred simultaneously instead of in their original execution order.
18
+
19
+ ## [18.5.0] - 2026-10-03
20
+
21
+ ### Fixed
22
+
23
+ - Fixed AWS `credential_process` on Windows stripping backslashes from unquoted paths such as `C:\Users\me\helper.exe`; commands are now split with Windows command-line rules there, matching the AWS CLI.
24
+
5
25
  ## [18.4.12] - 2026-10-02
6
26
 
7
27
  ### Added
@@ -2294,29 +2314,4 @@
2294
2314
 
2295
2315
  - Fixed the auth-broker (`OMP_AUTH_BROKER_URL`) rejecting OAuth credentials that carry provider-specific extension fields (e.g. an MCP server's `tokenUrl`/`clientId`/`clientSecret`/`resource` embedded for self-contained token refresh): the OAuth credential wire schema was `.strict()`, so `POST /v1/credential` failed with `400 unrecognized_keys` and a broker-backed MCP reauth reported success while the reloaded credential lacked its refresh material and could no longer refresh. The OAuth wire schema now uses `.loose()` to preserve unknown fields — matching the field-preserving local SQLite store — so extra OAuth fields round-trip through broker set->get (envelope and API-key schemas stay strict).
2296
2316
 
2297
- ## [15.13.0] - 2026-06-14
2298
-
2299
- ### Fixed
2300
-
2301
- - Fixed OpenAI Responses/Realtime SSE stream handler crashing with "Error Code undefined: undefined" when parsing error events with nested error details by falling back to the nested error object fields.
2302
- - Fixed OpenAI-compatible providers that reject forced `tool_choice` on thinking-required models by downgrading unsupported forced choices to `auto` while keeping tools available ([#2546](https://github.com/can1357/oh-my-pi/issues/2546)).
2303
- - Fixed GitHub Copilot Anthropic transport (`api.githubcopilot.com/v1/messages`) returning `400 tools.0.custom.eager_input_streaming: Extra inputs are not permitted` on every tool-bearing turn by stopping the emission of the per-tool `eager_input_streaming` flag and the `fine-grained-tool-streaming-2025-05-14` beta header on the Copilot transport — the proxy whitelists neither ([#2558](https://github.com/can1357/oh-my-pi/issues/2558)).
2304
- - Disabled Bun's native ~300s pre-response `fetch` timeout in every streaming provider (OpenAI completions/responses, Azure responses, Anthropic, Codex SSE, Bedrock, Gemini CLI, Ollama). The configurable first-event/idle/SDK watchdogs (`PI_STREAM_FIRST_EVENT_TIMEOUT_MS`, `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS`, `compat.streamIdleTimeoutMs`) were silently capped by Bun's hidden ceiling, so cold large-context streams (e.g. self-hosted vLLM at multi-hundred-K prompts) died at exactly 300s with `TimeoutError: The operation timed out.` Direct callers of `./providers/{amazon-bedrock,google-gemini-cli,ollama,openai-codex-responses}` (which bypass `register-builtins`' iterator-level watchdog) now install a pre-response `AbortSignal.timeout(firstEventTimeoutMs)` alongside the disable, so a stalled upstream still fails within the configured budget instead of hanging forever ([#2422](https://github.com/can1357/oh-my-pi/issues/2422))
2305
- - Fixed Gemini / Antigravity streams (Google Cloud Code Assist API) creating a trailing empty text block and emitting redundant `text_start`/`text_delta`/`text_end` events at the end of the turn when the final SSE chunk contains an empty text part (`text: ""`). The parser now ignores empty text parts, preserving the active transcript block state and ensuring proper nesting and rendering of subsequent background jobs or new turns.
2306
- - Preserved terminal Google `thoughtSignature`s by still extracting and applying the signature on the active block even when the text part is empty or undefined.
2307
- - Stopped Gemini Antigravity sessions (`gemini-3*` / Claude under Cloud Code Assist) from leaking system rule reminders and personality preambles into the final response, by appending an explicit 'do not output rule checks' instruction to the injected system parts.
2308
- - Fixed Gemini / Antigravity streams (Google Cloud Code Assist API) letting a `functionCall` part's own `thoughtSignature` clobber the preceding text or thinking block's signature on `think → tool` and `text → tool` turns. A signed function-call part has `text: undefined`, so it fell into the terminal-signature branch while the prior block was still active; that branch now skips function-call parts, leaving the tool call's signature on the tool call where it belongs and preventing corrupted signatures on same-model replay.
2309
- - Fixed MiniMax-M3 OpenAI-compatible streams rendering reasoning twice when the same chunk carried both `<think>…</think>` content and structured `reasoning_content`; structured reasoning now wins and cumulative MiniMax reasoning snapshots are collapsed to deltas using a per-signature snapshot tracker that survives the `</think>`-to-text block transition (so post-answer cumulative snapshots don't reinstate a duplicate thinking block). ([#2433](https://github.com/can1357/oh-my-pi/issues/2433))
2310
-
2311
- ## [15.12.6] - 2026-06-14
2312
-
2313
- ### Changed
2314
-
2315
- - Bumped Z.AI (GLM Coding Plan) API key validation probe to glm-5.2.
2316
-
2317
- ### Fixed
2318
-
2319
- - Fixed tool schema conversion for non-Cloud Code Assist Google Gemini models by normalizing parameters with `normalizeSchemaForGoogle` to prevent un-normalized schema properties (such as `additionalProperties: false` or type arrays) from causing Gemini API errors.
2320
- - Fixed OpenAI-family request builders dropping forced named `tool_choice` directives when the named tool is absent from the serialized `tools` array, preventing spec-strict providers from rejecting self-inconsistent requests. ([#1701](https://github.com/can1357/oh-my-pi/issues/1701))
2321
-
2322
- Older entries are archived in [packages/ai/CHANGELOG.md@edb740cbad49](https://github.com/can1357/oh-my-pi/blob/edb740cbad499dbc96f8b5b46ebf78f70d6af4d0/packages/ai/CHANGELOG.md).
2317
+ Older entries are archived in [packages/ai/CHANGELOG.md@bac7e83b5b0e](https://github.com/can1357/oh-my-pi/blob/bac7e83b5b0eb86c909c17830a6666efc359578b/packages/ai/CHANGELOG.md).
@@ -29,14 +29,10 @@ export interface CredentialResolveOptions {
29
29
  fetch?: FetchImpl;
30
30
  }
31
31
  export declare function resolveAwsCredentials(opts?: CredentialResolveOptions): Promise<ResolvedCredentials>;
32
- /** POSIX-shell-style tokenizer used by the AWS CLI for `credential_process`.
33
- *
34
- * Outside quotes a backslash escapes the next character. Inside single quotes
35
- * everything is literal (no escapes, cannot contain `'`). Inside double quotes
36
- * a backslash only escapes `$`, `` ` ``, `"`, and `\` — every other backslash
37
- * is preserved verbatim, which is what makes Windows paths like
38
- * `"C:\Program Files\tool\auth.exe"` survive tokenization. */
39
- export declare function tokenizeCredentialProcessCommand(cmd: string): string[];
32
+ /** Split a `credential_process` command into argv the way botocore's
33
+ * `compat_shell_split` does: Windows command-line (CRT) rules on `win32`,
34
+ * POSIX shell rules everywhere else. */
35
+ export declare function tokenizeCredentialProcessCommand(cmd: string, platform?: NodeJS.Platform): string[];
40
36
  /** Test/diagnostic helper — drops cached credentials. */
41
37
  export declare function clearAwsCredentialCache(): void;
42
38
  /**
@@ -4,6 +4,15 @@
4
4
  * usable is present.
5
5
  */
6
6
  export declare function summarizeConnectErrorDetails(details: unknown): string | undefined;
7
+ /**
8
+ * True when Cursor's `aiserver.v1.ErrorDetails` entry declares the failure
9
+ * retryable. Its JSON `debug` form carries `details.isRetryable` — set on
10
+ * upstream outages such as ERROR_OPENAI ("Unable to reach the model
11
+ * provider") whose bare `code: message` text (`unavailable: Error`)
12
+ * classifies as a hard failure on its own. Other detail types are ignored
13
+ * even when they share the `debug.details` shape.
14
+ */
15
+ export declare function hasRetryableCursorErrorDetail(details: unknown): boolean;
7
16
  /**
8
17
  * Formats a Connect end-stream error object into a diagnosable message.
9
18
  * The "Connect error code: message" prefix is preserved exactly; detail
@@ -78,7 +78,13 @@ export declare function applyOpenAIServiceTier(params: {
78
78
  * proxy can never skew those costs.
79
79
  */
80
80
  export declare function applyOpenAIResponsesServiceTierCost(model: Pick<Model, "provider" | "serviceTierCost">, usage: AssistantMessage["usage"], responseServiceTier: unknown, requestServiceTier: ServiceTier | null | undefined): void;
81
- /** Reconcile token-price estimates with a gateway's authoritative account charge. */
81
+ /**
82
+ * Reconcile token-price estimates with a gateway's authoritative account charge.
83
+ * BYOK turns (`is_byok: true`) price from the provider spend in
84
+ * `cost_details.upstream_inference_cost` plus whatever credits charge
85
+ * OpenRouter reports in `cost` (its BYOK fee is plan-dependent and can be $0),
86
+ * so both are covered.
87
+ */
82
88
  export declare function applyProviderReportedCost(model: Pick<Model, "provider">, usage: Usage, rawUsage: unknown): void;
83
89
  export interface OpenAIUsageAccountingInput {
84
90
  promptTokens: number;
@@ -470,6 +476,13 @@ export interface BuildResponsesInputOptions<TApi extends Api> {
470
476
  supportsImageDetailOriginal: boolean;
471
477
  systemRole?: "system" | "developer";
472
478
  nativeHistory?: {
479
+ /**
480
+ * Replay same-provider native history. `false` marks a cold provider
481
+ * session (#489): native items are withheld except remote-compaction
482
+ * history and assistant turns that carry no server-issued state
483
+ * ({@link isColdReplayableResponsesTurn}); other turns are rebuilt from
484
+ * message content.
485
+ */
473
486
  replay: boolean;
474
487
  filterReasoning: boolean;
475
488
  };
@@ -1,4 +1,15 @@
1
1
  import type { Api, AssistantMessage, Message, Model } from "../types.js";
2
+ /**
3
+ * Maximum tool-call id length the strictest replay provider accepts.
4
+ *
5
+ * Anthropic requires `^[a-zA-Z0-9_-]+$` with a 64-char cap; Google and Codex
6
+ * `normalizeToolCallId` implementations cap individual id segments to the same
7
+ * 64-char ceiling. Replacement ids minted here flow back through
8
+ * `convertAnthropicMessages` (and friends) unchanged, so the `_dupN` suffix
9
+ * MUST not push a normalized id past this bound.
10
+ */
11
+ export declare const MAX_TOOL_CALL_ID_LENGTH = 64;
12
+ export declare function appendDuplicateSuffix(originalId: string, suffix: string, maxLength: number): string;
2
13
  /**
3
14
  * Whether a tool-call name cannot belong to any declared tool: missing, empty,
4
15
  * longer than OpenAI's 128-character replay limit, or containing whitespace or
@@ -1,6 +1,14 @@
1
1
  import type { ToolCall } from "../types.js";
2
2
  /** Final arguments must not reuse the streaming parser's auto-closed preview. */
3
3
  export declare function parseToolCallArguments(json: string | undefined): ToolCall["arguments"];
4
+ /**
5
+ * Native Responses `function_call.arguments` to persist for replay.
6
+ *
7
+ * {@link parseToolCallArguments} repairs some invalid JSON (e.g. `"name": ,`) and the tool runs with the
8
+ * repaired value; replay drops a call whose stored arguments do not parse, orphaning its output (#14155).
9
+ * Returns `raw` when it already parses or the arguments were unrepairable, else the executed arguments.
10
+ */
11
+ export declare function replayableToolCallArguments(raw: string, executed: ToolCall["arguments"]): string;
4
12
  /** Longest raw argument text kept in the parse-error diagnostic. */
5
13
  export declare const INVALID_ARGUMENTS_RAW_LIMIT = 512;
6
14
  /** Bound raw argument text for a diagnostic. Text this function already bounded is returned unchanged. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@oh-my-pi/pi-ai",
3
- "version": "18.4.12",
3
+ "version": "18.5.1",
4
4
  "description": "Unified LLM API with automatic model discovery and provider configuration",
5
5
  "keywords": [
6
6
  "ai",
@@ -155,11 +155,11 @@
155
155
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
156
156
  },
157
157
  "dependencies": {
158
- "@oh-my-pi/omptype": "18.4.12",
159
- "@oh-my-pi/pi-catalog": "18.4.12",
160
- "@oh-my-pi/pi-natives": "18.4.12",
161
- "@oh-my-pi/pi-utils": "18.4.12",
162
- "@oh-my-pi/pi-wire": "18.4.12"
158
+ "@oh-my-pi/omptype": "18.5.1",
159
+ "@oh-my-pi/pi-catalog": "18.5.1",
160
+ "@oh-my-pi/pi-natives": "18.5.1",
161
+ "@oh-my-pi/pi-utils": "18.5.1",
162
+ "@oh-my-pi/pi-wire": "18.5.1"
163
163
  },
164
164
  "devDependencies": {
165
165
  "@types/bun": "^1.3.14"
@@ -140,13 +140,16 @@ export class CredentialBlocks implements BlocksApi {
140
140
  if (!block) return undefined;
141
141
  // Once mirrored successfully, the store is authoritative for cross-process
142
142
  // deletion. A reset in a sibling must also retire this process's old map entry.
143
+ // Upserts are longest-wins, so a persisted row shorter than the deadline this
144
+ // process mirrored means the row was deleted and replaced: the old deadline
145
+ // was superseded and must not win the next longest-wins merge.
143
146
  if (block.persisted && !this.#deps.health.damaged && this.#deps.store.getCredentialBlock) {
144
147
  const separator = backoffKey.indexOf("\0");
145
148
  const providerKey = separator < 0 ? backoffKey : backoffKey.slice(0, separator);
146
149
  const scope = separator < 0 ? "" : backoffKey.slice(separator + 1);
147
150
  try {
148
151
  const persisted = this.#deps.store.getCredentialBlock(credentialId, providerKey, scope);
149
- if (persisted === undefined) {
152
+ if (persisted === undefined || persisted < block.until) {
150
153
  this.#deleteCredentialBackoff(backoffKey, credentialId);
151
154
  return undefined;
152
155
  }
@@ -552,28 +552,31 @@ function classifyText(
552
552
  ) {
553
553
  kinds |= Flag.UsageLimit;
554
554
  }
555
- if (isTimeoutText(errorMessage)) kinds |= Flag.Transient | Flag.Timeout;
556
- else if (isTransientErrorText(errorMessage)) kinds |= Flag.Transient;
557
- // A stream truncation, transport-level stream drop, or forwarded Codex HTTP
558
- // body-read failure may not match TRANSIENT_TRANSPORT_PATTERN. Flag it
559
- // explicitly so AIError.retriable and the turn-recovery layer treat it as
560
- // retryable, matching the provider retry path (isProviderRetryableError).
561
- // Separate `if` (not chained onto the else-if) so a timeout whose text also
562
- // reads as a truncation keeps Flag.Timeout alongside Flag.Transient. The
563
- // string arm applies the strict STREAM_PARSE_DIAGNOSTIC_PATTERN, per the
564
- // rationale on isTransientStreamParseError. Skip a phrase that rides on a
565
- // terminal 4xx (e.g. a malformed request rejected as "400 unexpected EOF"):
566
- // that is a deterministic client error that replays identically, so keep it
567
- // terminal. classify() carries the outer terminal status down the cause
568
- // chain so a wrapped truncation (ProviderHttpError 400 → cause "unexpected
569
- // EOF") is caught here too.
570
- if (
571
- !isTerminalClientErrorStatus(statusClean) &&
572
- (isTransientStreamParseError(errorMessage) ||
555
+ // Transport/timeout/truncation wording that rides on a terminal 4xx (e.g. a
556
+ // "400 unexpected EOF" malformed request, or a region/entitlement denial
557
+ // whose body carries `type=server_error`) describes a deterministic client
558
+ // error that replays identically, so keep it terminal — the same 4xx policy
559
+ // as isProviderRetryableError. classify() carries the outer terminal status
560
+ // down the cause chain so a wrapped phrase (ProviderHttpError 400 → cause
561
+ // "unexpected EOF") is caught here too.
562
+ if (!isTerminalClientErrorStatus(statusClean)) {
563
+ if (isTimeoutText(errorMessage)) kinds |= Flag.Transient | Flag.Timeout;
564
+ else if (isTransientErrorText(errorMessage)) kinds |= Flag.Transient;
565
+ // A stream truncation, transport-level stream drop, or forwarded Codex
566
+ // HTTP body-read failure may not match TRANSIENT_TRANSPORT_PATTERN. Flag
567
+ // it explicitly so AIError.retriable and the turn-recovery layer treat it
568
+ // as retryable, matching the provider retry path. Separate `if` (not
569
+ // chained onto the else-if) so a timeout whose text also reads as a
570
+ // truncation keeps Flag.Timeout alongside Flag.Transient. The string arm
571
+ // applies the strict STREAM_PARSE_DIAGNOSTIC_PATTERN, per the rationale
572
+ // on isTransientStreamParseError.
573
+ if (
574
+ isTransientStreamParseError(errorMessage) ||
573
575
  isTransientStreamDropError(errorMessage) ||
574
- CODEX_HTTP_BODY_READ_ERROR_PATTERN.test(errorMessage))
575
- ) {
576
- kinds |= Flag.Transient;
576
+ CODEX_HTTP_BODY_READ_ERROR_PATTERN.test(errorMessage)
577
+ ) {
578
+ kinds |= Flag.Transient;
579
+ }
577
580
  }
578
581
  // A concurrency cap (e.g. Vertex "Online prediction concurrent requests
579
582
  // quota exceeded") is transient — shed-and-backoff. The bare wording need
@@ -614,9 +617,10 @@ export function classify(error: unknown, api?: Api): number {
614
617
  const causeTokenEvidence = hasCauseTokenContextOverflowEvidence(error);
615
618
  let link: unknown = error;
616
619
  // A terminal 4xx on an outer link governs its own cause diagnostics: a
617
- // wrapped truncation is describing why the deterministic request failed,
618
- // not an independently retryable transport fault. Carry it down so the
619
- // stream-parse guard in classifyText sees it on the status-less cause.
620
+ // wrapped truncation or transport phrase is describing why the deterministic
621
+ // request failed, not an independently retryable transport fault. Carry it
622
+ // down so the terminal-4xx guard in classifyText sees it on the status-less
623
+ // cause.
620
624
  let governingTerminalStatus: number | undefined;
621
625
  while (link !== undefined && link !== null) {
622
626
  if (typeof link === "object") {
@@ -5530,7 +5530,10 @@ const ANTHROPIC_TOOL_SCHEMA_STRING_FORMATS = new Set([
5530
5530
  "ipv6",
5531
5531
  "uuid",
5532
5532
  ]);
5533
- const ANTHROPIC_STRICT_TOOL_ALLOWLIST = new Set(["bash", "python", "edit", "find"]);
5533
+ // Not `bash`: strict decoding fixes property order, so an optional key declared
5534
+ // before one the model has already written can no longer be emitted, and bash's
5535
+ // `timeout` vanished from every `async`-first call.
5536
+ const ANTHROPIC_STRICT_TOOL_ALLOWLIST = new Set(["python", "edit", "find"]);
5534
5537
  const MAX_ANTHROPIC_STRICT_TOOLS = 20;
5535
5538
  const MAX_ANTHROPIC_STRICT_OPTIONAL_PARAMETERS = 24;
5536
5539
  const MAX_ANTHROPIC_STRICT_UNION_PARAMETERS = 16;
@@ -737,14 +737,74 @@ function isBatchScript(executable: string): boolean {
737
737
  return lower.endsWith(".cmd") || lower.endsWith(".bat");
738
738
  }
739
739
 
740
+ /** Split a `credential_process` command into argv the way botocore's
741
+ * `compat_shell_split` does: Windows command-line (CRT) rules on `win32`,
742
+ * POSIX shell rules everywhere else. */
743
+ export function tokenizeCredentialProcessCommand(cmd: string, platform: NodeJS.Platform = process.platform): string[] {
744
+ return platform === "win32" ? tokenizeWindowsCommand(cmd) : tokenizePosixCommand(cmd);
745
+ }
746
+
747
+ /** Windows C-runtime argv rules (botocore `_windows_shell_split`): only space
748
+ * and tab delimit, only double quotes group, and backslashes are literal unless
749
+ * a run of them precedes a `"` — then each pair yields one backslash and an odd
750
+ * trailing backslash escapes the quote. Keeps `C:\path\tool.exe` intact. */
751
+ function tokenizeWindowsCommand(cmd: string): string[] {
752
+ const tokens: string[] = [];
753
+ let current = "";
754
+ let hasToken = false;
755
+ let quoted = false;
756
+ let backslashes = 0;
757
+ for (const ch of cmd) {
758
+ if (ch === "\\") {
759
+ backslashes++;
760
+ continue;
761
+ }
762
+ if (ch === '"') {
763
+ current += "\\".repeat(backslashes >> 1);
764
+ hasToken = true;
765
+ const escaped = (backslashes & 1) === 1;
766
+ backslashes = 0;
767
+ if (escaped) current += '"';
768
+ else quoted = !quoted;
769
+ continue;
770
+ }
771
+ if (backslashes > 0) {
772
+ current += "\\".repeat(backslashes);
773
+ hasToken = true;
774
+ backslashes = 0;
775
+ }
776
+ if ((ch === " " || ch === "\t") && !quoted) {
777
+ if (hasToken) {
778
+ tokens.push(current);
779
+ current = "";
780
+ hasToken = false;
781
+ }
782
+ continue;
783
+ }
784
+ current += ch;
785
+ hasToken = true;
786
+ }
787
+ if (quoted) {
788
+ throw new AIError.AwsCredentialsError(
789
+ "AWS credential_process command has an unterminated quote.",
790
+ "credential-process",
791
+ );
792
+ }
793
+ if (backslashes > 0) {
794
+ current += "\\".repeat(backslashes);
795
+ hasToken = true;
796
+ }
797
+ if (hasToken) tokens.push(current);
798
+ return tokens;
799
+ }
800
+
740
801
  /** POSIX-shell-style tokenizer used by the AWS CLI for `credential_process`.
741
802
  *
742
803
  * Outside quotes a backslash escapes the next character. Inside single quotes
743
804
  * everything is literal (no escapes, cannot contain `'`). Inside double quotes
744
805
  * a backslash only escapes `$`, `` ` ``, `"`, and `\` — every other backslash
745
- * is preserved verbatim, which is what makes Windows paths like
746
- * `"C:\Program Files\tool\auth.exe"` survive tokenization. */
747
- export function tokenizeCredentialProcessCommand(cmd: string): string[] {
806
+ * is preserved verbatim. */
807
+ function tokenizePosixCommand(cmd: string): string[] {
748
808
  const tokens: string[] = [];
749
809
  let current = "";
750
810
  let hasToken = false;
@@ -52,6 +52,27 @@ export function summarizeConnectErrorDetails(details: unknown): string | undefin
52
52
  return truncate(parts.join("; "), MAX_EXTRA_DETAIL_CHARS);
53
53
  }
54
54
 
55
+ /** Connect detail type Cursor uses for its own error envelope. */
56
+ const CURSOR_ERROR_DETAILS_TYPE = "aiserver.v1.ErrorDetails";
57
+
58
+ /**
59
+ * True when Cursor's `aiserver.v1.ErrorDetails` entry declares the failure
60
+ * retryable. Its JSON `debug` form carries `details.isRetryable` — set on
61
+ * upstream outages such as ERROR_OPENAI ("Unable to reach the model
62
+ * provider") whose bare `code: message` text (`unavailable: Error`)
63
+ * classifies as a hard failure on its own. Other detail types are ignored
64
+ * even when they share the `debug.details` shape.
65
+ */
66
+ export function hasRetryableCursorErrorDetail(details: unknown): boolean {
67
+ if (!Array.isArray(details)) return false;
68
+ for (const entry of details) {
69
+ if (!isRecord(entry) || entry.type !== CURSOR_ERROR_DETAILS_TYPE || !isRecord(entry.debug)) continue;
70
+ const inner = entry.debug.details;
71
+ if (isRecord(inner) && inner.isRetryable === true) return true;
72
+ }
73
+ return false;
74
+ }
75
+
55
76
  /**
56
77
  * Formats a Connect end-stream error object into a diagnosable message.
57
78
  * The "Connect error code: message" prefix is preserved exactly; detail
@@ -226,7 +226,7 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
226
226
  import { connectProxiedSocket, getProxyForUrl, wrapFetchForProxy } from "../utils/proxy";
227
227
  import { createRequestDebugSession, isRequestDebugEnabled, type RequestDebugResponseLog } from "../utils/request-debug";
228
228
  import { sanitizeSchemaForCursor, toolWireSchema } from "../utils/schema";
229
- import { formatConnectEndStreamError } from "./connect-error-detail";
229
+ import { formatConnectEndStreamError, hasRetryableCursorErrorDetail } from "./connect-error-detail";
230
230
  import mcpExternalHandoffMessage from "./cursor-external-tool-handoff.md" with { type: "text" };
231
231
  import {
232
232
  buildMcpStateResult,
@@ -613,7 +613,17 @@ function classifyConnectError(error: Record<string, unknown>): Error {
613
613
  if (structured) return classifyCursorStructuredError(structured);
614
614
  const code = typeof error.code === "string" ? error.code : "unknown";
615
615
  const message = typeof error.message === "string" ? error.message : "Unknown error";
616
- return new ConnectEndStreamError(`Connect error ${code}: ${message}`, formatConnectEndStreamError(error));
616
+ const endStreamError = new ConnectEndStreamError(
617
+ `Connect error ${code}: ${message}`,
618
+ formatConnectEndStreamError(error),
619
+ );
620
+ // Without a decodable binary detail, Cursor's retry verdict survives only in
621
+ // the detail's debug JSON; the classification text drops it, so carry it as
622
+ // a structured flag.
623
+ if (hasRetryableCursorErrorDetail(error.details)) {
624
+ AIError.attach(endStreamError, AIError.create(AIError.Flag.Transient));
625
+ }
626
+ return endStreamError;
617
627
  }
618
628
 
619
629
  function parseConnectEndStream(data: Uint8Array): Error | null {
@@ -5566,18 +5576,33 @@ function canReplayCursorThinking(msg: AssistantMessage, targetModelId: string |
5566
5576
  );
5567
5577
  }
5568
5578
 
5569
- function buildCursorAssistantContent(
5570
- msg: AssistantMessage,
5571
- targetModelId: string | undefined,
5572
- ): CursorRootPromptAssistantContentPart[] {
5573
- const content: CursorRootPromptAssistantContentPart[] = [];
5579
+ interface CursorAssistantStep {
5580
+ content: CursorRootPromptAssistantContentPart[];
5581
+ /** Raw (un-normalized) ids of the calls this round issued, in order. */
5582
+ callIds: string[];
5583
+ }
5584
+
5585
+ /**
5586
+ * Split one assistant message into the model rounds it recorded. A Cursor
5587
+ * server turn persists as a single message whose tool calls interleave with
5588
+ * the text and reasoning that followed each result; a round ends at a call
5589
+ * followed by anything other than another call. Consecutive calls stay in one
5590
+ * round — they were issued together. Hidden reasoning still marks a boundary.
5591
+ */
5592
+ function buildCursorAssistantSteps(msg: AssistantMessage, targetModelId: string | undefined): CursorAssistantStep[] {
5593
+ const steps: CursorAssistantStep[] = [];
5594
+ let step: CursorAssistantStep = { content: [], callIds: [] };
5574
5595
  const replayThinking = canReplayCursorThinking(msg, targetModelId);
5575
5596
  for (const item of msg.content) {
5597
+ if (item.type !== "toolCall" && step.callIds.length > 0) {
5598
+ steps.push(step);
5599
+ step = { content: [], callIds: [] };
5600
+ }
5576
5601
  if (item.type === "text") {
5577
- if (item.text) content.push({ type: "text", text: item.text });
5602
+ if (item.text) step.content.push({ type: "text", text: item.text });
5578
5603
  } else if (item.type === "thinking") {
5579
5604
  if (replayThinking && item.thinking) {
5580
- content.push({
5605
+ step.content.push({
5581
5606
  type: "reasoning",
5582
5607
  text: item.thinking,
5583
5608
  providerOptions: { cursor: { modelName: msg.model } },
@@ -5591,15 +5616,17 @@ function buildCursorAssistantContent(
5591
5616
  // gets the whole Run rejected as opaque resource_exhausted. Sanitize the
5592
5617
  // id everywhere it reaches the wire; the tool-result side normalizes the
5593
5618
  // same id identically, so the call/result pairing stays intact.
5594
- content.push({
5619
+ step.content.push({
5595
5620
  type: "tool-call",
5596
5621
  toolCallId: normalizeToolCallId(item.id),
5597
5622
  toolName: item.name,
5598
5623
  args: normalizeCursorMcpArguments(item.arguments),
5599
5624
  });
5625
+ step.callIds.push(item.id);
5600
5626
  }
5601
5627
  }
5602
- return content;
5628
+ steps.push(step);
5629
+ return steps.filter(({ content }) => content.length > 0);
5603
5630
  }
5604
5631
 
5605
5632
  function assertCursorKimiK3HistoryReplayable(
@@ -5712,25 +5739,61 @@ function buildRootPromptMessagesJson(
5712
5739
  ): Uint8Array[] {
5713
5740
  assertCursorKimiK3HistoryReplayable(messages, activeUserMessageIndex, targetModelId);
5714
5741
  const historyEnd = activeUserMessageIndex >= 0 ? activeUserMessageIndex : messages.length;
5715
- const { pairedToolCallIds } = collectCursorToolHistory(messages, historyEnd);
5742
+ const { toolResults, pairedToolCallIds } = collectCursorToolHistory(messages, historyEnd);
5716
5743
  const entries: Uint8Array[] = [...systemPromptIds];
5717
5744
  const pushJson = (obj: unknown) => {
5718
5745
  const bytes = new TextEncoder().encode(JSON.stringify(obj));
5719
5746
  entries.push(storeCursorBlob(blobStore, bytes));
5720
5747
  };
5748
+ // Results already replayed under the step that issued their call; the
5749
+ // message-order pass below skips them.
5750
+ const emittedResults = new Set<string>();
5751
+ // Emit even when the result text is empty: the assistant `tool-call` is
5752
+ // already in history, so dropping the pair would replay an orphaned call.
5753
+ const pushToolResult = (result: ToolResultMessage) => {
5754
+ const toolCallId = normalizeToolCallId(result.toolCallId);
5755
+ pushJson({
5756
+ role: "tool",
5757
+ id: toolCallId,
5758
+ content: [
5759
+ {
5760
+ type: "tool-result",
5761
+ toolName: result.toolName,
5762
+ toolCallId,
5763
+ result: toolResultToText(result),
5764
+ ...(result.isError ? { isError: true } : {}),
5765
+ },
5766
+ ],
5767
+ });
5768
+ emittedResults.add(result.toolCallId);
5769
+ };
5721
5770
 
5722
- for (let i = 0; i < messages.length; i++) {
5723
- if (i === activeUserMessageIndex) break;
5771
+ for (let i = 0; i < historyEnd; i++) {
5724
5772
  const msg = messages[i];
5725
5773
  if (msg.role === "user" || msg.role === "developer") {
5726
5774
  const content = buildCursorRootPromptContent(msg.content);
5727
5775
  if (content.length === 0) continue;
5728
5776
  pushJson({ role: "user", content });
5729
5777
  } else if (msg.role === "assistant") {
5730
- const content = buildCursorAssistantContent(msg, targetModelId);
5731
- if (content.length === 0) continue;
5732
- pushJson({ role: "assistant", content });
5778
+ const steps = buildCursorAssistantSteps(msg, targetModelId);
5779
+ for (const [stepIndex, step] of steps.entries()) {
5780
+ pushJson({ role: "assistant", content: step.content });
5781
+ // The final step's results stay in message order below: they are
5782
+ // what the following turn responds to, and may arrive out of order.
5783
+ if (stepIndex === steps.length - 1) continue;
5784
+ // Replay this round's results in the order they arrived: calls issued
5785
+ // together can finish out of call order.
5786
+ const roundCallIds = new Set(step.callIds);
5787
+ for (let j = i + 1; j < historyEnd; j++) {
5788
+ const later = messages[j];
5789
+ if (later.role !== "toolResult" || !roundCallIds.has(later.toolCallId)) continue;
5790
+ if (emittedResults.has(later.toolCallId)) continue;
5791
+ const result = toolResults.get(later.toolCallId);
5792
+ if (result) pushToolResult(result);
5793
+ }
5794
+ }
5733
5795
  } else if (msg.role === "toolResult") {
5796
+ if (emittedResults.has(msg.toolCallId)) continue;
5734
5797
  if (!pairedToolCallIds.has(msg.toolCallId)) {
5735
5798
  pushJson({
5736
5799
  role: "assistant",
@@ -5738,22 +5801,7 @@ function buildRootPromptMessagesJson(
5738
5801
  });
5739
5802
  continue;
5740
5803
  }
5741
- // Emit even when the result text is empty: the assistant `tool-call` is
5742
- // already in history, so dropping the pair would replay an orphaned call.
5743
- const toolCallId = normalizeToolCallId(msg.toolCallId);
5744
- pushJson({
5745
- role: "tool",
5746
- id: toolCallId,
5747
- content: [
5748
- {
5749
- type: "tool-result",
5750
- toolName: msg.toolName,
5751
- toolCallId,
5752
- result: toolResultToText(msg),
5753
- ...(msg.isError ? { isError: true } : {}),
5754
- },
5755
- ],
5756
- });
5804
+ pushToolResult(msg);
5757
5805
  }
5758
5806
  }
5759
5807
 
@@ -22,7 +22,7 @@ import {
22
22
  USER_AGENT,
23
23
  } from "@oh-my-pi/pi-utils";
24
24
  import * as AIError from "../error";
25
- import { parseToolCallArguments } from "../utils/tool-call-arguments";
25
+ import { parseToolCallArguments, replayableToolCallArguments } from "../utils/tool-call-arguments";
26
26
  import { getEnvApiKey, isOfficialCodexApiUrl } from "../stream";
27
27
  import type {
28
28
  Api,
@@ -273,7 +273,19 @@ const CODEX_WEBSOCKET_FIRST_EVENT_TIMEOUT_MS = Number($env.PI_CODEX_WEBSOCKET_FI
273
273
  const CODEX_WEBSOCKET_RETRY_BUDGET = Number($env.PI_CODEX_WEBSOCKET_RETRY_BUDGET || CODEX_MAX_RETRIES);
274
274
  const CODEX_WEBSOCKET_RETRY_DELAY_MS = Number($env.PI_CODEX_WEBSOCKET_RETRY_DELAY_MS || CODEX_RETRY_DELAY_MS);
275
275
  const CODEX_WEBSOCKET_TRANSPORT_ERROR_PREFIX = "Codex websocket transport error";
276
- const CODEX_RETRYABLE_EVENT_CODES = new Set(["model_error", "server_error", "internal_error"]);
276
+ /**
277
+ * The server's experimental native turn lane refuses `response.steer` with
278
+ * this code, then drops the in-flight response and closes the socket.
279
+ */
280
+ const CODEX_NATIVE_LANE_STEER_REJECTED_CODE = "unsupported_native_inflight_message";
281
+ // The native-lane steering rejection is replayable: the session stops steering
282
+ // first, so the replay cannot trip it again.
283
+ const CODEX_RETRYABLE_EVENT_CODES = new Set([
284
+ "model_error",
285
+ "server_error",
286
+ "internal_error",
287
+ CODEX_NATIVE_LANE_STEER_REJECTED_CODE,
288
+ ]);
277
289
  const CODEX_RETRYABLE_EVENT_MESSAGE =
278
290
  /processing your request|retry your request|temporar(?:y|ily)|overloaded|service.?unavailable|internal error|server error/i;
279
291
  const CODEX_PROVIDER_SESSION_STATE_KEY = "openai-codex-responses";
@@ -437,6 +449,8 @@ type CodexWebSocketSessionState = {
437
449
  * pending tool output; see {@link planSteeredRequest}.
438
450
  */
439
451
  acceptedSteering?: { connection: CodexWebSocketConnection; steers: CodexAcceptedSteer[] };
452
+ /** The server refused steering for this session (native turn lane); later responses are not steered. */
453
+ steeringUnsupported?: boolean;
440
454
  lastTransport?: CodexTransport;
441
455
  fallbackCount: number;
442
456
  lastFallbackAt?: number;
@@ -2454,11 +2468,13 @@ class CodexStreamProcessor {
2454
2468
  /** Start delivering caller steering into the response `response.created` announced. */
2455
2469
  #startSteering(rawEvent: Record<string, unknown>): void {
2456
2470
  const source = this.options?.liveSteering;
2457
- const connection = this.runtime.websocketState?.connection;
2471
+ const state = this.runtime.websocketState;
2472
+ const connection = state?.connection;
2458
2473
  const responseId = asRecord(rawEvent.response)?.id;
2459
2474
  if (
2460
2475
  !source ||
2461
2476
  !connection ||
2477
+ state?.steeringUnsupported ||
2462
2478
  this.#steerPump ||
2463
2479
  typeof responseId !== "string" ||
2464
2480
  this.runtime.transport !== "websocket" ||
@@ -2589,6 +2605,7 @@ class CodexStreamProcessor {
2589
2605
  name: item.name,
2590
2606
  arguments: parseToolCallArguments(item.arguments),
2591
2607
  };
2608
+ item.arguments = replayableToolCallArguments(item.arguments, toolCall.arguments);
2592
2609
  if (block?.type === "toolCall") {
2593
2610
  // Persist the authoritative final args on the stored block; the throttled
2594
2611
  // delta parser may have left block.arguments stale (often `{}`).
@@ -2722,7 +2739,28 @@ class CodexStreamProcessor {
2722
2739
  );
2723
2740
  }
2724
2741
 
2742
+ /**
2743
+ * The native turn lane rejected our steering and dropped the response with
2744
+ * its socket: stop steering this session and forget the dead socket so the
2745
+ * provider retry replays on a fresh one.
2746
+ */
2747
+ #stopSteeringOnNativeLaneRejection(error: unknown): void {
2748
+ const state = this.runtime.websocketState;
2749
+ if (
2750
+ !state ||
2751
+ !(error instanceof CodexProviderStreamError) ||
2752
+ error.code !== CODEX_NATIVE_LANE_STEER_REJECTED_CODE
2753
+ ) {
2754
+ return;
2755
+ }
2756
+ state.steeringUnsupported = true;
2757
+ state.connection?.close("native-lane-steer-rejected");
2758
+ state.connection = undefined;
2759
+ resetCodexWebSocketAppendState(state);
2760
+ }
2761
+
2725
2762
  async #recoverStreamError(error: unknown): Promise<boolean> {
2763
+ this.#stopSteeringOnNativeLaneRejection(error);
2726
2764
  if (
2727
2765
  error instanceof CodexSteerCommitError &&
2728
2766
  this.runtime.websocketState &&
@@ -4043,6 +4081,11 @@ class CodexWebSocketConnection {
4043
4081
  this.#handleSteerEvent(parsed);
4044
4082
  return;
4045
4083
  }
4084
+ // The native lane answers `response.steer` with a plain `error` that
4085
+ // also ends the response: refuse the submission, then fail the stream.
4086
+ if (parsed.type === "error" && parsed.code === CODEX_NATIVE_LANE_STEER_REJECTED_CODE) {
4087
+ this.#refuseSteerWaiters(parsed.code, typeof parsed.message === "string" ? parsed.message : undefined);
4088
+ }
4046
4089
  this.#push(parsed);
4047
4090
  } catch (error) {
4048
4091
  notifyCodexWebSocketMalformed(this.#streamObserver, event.data, error);
@@ -4336,6 +4379,12 @@ class CodexWebSocketConnection {
4336
4379
  return index < 0 ? undefined : this.#steerWaiters.splice(index, 1)[0];
4337
4380
  }
4338
4381
 
4382
+ #refuseSteerWaiters(code: string, message: string | undefined): void {
4383
+ const waiters = this.#steerWaiters;
4384
+ this.#steerWaiters = [];
4385
+ for (const waiter of waiters) waiter.resolve({ accepted: false, code, message });
4386
+ }
4387
+
4339
4388
  #rejectSteerWaiters(reason: string): void {
4340
4389
  const waiters = this.#steerWaiters;
4341
4390
  if (waiters.length === 0) return;
@@ -35,7 +35,7 @@ import {
35
35
  } from "@oh-my-pi/pi-utils";
36
36
  import { NO_AUTH_SENTINEL } from "../auth-retry";
37
37
  import * as AIError from "../error";
38
- import { parseToolCallArguments } from "../utils/tool-call-arguments";
38
+ import { parseToolCallArguments, replayableToolCallArguments } from "../utils/tool-call-arguments";
39
39
  import {
40
40
  type Api,
41
41
  type AssistantMessage,
@@ -406,7 +406,13 @@ export function applyOpenAIResponsesServiceTierCost(
406
406
  usage.cost.total = usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite;
407
407
  }
408
408
 
409
- /** Reconcile token-price estimates with a gateway's authoritative account charge. */
409
+ /**
410
+ * Reconcile token-price estimates with a gateway's authoritative account charge.
411
+ * BYOK turns (`is_byok: true`) price from the provider spend in
412
+ * `cost_details.upstream_inference_cost` plus whatever credits charge
413
+ * OpenRouter reports in `cost` (its BYOK fee is plan-dependent and can be $0),
414
+ * so both are covered.
415
+ */
410
416
  export function applyProviderReportedCost(model: Pick<Model, "provider">, usage: Usage, rawUsage: unknown): void {
411
417
  if (
412
418
  (model.provider !== "openrouter" && model.provider !== "cline-pass") ||
@@ -414,7 +420,23 @@ export function applyProviderReportedCost(model: Pick<Model, "provider">, usage:
414
420
  rawUsage === null
415
421
  )
416
422
  return;
417
- const reportedCost = Reflect.get(rawUsage, "cost");
423
+ let reportedCost = Reflect.get(rawUsage, "cost");
424
+ // BYOK turns run on the account's own provider key: `cost` carries only the
425
+ // credits charge OpenRouter bills the turn (its BYOK fee, plan-dependent and
426
+ // $0 inside the free allowance) while `cost_details.upstream_inference_cost`
427
+ // carries the provider spend. Both are real charges, so add them (verified
428
+ // live 2026-10-02: openrouter/openai/gpt-6.1-sol returned `cost: 0,
429
+ // is_byok: true, cost_details.upstream_inference_cost: 6.6e-05`).
430
+ if (Reflect.get(rawUsage, "is_byok") === true) {
431
+ const details = Reflect.get(rawUsage, "cost_details");
432
+ const upstreamCost =
433
+ typeof details === "object" && details !== null ? Reflect.get(details, "upstream_inference_cost") : undefined;
434
+ if (typeof upstreamCost === "number" && Number.isFinite(upstreamCost) && upstreamCost >= 0) {
435
+ const creditsCharge =
436
+ typeof reportedCost === "number" && Number.isFinite(reportedCost) && reportedCost >= 0 ? reportedCost : 0;
437
+ reportedCost = creditsCharge + upstreamCost;
438
+ }
439
+ }
418
440
  if (typeof reportedCost !== "number" || !Number.isFinite(reportedCost) || reportedCost < 0) return;
419
441
 
420
442
  const estimatedCost = usage.cost.total;
@@ -1902,6 +1924,13 @@ export interface BuildResponsesInputOptions<TApi extends Api> {
1902
1924
  supportsImageDetailOriginal: boolean;
1903
1925
  systemRole?: "system" | "developer";
1904
1926
  nativeHistory?: {
1927
+ /**
1928
+ * Replay same-provider native history. `false` marks a cold provider
1929
+ * session (#489): native items are withheld except remote-compaction
1930
+ * history and assistant turns that carry no server-issued state
1931
+ * ({@link isColdReplayableResponsesTurn}); other turns are rebuilt from
1932
+ * message content.
1933
+ */
1905
1934
  replay: boolean;
1906
1935
  filterReasoning: boolean;
1907
1936
  };
@@ -2008,6 +2037,31 @@ export function escapeReplayedControlTokens(items: ResponseInput): ResponseInput
2008
2037
  });
2009
2038
  }
2010
2039
 
2040
+ /**
2041
+ * Whether a same-provider assistant turn may replay its native items while the
2042
+ * provider session is still cold (#489). A cold session rebuilds turns from
2043
+ * message content because some backends bind native items to one connection
2044
+ * (GitHub Copilot: `401 input item does not belong to this connection`, #488).
2045
+ * Binding needs server-issued state that survives replay sanitization: an
2046
+ * `encrypted_content` blob or an item id. A turn with neither, whose every
2047
+ * reasoning item carries plaintext `reasoning_text`, has nothing to bind and is
2048
+ * exactly what the server receives once the session warms; rebuilding it would
2049
+ * drop that reasoning and change the prompt prefix the server cached.
2050
+ * Summary-only reasoning keeps the rebuild: it is no evidence of a server that
2051
+ * returns plaintext reasoning.
2052
+ */
2053
+ function isColdReplayableResponsesTurn(items: ResponseInput): boolean {
2054
+ let hasReasoning = false;
2055
+ for (const item of items) {
2056
+ if ("id" in item && typeof item.id === "string") return false;
2057
+ if ("encrypted_content" in item && typeof item.encrypted_content === "string") return false;
2058
+ if (item.type !== "reasoning") continue;
2059
+ if (!item.content?.some(part => part.type === "reasoning_text")) return false;
2060
+ hasReasoning = true;
2061
+ }
2062
+ return hasReasoning;
2063
+ }
2064
+
2011
2065
  export function buildResponsesInput<TApi extends Api>(options: BuildResponsesInputOptions<TApi>): ResponseInput {
2012
2066
  const messages: ResponseInput = [];
2013
2067
  const systemPrompts = options.systemRole ? normalizeSystemPrompts(options.context.systemPrompt) : [];
@@ -2132,7 +2186,12 @@ export function buildResponsesInput<TApi extends Api>(options: BuildResponsesInp
2132
2186
  options.requiresReasoningReplayForToolCalls ?? false,
2133
2187
  )
2134
2188
  : undefined;
2135
- if (nativeReplayEnabled && sanitizedHistoryItems) {
2189
+ const replayNativeItems =
2190
+ nativeReplayEnabled ||
2191
+ (options.nativeHistory !== undefined &&
2192
+ rawSanitizedHistoryItems !== undefined &&
2193
+ isColdReplayableResponsesTurn(rawSanitizedHistoryItems));
2194
+ if (replayNativeItems && sanitizedHistoryItems) {
2136
2195
  // Model-owned replay items can carry reserved control-token
2137
2196
  // spellings as data (the model writing *about* Harmony); escape the
2138
2197
  // transport copy just like client turns.
@@ -3464,7 +3523,6 @@ export async function processResponsesStream<TApi extends Api>(
3464
3523
  }
3465
3524
  } else if (event.type === "response.output_item.done") {
3466
3525
  const item = structuredCloneJSON(event.item);
3467
- options?.onOutputItemDone?.(item);
3468
3526
  const entry =
3469
3527
  item.type === "function_call" || item.type === "custom_tool_call"
3470
3528
  ? lookupOpenItem({ output_index: event.output_index, item_id: item.id ?? item.call_id })
@@ -3516,6 +3574,7 @@ export async function processResponsesStream<TApi extends Api>(
3516
3574
  : item.arguments
3517
3575
  ? parseToolCallArguments(item.arguments)
3518
3576
  : parseToolCallArguments(block?.[kStreamingPartialJson]);
3577
+ item.arguments = replayableToolCallArguments(item.arguments, args);
3519
3578
  const toolCall: ToolCall = {
3520
3579
  type: "toolCall",
3521
3580
  id: encodeResponsesToolCallId(item.call_id, item.id),
@@ -3596,6 +3655,8 @@ export async function processResponsesStream<TApi extends Api>(
3596
3655
  } else if (item.type === "image_generation_call" && item.status === "completed" && item.result) {
3597
3656
  appendResponsesImageResult(output, stream, item.result);
3598
3657
  }
3658
+ // After the branches so the native history item carries any normalization above.
3659
+ options?.onOutputItemDone?.(item);
3599
3660
  } else if (terminalEvent) {
3600
3661
  const response = terminalEvent.response;
3601
3662
  const shouldPromoteIncompleteToolUse =
@@ -28,7 +28,7 @@ const enum ToolCallStatus {
28
28
  * `convertAnthropicMessages` (and friends) unchanged, so the `_dupN` suffix
29
29
  * MUST not push a normalized id past this bound.
30
30
  */
31
- const MAX_TOOL_CALL_ID_LENGTH = 64;
31
+ export const MAX_TOOL_CALL_ID_LENGTH = 64;
32
32
 
33
33
  /**
34
34
  * OpenAI Responses-family APIs mint composite tool ids (`call_id|item_id`);
@@ -128,7 +128,7 @@ function toolCallPairingKey(id: string, originScope: ToolCallOriginScope): strin
128
128
  return originScope.responsesComponents.has(prefix) ? prefix : id;
129
129
  }
130
130
 
131
- function appendDuplicateSuffix(originalId: string, suffix: string, maxLength: number): string {
131
+ export function appendDuplicateSuffix(originalId: string, suffix: string, maxLength: number): string {
132
132
  // Responses-family ids are composites (`callId|itemId`): the wire call_id is
133
133
  // the FIRST segment (normalizeResponsesToolCallId splits on `|`), so the
134
134
  // suffix must land on every segment or the duplicate collapses back onto the
package/src/stream.ts CHANGED
@@ -989,7 +989,10 @@ function streamDispatch<TApi extends Api>(
989
989
  context: Context,
990
990
  options?: OptionsForApi<TApi>,
991
991
  ): AssistantMessageEventStream {
992
- const requestOptions = withTransportFetch(model, (options || {}) as StreamOptions) as OptionsForApi<TApi>;
992
+ const requestOptions = withSupportedSamplingParams(
993
+ model,
994
+ withTransportFetch(model, (options || {}) as StreamOptions),
995
+ ) as OptionsForApi<TApi>;
993
996
  assertExplicitOpenAIResponsesPromptCacheSupport(model, requestOptions);
994
997
 
995
998
  // Check custom API registry first (extension-provided APIs like "vertex-claude-api")
@@ -1264,6 +1267,43 @@ function withInferenceSessionId(options?: SimpleStreamOptions): SimpleStreamOpti
1264
1267
  return { ...options, sessionId: crypto.randomUUID() };
1265
1268
  }
1266
1269
 
1270
+ type SamplingOptions = Pick<
1271
+ StreamOptions,
1272
+ "temperature" | "topP" | "topK" | "minP" | "presencePenalty" | "repetitionPenalty" | "frequencyPenalty"
1273
+ >;
1274
+
1275
+ /**
1276
+ * Drop explicit sampling parameters before any provider builds its payload
1277
+ * when the model's resolved `compat.supportsSamplingParams` is `false`. The
1278
+ * catalog's class rules assign that per model lineage on every compat record,
1279
+ * and an explicit compat override still wins, so this one check covers every
1280
+ * provider.
1281
+ */
1282
+ function withSupportedSamplingParams<T extends SamplingOptions>(model: Model<Api>, options: T): T {
1283
+ if (
1284
+ options.temperature === undefined &&
1285
+ options.topP === undefined &&
1286
+ options.topK === undefined &&
1287
+ options.minP === undefined &&
1288
+ options.presencePenalty === undefined &&
1289
+ options.repetitionPenalty === undefined &&
1290
+ options.frequencyPenalty === undefined
1291
+ ) {
1292
+ return options;
1293
+ }
1294
+ const compat = model.compat;
1295
+ if (!compat || !("supportsSamplingParams" in compat) || compat.supportsSamplingParams !== false) return options;
1296
+ const supported = { ...options };
1297
+ delete supported.temperature;
1298
+ delete supported.topP;
1299
+ delete supported.topK;
1300
+ delete supported.minP;
1301
+ delete supported.presencePenalty;
1302
+ delete supported.repetitionPenalty;
1303
+ delete supported.frequencyPenalty;
1304
+ return supported;
1305
+ }
1306
+
1267
1307
  export function streamSimple<TApi extends Api>(
1268
1308
  model: Model<TApi>,
1269
1309
  context: Context,
@@ -1312,7 +1352,10 @@ function streamSimpleRequest<TApi extends Api>(
1312
1352
  context: Context,
1313
1353
  options?: SimpleStreamOptions,
1314
1354
  ): AssistantMessageEventStream {
1315
- const requestOptions = withTransportFetch(model, (options || {}) as SimpleStreamOptions);
1355
+ const requestOptions = withSupportedSamplingParams(
1356
+ model,
1357
+ withTransportFetch(model, (options || {}) as SimpleStreamOptions),
1358
+ );
1316
1359
 
1317
1360
  const apiKeyResolver = isApiKeyResolver(requestOptions?.apiKey) ? requestOptions.apiKey : undefined;
1318
1361
  if (apiKeyResolver) {
@@ -12,6 +12,22 @@ export function parseToolCallArguments(json: string | undefined): ToolCall["argu
12
12
  }
13
13
  }
14
14
 
15
+ /**
16
+ * Native Responses `function_call.arguments` to persist for replay.
17
+ *
18
+ * {@link parseToolCallArguments} repairs some invalid JSON (e.g. `"name": ,`) and the tool runs with the
19
+ * repaired value; replay drops a call whose stored arguments do not parse, orphaning its output (#14155).
20
+ * Returns `raw` when it already parses or the arguments were unrepairable, else the executed arguments.
21
+ */
22
+ export function replayableToolCallArguments(raw: string, executed: ToolCall["arguments"]): string {
23
+ try {
24
+ JSON.parse(raw);
25
+ return raw;
26
+ } catch {
27
+ return executed && typeof executed === "object" && "__parseError" in executed ? raw : JSON.stringify(executed);
28
+ }
29
+ }
30
+
15
31
  /** Longest raw argument text kept in the parse-error diagnostic. */
16
32
  export const INVALID_ARGUMENTS_RAW_LIMIT = 512;
17
33
  const TRUNCATED_SUFFIX = /… \[truncated \d+ chars\]$/;