@oh-my-pi/pi-ai 18.2.0 → 18.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/CHANGELOG.md +53 -0
  2. package/README.md +2 -0
  3. package/dist/types/auth/sqlite-credential-store.d.ts +2 -1
  4. package/dist/types/auth-broker/remote-store.d.ts +17 -0
  5. package/dist/types/auth-gateway/index.d.ts +1 -0
  6. package/dist/types/auth-gateway/session-state.d.ts +118 -0
  7. package/dist/types/auth-storage.d.ts +17 -0
  8. package/dist/types/error/body-error.d.ts +15 -0
  9. package/dist/types/error/flags.d.ts +16 -0
  10. package/dist/types/error/index.d.ts +1 -0
  11. package/dist/types/index.d.ts +1 -0
  12. package/dist/types/oneshot-retry.d.ts +6 -0
  13. package/dist/types/provider-session-state.d.ts +46 -0
  14. package/dist/types/providers/amazon-bedrock.d.ts +3 -0
  15. package/dist/types/providers/aws-sigv4.d.ts +12 -0
  16. package/dist/types/providers/openai-codex/request-transformer.d.ts +27 -0
  17. package/dist/types/providers/openai-responses.d.ts +15 -0
  18. package/dist/types/providers/openai-shared.d.ts +20 -3
  19. package/dist/types/registry/oauth/perplexity.d.ts +1 -7
  20. package/dist/types/registry/oauth/types.d.ts +8 -0
  21. package/dist/types/stream.d.ts +2 -0
  22. package/dist/types/types.d.ts +3 -1
  23. package/dist/types/usage/openai-codex.d.ts +3 -1
  24. package/dist/types/usage.d.ts +11 -1
  25. package/dist/types/utils/block-symbols.d.ts +36 -0
  26. package/dist/types/utils/openai-http.d.ts +2 -0
  27. package/dist/types/utils/retry-after.d.ts +2 -0
  28. package/dist/types/utils/schema/wire.d.ts +4 -5
  29. package/dist/types/utils.d.ts +9 -0
  30. package/package.json +6 -6
  31. package/src/auth/sqlite-credential-store.ts +8 -33
  32. package/src/auth-broker/remote-store.ts +73 -8
  33. package/src/auth-broker/wire-schemas.ts +1 -0
  34. package/src/auth-gateway/index.ts +1 -0
  35. package/src/auth-gateway/server.ts +186 -74
  36. package/src/auth-gateway/session-state.ts +312 -0
  37. package/src/auth-storage.ts +146 -15
  38. package/src/error/body-error.ts +310 -0
  39. package/src/error/flags.ts +63 -13
  40. package/src/error/index.ts +1 -0
  41. package/src/error/retryable.ts +2 -0
  42. package/src/index.ts +1 -0
  43. package/src/oneshot-retry.ts +13 -3
  44. package/src/provider-session-state.ts +56 -0
  45. package/src/providers/amazon-bedrock.ts +20 -3
  46. package/src/providers/anthropic-messages-server.ts +104 -23
  47. package/src/providers/anthropic-signature.ts +5 -2
  48. package/src/providers/anthropic.ts +101 -15
  49. package/src/providers/aws-sigv4.ts +16 -5
  50. package/src/providers/cursor.ts +60 -10
  51. package/src/providers/devin.ts +82 -28
  52. package/src/providers/openai-chat-server.ts +4 -0
  53. package/src/providers/openai-codex/request-transformer.ts +36 -0
  54. package/src/providers/openai-codex-responses.ts +35 -12
  55. package/src/providers/openai-completions.ts +49 -12
  56. package/src/providers/openai-reasoning-fallback.ts +6 -6
  57. package/src/providers/openai-responses-server.ts +2 -1
  58. package/src/providers/openai-responses.ts +52 -4
  59. package/src/providers/openai-shared.ts +199 -51
  60. package/src/registry/oauth/perplexity.ts +94 -28
  61. package/src/registry/oauth/types.ts +9 -0
  62. package/src/stream.ts +23 -2
  63. package/src/types.ts +3 -0
  64. package/src/usage/claude.ts +33 -0
  65. package/src/usage/google-antigravity.ts +8 -2
  66. package/src/usage/openai-codex.ts +94 -11
  67. package/src/usage.ts +8 -1
  68. package/src/utils/block-symbols.ts +57 -0
  69. package/src/utils/http-inspector.ts +20 -0
  70. package/src/utils/openai-http.ts +39 -3
  71. package/src/utils/retry-after.ts +12 -0
  72. package/src/utils/schema/normalize.ts +3 -3
  73. package/src/utils/schema/stamps.ts +33 -45
  74. package/src/utils/schema/wire.ts +9 -7
  75. package/src/utils.ts +67 -22
@@ -99,7 +99,7 @@ import {
99
99
  } from "./github-copilot-headers";
100
100
  import { servedModelFromOpenRouterReasoning } from "./anthropic-signature";
101
101
  import type { ChatCompletionCreateParamsStreaming } from "./openai-chat-wire";
102
- import type { InputItem } from "./openai-codex/request-transformer";
102
+ import { type InputItem, sanitizeCodexCallId } from "./openai-codex/request-transformer";
103
103
  import type {
104
104
  Response as OpenAIResponse,
105
105
  ResponseComputerToolCall,
@@ -770,6 +770,32 @@ export function applyOpenAIExtraBody<P extends object>(
770
770
  }
771
771
  }
772
772
 
773
+ /**
774
+ * Normalize `content: null` to `[]` in place across a list of inbound wire
775
+ * message items.
776
+ *
777
+ * Codex (and other OpenAI clients) emit `content: null` on empty message items
778
+ * during multi-turn/tool turns; OpenAI tolerates it. Normalizing before the
779
+ * auth-gateway's request schema validation lets the item take the same path as
780
+ * an explicit empty content array — for both validation and any native
781
+ * history-replay clone — instead of 400ing. Shared by the `/v1/responses`
782
+ * (`input[]`) and `/v1/chat/completions` (`messages[]`) routes.
783
+ *
784
+ * `isEligible` lets a route skip items whose content is not array-typed, e.g.
785
+ * the chat `function` role whose content is `string | null`. See issue #10956.
786
+ */
787
+ export function coerceNullMessageContentInPlace(
788
+ items: unknown,
789
+ isEligible?: (item: Record<string, unknown>) => boolean,
790
+ ): void {
791
+ if (!Array.isArray(items)) return;
792
+ for (const item of items) {
793
+ if (typeof item !== "object" || item === null || Array.isArray(item)) continue;
794
+ const record = item as Record<string, unknown>;
795
+ if (record.content === null && (isEligible?.(record) ?? true)) record.content = [];
796
+ }
797
+ }
798
+
773
799
  /**
774
800
  * Chat Completions streaming request body shaped by the OpenAI-family providers.
775
801
  * (binary `thinking`, Qwen `enable_thinking`/`chat_template_kwargs`, Venice
@@ -808,7 +834,7 @@ export interface ChatCompletionsReasoningOptions {
808
834
 
809
835
  export type OpenAICompatEndpoint = "chat-completions" | "responses";
810
836
 
811
- export type OpenAIReasoningDisableReason = "caller" | "forced-tool-choice" | "tool-choice" | "not-requested";
837
+ export type OpenAIReasoningDisableReason = "caller" | "forced-tool-choice" | "tool-choice" | "tools" | "not-requested";
812
838
 
813
839
  export type OpenAICompatPolicyCompat = ResolvedOpenAISharedCompat &
814
840
  Partial<ResolvedOpenAICompat> &
@@ -820,6 +846,7 @@ export interface ResolveOpenAICompatPolicyOptions {
820
846
  reasoning?: string;
821
847
  disableReasoning?: boolean;
822
848
  toolChoice?: unknown;
849
+ hasTools?: boolean;
823
850
  strictResponsesPairing?: boolean;
824
851
  includeEncryptedReasoning?: boolean;
825
852
  filterReasoningHistory?: boolean;
@@ -927,12 +954,19 @@ export function resolveOpenAICompatPolicy<TApi extends Api>(
927
954
  !forcedToolChoiceSuppressesReasoning &&
928
955
  baseCompat.disableReasoningOnToolChoice &&
929
956
  options.toolChoice !== undefined;
957
+ const toolsSuppressReasoning =
958
+ !forcedToolChoiceSuppressesReasoning &&
959
+ !anyToolChoiceSuppressesReasoning &&
960
+ baseCompat.disableReasoningWithTools &&
961
+ options.hasTools === true;
930
962
  const requestedAndAllowed = requestedEffort !== undefined && !options.disableReasoning && modelSupported;
931
963
  const conflictDisableReason: OpenAIReasoningDisableReason | undefined = forcedToolChoiceSuppressesReasoning
932
964
  ? "forced-tool-choice"
933
965
  : anyToolChoiceSuppressesReasoning
934
966
  ? "tool-choice"
935
- : undefined;
967
+ : toolsSuppressReasoning
968
+ ? "tools"
969
+ : undefined;
936
970
  const disableReason: OpenAIReasoningDisableReason | undefined = options.disableReasoning
937
971
  ? "caller"
938
972
  : conflictDisableReason;
@@ -1190,7 +1224,7 @@ export function applyChatCompletionsReasoningParams(
1190
1224
  params: OpenAICompletionsParams,
1191
1225
  model: Model<"openai-completions">,
1192
1226
  compat: ResolvedOpenAICompat,
1193
- options: (ChatCompletionsReasoningOptions & { toolChoice?: unknown }) | undefined,
1227
+ options: (ChatCompletionsReasoningOptions & { toolChoice?: unknown; hasTools?: boolean }) | undefined,
1194
1228
  ): void {
1195
1229
  const policy = resolveOpenAICompatPolicy(model, {
1196
1230
  endpoint: "chat-completions",
@@ -1198,6 +1232,7 @@ export function applyChatCompletionsReasoningParams(
1198
1232
  reasoning: options?.reasoning,
1199
1233
  disableReasoning: options?.disableReasoning,
1200
1234
  toolChoice: options?.toolChoice,
1235
+ hasTools: options?.hasTools,
1201
1236
  });
1202
1237
  applyChatCompletionsCompatPolicy(params, policy);
1203
1238
  if (
@@ -1420,18 +1455,14 @@ export function normalizeResponsesToolCallIdForTransform(
1420
1455
  model?: Model<Api>,
1421
1456
  source?: AssistantMessage,
1422
1457
  ): string {
1423
- if (!id.includes("|")) return id;
1458
+ const sep = id.search(/[\n|]/);
1459
+ if (sep < 0 && id.length <= 64 && /^[a-zA-Z0-9_-]+$/.test(id)) return id;
1424
1460
  const isForeignToolCall =
1425
1461
  source != null && model != null && (source.provider !== model.provider || source.api !== model.api);
1426
- if (isForeignToolCall) {
1427
- const [callId, itemId] = id.split("|");
1428
- const normalizeIdPart = (part: string): string => {
1429
- const sanitized = part.replace(/[^a-zA-Z0-9_-]/g, "_");
1430
- const truncated = sanitized.length > 64 ? sanitized.slice(0, 64) : sanitized;
1431
- return truncated.replace(/_+$/, "");
1432
- };
1433
- const normalizedCallId = normalizeIdPart(callId);
1434
- let normalizedItemId = `fc_${Bun.hash(itemId).toString(36)}`;
1462
+ if (isForeignToolCall || sep >= 0 || id.length > 64) {
1463
+ const [callId, itemId] = sep > 0 ? [id.slice(0, sep), id.slice(sep + 1)] : [id, undefined];
1464
+ const normalizedCallId = sanitizeCodexCallId(callId);
1465
+ let normalizedItemId = itemId ? `fc_${Bun.hash(itemId).toString(36)}` : `fc_${Bun.hash(id).toString(36)}`;
1435
1466
  if (normalizedItemId.length > 64) normalizedItemId = normalizedItemId.slice(0, 64);
1436
1467
  return `${normalizedCallId}|${normalizedItemId}`;
1437
1468
  }
@@ -2113,10 +2144,16 @@ export function buildResponsesInput<TApi extends Api>(options: BuildResponsesInp
2113
2144
  msgIndex++;
2114
2145
  }
2115
2146
 
2116
- const hoisted = hoistInterleavedResponsesToolBatchMessages(messages);
2117
- const withRepairedOutputs = options.repairOrphanOutputs ? repairOrphanResponsesToolOutputs(hoisted) : hoisted;
2147
+ // Repair orphan outputs/calls first: both can inject an assistant `message`
2148
+ // (an `[Orphan … result]` note, a `[Computer call interrupted …]` note) in
2149
+ // place of, or beside, a tool item — wedging it between another call's
2150
+ // `function_call` and `function_call_output`. Hoist runs last so it relocates
2151
+ // any wedged message — model-streamed or repair-injected — out of the batch,
2152
+ // preserving the Responses call→output pairing (#11473, extends #8789).
2153
+ const withRepairedOutputs = options.repairOrphanOutputs ? repairOrphanResponsesToolOutputs(messages) : messages;
2118
2154
  const withRepairedCalls = repairOrphanResponsesToolCalls(withRepairedOutputs);
2119
- return stripUnpairedOpenAIResponsesComputerReasoningIdsForReplay(withRepairedCalls);
2155
+ const hoisted = hoistInterleavedResponsesToolBatchMessages(withRepairedCalls);
2156
+ return stripUnpairedOpenAIResponsesComputerReasoningIdsForReplay(hoisted);
2120
2157
  }
2121
2158
 
2122
2159
  type ResponsesReplayAssistantMessage = Omit<ResponseOutputMessage, "id"> & { id?: string };
@@ -2309,19 +2346,22 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
2309
2346
  // reasoning dropped by compaction/archive budget) the carried text is
2310
2347
  // empty — and DeepSeek-family targets reject an empty `reasoning_text`
2311
2348
  // exactly like a missing item (#10690), so substitute a non-empty
2312
- // placeholder. The `id` still prefers a surviving upstream item id.
2349
+ // placeholder. The `id` is only set when an upstream item id survived:
2350
+ // providers that validate reasoning ids against their own store (Meta via
2351
+ // OpenRouter) reject a fabricated `rs_…` with "Referenced reasoning item
2352
+ // … was not found or has expired", and the targets that need the item
2353
+ // (DeepSeek, Kimi, Meta) all accept it without an id.
2313
2354
  const carriedReasoningText = carriedReasoningTexts.join("\n");
2314
2355
  const reasoningText =
2315
2356
  carriedReasoningText.length > 0 ? carriedReasoningText : SYNTHETIC_REASONING_REPLAY_PLACEHOLDER;
2316
- const reasoningId =
2317
- synthesizedReasoningItemId ?? `rs_${Bun.hash(`${model.id}:${msgIndex}:${reasoningText}`).toString(36)}`;
2318
- const reasoningItem: ResponseReasoningItem = {
2357
+ const reasoningItem = {
2319
2358
  type: "reasoning",
2320
- id: reasoningId,
2359
+ ...(synthesizedReasoningItemId ? { id: synthesizedReasoningItemId } : {}),
2321
2360
  summary: [],
2322
2361
  content: [{ type: "reasoning_text", text: reasoningText }],
2323
- };
2324
- outputItems.unshift(reasoningItem);
2362
+ } satisfies Omit<ResponseReasoningItem, "id"> & Partial<Pick<ResponseReasoningItem, "id">>;
2363
+ // The vendored SDK type marks `id` required; the wire accepts its absence.
2364
+ outputItems.unshift(reasoningItem as ResponseReasoningItem);
2325
2365
  }
2326
2366
 
2327
2367
  return outputItems;
@@ -2475,21 +2515,38 @@ export function appendResponsesToolResultMessages<TApi extends Api>(
2475
2515
  */
2476
2516
  type ResponsesToolCallBlock = ToolCall & { [kStreamingPartialJson]: string; [kStreamingLastParseLen]?: number };
2477
2517
 
2518
+ // Proxies sometimes send payloadless progress frames. They carry no text; a
2519
+ // supplied non-string payload is malformed output, not an empty delta.
2520
+ function optionalResponsesText(value: unknown, field: string): string | undefined {
2521
+ if (value === undefined) return undefined;
2522
+ if (typeof value !== "string") throw new TypeError(`Invalid Responses ${field}: expected a string`);
2523
+ return value;
2524
+ }
2525
+
2478
2526
  function ensureReasoningSummaryPart(
2479
2527
  item: ResponseReasoningItem,
2480
- summaryIndex: number,
2528
+ summaryIndex: number | undefined,
2481
2529
  ): ResponseReasoningItem["summary"][number] {
2482
2530
  item.summary = item.summary || [];
2531
+ if (summaryIndex === undefined) summaryIndex = Math.max(0, item.summary.length - 1);
2532
+ if (!Number.isSafeInteger(summaryIndex) || summaryIndex < 0) {
2533
+ throw new TypeError("Invalid Responses summary_index: expected a non-negative integer");
2534
+ }
2483
2535
  while (item.summary.length <= summaryIndex) {
2484
2536
  item.summary.push({ type: "summary_text", text: "" });
2485
2537
  }
2486
- return item.summary[summaryIndex]!;
2538
+ const part = item.summary[summaryIndex]!;
2539
+ part.text = optionalResponsesText(part.text, "summary text") ?? "";
2540
+ return part;
2487
2541
  }
2488
2542
 
2489
2543
  export function appendReasoningSummaryPart(
2490
2544
  item: ResponseReasoningItem,
2491
- part: ResponseReasoningItem["summary"][number],
2545
+ part: ResponseReasoningItem["summary"][number] | undefined,
2492
2546
  ): void {
2547
+ if (part === undefined) return;
2548
+ if (part?.type !== "summary_text") throw new TypeError("Invalid Responses reasoning summary part");
2549
+ part.text = optionalResponsesText(part.text, "summary text") ?? "";
2493
2550
  item.summary = item.summary || [];
2494
2551
  item.summary.push(part);
2495
2552
  }
@@ -2573,8 +2630,10 @@ export function appendReasoningSummaryTextDelta(
2573
2630
  stream: AssistantMessageEventStream,
2574
2631
  output: AssistantMessage,
2575
2632
  contentIndex: number,
2576
- summaryIndex = 0,
2633
+ summaryIndex?: number,
2577
2634
  ): void {
2635
+ delta = optionalResponsesText(delta, "reasoning summary delta") ?? "";
2636
+ if (!delta) return;
2578
2637
  const part = ensureReasoningSummaryPart(item, summaryIndex);
2579
2638
  block.thinking += delta;
2580
2639
  part.text += delta;
@@ -2594,6 +2653,9 @@ export function applyReasoningSummaryTextDone(
2594
2653
  output: AssistantMessage,
2595
2654
  contentIndex: number,
2596
2655
  ): void {
2656
+ const snapshot = optionalResponsesText(text, "reasoning summary text");
2657
+ if (snapshot === undefined) return;
2658
+ text = snapshot;
2597
2659
  const part = ensureReasoningSummaryPart(item, summaryIndex);
2598
2660
  const previous = part.text;
2599
2661
  part.text = text;
@@ -2603,8 +2665,8 @@ export function applyReasoningSummaryTextDone(
2603
2665
  stream.push({ type: "thinking_delta", contentIndex, delta: text, partial: output });
2604
2666
  return;
2605
2667
  }
2606
- if (text.startsWith(block.thinking)) {
2607
- const delta = text.slice(block.thinking.length);
2668
+ if (text.startsWith(previous) && block.thinking.endsWith(previous)) {
2669
+ const delta = text.slice(previous.length);
2608
2670
  if (!delta) return;
2609
2671
  block.thinking += delta;
2610
2672
  stream.push({ type: "thinking_delta", contentIndex, delta, partial: output });
@@ -2681,6 +2743,8 @@ export function appendMessageTextDelta(
2681
2743
  contentIndex: number,
2682
2744
  partType: "output_text" | "refusal",
2683
2745
  ): void {
2746
+ delta = optionalResponsesText(delta, "message delta") ?? "";
2747
+ if (!delta) return;
2684
2748
  item.content = item.content || [];
2685
2749
  let lastPart = item.content[item.content.length - 1];
2686
2750
  if (lastPart?.type !== partType) {
@@ -2694,12 +2758,41 @@ export function appendMessageTextDelta(
2694
2758
  }
2695
2759
  block.text += delta;
2696
2760
  if (lastPart.type === "output_text") {
2697
- lastPart.text += delta;
2761
+ lastPart.text = (optionalResponsesText(lastPart.text, "output text") ?? "") + delta;
2698
2762
  } else {
2699
- lastPart.refusal += delta;
2763
+ lastPart.refusal = (optionalResponsesText(lastPart.refusal, "refusal") ?? "") + delta;
2700
2764
  }
2701
2765
  stream.push({ type: "text_delta", contentIndex, delta, partial: output });
2702
2766
  }
2767
+
2768
+ /** Recover text omitted from delta frames without replaying an already streamed prefix. */
2769
+ function applyMessageTextDone(
2770
+ item: ResponseOutputMessage,
2771
+ block: TextContent,
2772
+ text: string,
2773
+ stream: AssistantMessageEventStream,
2774
+ output: AssistantMessage,
2775
+ contentIndex: number,
2776
+ partType: "output_text" | "refusal",
2777
+ ): void {
2778
+ const snapshot = optionalResponsesText(text, "message text");
2779
+ if (snapshot === undefined) return;
2780
+ const lastPart = item.content?.[item.content.length - 1];
2781
+ const previous =
2782
+ lastPart?.type === partType
2783
+ ? (optionalResponsesText(lastPart.type === "output_text" ? lastPart.text : lastPart.refusal, "message text") ??
2784
+ "")
2785
+ : "";
2786
+ if (snapshot.startsWith(previous)) {
2787
+ appendMessageTextDelta(item, block, snapshot.slice(previous.length), stream, output, contentIndex, partType);
2788
+ } else if (lastPart?.type === partType) {
2789
+ // A correction cannot be represented as an append-only delta. Keep it for
2790
+ // the completed block rather than appending contradictory text.
2791
+ if (lastPart.type === "output_text") lastPart.text = snapshot;
2792
+ else lastPart.refusal = snapshot;
2793
+ block.text = finalizeMessageText(item, block.text);
2794
+ }
2795
+ }
2703
2796
  /** Chooses final message text while treating non-empty terminal content as authoritative. */
2704
2797
  export function finalizeMessageText(item: ResponseOutputMessage, streamedText: string): string {
2705
2798
  if (!item.content?.length) return streamedText || "";
@@ -2727,6 +2820,8 @@ export function accumulateToolCallArgumentsDelta(
2727
2820
  output: AssistantMessage,
2728
2821
  contentIndex: number,
2729
2822
  ): void {
2823
+ delta = optionalResponsesText(delta, "function call arguments delta") ?? "";
2824
+ if (!delta) return;
2730
2825
  block[kStreamingPartialJson] += delta;
2731
2826
  const throttled = parseStreamingJsonThrottled(block[kStreamingPartialJson], block[kStreamingLastParseLen] ?? 0);
2732
2827
  if (throttled) {
@@ -2755,6 +2850,8 @@ export function accumulateCustomToolCallInputDelta(
2755
2850
  output: AssistantMessage,
2756
2851
  contentIndex: number,
2757
2852
  ): void {
2853
+ delta = optionalResponsesText(delta, "custom tool input delta") ?? "";
2854
+ if (!delta) return;
2758
2855
  block[kStreamingPartialJson] += delta;
2759
2856
  block.arguments = { input: block[kStreamingPartialJson] };
2760
2857
  stream.push({ type: "toolcall_delta", contentIndex, delta, partial: output });
@@ -3046,8 +3143,23 @@ export async function processResponsesStream<TApi extends Api>(
3046
3143
  if (entry && identifierlessFunctionDeltaTarget === entry) identifierlessFunctionDeltaTarget = undefined;
3047
3144
  if (entry && lastOpenItem === entry) lastOpenItem = null;
3048
3145
  };
3049
- const contentIndexOf = (block: ThinkingContent | TextContent | StreamingToolCallBlock): number =>
3050
- output.content.indexOf(block);
3146
+ // Content blocks are append-only for the lifetime of the stream, so each
3147
+ // block's index is stable once pushed. Map block → index to keep the
3148
+ // per-delta contentIndex lookup O(1); the legacy linear scan would turn a
3149
+ // long turn (many blocks × many deltas) quadratic and is what pegs the
3150
+ // shared JS main thread on streaming-heavy sessions (issue #10605).
3151
+ const contentIndexByBlock = new Map<object, number>();
3152
+ const pushContentBlock = (block: ThinkingContent | TextContent | StreamingToolCallBlock | ToolCall): void => {
3153
+ contentIndexByBlock.set(block, output.content.length);
3154
+ output.content.push(block);
3155
+ };
3156
+ const contentIndexOf = (block: ThinkingContent | TextContent | StreamingToolCallBlock): number => {
3157
+ const index = contentIndexByBlock.get(block);
3158
+ // Blocks appended outside this closure (e.g. images via
3159
+ // appendResponsesImageResult) never route through contentIndexOf, but
3160
+ // keep the fallback so an unexpected lookup stays correct.
3161
+ return index ?? output.content.indexOf(block);
3162
+ };
3051
3163
 
3052
3164
  let sawFirstToken = false;
3053
3165
  // Whether the current stream produced a completed native `web_search_call`
@@ -3067,7 +3179,7 @@ export async function processResponsesStream<TApi extends Api>(
3067
3179
  const item = event.item;
3068
3180
  if (item.type === "reasoning") {
3069
3181
  const block: ThinkingContent = { type: "thinking", thinking: "", itemId: item.id };
3070
- output.content.push(block);
3182
+ pushContentBlock(block);
3071
3183
  registerOpenItem(event.output_index, item.id, { item, block });
3072
3184
  stream.push({ type: "thinking_start", contentIndex: contentIndexOf(block), partial: output });
3073
3185
  } else if (item.type === "message") {
@@ -3076,7 +3188,7 @@ export async function processResponsesStream<TApi extends Api>(
3076
3188
  text: "",
3077
3189
  textSignature: encodeTextSignatureV1(item.id, item.phase ?? undefined),
3078
3190
  };
3079
- output.content.push(block);
3191
+ pushContentBlock(block);
3080
3192
  registerOpenItem(event.output_index, item.id, { item, block });
3081
3193
  stream.push({ type: "text_start", contentIndex: contentIndexOf(block), partial: output });
3082
3194
  } else if (item.type === "function_call") {
@@ -3087,7 +3199,7 @@ export async function processResponsesStream<TApi extends Api>(
3087
3199
  arguments: {},
3088
3200
  [kStreamingPartialJson]: item.arguments || "",
3089
3201
  };
3090
- output.content.push(block);
3202
+ pushContentBlock(block);
3091
3203
  registerOpenItem(
3092
3204
  event.output_index,
3093
3205
  item.id,
@@ -3105,7 +3217,7 @@ export async function processResponsesStream<TApi extends Api>(
3105
3217
  providerMetadata: computerCallMetadata(item),
3106
3218
  [kStreamingPartialJson]: "",
3107
3219
  };
3108
- output.content.push(block);
3220
+ pushContentBlock(block);
3109
3221
  registerOpenItem(event.output_index, item.id, { item, block }, item.call_id);
3110
3222
  stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output });
3111
3223
  } else if (item.type === "custom_tool_call") {
@@ -3123,7 +3235,7 @@ export async function processResponsesStream<TApi extends Api>(
3123
3235
  // accumulation buffer so later code that inspects the field still works.
3124
3236
  [kStreamingPartialJson]: item.input ?? "",
3125
3237
  };
3126
- output.content.push(block);
3238
+ pushContentBlock(block);
3127
3239
  registerOpenItem(
3128
3240
  event.output_index,
3129
3241
  item.id,
@@ -3171,12 +3283,13 @@ export async function processResponsesStream<TApi extends Api>(
3171
3283
  // Raw reasoning text delta from local providers that stream thinking
3172
3284
  // directly rather than via the OpenAI summary tracking protocol.
3173
3285
  const entry = lookupOpenItem(event);
3174
- if (entry?.item.type === "reasoning" && entry.block.type === "thinking") {
3175
- entry.block.thinking += event.delta;
3286
+ const delta = optionalResponsesText(event.delta, "reasoning delta");
3287
+ if (entry?.item.type === "reasoning" && entry.block.type === "thinking" && delta) {
3288
+ entry.block.thinking += delta;
3176
3289
  stream.push({
3177
3290
  type: "thinking_delta",
3178
3291
  contentIndex: contentIndexOf(entry.block),
3179
- delta: event.delta,
3292
+ delta,
3180
3293
  partial: output,
3181
3294
  });
3182
3295
  }
@@ -3209,6 +3322,19 @@ export async function processResponsesStream<TApi extends Api>(
3209
3322
  "refusal",
3210
3323
  );
3211
3324
  }
3325
+ } else if (event.type === "response.output_text.done" || event.type === "response.refusal.done") {
3326
+ const entry = lookupOpenItem(event);
3327
+ if (entry?.item.type === "message" && entry.block.type === "text") {
3328
+ applyMessageTextDone(
3329
+ entry.item,
3330
+ entry.block,
3331
+ event.type === "response.output_text.done" ? event.text : event.refusal,
3332
+ stream,
3333
+ output,
3334
+ contentIndexOf(entry.block),
3335
+ event.type === "response.output_text.done" ? "output_text" : "refusal",
3336
+ );
3337
+ }
3212
3338
  } else if (event.type === "response.function_call_arguments.delta") {
3213
3339
  const entry = lookupOpenFunctionCallItem(event);
3214
3340
  if (entry?.item.type === "function_call" && entry.block.type === "toolCall") {
@@ -3216,8 +3342,9 @@ export async function processResponsesStream<TApi extends Api>(
3216
3342
  }
3217
3343
  } else if (event.type === "response.function_call_arguments.done") {
3218
3344
  const entry = lookupOpenFunctionCallItem(event);
3219
- if (entry?.item.type === "function_call" && entry.block.type === "toolCall") {
3220
- finalizeToolCallArgumentsDone(entry.block, event.arguments);
3345
+ const args = optionalResponsesText(event.arguments, "function call arguments");
3346
+ if (entry?.item.type === "function_call" && entry.block.type === "toolCall" && args !== undefined) {
3347
+ finalizeToolCallArgumentsDone(entry.block, args);
3221
3348
  entry.block[kStreamingArgumentsDone] = true;
3222
3349
  }
3223
3350
  } else if (event.type === "response.custom_tool_call_input.delta") {
@@ -3227,8 +3354,9 @@ export async function processResponsesStream<TApi extends Api>(
3227
3354
  }
3228
3355
  } else if (event.type === "response.custom_tool_call_input.done") {
3229
3356
  const entry = lookupOpenToolCallAlias(event, "custom_tool_call");
3230
- if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") {
3231
- finalizeCustomToolCallInputDone(entry.block, event.input);
3357
+ const input = optionalResponsesText(event.input, "custom tool input");
3358
+ if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall" && input !== undefined) {
3359
+ finalizeCustomToolCallInputDone(entry.block, input);
3232
3360
  entry.block[kStreamingArgumentsDone] = true;
3233
3361
  }
3234
3362
  } else if (event.type === "response.output_item.done") {
@@ -3273,7 +3401,7 @@ export async function processResponsesStream<TApi extends Api>(
3273
3401
  // `output_item.added` never arrived (lossy proxy) — synthesize the
3274
3402
  // block so the final message still carries the authoritative text.
3275
3403
  const synthesized: TextContent = { type: "text", text, textSignature };
3276
- output.content.push(synthesized);
3404
+ pushContentBlock(synthesized);
3277
3405
  contentIndex = output.content.length - 1;
3278
3406
  }
3279
3407
  stream.push({ type: "text_end", contentIndex, content: text, partial: output });
@@ -3306,7 +3434,7 @@ export async function processResponsesStream<TApi extends Api>(
3306
3434
  // `output_item.added` never arrived (lossy proxy) — synthesize the
3307
3435
  // block so the final message carries the call the consumer was told
3308
3436
  // completed (the agent loop executes tools from message.content).
3309
- output.content.push(toolCall);
3437
+ pushContentBlock(toolCall);
3310
3438
  contentIndex = output.content.length - 1;
3311
3439
  }
3312
3440
  closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id));
@@ -3327,14 +3455,19 @@ export async function processResponsesStream<TApi extends Api>(
3327
3455
  clearStreamingPartialJson(block);
3328
3456
  contentIndex = contentIndexOf(block);
3329
3457
  } else {
3330
- output.content.push(toolCall);
3458
+ pushContentBlock(toolCall);
3331
3459
  contentIndex = output.content.length - 1;
3332
3460
  }
3333
3461
  closeOpenItem(event.output_index, item.id, entry, item.call_id);
3334
3462
  stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
3335
3463
  } else if (item.type === "custom_tool_call") {
3336
3464
  const block = entry?.block.type === "toolCall" ? entry.block : undefined;
3337
- const rawInput = block?.[kStreamingPartialJson] ? block[kStreamingPartialJson] : (item.input ?? "");
3465
+ const rawInput =
3466
+ optionalResponsesText(item.input, "custom tool input") ??
3467
+ (block?.[kStreamingArgumentsDone]
3468
+ ? optionalResponsesText(block.arguments.input, "custom tool input")
3469
+ : block?.[kStreamingPartialJson]) ??
3470
+ "";
3338
3471
  const toolCall: ToolCall = {
3339
3472
  type: "toolCall",
3340
3473
  id: encodeResponsesToolCallId(item.call_id, item.id),
@@ -3350,7 +3483,7 @@ export async function processResponsesStream<TApi extends Api>(
3350
3483
  clearStreamingPartialJson(block);
3351
3484
  contentIndex = contentIndexOf(block);
3352
3485
  } else {
3353
- output.content.push(toolCall);
3486
+ pushContentBlock(toolCall);
3354
3487
  contentIndex = output.content.length - 1;
3355
3488
  }
3356
3489
  closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id));
@@ -3386,6 +3519,13 @@ export async function processResponsesStream<TApi extends Api>(
3386
3519
  const error = response?.error ?? (response as any)?.status_details?.error;
3387
3520
  const details = response?.incomplete_details;
3388
3521
  const statusDetailsReason = (response as any)?.status_details?.reason;
3522
+ // A rate-limit/overload body inside a terminal response envelope must
3523
+ // advance the fallback chain exactly like an HTTP-status 429 would
3524
+ // (body-error.ts). The whole envelope goes to the probe so a status or
3525
+ // code carried beside `error` is visible, not only the inner object.
3526
+ // Non-retryable failures keep their existing message.
3527
+ const inBand = details ? undefined : AIError.createInBandProviderError({ ...response, error });
3528
+ if (inBand) throw inBand;
3389
3529
  const message = error
3390
3530
  ? `${error.code || "unknown"}: ${error.message || "no message"}`
3391
3531
  : details?.reason
@@ -3428,6 +3568,12 @@ export async function processResponsesStream<TApi extends Api>(
3428
3568
  break;
3429
3569
  } else if (event.type === "error") {
3430
3570
  const err = (event as any).error ?? event;
3571
+ // An in-band rate-limit/overload `error` event advances the fallback chain
3572
+ // like an HTTP-status 429 (body-error.ts); the whole event is passed so
3573
+ // event-level status/code fields count too. Other codes keep the existing
3574
+ // `Error Code <code>: <message>` message that error tests pin on.
3575
+ const inBand = AIError.createInBandProviderError(event);
3576
+ if (inBand) throw inBand;
3431
3577
  const code = err.code ?? "unknown";
3432
3578
  const message = err.message ?? "no message";
3433
3579
  throw new AIError.ProviderResponseError(`Error Code ${code}: ${message}`, {
@@ -3438,6 +3584,8 @@ export async function processResponsesStream<TApi extends Api>(
3438
3584
  populateResponsesUsageFromResponse(output, event.response?.usage);
3439
3585
  const error = event.response?.error ?? (event.response as any)?.status_details?.error;
3440
3586
  const details = event.response?.incomplete_details;
3587
+ const inBand = details ? undefined : AIError.createInBandProviderError({ ...event.response, error });
3588
+ if (inBand) throw inBand;
3441
3589
  const message = error
3442
3590
  ? `${error.code || "unknown"}: ${error.message || "no message"}`
3443
3591
  : details?.reason