@oh-my-pi/pi-ai 18.2.0 → 18.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +35 -0
  2. package/dist/types/auth-broker/remote-store.d.ts +17 -0
  3. package/dist/types/auth-gateway/index.d.ts +1 -0
  4. package/dist/types/auth-gateway/session-state.d.ts +65 -0
  5. package/dist/types/auth-storage.d.ts +16 -0
  6. package/dist/types/error/body-error.d.ts +15 -0
  7. package/dist/types/error/flags.d.ts +16 -0
  8. package/dist/types/error/index.d.ts +1 -0
  9. package/dist/types/oneshot-retry.d.ts +6 -0
  10. package/dist/types/providers/openai-codex/request-transformer.d.ts +27 -0
  11. package/dist/types/providers/openai-shared.d.ts +20 -3
  12. package/dist/types/registry/oauth/perplexity.d.ts +1 -7
  13. package/dist/types/registry/oauth/types.d.ts +8 -0
  14. package/dist/types/stream.d.ts +2 -0
  15. package/dist/types/types.d.ts +3 -1
  16. package/dist/types/usage.d.ts +8 -0
  17. package/dist/types/utils/block-symbols.d.ts +36 -0
  18. package/dist/types/utils/openai-http.d.ts +2 -0
  19. package/dist/types/utils/retry-after.d.ts +2 -0
  20. package/dist/types/utils/schema/wire.d.ts +4 -5
  21. package/dist/types/utils.d.ts +9 -0
  22. package/package.json +6 -6
  23. package/src/auth-broker/remote-store.ts +73 -8
  24. package/src/auth-broker/wire-schemas.ts +1 -0
  25. package/src/auth-gateway/index.ts +1 -0
  26. package/src/auth-gateway/server.ts +48 -11
  27. package/src/auth-gateway/session-state.ts +114 -0
  28. package/src/auth-storage.ts +144 -13
  29. package/src/error/body-error.ts +310 -0
  30. package/src/error/flags.ts +63 -13
  31. package/src/error/index.ts +1 -0
  32. package/src/error/retryable.ts +2 -0
  33. package/src/oneshot-retry.ts +13 -3
  34. package/src/providers/anthropic-messages-server.ts +24 -3
  35. package/src/providers/anthropic.ts +101 -15
  36. package/src/providers/cursor.ts +7 -1
  37. package/src/providers/devin.ts +82 -28
  38. package/src/providers/openai-chat-server.ts +4 -0
  39. package/src/providers/openai-codex/request-transformer.ts +36 -0
  40. package/src/providers/openai-codex-responses.ts +35 -12
  41. package/src/providers/openai-completions.ts +43 -12
  42. package/src/providers/openai-reasoning-fallback.ts +6 -6
  43. package/src/providers/openai-responses-server.ts +2 -1
  44. package/src/providers/openai-responses.ts +25 -4
  45. package/src/providers/openai-shared.ts +199 -51
  46. package/src/registry/oauth/perplexity.ts +94 -28
  47. package/src/registry/oauth/types.ts +9 -0
  48. package/src/stream.ts +23 -2
  49. package/src/types.ts +3 -0
  50. package/src/usage/claude.ts +33 -0
  51. package/src/usage/google-antigravity.ts +8 -2
  52. package/src/usage.ts +3 -0
  53. package/src/utils/block-symbols.ts +57 -0
  54. package/src/utils/openai-http.ts +39 -3
  55. package/src/utils/retry-after.ts +12 -0
  56. package/src/utils/schema/normalize.ts +3 -3
  57. package/src/utils/schema/stamps.ts +33 -45
  58. package/src/utils/schema/wire.ts +9 -7
  59. package/src/utils.ts +67 -22
@@ -237,6 +237,40 @@ function toolOutputKind(type: unknown): ToolCallKind | undefined {
237
237
  * tool-result child is dropped from the reconstructed history) or when a turn
238
238
  * is aborted/crashes after the call streamed but before its result persisted.
239
239
  */
240
+
241
+ /**
242
+ * Sanitize an OpenAI Responses/Codex tool call ID to <= 64 characters and valid charset.
243
+ * Composite IDs with '|' or '\n' have their secondary/item part stripped.
244
+ * Hashing is anchored on the canonical base part so assistant and result composites
245
+ * with different item halves stay identical. Short lossy changes include a hash suffix
246
+ * to preserve collision resistance across distinct IDs.
247
+ */
248
+ export function sanitizeCodexCallId(rawCallId: string): string {
249
+ if (!rawCallId) return `call_${Bun.hash("empty").toString(36)}`;
250
+ const sep = rawCallId.search(/[\n|]/);
251
+ const base = sep > 0 ? rawCallId.slice(0, sep) : sep === 0 ? rawCallId.slice(1) : rawCallId;
252
+ const sanitized = base.replace(/[^a-zA-Z0-9_-]/g, "_").replace(/_+$/, "");
253
+ if (sanitized.length > 0 && sanitized.length <= 64 && sanitized === base) {
254
+ return sanitized;
255
+ }
256
+ const hash = Bun.hash(base || rawCallId).toString(36);
257
+ const effectiveBase = sanitized.length > 0 ? sanitized : "call";
258
+ const prefixLen = Math.max(0, 63 - hash.length);
259
+ return `${effectiveBase.slice(0, prefixLen)}_${hash}`.slice(0, 64);
260
+ }
261
+
262
+ /**
263
+ * In-place mutates the `call_id` property on every input item in the array to conform
264
+ * to the OpenAI Responses/Codex 64-character limit and valid charset constraints.
265
+ */
266
+ export function sanitizeInputCallIds(input: InputItem[]): void {
267
+ for (const item of input) {
268
+ if (typeof item.call_id === "string") {
269
+ item.call_id = sanitizeCodexCallId(item.call_id);
270
+ }
271
+ }
272
+ }
273
+
240
274
  function repairToolCallPairs(input: InputItem[]): InputItem[] {
241
275
  const callKinds = new Map<string, ToolCallKind>();
242
276
  const outputKinds = new Map<string, ToolCallKind>();
@@ -332,6 +366,7 @@ export interface CodexLiteShapedBody {
332
366
  export function applyCodexResponsesLiteShape(body: CodexLiteShapedBody): void {
333
367
  const input = Array.isArray(body.input) ? body.input : [];
334
368
  stripImageDetails(input);
369
+ sanitizeInputCallIds(input as InputItem[]);
335
370
  body.parallel_tool_calls = false;
336
371
  const declaredTools = Array.isArray(body.tools) ? body.tools : [];
337
372
  let additionalTools = declaredTools;
@@ -382,6 +417,7 @@ export async function transformRequestBody(
382
417
  if (body.input && Array.isArray(body.input)) {
383
418
  body.input = filterInput(body.input);
384
419
  if (body.input) {
420
+ sanitizeInputCallIds(body.input);
385
421
  body.input = repairToolCallPairs(body.input);
386
422
  }
387
423
  }
@@ -78,6 +78,7 @@ import {
78
78
  type ReasoningConfig,
79
79
  type RequestBody,
80
80
  resolveCodexResponsesLite,
81
+ sanitizeCodexCallId,
81
82
  transformRequestBody,
82
83
  } from "./openai-codex/request-transformer";
83
84
  import { CodexApiError } from "./openai-codex/response-handler";
@@ -811,6 +812,7 @@ interface CodexOpenItem {
811
812
  contentIndex: number;
812
813
  itemId?: string;
813
814
  outputIndex?: number;
815
+ nativeOutputItem?: Record<string, unknown>;
814
816
  }
815
817
 
816
818
  class CodexStreamRuntime {
@@ -840,6 +842,7 @@ class CodexStreamRuntime {
840
842
  currentItem: CodexEventItem | null = null;
841
843
  currentBlock: CodexOutputBlock | null = null;
842
844
  nativeOutputItems: Array<Record<string, unknown>> = [];
845
+ nativeOutputEntries: CodexOpenItem[] = [];
843
846
  /** Sequential-cutoff summary sections/emitted text, global to the response (indices span reasoning items). */
844
847
  cutoffSummaries: SequentialCutoffSummaryState = createSequentialCutoffSummaryState();
845
848
  /** Summary deltas buffered while waiting to see whether atomic `.done` events arrive. */
@@ -876,10 +879,23 @@ class CodexStreamRuntime {
876
879
  this.currentItem = null;
877
880
  this.currentBlock = null;
878
881
  this.nativeOutputItems.length = 0;
882
+ this.nativeOutputEntries.length = 0;
879
883
  this.pendingSummaryDeltas.clear();
880
884
  this.cutoffSummaries = createSequentialCutoffSummaryState();
881
885
  }
882
886
 
887
+ finalizeNativeOutputItems(): Array<Record<string, unknown>> {
888
+ if (this.nativeOutputEntries.length === 0) return this.nativeOutputItems;
889
+ const ordered: Array<Record<string, unknown>> = [];
890
+ for (const entry of this.nativeOutputEntries) {
891
+ if (entry.nativeOutputItem) ordered.push(entry.nativeOutputItem);
892
+ }
893
+ ordered.push(...this.nativeOutputItems);
894
+ this.nativeOutputEntries.length = 0;
895
+ this.nativeOutputItems = ordered;
896
+ return ordered;
897
+ }
898
+
883
899
  /**
884
900
  * Look up the open item a Codex stream event targets. `item_id` wins because it
885
901
  * uniquely identifies a response item; `output_index` covers idless function
@@ -2195,6 +2211,7 @@ class CodexStreamProcessor {
2195
2211
  ? Math.trunc(rawEvent.output_index)
2196
2212
  : undefined;
2197
2213
  const entry: CodexOpenItem = { item, block: this.runtime.currentBlock, contentIndex, itemId, outputIndex };
2214
+ this.runtime.nativeOutputEntries.push(entry);
2198
2215
  this.runtime.currentEntry = entry;
2199
2216
  if (itemId) this.runtime.openItems.set(itemId, entry);
2200
2217
  if (outputIndex !== undefined) this.runtime.openItemsByOutputIndex.set(outputIndex, entry);
@@ -2401,7 +2418,6 @@ class CodexStreamProcessor {
2401
2418
  if (!rawItem || typeof rawItem !== "object") return;
2402
2419
  const item = structuredCloneJSON(rawItem) as CodexEventItem;
2403
2420
  if (item.type === "image_generation_call" && item.result) item.status = "completed";
2404
- runtime.nativeOutputItems.push(item as unknown as Record<string, unknown>);
2405
2421
 
2406
2422
  // Match the finalization to the OPEN ITEM that started this block, not the
2407
2423
  // singleton current — interleaved items can finish out of order, so the
@@ -2410,6 +2426,9 @@ class CodexStreamProcessor {
2410
2426
  // routes `output_item.done` to the block that received `output_item.added`.
2411
2427
  const itemId = "id" in item && typeof item.id === "string" ? item.id : "";
2412
2428
  const entry = (itemId ? runtime.openItems.get(itemId) : null) ?? runtime.openItemForEvent(rawEvent);
2429
+ const nativeOutputItem = item as unknown as Record<string, unknown>;
2430
+ if (entry) entry.nativeOutputItem = nativeOutputItem;
2431
+ else runtime.nativeOutputItems.push(nativeOutputItem);
2413
2432
  const block = entry?.block ?? null;
2414
2433
  const contentIndex = entry?.contentIndex ?? output.content.length - 1;
2415
2434
 
@@ -2543,16 +2562,19 @@ class CodexStreamProcessor {
2543
2562
  resetCodexWebSocketAppendState(state);
2544
2563
  } else {
2545
2564
  state.lastRequest = structuredCloneJSON(runtime.requestBodyForState);
2565
+ const nativeOutputItems = runtime.finalizeNativeOutputItems();
2546
2566
  const replayableResponseItems = sanitizeOpenAIResponsesAssistantHistoryItemsForReplay(
2547
- structuredCloneJSON(runtime.nativeOutputItems),
2567
+ structuredCloneJSON(nativeOutputItems),
2548
2568
  );
2549
- if (responseId && replayableResponseItems) {
2569
+ if (responseId && replayableResponseItems && replayableResponseItems.length === nativeOutputItems.length) {
2550
2570
  state.lastResponseId = responseId;
2551
2571
  state.lastResponseItems = replayableResponseItems;
2552
2572
  state.canAppend = rawEvent.type === "response.done" || rawEvent.type === "response.completed";
2553
2573
  } else {
2554
- // Without both a response id and replayable output, the append baseline cannot be trusted.
2555
- state.canAppend = false;
2574
+ // No response id, or replay sanitization dropped an item the server
2575
+ // still holds. Sanitization is 1:1-or-fewer, so either case makes the
2576
+ // append baseline untrustworthy; next turn must replay in full.
2577
+ resetCodexWebSocketAppendState(state);
2556
2578
  }
2557
2579
  }
2558
2580
  }
@@ -2967,7 +2989,10 @@ class CodexStreamProcessor {
2967
2989
  throw new CodexProviderStreamError("Codex response failed", false);
2968
2990
  }
2969
2991
 
2970
- output.providerPayload = createOpenAIResponsesHistoryPayload(this.model.provider, this.runtime.nativeOutputItems);
2992
+ output.providerPayload = createOpenAIResponsesHistoryPayload(
2993
+ this.model.provider,
2994
+ this.runtime.finalizeNativeOutputItems(),
2995
+ );
2971
2996
  output.duration = performance.now() - this.startTime;
2972
2997
  if (completion.firstTokenTime) {
2973
2998
  output.ttft = completion.firstTokenTime - this.startTime;
@@ -4528,16 +4553,14 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
4528
4553
  const messages: ResponseInput = [];
4529
4554
 
4530
4555
  const normalizeToolCallId = (id: string): string => {
4531
- if (!id.includes("|")) return id;
4532
- const [callId, itemId] = id.split("|");
4533
- const sanitizedCallId = callId.replace(/[^a-zA-Z0-9_-]/g, "_");
4534
- let sanitizedItemId = itemId.replace(/[^a-zA-Z0-9_-]/g, "_");
4556
+ const sep = id.search(/[\n|]/);
4557
+ const [callId, itemId] = sep > 0 ? [id.slice(0, sep), id.slice(sep + 1)] : [id, undefined];
4558
+ const normalizedCallId = sanitizeCodexCallId(callId);
4559
+ let sanitizedItemId = (itemId ?? Bun.hash(id).toString(36)).replace(/[^a-zA-Z0-9_-]/g, "_");
4535
4560
  if (!sanitizedItemId.startsWith("fc")) {
4536
4561
  sanitizedItemId = `fc_${sanitizedItemId}`;
4537
4562
  }
4538
- let normalizedCallId = sanitizedCallId.length > 64 ? sanitizedCallId.slice(0, 64) : sanitizedCallId;
4539
4563
  let normalizedItemId = sanitizedItemId.length > 64 ? sanitizedItemId.slice(0, 64) : sanitizedItemId;
4540
- normalizedCallId = normalizedCallId.replace(/_+$/, "");
4541
4564
  normalizedItemId = normalizedItemId.replace(/_+$/, "");
4542
4565
  return `${normalizedCallId}|${normalizedItemId}`;
4543
4566
  };
@@ -677,7 +677,7 @@ const streamOpenAICompletionsOnce = (
677
677
  (async () => {
678
678
  const startTime = performance.now();
679
679
  let firstTokenTime: number | undefined;
680
- const policy = resolveOpenAICompatForRequest(model, options);
680
+ const policy = resolveOpenAICompatForRequest(model, options, Boolean(context.tools?.length));
681
681
 
682
682
  const output: AssistantMessage = createInitialResponsesAssistantMessage(model.api, model.provider, model.id);
683
683
  let rawRequestDump: RawHttpRequestDump | undefined;
@@ -765,14 +765,17 @@ const streamOpenAICompletionsOnce = (
765
765
  const builtParams = buildParams(model, context, options, effectiveToolStrictModeOverride);
766
766
  appliedStrictTools = builtParams.strictToolsApplied;
767
767
  let params = builtParams.params;
768
- const reasoningEffortFallbackKey = createOpenAIReasoningEffortFallbackKey(
769
- "chat-completions",
770
- trimmedBaseUrl,
771
- params.model,
772
- );
773
- const requestReasoningEffortFallback = requestReasoningEffortFallbacks.has(reasoningEffortFallbackKey)
774
- ? requestReasoningEffortFallbacks.get(reasoningEffortFallbackKey)
775
- : getOpenAIReasoningEffortFallback(providerSessionState, reasoningEffortFallbackKey);
768
+ // Tool-triggered suppression is a hard wire constraint; cached
769
+ // enabled-effort negotiation must not overwrite its `none`.
770
+ const reasoningEffortFallbackKey = builtParams.reasoningEffortFallbackAllowed
771
+ ? createOpenAIReasoningEffortFallbackKey("chat-completions", trimmedBaseUrl, params.model)
772
+ : undefined;
773
+ const requestReasoningEffortFallback =
774
+ reasoningEffortFallbackKey === undefined
775
+ ? undefined
776
+ : requestReasoningEffortFallbacks.has(reasoningEffortFallbackKey)
777
+ ? requestReasoningEffortFallbacks.get(reasoningEffortFallbackKey)
778
+ : getOpenAIReasoningEffortFallback(providerSessionState, reasoningEffortFallbackKey);
776
779
  if (requestReasoningEffortFallback !== undefined) {
777
780
  applyOpenAIReasoningEffortFallback(params, requestReasoningEffortFallback);
778
781
  }
@@ -1166,6 +1169,21 @@ const streamOpenAICompletionsOnce = (
1166
1169
  });
1167
1170
  for await (const chunk of terminalAwareStream) {
1168
1171
  if (!chunk || typeof chunk !== "object") continue;
1172
+ // Rate-limit/overload bodies sent inside an HTTP 200 stream (Azure,
1173
+ // LiteLLM-style aggregators, some gates) arrive as an `error` member or
1174
+ // a bare `{ code, status }` chunk. This probe runs first: the legacy
1175
+ // stream-error guard below turns *any* object `error` member into a
1176
+ // statusless `ProviderResponseError`, so if it went first no throttle
1177
+ // envelope would ever reach the in-band classifier.
1178
+ //
1179
+ // Invariants (body-error.ts): the status is read only from error
1180
+ // `status`/`code` fields and restricted to 429/5xx — never derived from
1181
+ // prose, so a body mentioning 401/403 stays out of the auth lane — and a
1182
+ // synthesized message is never opaque, so an unreadable body cannot burn
1183
+ // a credential. Envelopes that are not a recognised throttle return
1184
+ // `undefined` and keep their pre-existing handling.
1185
+ const inBand = AIError.createInBandProviderError(chunk);
1186
+ if (inBand) throw inBand;
1169
1187
  const streamError = createOpenAICompletionsStreamError(chunk, model.provider);
1170
1188
  if (streamError) throw streamError;
1171
1189
 
@@ -1570,12 +1588,14 @@ function createRequestSetup(
1570
1588
  function resolveOpenAICompatForRequest(
1571
1589
  model: Model<"openai-completions">,
1572
1590
  options: OpenAICompletionsOptions | undefined,
1591
+ hasTools: boolean,
1573
1592
  ): OpenAICompatPolicy {
1574
1593
  return resolveOpenAICompatPolicy(model, {
1575
1594
  endpoint: "chat-completions",
1576
1595
  reasoning: options?.reasoning,
1577
1596
  disableReasoning: options?.disableReasoning,
1578
1597
  toolChoice: mapToOpenAICompletionsToolChoice(options?.toolChoice),
1598
+ hasTools,
1579
1599
  });
1580
1600
  }
1581
1601
 
@@ -1678,8 +1698,9 @@ function buildParams(
1678
1698
  params: OpenAICompletionsParams;
1679
1699
  toolStrictMode: AppliedToolStrictMode;
1680
1700
  strictToolsApplied: boolean;
1701
+ reasoningEffortFallbackAllowed: boolean;
1681
1702
  } {
1682
- const initialPolicy = resolveOpenAICompatForRequest(model, options);
1703
+ const initialPolicy = resolveOpenAICompatForRequest(model, options, Boolean(context.tools?.length));
1683
1704
  const initialCompat = initialPolicy.compat as ResolvedOpenAICompat;
1684
1705
  const cacheRetention = resolveCacheRetention(options?.cacheRetention);
1685
1706
 
@@ -1833,6 +1854,7 @@ function buildParams(
1833
1854
  reasoning: options?.reasoning,
1834
1855
  disableReasoning: options?.disableReasoning,
1835
1856
  toolChoice: params.tool_choice,
1857
+ hasTools: Array.isArray(params.tools) && params.tools.length > 0,
1836
1858
  });
1837
1859
  const compat = finalPolicy.compat as ResolvedOpenAICompat;
1838
1860
  const messages = convertMessages(model, context, compat);
@@ -1857,7 +1879,11 @@ function buildParams(
1857
1879
  }
1858
1880
  applyChatCompletionsToolStream(params, model, compat);
1859
1881
 
1860
- applyChatCompletionsReasoningParams(params, model, compat, { ...options, toolChoice: params.tool_choice });
1882
+ applyChatCompletionsReasoningParams(params, model, compat, {
1883
+ ...options,
1884
+ toolChoice: params.tool_choice,
1885
+ hasTools: Array.isArray(params.tools) && params.tools.length > 0,
1886
+ });
1861
1887
  dropOpenRouterKimiForcedToolReasoning(params, model, finalPolicy);
1862
1888
 
1863
1889
  applyOpenAIGatewayRouting(params, compat, cacheRetention !== "none");
@@ -1867,7 +1893,12 @@ function buildParams(
1867
1893
  });
1868
1894
  applyOpenAIChatCompletionsPromptCachePolicy(params, model, options);
1869
1895
 
1870
- return { params, toolStrictMode, strictToolsApplied };
1896
+ return {
1897
+ params,
1898
+ toolStrictMode,
1899
+ strictToolsApplied,
1900
+ reasoningEffortFallbackAllowed: finalPolicy.reasoning.disableReason !== "tools",
1901
+ };
1871
1902
  }
1872
1903
 
1873
1904
  export function parseChunkUsage(
@@ -187,7 +187,7 @@ function collectMessageParts(error: unknown, captured: CapturedHttpErrorResponse
187
187
  * the same rejection with `Supported values are: …`, so value lists count too.
188
188
  */
189
189
  const REASONING_EFFORT_FIELD_PATTERN =
190
- /reasoning[_. ]effort|reasoning value|(?:valid|supported|allowed) (?:levels?|values?)/i;
190
+ /reasoning[_. ]?effort|reasoning value|(?:valid|supported|allowed) (?:levels?|values?)/i;
191
191
 
192
192
  function mentionsReasoningEffort(error: unknown, captured: CapturedHttpErrorResponse | undefined): boolean {
193
193
  const param = capturedStringField(captured, "param");
@@ -224,17 +224,17 @@ interface EffortRejectionSignal {
224
224
  rejectedMatches: boolean;
225
225
  }
226
226
 
227
- const EFFORT_FIELD_PATTERN = /reasoning[_. ]effort|reasoning value/i;
227
+ const EFFORT_FIELD_PATTERN = /reasoning[_. ]?effort|reasoning value/i;
228
228
  const ALLOWED_LEVELS_PATTERN = /(?:valid|supported|allowed) levels?/i;
229
229
 
230
230
  /** Fielded rejection verdicts in any word order: verdict-first, field-first, or bare mention plus verdict. */
231
231
  function messageCarriesEffortVerdict(message: string): boolean {
232
232
  return (
233
- /invalid[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(message) ||
234
- /(?:reasoning[_. ]effort|reasoning value)[^\n]*(?:invalid|unsupported|not supported|not permitted|must be|expected|unknown|unexpected|unrecognized)/i.test(
233
+ /invalid[^\n]*(?:reasoning[_. ]?effort|reasoning value)/i.test(message) ||
234
+ /(?:reasoning[_. ]?effort|reasoning value)[^\n]*(?:invalid|unsupported|not supported|not permitted|must be|expected|unknown|unexpected|unrecognized)/i.test(
235
235
  message,
236
236
  ) ||
237
- /(?:unsupported|not supported|not permitted|unknown|unexpected|unrecognized|extra)[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(
237
+ /(?:unsupported|not supported|not permitted|unknown|unexpected|unrecognized|extra)[^\n]*(?:reasoning[_. ]?effort|reasoning value)/i.test(
238
238
  message,
239
239
  )
240
240
  );
@@ -367,7 +367,7 @@ function nearestEnabledReasoningFallback(currentEffort: string, allowed: Set<str
367
367
  * supported`).
368
368
  */
369
369
  const TEMPLATE_KWARG_EFFORT_PATTERN =
370
- /chat_template_kwargs[^\n]{0,120}reasoning[_. ]effort|reasoning[_. ]effort[^\n]{0,120}chat_template_kwargs/i;
370
+ /chat_template_kwargs[^\n]{0,120}reasoning[_. ]?effort|reasoning[_. ]?effort[^\n]{0,120}chat_template_kwargs/i;
371
371
  const FIELD_REJECTION_PATTERN =
372
372
  /invalid|unsupported|not supported|not permitted|unknown|unexpected|unrecognized|rejected|extra input/i;
373
373
 
@@ -43,7 +43,7 @@ import {
43
43
  type OpenAIResponsesTool,
44
44
  openaiResponsesRequestSchema,
45
45
  } from "./openai-responses-server-schema";
46
- import { encodeTextSignatureV1, parseTextSignature } from "./openai-shared";
46
+ import { coerceNullMessageContentInPlace, encodeTextSignatureV1, parseTextSignature } from "./openai-shared";
47
47
 
48
48
  export type { ParsedRequest };
49
49
 
@@ -365,6 +365,7 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
365
365
  // `resolvePromptCacheKey` call further down.
366
366
 
367
367
  rejectUnsupportedExplicitPromptCacheFields(body);
368
+ coerceNullMessageContentInPlace(isObj(body) ? body.input : undefined);
368
369
  const data = openaiResponsesRequestSchema(body);
369
370
  if (data instanceof type.errors) {
370
371
  throw new AIError.ValidationError(`openai-responses: ${data.summary}`);
@@ -422,6 +422,7 @@ const streamOpenAIResponsesOnce = (
422
422
  let rawRequestDump: RawHttpRequestDump | undefined;
423
423
  let chainState: OpenAIResponsesChainState | undefined;
424
424
  let sentPreviousResponseId: string | undefined;
425
+ let lastSubmittedRequestWasFullReplay: boolean | undefined;
425
426
  const abortTracker = createAbortSourceTracker(options?.signal);
426
427
  const firstEventTimeoutAbortError = new AIError.StreamTimeoutError(OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE);
427
428
  const { requestAbortController, requestSignal } = abortTracker;
@@ -552,6 +553,7 @@ const streamOpenAIResponsesOnce = (
552
553
  typeof requestParams.model === "string" ? requestParams.model : model.id,
553
554
  );
554
555
  activeRequestParams = requestParams;
556
+ lastSubmittedRequestWasFullReplay = requestParams.previous_response_id === undefined;
555
557
  let requestTimeout: NodeJS.Timeout | undefined;
556
558
  if (requestTimeoutMs !== undefined) {
557
559
  requestTimeout = setTimeout(
@@ -576,6 +578,9 @@ const streamOpenAIResponsesOnce = (
576
578
  copilotCacheKey,
577
579
  copilotCacheSnapshot,
578
580
  ),
581
+ shouldRetryResponse: (response, bodyText) =>
582
+ !AIError.isRequestBodyReadTimeout(response.status, bodyText) ||
583
+ lastSubmittedRequestWasFullReplay !== true,
579
584
  // Transient 408/429/5xx get Retry-After-aware transport
580
585
  // retries; the first-event watchdog aborts `requestSignal`,
581
586
  // so retries cannot extend the caller's deadline.
@@ -878,7 +883,7 @@ const streamOpenAIResponsesOnce = (
878
883
  : activeParams,
879
884
  );
880
885
  chainState.lastPromptCacheBreakpointPolicy = promptCacheBreakpointPolicy;
881
- if (output.responseId) {
886
+ if (output.responseId && replayableResponseItems.length === nativeOutputItems.length) {
882
887
  chainState.lastResponseId = output.responseId;
883
888
  chainState.lastResponseItems = replayableResponseItems;
884
889
  chainState.canAppend = true;
@@ -886,8 +891,12 @@ const streamOpenAIResponsesOnce = (
886
891
  // full-context success must not mask categorical rejection.
887
892
  if (sentPreviousResponseId) chainState.staleFailures = 0;
888
893
  } else {
889
- // Without a response id the append baseline cannot be trusted.
894
+ // No response id, or replay sanitization dropped an item the server
895
+ // still holds. Sanitization is 1:1-or-fewer, so either case makes the
896
+ // append baseline untrustworthy; next turn must replay in full.
890
897
  chainState.canAppend = false;
898
+ chainState.lastResponseId = undefined;
899
+ chainState.lastResponseItems = undefined;
891
900
  }
892
901
  }
893
902
  } else if (chainState) {
@@ -926,6 +935,9 @@ const streamOpenAIResponsesOnce = (
926
935
  output.errorStatus = result.status;
927
936
  output.errorId = result.id;
928
937
  output.errorMessage = result.message;
938
+ if (AIError.isRequestBodyReadTimeout(result.status, result.message) && lastSubmittedRequestWasFullReplay) {
939
+ output.requestBodyReadTimeoutFullReplay = true;
940
+ }
929
941
  // Some providers via OpenRouter include extra details here.
930
942
  const rawMetadata = (error as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
931
943
  if (rawMetadata) output.errorMessage += `\n${rawMetadata}`;
@@ -1153,6 +1165,11 @@ export function buildParams(
1153
1165
  });
1154
1166
  const strictResponsesPairing = policy.tools.strictResponsesPairing;
1155
1167
  const shouldReplayNativeHistory = providerSessionState?.nativeHistoryReplayWarmed ?? true;
1168
+ // Filtering native reasoning must not be undone by reconstruction when the
1169
+ // target also rejects synthetic items (Muse on OpenRouter). Unfiltered targets
1170
+ // retain required text/placeholder replay, including DeepSeek's #10690 fallback.
1171
+ const canReconstructReasoningReplay =
1172
+ !policy.reasoning.filterReasoningHistory || policy.reasoning.allowsSyntheticReasoningContentForToolCalls;
1156
1173
  const messages = buildResponsesInput({
1157
1174
  model,
1158
1175
  context,
@@ -1164,9 +1181,13 @@ export function buildParams(
1164
1181
  },
1165
1182
  includeThinkingSignatures: shouldReplayNativeHistory && !policy.reasoning.filterReasoningHistory,
1166
1183
  requiresReasoningReplayForAllTurns:
1167
- policy.reasoning.enabled && policy.reasoning.requiresReasoningContentForAllAssistantTurns,
1184
+ policy.reasoning.enabled &&
1185
+ policy.reasoning.requiresReasoningContentForAllAssistantTurns &&
1186
+ canReconstructReasoningReplay,
1168
1187
  requiresReasoningReplayForToolCalls:
1169
- policy.reasoning.enabled && policy.reasoning.requiresReasoningContentForToolCalls,
1188
+ policy.reasoning.enabled &&
1189
+ policy.reasoning.requiresReasoningContentForToolCalls &&
1190
+ canReconstructReasoningReplay,
1170
1191
  repairOrphanOutputs: true,
1171
1192
  });
1172
1193