@oh-my-pi/pi-ai 18.2.0 → 18.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/CHANGELOG.md +53 -0
  2. package/README.md +2 -0
  3. package/dist/types/auth/sqlite-credential-store.d.ts +2 -1
  4. package/dist/types/auth-broker/remote-store.d.ts +17 -0
  5. package/dist/types/auth-gateway/index.d.ts +1 -0
  6. package/dist/types/auth-gateway/session-state.d.ts +118 -0
  7. package/dist/types/auth-storage.d.ts +17 -0
  8. package/dist/types/error/body-error.d.ts +15 -0
  9. package/dist/types/error/flags.d.ts +16 -0
  10. package/dist/types/error/index.d.ts +1 -0
  11. package/dist/types/index.d.ts +1 -0
  12. package/dist/types/oneshot-retry.d.ts +6 -0
  13. package/dist/types/provider-session-state.d.ts +46 -0
  14. package/dist/types/providers/amazon-bedrock.d.ts +3 -0
  15. package/dist/types/providers/aws-sigv4.d.ts +12 -0
  16. package/dist/types/providers/openai-codex/request-transformer.d.ts +27 -0
  17. package/dist/types/providers/openai-responses.d.ts +15 -0
  18. package/dist/types/providers/openai-shared.d.ts +20 -3
  19. package/dist/types/registry/oauth/perplexity.d.ts +1 -7
  20. package/dist/types/registry/oauth/types.d.ts +8 -0
  21. package/dist/types/stream.d.ts +2 -0
  22. package/dist/types/types.d.ts +3 -1
  23. package/dist/types/usage/openai-codex.d.ts +3 -1
  24. package/dist/types/usage.d.ts +11 -1
  25. package/dist/types/utils/block-symbols.d.ts +36 -0
  26. package/dist/types/utils/openai-http.d.ts +2 -0
  27. package/dist/types/utils/retry-after.d.ts +2 -0
  28. package/dist/types/utils/schema/wire.d.ts +4 -5
  29. package/dist/types/utils.d.ts +9 -0
  30. package/package.json +6 -6
  31. package/src/auth/sqlite-credential-store.ts +8 -33
  32. package/src/auth-broker/remote-store.ts +73 -8
  33. package/src/auth-broker/wire-schemas.ts +1 -0
  34. package/src/auth-gateway/index.ts +1 -0
  35. package/src/auth-gateway/server.ts +186 -74
  36. package/src/auth-gateway/session-state.ts +312 -0
  37. package/src/auth-storage.ts +146 -15
  38. package/src/error/body-error.ts +310 -0
  39. package/src/error/flags.ts +63 -13
  40. package/src/error/index.ts +1 -0
  41. package/src/error/retryable.ts +2 -0
  42. package/src/index.ts +1 -0
  43. package/src/oneshot-retry.ts +13 -3
  44. package/src/provider-session-state.ts +56 -0
  45. package/src/providers/amazon-bedrock.ts +20 -3
  46. package/src/providers/anthropic-messages-server.ts +104 -23
  47. package/src/providers/anthropic-signature.ts +5 -2
  48. package/src/providers/anthropic.ts +101 -15
  49. package/src/providers/aws-sigv4.ts +16 -5
  50. package/src/providers/cursor.ts +60 -10
  51. package/src/providers/devin.ts +82 -28
  52. package/src/providers/openai-chat-server.ts +4 -0
  53. package/src/providers/openai-codex/request-transformer.ts +36 -0
  54. package/src/providers/openai-codex-responses.ts +35 -12
  55. package/src/providers/openai-completions.ts +49 -12
  56. package/src/providers/openai-reasoning-fallback.ts +6 -6
  57. package/src/providers/openai-responses-server.ts +2 -1
  58. package/src/providers/openai-responses.ts +52 -4
  59. package/src/providers/openai-shared.ts +199 -51
  60. package/src/registry/oauth/perplexity.ts +94 -28
  61. package/src/registry/oauth/types.ts +9 -0
  62. package/src/stream.ts +23 -2
  63. package/src/types.ts +3 -0
  64. package/src/usage/claude.ts +33 -0
  65. package/src/usage/google-antigravity.ts +8 -2
  66. package/src/usage/openai-codex.ts +94 -11
  67. package/src/usage.ts +8 -1
  68. package/src/utils/block-symbols.ts +57 -0
  69. package/src/utils/http-inspector.ts +20 -0
  70. package/src/utils/openai-http.ts +39 -3
  71. package/src/utils/retry-after.ts +12 -0
  72. package/src/utils/schema/normalize.ts +3 -3
  73. package/src/utils/schema/stamps.ts +33 -45
  74. package/src/utils/schema/wire.ts +9 -7
  75. package/src/utils.ts +67 -22
@@ -1,6 +1,7 @@
1
1
  import { createHash } from "node:crypto";
2
2
  import * as fs from "node:fs/promises";
3
3
  import http2 from "node:http2";
4
+ import { isCursorMaxModeWireId } from "@oh-my-pi/pi-catalog/compat/collapse";
4
5
  import { classifyModel, collapseVariantId } from "@oh-my-pi/pi-catalog/compat/taxonomy";
5
6
  import type {
6
7
  ConversationStep,
@@ -4504,7 +4505,13 @@ export function processInteractionUpdate(
4504
4505
  let persisted: ToolResultMessage | undefined;
4505
4506
  let hostError: string | null = null;
4506
4507
  try {
4507
- persisted = state.onTodoSnapshot?.(snapshot, settled.id, error) ?? undefined;
4508
+ persisted =
4509
+ state.onTodoSnapshot?.(
4510
+ snapshot,
4511
+ settled.id,
4512
+ error,
4513
+ toolCall && selectTodoCalls(toolCall).read ? "read" : "update",
4514
+ ) ?? undefined;
4508
4515
  } catch (callbackError) {
4509
4516
  // A throwing host callback (e.g. session persistence failing on
4510
4517
  // disk error) must not leave the resolved block unpaired: the
@@ -5228,6 +5235,46 @@ function extractImages(content: (TextContent | ImageContent)[]) {
5228
5235
  );
5229
5236
  }
5230
5237
 
5238
+ /**
5239
+ * Resolve `max_mode` for the wire id a request actually routes to.
5240
+ *
5241
+ * `GetUsableModels` marks max-mode models per raw row and discovery copies that
5242
+ * onto `cursorMaxMode`, so on a row that puts its own id on the wire the marker
5243
+ * is the authority — Cursor serves the whole Opus `-fast` lane in max mode
5244
+ * (`claude-opus-4-8-high-fast` included) and leaves reasoning tiers such as
5245
+ * `claude-4.6-opus-max` out of it, neither of which the wire slug can tell.
5246
+ *
5247
+ * Collapsing a family ORs the members' markers onto the logical row, so there
5248
+ * `cursorMaxMode: true` only means *some* tier needs max mode; sending it for
5249
+ * every tier is the refused `-low` request of issue #9478. The members' own
5250
+ * markers survive per wire id in `cursorMaxModeRoutes`, so the routed id is
5251
+ * looked up there first.
5252
+ *
5253
+ * A row's own wire id still owns its marker even when it has effort routing
5254
+ * (for example a bare/thinking pair). Logical-only bundled rows and routes
5255
+ * discovery never advertised have no per-id marker; only those use the suffix.
5256
+ * A collapsed row whose `true` no route's suffix can explain keeps it for every
5257
+ * route: the marker came from a member the suffix rule cannot see.
5258
+ */
5259
+ function resolveCursorMaxMode(model: Model<"cursor-agent">, wireModelId: string): boolean {
5260
+ const discovered = model.cursorMaxModeRoutes?.[wireModelId];
5261
+ if (discovered !== undefined) return discovered;
5262
+ const routing = model.thinking?.effortRouting;
5263
+ if (routing === undefined || wireModelId === model.id) {
5264
+ return model.cursorMaxMode ?? isCursorMaxModeWireId(wireModelId);
5265
+ }
5266
+ let routesOwnId = routing.off === model.id;
5267
+ let hasInferredMaxRoute = typeof routing.off === "string" && isCursorMaxModeWireId(routing.off);
5268
+ for (const effort of THINKING_EFFORTS) {
5269
+ const target = routing[effort];
5270
+ if (target === model.id) routesOwnId = true;
5271
+ if (typeof target === "string" && isCursorMaxModeWireId(target)) hasInferredMaxRoute = true;
5272
+ }
5273
+ if (routesOwnId) return model.cursorMaxMode ?? isCursorMaxModeWireId(wireModelId);
5274
+ if (model.cursorMaxMode === true && !hasInferredMaxRoute) return true;
5275
+ return isCursorMaxModeWireId(wireModelId);
5276
+ }
5277
+
5231
5278
  /**
5232
5279
  * Resolve the Cursor Run wire model id and its parameter list.
5233
5280
  *
@@ -5255,9 +5302,11 @@ function resolveCursorWireModel(
5255
5302
  ): {
5256
5303
  modelId: string;
5257
5304
  parameters: RequestedModel_ModelParameterbytes[];
5305
+ maxMode: boolean;
5258
5306
  } {
5259
5307
  const wireModelId = requestModelId ?? model.requestModelId ?? model.id;
5260
- if (wireMode === "discovered") return { modelId: wireModelId, parameters: [] };
5308
+ const maxMode = resolveCursorMaxMode(model, wireModelId);
5309
+ if (wireMode === "discovered") return { modelId: wireModelId, parameters: [], maxMode };
5261
5310
  // `collapseVariantId` keeps the lane in the logical id (`-high-fast` →
5262
5311
  // base `-fast`) and decodes the KDL effort (`-none` → `off`).
5263
5312
  const collapsed = collapseVariantId("cursor", wireModelId);
@@ -5265,11 +5314,12 @@ function resolveCursorWireModel(
5265
5314
  const base = effort !== undefined ? collapsed.logicalId : undefined;
5266
5315
  if (effort !== undefined && base && classifyModel("cursor", base).class === "openai") {
5267
5316
  if (effort === "off") {
5268
- return { modelId: base, parameters: [] };
5317
+ return { modelId: base, parameters: [], maxMode };
5269
5318
  }
5270
5319
  if ((THINKING_EFFORTS as readonly string[]).includes(effort)) {
5271
5320
  return {
5272
5321
  modelId: base,
5322
+ maxMode,
5273
5323
  parameters: [
5274
5324
  create(RequestedModel_ModelParameterbytesSchema, { id: "reasoning", value: collapsed.effort }),
5275
5325
  ],
@@ -5283,9 +5333,10 @@ function resolveCursorWireModel(
5283
5333
  return {
5284
5334
  modelId: wireModelId,
5285
5335
  parameters: [create(RequestedModel_ModelParameterbytesSchema, { id: "fast", value: "false" })],
5336
+ maxMode,
5286
5337
  };
5287
5338
  }
5288
- return { modelId: wireModelId, parameters: [] };
5339
+ return { modelId: wireModelId, parameters: [], maxMode };
5289
5340
  }
5290
5341
 
5291
5342
  async function buildGrpcRequestForWireMode(
@@ -5395,12 +5446,11 @@ async function buildGrpcRequestForWireMode(
5395
5446
  turns,
5396
5447
  });
5397
5448
 
5398
- const { modelId: wireModelId, parameters: wireParameters } = resolveCursorWireModel(
5399
- model,
5400
- options?.wireModelId,
5401
- wireMode,
5402
- );
5403
- const cursorMaxMode = model.cursorMaxMode === true;
5449
+ const {
5450
+ modelId: wireModelId,
5451
+ parameters: wireParameters,
5452
+ maxMode: cursorMaxMode,
5453
+ } = resolveCursorWireModel(model, options?.wireModelId, wireMode);
5404
5454
  const modelDetails = create(ModelDetailsSchema, {
5405
5455
  modelId: wireModelId,
5406
5456
  displayModelId: model.id,
@@ -1,5 +1,6 @@
1
1
  import { gunzipSync, gzipSync } from "node:zlib";
2
2
 
3
+ import { classifyModel } from "@oh-my-pi/pi-catalog/compat/taxonomy";
3
4
  import {
4
5
  AssignModelRequestSchema,
5
6
  AssignModelResponseSchema,
@@ -29,8 +30,9 @@ import { create, fromBinary, toBinary } from "@oh-my-pi/pi-catalog/discovery/pro
29
30
  import { calculateCost } from "@oh-my-pi/pi-catalog/models";
30
31
  import { DEVIN_DEFAULT_BASE_URL, devinCliMetadata } from "@oh-my-pi/pi-catalog/wire/devin";
31
32
  import { decodeDevinUnaryMessage } from "@oh-my-pi/pi-catalog/wire/devin-proto";
32
- import { logger, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
33
+ import { isRecord, logger, parseStreamingJson, parseStreamingJsonThrottled, sanitizeText } from "@oh-my-pi/pi-utils";
33
34
  import * as AIError from "../error";
35
+
34
36
  import type {
35
37
  Api,
36
38
  AssistantMessage,
@@ -50,7 +52,7 @@ import { normalizeSystemPrompts } from "../utils";
50
52
  import { isDemotedThinking } from "../utils/block-symbols";
51
53
  import { deterministicUuid } from "../utils/deterministic-id";
52
54
  import { AssistantMessageEventStream } from "../utils/event-stream";
53
- import { toolWireSchema } from "../utils/schema/wire";
55
+ import { normalizeSchemaForGoogle, toolWireSchema } from "../utils/schema";
54
56
  import { transformMessages } from "./transform-messages";
55
57
 
56
58
  /** Base host for Codeium/Windsurf's Cascade chat API (Connect protocol over HTTP/1.1). */
@@ -89,6 +91,58 @@ const MAX_CONNECT_FRAME_PAYLOAD = 16 * 1024 * 1024;
89
91
  * to benefit from the existing context-overflow maintenance path.
90
92
  */
91
93
  const LARGE_HISTORY_RECOVERY_BYTES = 512 * 1024;
94
+ const MAX_DEVIN_ERROR_DETAIL_CHARS = 4096;
95
+ const HTML_ERROR_BODY_PATTERN = /^\s*(?:<!doctype\s+html\b|<html\b)/i;
96
+
97
+ /** Extract a bounded human error without leaking proxy HTML or binary protobuf. */
98
+ function devinErrorDetail(response: Response, payload: Uint8Array): string | undefined {
99
+ let text: string;
100
+ try {
101
+ text = new TextDecoder("utf-8", { fatal: true }).decode(payload).trim();
102
+ } catch {
103
+ return undefined;
104
+ }
105
+ if (response.headers.get("content-type")?.toLowerCase().includes("text/html")) return undefined;
106
+ try {
107
+ const decoded: unknown = JSON.parse(text);
108
+ if (isRecord(decoded)) {
109
+ const error = decoded.error;
110
+ if (isRecord(error) && typeof error.message === "string") text = error.message.trim();
111
+ else if (typeof error === "string") text = error.trim();
112
+ else if (typeof decoded.message === "string") text = decoded.message.trim();
113
+ }
114
+ } catch {}
115
+ // Validate after envelope extraction: JSON escapes (`\u001b`, `<html>` inside a
116
+ // message) materialize bytes the raw-source scan cannot see. Whitespace is
117
+ // collapsed first so benign CRLF does not trip the control detection; after
118
+ // that, any text `sanitizeText` would alter (C0/C1 controls, DEL, malformed
119
+ // Unicode) is untrustworthy diagnostics and suppresses to status-only.
120
+ const normalized = text.replace(/\s+/g, " ").trim();
121
+ if (normalized.length === 0 || HTML_ERROR_BODY_PATTERN.test(normalized) || sanitizeText(normalized) !== normalized) {
122
+ return undefined;
123
+ }
124
+ if (normalized.length <= MAX_DEVIN_ERROR_DETAIL_CHARS) return normalized;
125
+ // Truncate on a code-point boundary: slicing through a surrogate pair would
126
+ // re-introduce the malformed text the sanitizeText gate just ruled out.
127
+ const boundaryUnit = normalized.charCodeAt(MAX_DEVIN_ERROR_DETAIL_CHARS - 1);
128
+ const cut =
129
+ boundaryUnit >= 0xd800 && boundaryUnit <= 0xdbff
130
+ ? MAX_DEVIN_ERROR_DETAIL_CHARS - 1
131
+ : MAX_DEVIN_ERROR_DETAIL_CHARS;
132
+ return normalized.slice(0, cut);
133
+ }
134
+
135
+ function createDevinHttpError(operation: string, response: Response, payload: Uint8Array): AIError.DevinApiError {
136
+ const status = `${response.status}${response.statusText ? ` ${response.statusText}` : ""}`;
137
+ const detail = devinErrorDetail(response, payload);
138
+ return new AIError.DevinApiError(
139
+ `Devin ${operation} error ${status}${detail ? `: ${detail}` : ""}`,
140
+ response.status,
141
+ {
142
+ headers: response.headers,
143
+ },
144
+ );
145
+ }
92
146
 
93
147
  export const streamDevin: StreamFunction<"devin-agent"> = (
94
148
  model: Model<"devin-agent">,
@@ -210,11 +264,8 @@ export const streamDevin: StreamFunction<"devin-agent"> = (
210
264
  });
211
265
 
212
266
  if (!response.ok) {
213
- const text = await response.text();
214
- throw new AIError.DevinApiError(
215
- `Devin API error ${response.status} ${response.statusText}: ${text}`,
216
- response.status,
217
- );
267
+ const payload = new Uint8Array(await response.arrayBuffer());
268
+ throw createDevinHttpError("API", response, payload);
218
269
  }
219
270
  if (!response.body) {
220
271
  throw new AIError.ProviderResponseError("Devin API error: response body is empty", {
@@ -501,12 +552,7 @@ async function fetchDevinAuthMetadata(
501
552
  signal,
502
553
  });
503
554
  const payload = new Uint8Array(await response.arrayBuffer());
504
- if (!response.ok) {
505
- throw new AIError.DevinApiError(
506
- `Devin auth error ${response.status} ${response.statusText}: ${new TextDecoder().decode(payload)}`,
507
- response.status,
508
- );
509
- }
555
+ if (!response.ok) throw createDevinHttpError("auth", response, payload);
510
556
  const decoded = decodeDevinUnaryMessage(GetUserJwtResponseSchema, payload);
511
557
  if (!decoded?.userJwt) {
512
558
  throw new AIError.ProviderResponseError("Devin auth error: GetUserJwt returned an empty user JWT", {
@@ -548,12 +594,7 @@ async function assignDevinModel(
548
594
  signal,
549
595
  });
550
596
  const payload = new Uint8Array(await response.arrayBuffer());
551
- if (!response.ok) {
552
- throw new AIError.DevinApiError(
553
- `Devin AssignModel error ${response.status} ${response.statusText}: ${new TextDecoder().decode(payload)}`,
554
- response.status,
555
- );
556
- }
597
+ if (!response.ok) throw createDevinHttpError("AssignModel", response, payload);
557
598
  const assignment = decodeDevinUnaryMessage(AssignModelResponseSchema, payload)?.assignment;
558
599
  if (!assignment?.assignmentJwt || !assignment.modelUid) {
559
600
  throw new AIError.ProviderResponseError(
@@ -598,11 +639,31 @@ function buildDevinChatRequest(
598
639
  options?.stopSequences && options.stopSequences.length > 0
599
640
  ? [...DEVIN_DEFAULT_STOP_PATTERNS, ...options.stopSequences]
600
641
  : DEVIN_DEFAULT_STOP_PATTERNS;
642
+ const chatModelUid = assignment?.modelUid ?? options?.chatModelUid ?? model.requestModelId ?? model.id;
643
+ // Devin routes multiple provider families through one Cascade envelope. Its
644
+ // Gemini backend applies Google's tool-schema constraints and rejects JSON
645
+ // Schema type arrays (e.g. `["number", "null"]`) as an opaque internal
646
+ // `invalid_argument`; normalize both direct Gemini models and router-assigned
647
+ // enum-style UIDs (`MODEL_GOOGLE_GEMINI_*`) before serializing tools. The UID
648
+ // prefix is the server's own enum namespace, which classifyModel cannot parse.
649
+ const googleToolSchema =
650
+ classifyModel("devin", model.id, { lenient: true }).class === "gemini" ||
651
+ classifyModel("devin", chatModelUid, { lenient: true }).class === "gemini" ||
652
+ chatModelUid.startsWith("MODEL_GOOGLE_GEMINI_");
653
+ const tools = (context.tools ?? []).map((tool: Tool) => {
654
+ const schema = toolWireSchema(tool);
655
+ return create(ChatToolDefinitionSchema, {
656
+ name: tool.name,
657
+ description: tool.description,
658
+ jsonSchemaString: JSON.stringify(googleToolSchema ? normalizeSchemaForGoogle(schema) : schema),
659
+ strict: tool.strict ?? false,
660
+ });
661
+ });
601
662
  return create(GetChatMessageRequestSchema, {
602
663
  metadata: create(MetadataSchema, devinCliMetadata(turn.apiKey, turn.userJwt)),
603
664
  prompt: normalizeSystemPrompts(context.systemPrompt).join("\n\n"),
604
665
  chatMessagePrompts: buildChatMessagePrompts(turn.messages, turn.cascadeId, model),
605
- chatModelUid: assignment?.modelUid ?? options?.chatModelUid ?? model.requestModelId ?? model.id,
666
+ chatModelUid,
606
667
  ...(assignment ? { modelAssignmentJwt: assignment.assignmentJwt } : undefined),
607
668
  requestType: ChatMessageRequestType.CASCADE,
608
669
  plannerMode: ConversationalPlannerMode.DEFAULT,
@@ -622,14 +683,7 @@ function buildDevinChatRequest(
622
683
  stopPatterns,
623
684
  fimEotProbThreshold: 1,
624
685
  }),
625
- tools: (context.tools ?? []).map((tool: Tool) =>
626
- create(ChatToolDefinitionSchema, {
627
- name: tool.name,
628
- description: tool.description,
629
- jsonSchemaString: JSON.stringify(toolWireSchema(tool)),
630
- strict: tool.strict ?? false,
631
- }),
632
- ),
686
+ tools,
633
687
  });
634
688
  }
635
689
 
@@ -30,6 +30,7 @@ import {
30
30
  openaiChatRequestSchema,
31
31
  } from "./openai-chat-server-schema";
32
32
  import { decodeDataUri } from "./openai-data-uri";
33
+ import { coerceNullMessageContentInPlace } from "./openai-shared";
33
34
 
34
35
  export type { ParsedRequest };
35
36
 
@@ -90,6 +91,9 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
90
91
  // for `resolvePromptCacheKey` to pull a cache identity out of inbound
91
92
  // vendor-neutral headers when the body doesn't carry one.
92
93
  rejectUnsupportedExplicitPromptCacheFields(body);
94
+ const request =
95
+ typeof body === "object" && body !== null && !Array.isArray(body) ? (body as Record<string, unknown>) : undefined;
96
+ coerceNullMessageContentInPlace(request?.messages, message => message.role !== "function");
93
97
  const parsed = openaiChatRequestSchema(body);
94
98
  if (parsed instanceof type.errors) {
95
99
  throw new AIError.ValidationError(`openai-chat: ${parsed.summary}`);
@@ -237,6 +237,40 @@ function toolOutputKind(type: unknown): ToolCallKind | undefined {
237
237
  * tool-result child is dropped from the reconstructed history) or when a turn
238
238
  * is aborted/crashes after the call streamed but before its result persisted.
239
239
  */
240
+
241
+ /**
242
+ * Sanitize an OpenAI Responses/Codex tool call ID to <= 64 characters and valid charset.
243
+ * Composite IDs with '|' or '\n' have their secondary/item part stripped.
244
+ * Hashing is anchored on the canonical base part so assistant and result composites
245
+ * with different item halves stay identical. Short lossy changes include a hash suffix
246
+ * to preserve collision resistance across distinct IDs.
247
+ */
248
+ export function sanitizeCodexCallId(rawCallId: string): string {
249
+ if (!rawCallId) return `call_${Bun.hash("empty").toString(36)}`;
250
+ const sep = rawCallId.search(/[\n|]/);
251
+ const base = sep > 0 ? rawCallId.slice(0, sep) : sep === 0 ? rawCallId.slice(1) : rawCallId;
252
+ const sanitized = base.replace(/[^a-zA-Z0-9_-]/g, "_").replace(/_+$/, "");
253
+ if (sanitized.length > 0 && sanitized.length <= 64 && sanitized === base) {
254
+ return sanitized;
255
+ }
256
+ const hash = Bun.hash(base || rawCallId).toString(36);
257
+ const effectiveBase = sanitized.length > 0 ? sanitized : "call";
258
+ const prefixLen = Math.max(0, 63 - hash.length);
259
+ return `${effectiveBase.slice(0, prefixLen)}_${hash}`.slice(0, 64);
260
+ }
261
+
262
+ /**
263
+ * In-place mutates the `call_id` property on every input item in the array to conform
264
+ * to the OpenAI Responses/Codex 64-character limit and valid charset constraints.
265
+ */
266
+ export function sanitizeInputCallIds(input: InputItem[]): void {
267
+ for (const item of input) {
268
+ if (typeof item.call_id === "string") {
269
+ item.call_id = sanitizeCodexCallId(item.call_id);
270
+ }
271
+ }
272
+ }
273
+
240
274
  function repairToolCallPairs(input: InputItem[]): InputItem[] {
241
275
  const callKinds = new Map<string, ToolCallKind>();
242
276
  const outputKinds = new Map<string, ToolCallKind>();
@@ -332,6 +366,7 @@ export interface CodexLiteShapedBody {
332
366
  export function applyCodexResponsesLiteShape(body: CodexLiteShapedBody): void {
333
367
  const input = Array.isArray(body.input) ? body.input : [];
334
368
  stripImageDetails(input);
369
+ sanitizeInputCallIds(input as InputItem[]);
335
370
  body.parallel_tool_calls = false;
336
371
  const declaredTools = Array.isArray(body.tools) ? body.tools : [];
337
372
  let additionalTools = declaredTools;
@@ -382,6 +417,7 @@ export async function transformRequestBody(
382
417
  if (body.input && Array.isArray(body.input)) {
383
418
  body.input = filterInput(body.input);
384
419
  if (body.input) {
420
+ sanitizeInputCallIds(body.input);
385
421
  body.input = repairToolCallPairs(body.input);
386
422
  }
387
423
  }
@@ -78,6 +78,7 @@ import {
78
78
  type ReasoningConfig,
79
79
  type RequestBody,
80
80
  resolveCodexResponsesLite,
81
+ sanitizeCodexCallId,
81
82
  transformRequestBody,
82
83
  } from "./openai-codex/request-transformer";
83
84
  import { CodexApiError } from "./openai-codex/response-handler";
@@ -811,6 +812,7 @@ interface CodexOpenItem {
811
812
  contentIndex: number;
812
813
  itemId?: string;
813
814
  outputIndex?: number;
815
+ nativeOutputItem?: Record<string, unknown>;
814
816
  }
815
817
 
816
818
  class CodexStreamRuntime {
@@ -840,6 +842,7 @@ class CodexStreamRuntime {
840
842
  currentItem: CodexEventItem | null = null;
841
843
  currentBlock: CodexOutputBlock | null = null;
842
844
  nativeOutputItems: Array<Record<string, unknown>> = [];
845
+ nativeOutputEntries: CodexOpenItem[] = [];
843
846
  /** Sequential-cutoff summary sections/emitted text, global to the response (indices span reasoning items). */
844
847
  cutoffSummaries: SequentialCutoffSummaryState = createSequentialCutoffSummaryState();
845
848
  /** Summary deltas buffered while waiting to see whether atomic `.done` events arrive. */
@@ -876,10 +879,23 @@ class CodexStreamRuntime {
876
879
  this.currentItem = null;
877
880
  this.currentBlock = null;
878
881
  this.nativeOutputItems.length = 0;
882
+ this.nativeOutputEntries.length = 0;
879
883
  this.pendingSummaryDeltas.clear();
880
884
  this.cutoffSummaries = createSequentialCutoffSummaryState();
881
885
  }
882
886
 
887
+ finalizeNativeOutputItems(): Array<Record<string, unknown>> {
888
+ if (this.nativeOutputEntries.length === 0) return this.nativeOutputItems;
889
+ const ordered: Array<Record<string, unknown>> = [];
890
+ for (const entry of this.nativeOutputEntries) {
891
+ if (entry.nativeOutputItem) ordered.push(entry.nativeOutputItem);
892
+ }
893
+ ordered.push(...this.nativeOutputItems);
894
+ this.nativeOutputEntries.length = 0;
895
+ this.nativeOutputItems = ordered;
896
+ return ordered;
897
+ }
898
+
883
899
  /**
884
900
  * Look up the open item a Codex stream event targets. `item_id` wins because it
885
901
  * uniquely identifies a response item; `output_index` covers idless function
@@ -2195,6 +2211,7 @@ class CodexStreamProcessor {
2195
2211
  ? Math.trunc(rawEvent.output_index)
2196
2212
  : undefined;
2197
2213
  const entry: CodexOpenItem = { item, block: this.runtime.currentBlock, contentIndex, itemId, outputIndex };
2214
+ this.runtime.nativeOutputEntries.push(entry);
2198
2215
  this.runtime.currentEntry = entry;
2199
2216
  if (itemId) this.runtime.openItems.set(itemId, entry);
2200
2217
  if (outputIndex !== undefined) this.runtime.openItemsByOutputIndex.set(outputIndex, entry);
@@ -2401,7 +2418,6 @@ class CodexStreamProcessor {
2401
2418
  if (!rawItem || typeof rawItem !== "object") return;
2402
2419
  const item = structuredCloneJSON(rawItem) as CodexEventItem;
2403
2420
  if (item.type === "image_generation_call" && item.result) item.status = "completed";
2404
- runtime.nativeOutputItems.push(item as unknown as Record<string, unknown>);
2405
2421
 
2406
2422
  // Match the finalization to the OPEN ITEM that started this block, not the
2407
2423
  // singleton current — interleaved items can finish out of order, so the
@@ -2410,6 +2426,9 @@ class CodexStreamProcessor {
2410
2426
  // routes `output_item.done` to the block that received `output_item.added`.
2411
2427
  const itemId = "id" in item && typeof item.id === "string" ? item.id : "";
2412
2428
  const entry = (itemId ? runtime.openItems.get(itemId) : null) ?? runtime.openItemForEvent(rawEvent);
2429
+ const nativeOutputItem = item as unknown as Record<string, unknown>;
2430
+ if (entry) entry.nativeOutputItem = nativeOutputItem;
2431
+ else runtime.nativeOutputItems.push(nativeOutputItem);
2413
2432
  const block = entry?.block ?? null;
2414
2433
  const contentIndex = entry?.contentIndex ?? output.content.length - 1;
2415
2434
 
@@ -2543,16 +2562,19 @@ class CodexStreamProcessor {
2543
2562
  resetCodexWebSocketAppendState(state);
2544
2563
  } else {
2545
2564
  state.lastRequest = structuredCloneJSON(runtime.requestBodyForState);
2565
+ const nativeOutputItems = runtime.finalizeNativeOutputItems();
2546
2566
  const replayableResponseItems = sanitizeOpenAIResponsesAssistantHistoryItemsForReplay(
2547
- structuredCloneJSON(runtime.nativeOutputItems),
2567
+ structuredCloneJSON(nativeOutputItems),
2548
2568
  );
2549
- if (responseId && replayableResponseItems) {
2569
+ if (responseId && replayableResponseItems && replayableResponseItems.length === nativeOutputItems.length) {
2550
2570
  state.lastResponseId = responseId;
2551
2571
  state.lastResponseItems = replayableResponseItems;
2552
2572
  state.canAppend = rawEvent.type === "response.done" || rawEvent.type === "response.completed";
2553
2573
  } else {
2554
- // Without both a response id and replayable output, the append baseline cannot be trusted.
2555
- state.canAppend = false;
2574
+ // No response id, or replay sanitization dropped an item the server
2575
+ // still holds. Sanitization is 1:1-or-fewer, so either case makes the
2576
+ // append baseline untrustworthy; next turn must replay in full.
2577
+ resetCodexWebSocketAppendState(state);
2556
2578
  }
2557
2579
  }
2558
2580
  }
@@ -2967,7 +2989,10 @@ class CodexStreamProcessor {
2967
2989
  throw new CodexProviderStreamError("Codex response failed", false);
2968
2990
  }
2969
2991
 
2970
- output.providerPayload = createOpenAIResponsesHistoryPayload(this.model.provider, this.runtime.nativeOutputItems);
2992
+ output.providerPayload = createOpenAIResponsesHistoryPayload(
2993
+ this.model.provider,
2994
+ this.runtime.finalizeNativeOutputItems(),
2995
+ );
2971
2996
  output.duration = performance.now() - this.startTime;
2972
2997
  if (completion.firstTokenTime) {
2973
2998
  output.ttft = completion.firstTokenTime - this.startTime;
@@ -4528,16 +4553,14 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
4528
4553
  const messages: ResponseInput = [];
4529
4554
 
4530
4555
  const normalizeToolCallId = (id: string): string => {
4531
- if (!id.includes("|")) return id;
4532
- const [callId, itemId] = id.split("|");
4533
- const sanitizedCallId = callId.replace(/[^a-zA-Z0-9_-]/g, "_");
4534
- let sanitizedItemId = itemId.replace(/[^a-zA-Z0-9_-]/g, "_");
4556
+ const sep = id.search(/[\n|]/);
4557
+ const [callId, itemId] = sep > 0 ? [id.slice(0, sep), id.slice(sep + 1)] : [id, undefined];
4558
+ const normalizedCallId = sanitizeCodexCallId(callId);
4559
+ let sanitizedItemId = (itemId ?? Bun.hash(id).toString(36)).replace(/[^a-zA-Z0-9_-]/g, "_");
4535
4560
  if (!sanitizedItemId.startsWith("fc")) {
4536
4561
  sanitizedItemId = `fc_${sanitizedItemId}`;
4537
4562
  }
4538
- let normalizedCallId = sanitizedCallId.length > 64 ? sanitizedCallId.slice(0, 64) : sanitizedCallId;
4539
4563
  let normalizedItemId = sanitizedItemId.length > 64 ? sanitizedItemId.slice(0, 64) : sanitizedItemId;
4540
- normalizedCallId = normalizedCallId.replace(/_+$/, "");
4541
4564
  normalizedItemId = normalizedItemId.replace(/_+$/, "");
4542
4565
  return `${normalizedCallId}|${normalizedItemId}`;
4543
4566
  };
@@ -677,7 +677,7 @@ const streamOpenAICompletionsOnce = (
677
677
  (async () => {
678
678
  const startTime = performance.now();
679
679
  let firstTokenTime: number | undefined;
680
- const policy = resolveOpenAICompatForRequest(model, options);
680
+ const policy = resolveOpenAICompatForRequest(model, options, Boolean(context.tools?.length));
681
681
 
682
682
  const output: AssistantMessage = createInitialResponsesAssistantMessage(model.api, model.provider, model.id);
683
683
  let rawRequestDump: RawHttpRequestDump | undefined;
@@ -765,14 +765,17 @@ const streamOpenAICompletionsOnce = (
765
765
  const builtParams = buildParams(model, context, options, effectiveToolStrictModeOverride);
766
766
  appliedStrictTools = builtParams.strictToolsApplied;
767
767
  let params = builtParams.params;
768
- const reasoningEffortFallbackKey = createOpenAIReasoningEffortFallbackKey(
769
- "chat-completions",
770
- trimmedBaseUrl,
771
- params.model,
772
- );
773
- const requestReasoningEffortFallback = requestReasoningEffortFallbacks.has(reasoningEffortFallbackKey)
774
- ? requestReasoningEffortFallbacks.get(reasoningEffortFallbackKey)
775
- : getOpenAIReasoningEffortFallback(providerSessionState, reasoningEffortFallbackKey);
768
+ // Tool-triggered suppression is a hard wire constraint; cached
769
+ // enabled-effort negotiation must not overwrite its `none`.
770
+ const reasoningEffortFallbackKey = builtParams.reasoningEffortFallbackAllowed
771
+ ? createOpenAIReasoningEffortFallbackKey("chat-completions", trimmedBaseUrl, params.model)
772
+ : undefined;
773
+ const requestReasoningEffortFallback =
774
+ reasoningEffortFallbackKey === undefined
775
+ ? undefined
776
+ : requestReasoningEffortFallbacks.has(reasoningEffortFallbackKey)
777
+ ? requestReasoningEffortFallbacks.get(reasoningEffortFallbackKey)
778
+ : getOpenAIReasoningEffortFallback(providerSessionState, reasoningEffortFallbackKey);
776
779
  if (requestReasoningEffortFallback !== undefined) {
777
780
  applyOpenAIReasoningEffortFallback(params, requestReasoningEffortFallback);
778
781
  }
@@ -1166,6 +1169,21 @@ const streamOpenAICompletionsOnce = (
1166
1169
  });
1167
1170
  for await (const chunk of terminalAwareStream) {
1168
1171
  if (!chunk || typeof chunk !== "object") continue;
1172
+ // Rate-limit/overload bodies sent inside an HTTP 200 stream (Azure,
1173
+ // LiteLLM-style aggregators, some gates) arrive as an `error` member or
1174
+ // a bare `{ code, status }` chunk. This probe runs first: the legacy
1175
+ // stream-error guard below turns *any* object `error` member into a
1176
+ // statusless `ProviderResponseError`, so if it went first no throttle
1177
+ // envelope would ever reach the in-band classifier.
1178
+ //
1179
+ // Invariants (body-error.ts): the status is read only from error
1180
+ // `status`/`code` fields and restricted to 429/5xx — never derived from
1181
+ // prose, so a body mentioning 401/403 stays out of the auth lane — and a
1182
+ // synthesized message is never opaque, so an unreadable body cannot burn
1183
+ // a credential. Envelopes that are not a recognised throttle return
1184
+ // `undefined` and keep their pre-existing handling.
1185
+ const inBand = AIError.createInBandProviderError(chunk);
1186
+ if (inBand) throw inBand;
1169
1187
  const streamError = createOpenAICompletionsStreamError(chunk, model.provider);
1170
1188
  if (streamError) throw streamError;
1171
1189
 
@@ -1259,6 +1277,12 @@ const streamOpenAICompletionsOnce = (
1259
1277
 
1260
1278
  if (choice?.delta?.tool_calls && choice.delta.tool_calls.length > 0) {
1261
1279
  const toolCalls = choice.delta.tool_calls;
1280
+ // Pure tool-call responses never emit a text/thinking delta, so
1281
+ // without this stamp TTFT stays undefined for every turn that
1282
+ // begins with a structured call (measured: all toolUse-stop
1283
+ // rows on OpenAI-compatible gateways) and the usage row's
1284
+ // TTFT/tok/s figures silently degrade.
1285
+ if (!firstTokenTime) firstTokenTime = performance.now();
1262
1286
  for (let toolCallOffset = 0; toolCallOffset < toolCalls.length; toolCallOffset++) {
1263
1287
  const toolCall = toolCalls[toolCallOffset]!;
1264
1288
  const streamIndex = typeof toolCall.index === "number" ? toolCall.index : undefined;
@@ -1570,12 +1594,14 @@ function createRequestSetup(
1570
1594
  function resolveOpenAICompatForRequest(
1571
1595
  model: Model<"openai-completions">,
1572
1596
  options: OpenAICompletionsOptions | undefined,
1597
+ hasTools: boolean,
1573
1598
  ): OpenAICompatPolicy {
1574
1599
  return resolveOpenAICompatPolicy(model, {
1575
1600
  endpoint: "chat-completions",
1576
1601
  reasoning: options?.reasoning,
1577
1602
  disableReasoning: options?.disableReasoning,
1578
1603
  toolChoice: mapToOpenAICompletionsToolChoice(options?.toolChoice),
1604
+ hasTools,
1579
1605
  });
1580
1606
  }
1581
1607
 
@@ -1678,8 +1704,9 @@ function buildParams(
1678
1704
  params: OpenAICompletionsParams;
1679
1705
  toolStrictMode: AppliedToolStrictMode;
1680
1706
  strictToolsApplied: boolean;
1707
+ reasoningEffortFallbackAllowed: boolean;
1681
1708
  } {
1682
- const initialPolicy = resolveOpenAICompatForRequest(model, options);
1709
+ const initialPolicy = resolveOpenAICompatForRequest(model, options, Boolean(context.tools?.length));
1683
1710
  const initialCompat = initialPolicy.compat as ResolvedOpenAICompat;
1684
1711
  const cacheRetention = resolveCacheRetention(options?.cacheRetention);
1685
1712
 
@@ -1833,6 +1860,7 @@ function buildParams(
1833
1860
  reasoning: options?.reasoning,
1834
1861
  disableReasoning: options?.disableReasoning,
1835
1862
  toolChoice: params.tool_choice,
1863
+ hasTools: Array.isArray(params.tools) && params.tools.length > 0,
1836
1864
  });
1837
1865
  const compat = finalPolicy.compat as ResolvedOpenAICompat;
1838
1866
  const messages = convertMessages(model, context, compat);
@@ -1857,7 +1885,11 @@ function buildParams(
1857
1885
  }
1858
1886
  applyChatCompletionsToolStream(params, model, compat);
1859
1887
 
1860
- applyChatCompletionsReasoningParams(params, model, compat, { ...options, toolChoice: params.tool_choice });
1888
+ applyChatCompletionsReasoningParams(params, model, compat, {
1889
+ ...options,
1890
+ toolChoice: params.tool_choice,
1891
+ hasTools: Array.isArray(params.tools) && params.tools.length > 0,
1892
+ });
1861
1893
  dropOpenRouterKimiForcedToolReasoning(params, model, finalPolicy);
1862
1894
 
1863
1895
  applyOpenAIGatewayRouting(params, compat, cacheRetention !== "none");
@@ -1867,7 +1899,12 @@ function buildParams(
1867
1899
  });
1868
1900
  applyOpenAIChatCompletionsPromptCachePolicy(params, model, options);
1869
1901
 
1870
- return { params, toolStrictMode, strictToolsApplied };
1902
+ return {
1903
+ params,
1904
+ toolStrictMode,
1905
+ strictToolsApplied,
1906
+ reasoningEffortFallbackAllowed: finalPolicy.reasoning.disableReason !== "tools",
1907
+ };
1871
1908
  }
1872
1909
 
1873
1910
  export function parseChunkUsage(