@gajae-code/ai 0.12.11 → 0.12.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -395,8 +395,20 @@ export function isAnthropicFastModeUnsupportedError(error: unknown): boolean {
395
395
  return false;
396
396
  }
397
397
 
398
+ /**
399
+ * Proxies (e.g. CLIProxyAPI) can deliver Anthropic's 400 body as an in-stream
400
+ * SSE `error` event on an HTTP 200 response; the thrown error then carries no
401
+ * HTTP status at all (issue #3900). Accept both the direct 400 and the
402
+ * statusless SSE shape — the strict `invalid_request_error` message checks in
403
+ * each matcher keep the statusless branch from claiming unrelated failures.
404
+ */
405
+ function isAnthropicInvalidRequestStatus(error: unknown): boolean {
406
+ const status = extractHttpStatusFromError(error);
407
+ return status === 400 || status === undefined;
408
+ }
409
+
398
410
  export function isAnthropicThinkingBlockMutationError(error: unknown): boolean {
399
- if (extractHttpStatusFromError(error) !== 400) return false;
411
+ if (!isAnthropicInvalidRequestStatus(error)) return false;
400
412
  const message = error instanceof Error ? error.message : String(error);
401
413
  return (
402
414
  /invalid_request_error/i.test(message) &&
@@ -414,7 +426,7 @@ export function isAnthropicThinkingBlockMutationError(error: unknown): boolean {
414
426
  * than only the latest one.
415
427
  */
416
428
  export function isAnthropicThinkingSignatureInvalidError(error: unknown): boolean {
417
- if (extractHttpStatusFromError(error) !== 400) return false;
429
+ if (!isAnthropicInvalidRequestStatus(error)) return false;
418
430
  const message = error instanceof Error ? error.message : String(error);
419
431
  return (
420
432
  /invalid_request_error/i.test(message) &&
@@ -423,6 +435,27 @@ export function isAnthropicThinkingSignatureInvalidError(error: unknown): boolea
423
435
  );
424
436
  }
425
437
 
438
+ /**
439
+ * CLIProxyAPI replaces Anthropic's rejection body wholesale instead of forwarding
440
+ * it: the client only ever sees
441
+ * `{"type":"error","error":{"type":"api_error","message":"An error occurred while
442
+ * processing the request."}}`, delivered as an in-stream SSE `error` event on an
443
+ * HTTP 200 response, so neither the status nor the message survives. Captured CPA
444
+ * traces for that masked shape carry the thinking-integrity 400 upstream (issue
445
+ * #3900), and the generic body matches no transient phrase either, so the turn
446
+ * dies unrecoverably. Nothing in the payload names the cause; callers must pair
447
+ * this with a request that actually replays signed thinking blocks before
448
+ * treating it as a thinking-replay rejection.
449
+ */
450
+ export function isAnthropicMaskedProxyRejection(error: unknown): boolean {
451
+ const status = extractHttpStatusFromError(error);
452
+ if (status !== undefined && status !== 400) return false;
453
+ const message = error instanceof Error ? error.message : String(error);
454
+ // A body that still names its error type is classified by the strict matchers.
455
+ if (/invalid_request_error/i.test(message)) return false;
456
+ return /"type"\s*:\s*"api_error"/.test(message) && /an error occurred while processing/i.test(message);
457
+ }
458
+
426
459
  function hasStrictAnthropicTools(params: MessageCreateParamsStreaming): boolean {
427
460
  const tools = params.tools as Array<{ strict?: unknown }> | undefined;
428
461
  return tools?.some(tool => tool.strict === true) ?? false;
@@ -447,12 +480,21 @@ function dropAnthropicStrictTools(params: MessageCreateParamsStreaming): void {
447
480
  }
448
481
  }
449
482
 
483
+ function isClaudeFamilyModel(model: Model<"anthropic-messages">): boolean {
484
+ // Classify the same identifier the request body serializes (`params.model =
485
+ // model.id` in buildParams); a differing `wireModelId` is not dispatched by
486
+ // this transport, so it must not drive the cache decision either.
487
+ const id = model.id;
488
+ const shortId = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
489
+ return shortId.toLowerCase().startsWith("claude-");
490
+ }
491
+
450
492
  function getCacheControl(
451
493
  model: Model<"anthropic-messages">,
452
494
  baseUrl: string,
453
495
  cacheRetention?: CacheRetention,
454
496
  ): { mode: AnthropicCacheMode; cacheControl?: AnthropicCacheControl } {
455
- const retention = resolveCacheRetention(cacheRetention, "long");
497
+ const retention = resolveCacheRetention(cacheRetention ?? model.cacheRetention, "long");
456
498
  if (retention === "none") return { mode: "none" };
457
499
 
458
500
  const isCanonicalApi = isAnthropicApiBaseUrl(baseUrl);
@@ -462,7 +504,7 @@ function getCacheControl(
462
504
  ? "none"
463
505
  : promptCacheMode === "explicit"
464
506
  ? "explicit"
465
- : isCanonicalApi
507
+ : promptCacheMode === "automatic" || isCanonicalApi || isClaudeFamilyModel(model)
466
508
  ? "automatic"
467
509
  : "none";
468
510
  if (mode === "none") return { mode };
@@ -1847,7 +1889,12 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1847
1889
  !options?.fallbackManaged &&
1848
1890
  !repairAllAssistantThinking &&
1849
1891
  firstTokenTime === undefined &&
1850
- (thinkingSignatureInvalid || isAnthropicThinkingBlockMutationError(streamFailure))
1892
+ (thinkingSignatureInvalid ||
1893
+ isAnthropicThinkingBlockMutationError(streamFailure) ||
1894
+ // Masked proxy rejection: unclassifiable on its own, so the replayed
1895
+ // request shape is the evidence. Without signed thinking blocks in
1896
+ // flight there is nothing to repair and the error must surface.
1897
+ (isAnthropicMaskedProxyRejection(streamFailure) && hasNativeThinkingBlocks(params.messages)))
1851
1898
  ) {
1852
1899
  // The mutation 400 blames the "latest assistant message", but its cited
1853
1900
  // `messages.N.content.M` path can point at an EARLIER replayed turn, so the
@@ -2285,10 +2332,11 @@ function applyExplicitPromptCaching(params: AnthropicCacheParams, cacheControl:
2285
2332
  const currentUser = params.messages[currentUserIndex];
2286
2333
  if (!currentUser) return;
2287
2334
 
2288
- // A tool result is encoded as role "user" on the wire, but belongs to the
2289
- // preceding assistant turn. Anchor that assistant turn, not the tool result,
2290
- // so changing tool output does not invalidate the reusable conversation prefix.
2291
- for (let index = currentUserIndex - 1; index >= 0; index--) {
2335
+ // Tool results are encoded as role "user" on the wire but belong to the
2336
+ // assistant tool-use turn immediately before them. Anchor the latest completed
2337
+ // assistant turn so the reusable prefix advances during an agent tool loop,
2338
+ // while keeping the newest tool output outside the cache boundary.
2339
+ for (let index = params.messages.length - 1; index >= 0; index--) {
2292
2340
  const message = params.messages[index];
2293
2341
  if (message?.role !== "assistant" || !Array.isArray(message.content)) continue;
2294
2342
  if (
@@ -1,5 +1,5 @@
1
1
  import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
2
- import { AzureOpenAI } from "openai";
2
+ import { APIConnectionTimeoutError, AzureOpenAI } from "openai";
3
3
  import type {
4
4
  Tool as OpenAITool,
5
5
  ResponseCreateParamsStreaming,
@@ -22,9 +22,11 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
22
22
  import { transportFailureFacts } from "../utils/fallback-transport";
23
23
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
24
24
  import {
25
+ FirstEventTimeoutError,
25
26
  getOpenAIStreamIdleTimeoutMs,
26
27
  getStreamFirstEventTimeoutMs,
27
28
  iterateWithIdleTimeout,
29
+ resolveOpenAISdkRequestTimeoutMs,
28
30
  } from "../utils/idle-iterator";
29
31
  import { resolveRetryBudget } from "../utils/retry-budget";
30
32
  import { flattenToolRootCombinators, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
@@ -44,6 +46,7 @@ import {
44
46
  convertResponsesAssistantMessage,
45
47
  convertResponsesInputContent,
46
48
  createInitialResponsesAssistantMessage,
49
+ isOpenAIResponsesProgressEvent,
47
50
  normalizeResponsesToolCallIdForTransform,
48
51
  processResponsesStream,
49
52
  } from "./openai-responses-shared";
@@ -108,6 +111,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
108
111
  (async () => {
109
112
  const startTime = Date.now();
110
113
  let firstTokenTime: number | undefined;
114
+ let streamConnected = false;
111
115
  const deploymentName = resolveDeploymentName(model, options);
112
116
 
113
117
  const output: AssistantMessage = createInitialResponsesAssistantMessage(
@@ -162,6 +166,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
162
166
  rawRequestDump = { ...rawRequestDump, body: params };
163
167
  openaiStream = await client.responses.create(params, { signal: requestSignal });
164
168
  }
169
+ streamConnected = true;
165
170
  const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
166
171
  stream.push({ type: "start", partial: output });
167
172
 
@@ -173,6 +178,8 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
173
178
  errorMessage: "Azure OpenAI responses stream stalled while waiting for the next event",
174
179
  onIdle: () => requestAbortController.abort(),
175
180
  onFirstItemTimeout: () => requestAbortController.abort(),
181
+ isProgressItem: isOpenAIResponsesProgressEvent,
182
+ abortSignal: options?.signal,
176
183
  }),
177
184
  output,
178
185
  stream,
@@ -203,10 +210,15 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
203
210
  } catch (error) {
204
211
  for (const block of output.content) delete (block as { index?: number }).index;
205
212
  const firstEventTimeoutError = abortTracker.getLocalAbortReason();
213
+ const normalizedError =
214
+ !streamConnected && error instanceof APIConnectionTimeoutError
215
+ ? new FirstEventTimeoutError(AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE)
216
+ : error;
206
217
  output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
207
- output.errorStatus = extractHttpStatusFromError(error);
208
- output.transportFailure = transportFailureFacts(error);
209
- output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
218
+ output.errorStatus = extractHttpStatusFromError(firstEventTimeoutError ?? normalizedError);
219
+ output.transportFailure = transportFailureFacts(firstEventTimeoutError ?? normalizedError);
220
+ output.errorMessage =
221
+ firstEventTimeoutError?.message ?? (await finalizeErrorMessage(normalizedError, rawRequestDump));
210
222
  output.duration = Date.now() - startTime;
211
223
  if (firstTokenTime) output.ttft = firstTokenTime - startTime;
212
224
  stream.push({ type: "error", reason: output.stopReason, error: output });
@@ -305,6 +317,9 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
305
317
 
306
318
  const baseFetch = wrapOpenAIFetchForBoundedRateLimits(options?.fetch ?? fetch, options?.maxRetryDelayMs);
307
319
  const onSseEvent = options?.onSseEvent;
320
+ // Bound HTTP request timeout to the first-event window so a stalled-before-headers
321
+ // fetch cannot wait the SDK's 10-minute default before the transport watchdog arms.
322
+ const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, options?.streamFirstEventTimeoutMs);
308
323
  return new AzureOpenAI({
309
324
  apiKey,
310
325
  apiVersion,
@@ -315,6 +330,7 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
315
330
  fetch: onSseEvent
316
331
  ? wrapFetchForSseDebug(baseFetch, event => onSseEvent(event, model, options?.attemptScope))
317
332
  : baseFetch,
333
+ ...(sdkTimeoutMs !== undefined ? { timeout: sdkTimeoutMs } : {}),
318
334
  });
319
335
  }
320
336
 
@@ -66,6 +66,7 @@ import {
66
66
  toolWireSchema,
67
67
  } from "../utils/schema";
68
68
  import {
69
+ isCodexStatuslessNamedToolChoiceNotFoundError,
69
70
  isForcedToolChoiceUnsupportedError,
70
71
  markToolChoiceIncapability,
71
72
  resolveToolChoice,
@@ -247,9 +248,7 @@ async function retryCodexInitialTransportWithoutToolChoice(
247
248
  requestBodyForState: RequestBody;
248
249
  transport: CodexTransport;
249
250
  }> {
250
- if (
251
- !isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(requestContext.transformedBody.tool_choice))
252
- ) {
251
+ if (!isCodexForcedToolChoiceUnsupportedError(error, requestContext.transformedBody)) {
253
252
  throw error;
254
253
  }
255
254
  const reason = await finalizeErrorMessage(error, requestContext.rawRequestDump);
@@ -555,10 +554,12 @@ function getCodexServiceTierCostMultiplier(
555
554
 
556
555
  function resolveCodexCostServiceTier(res: unknown, req?: unknown): ServiceTier | "default" | undefined {
557
556
  switch (res) {
557
+ case "auto":
558
+ case "default":
558
559
  case "flex":
559
- return "flex";
560
+ case "scale":
560
561
  case "priority":
561
- return "priority";
562
+ return res;
562
563
  default:
563
564
  if (req === "flex" || req === "priority") {
564
565
  return req;
@@ -1525,7 +1526,7 @@ async function tryRetryWithoutForcedToolChoice(
1525
1526
  context.output.content.length > 0 ||
1526
1527
  context.firstTokenTime !== undefined ||
1527
1528
  context.options?.signal?.aborted ||
1528
- !isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(runtime.requestBodyForState.tool_choice))
1529
+ !isCodexForcedToolChoiceUnsupportedError(error, runtime.requestBodyForState)
1529
1530
  ) {
1530
1531
  return false;
1531
1532
  }
@@ -1577,6 +1578,30 @@ async function tryRetryWithoutForcedToolChoice(
1577
1578
  function isForcedCodexToolChoice(choice: RequestBody["tool_choice"]): boolean {
1578
1579
  return !!choice && choice !== "none" && choice !== "auto";
1579
1580
  }
1581
+ function isCodexForcedToolChoiceUnsupportedError(error: unknown, body: RequestBody): boolean {
1582
+ if (isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(body.tool_choice))) {
1583
+ return true;
1584
+ }
1585
+ return isCodexStatuslessNamedToolChoiceNotFoundError(
1586
+ error,
1587
+ codexNamedFunctionToolChoiceName(body.tool_choice),
1588
+ codexSerializedToolNames(body.tools),
1589
+ );
1590
+ }
1591
+
1592
+ function codexNamedFunctionToolChoiceName(choice: RequestBody["tool_choice"]): string | undefined {
1593
+ if (!choice || typeof choice !== "object") return undefined;
1594
+ const namedChoice = choice as { type?: unknown; name?: unknown };
1595
+ return namedChoice.type === "function" && typeof namedChoice.name === "string" ? namedChoice.name : undefined;
1596
+ }
1597
+
1598
+ function codexSerializedToolNames(tools: RequestBody["tools"]): string[] {
1599
+ if (!Array.isArray(tools)) return [];
1600
+ return tools.flatMap(tool => {
1601
+ const name = (tool as { name?: unknown }).name;
1602
+ return typeof name === "string" ? [name] : [];
1603
+ });
1604
+ }
1580
1605
 
1581
1606
  /**
1582
1607
  * Handles `websocket_connection_limit_reached` errors by closing the stale connection
@@ -52,6 +52,7 @@ import {
52
52
  getProviderFirstEventTimeoutFallbackMs,
53
53
  getStreamFirstEventTimeoutMs,
54
54
  iterateWithIdleTimeout,
55
+ resolveOpenAISdkRequestTimeoutMs,
55
56
  } from "../utils/idle-iterator";
56
57
  import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
57
58
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
@@ -524,7 +525,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
524
525
 
525
526
  try {
526
527
  const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
527
- const idleTimeoutMs = getOpenAIStreamIdleTimeoutMs();
528
+ const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
528
529
  const {
529
530
  client,
530
531
  copilotPremiumRequests,
@@ -1233,26 +1234,7 @@ async function createClient(
1233
1234
  // The OpenAI SDK's default is 10 minutes per attempt × `maxRetries`, which
1234
1235
  // turns a stalled-before-headers fetch into a multi-minute hang invisible
1235
1236
  // to the agent loop (the iterator watchdog only arms AFTER `create()` returns).
1236
- // Using the first-event timeout keeps both layers aligned: the SDK gives up
1237
- // before the agent watchdog would have, surfacing a real error to the catch
1238
- // in the IIFE.
1239
- // A caller may raise `StreamOptions.streamFirstEventTimeoutMs` for a slow-
1240
- // before-headers provider; respect it so the SDK doesn't give up before the
1241
- // wrapping watchdog arms. Provider-specific fallbacks apply only when the
1242
- // caller does not pin a value, so an explicit nonzero override must beat that
1243
- // fallback even when it is shorter. An explicit `0` disables the watchdog,
1244
- // and the SDK treats `timeout: 0` as an immediate timeout, so do not pass a
1245
- // request timeout in that case.
1246
- const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
1247
- const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
1248
- const sdkTimeoutMs =
1249
- streamFirstEventTimeoutOverride === 0
1250
- ? undefined
1251
- : streamFirstEventTimeoutOverride !== undefined
1252
- ? providerFirstEventFallbackMs !== undefined
1253
- ? streamFirstEventTimeoutOverride
1254
- : Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride)
1255
- : envSdkTimeoutMs;
1237
+ const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, streamFirstEventTimeoutOverride);
1256
1238
  return {
1257
1239
  client: new OpenAI({
1258
1240
  apiKey,
@@ -33,6 +33,32 @@ import type { AssistantMessageEventStream } from "../utils/event-stream";
33
33
  import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
34
34
  import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
35
35
 
36
+ const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
37
+ "response.created",
38
+ "response.output_item.added",
39
+ "response.reasoning_summary_part.added",
40
+ "response.reasoning_summary_text.delta",
41
+ "response.reasoning_summary_part.done",
42
+ "response.reasoning_text.delta",
43
+ "response.content_part.added",
44
+ "response.output_text.delta",
45
+ "response.refusal.delta",
46
+ "response.function_call_arguments.delta",
47
+ "response.function_call_arguments.done",
48
+ "response.custom_tool_call_input.delta",
49
+ "response.custom_tool_call_input.done",
50
+ "response.output_item.done",
51
+ "response.completed",
52
+ "response.failed",
53
+ "error",
54
+ ]);
55
+
56
+ export function isOpenAIResponsesProgressEvent(event: unknown): boolean {
57
+ if (!event || typeof event !== "object") return false;
58
+ const type = (event as { type?: unknown }).type;
59
+ return typeof type === "string" && OPENAI_RESPONSES_PROGRESS_EVENT_TYPES.has(type);
60
+ }
61
+
36
62
  export function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string {
37
63
  const payload: TextSignatureV1 = { v: 1, id };
38
64
  if (phase) payload.phase = phase;
@@ -43,6 +43,7 @@ import {
43
43
  getProviderFirstEventTimeoutFallbackMs,
44
44
  getStreamFirstEventTimeoutMs,
45
45
  iterateWithIdleTimeout,
46
+ resolveOpenAISdkRequestTimeoutMs,
46
47
  } from "../utils/idle-iterator";
47
48
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
48
49
  import { notifyProviderResponse } from "../utils/provider-response";
@@ -85,6 +86,7 @@ import {
85
86
  convertResponsesAssistantMessage,
86
87
  convertResponsesInputContent,
87
88
  createInitialResponsesAssistantMessage,
89
+ isOpenAIResponsesProgressEvent,
88
90
  normalizeResponsesToolCallIdForTransform,
89
91
  processResponsesStream,
90
92
  repairOrphanResponsesToolOutputs,
@@ -229,32 +231,6 @@ function appendQueryToRequest(input: string | URL | Request, query?: OpenAIRespo
229
231
  return url;
230
232
  }
231
233
 
232
- const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
233
- "response.created",
234
- "response.output_item.added",
235
- "response.reasoning_summary_part.added",
236
- "response.reasoning_summary_text.delta",
237
- "response.reasoning_summary_part.done",
238
- "response.reasoning_text.delta",
239
- "response.content_part.added",
240
- "response.output_text.delta",
241
- "response.refusal.delta",
242
- "response.function_call_arguments.delta",
243
- "response.function_call_arguments.done",
244
- "response.custom_tool_call_input.delta",
245
- "response.custom_tool_call_input.done",
246
- "response.output_item.done",
247
- "response.completed",
248
- "response.failed",
249
- "error",
250
- ]);
251
-
252
- function isOpenAIResponsesProgressEvent(event: unknown): boolean {
253
- if (!event || typeof event !== "object") return false;
254
- const type = (event as { type?: unknown }).type;
255
- return typeof type === "string" && OPENAI_RESPONSES_PROGRESS_EVENT_TYPES.has(type);
256
- }
257
-
258
234
  interface OpenAIResponsesProviderSessionState extends ProviderSessionState {
259
235
  nativeHistoryReplayWarmed: boolean;
260
236
  }
@@ -343,6 +319,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
343
319
  options?.requestMaxRetries,
344
320
  options?.maxRetryDelayMs,
345
321
  options?.attemptScope,
322
+ options?.streamFirstEventTimeoutMs,
346
323
  );
347
324
  const premiumRequestsTotal = copilotPremiumRequests;
348
325
  const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
@@ -498,6 +475,7 @@ function createClient(
498
475
  requestMaxRetries?: number,
499
476
  maxRetryDelayMs?: number,
500
477
  attemptScope?: import("../types.js").AttemptScopeRef,
478
+ streamFirstEventTimeoutOverride?: number,
501
479
  ): {
502
480
  client: OpenAI;
503
481
  copilotPremiumRequests: number | undefined;
@@ -568,6 +546,9 @@ function createClient(
568
546
  model.requestTransform,
569
547
  `Gajae-Code/${packageJson.version}`,
570
548
  );
549
+ // Bound HTTP request timeout to the first-event window so a stalled-before-headers
550
+ // fetch cannot wait the SDK's 10-minute default before the transport watchdog arms.
551
+ const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, streamFirstEventTimeoutOverride);
571
552
  return {
572
553
  client: new OpenAI({
573
554
  apiKey,
@@ -578,6 +559,7 @@ function createClient(
578
559
  fetch: onSseEvent
579
560
  ? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model, attemptScope))
580
561
  : transformedFetch,
562
+ ...(sdkTimeoutMs !== undefined ? { timeout: sdkTimeoutMs } : {}),
581
563
  }),
582
564
  copilotPremiumRequests,
583
565
  baseUrl,
@@ -167,6 +167,7 @@ export function setBedrockProviderModule(module: BedrockProviderModule): void {
167
167
 
168
168
  const LAZY_STREAM_IDLE_TIMEOUT_ERROR = "Provider stream stalled while waiting for the next event";
169
169
  const LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR = "Provider stream timed out while waiting for the first event";
170
+ const LAZY_STREAM_NON_PROGRESS_EVENT_TYPES = new Set(["start", "toolChoiceIncapability"]);
170
171
 
171
172
  function hasFinalResult(
172
173
  source: AsyncIterable<AssistantMessageEvent>,
@@ -184,8 +185,14 @@ function hasFinalResult(
184
185
  interface LazyStreamLimits {
185
186
  defaultFirstEventTimeoutMs?: number;
186
187
  defaultIdleTimeoutMs?: number;
188
+ /** The provider already watches raw transport events, which this normalized wrapper cannot observe. */
189
+ providerOwnsWatchdog?: boolean;
187
190
  }
188
191
 
192
+ const PROVIDER_OWNED_STREAM_WATCHDOG: LazyStreamLimits = {
193
+ providerOwnsWatchdog: true,
194
+ };
195
+
189
196
  /**
190
197
  * Cloud Code Assist (google-gemini-cli / google-antigravity) routinely takes
191
198
  * longer than the global 100s default to emit its first SSE event when serving
@@ -224,28 +231,34 @@ function forwardStream<TApi extends Api>(
224
231
  ): void {
225
232
  (async () => {
226
233
  try {
227
- const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
228
- const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
229
- model.provider,
230
- limits?.defaultFirstEventTimeoutMs,
231
- );
232
- const watchedSource = iterateWithIdleTimeout(source, {
233
- idleTimeoutMs,
234
- firstItemTimeoutMs:
235
- options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
236
- errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
237
- firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
238
- onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
239
- onFirstItemTimeout: () =>
240
- abortTracker.abortLocally(new FirstEventTimeoutError(LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR)),
241
- abortSignal: options.signal,
242
- // The synthetic `start` event is yielded immediately by every provider before
243
- // the upstream model has emitted any tokens. Treating it as the first "real"
244
- // item would flip the watchdog from `firstItemTimeoutMs` to the much shorter
245
- // `idleTimeoutMs` while we're still legitimately waiting on the model's
246
- // first response (slow first-token from reasoning models, cold proxies, etc.).
247
- isProgressItem: event => (event as AssistantMessageEvent).type !== "start",
248
- });
234
+ let watchedSource = source;
235
+ if (!limits?.providerOwnsWatchdog) {
236
+ const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
237
+ const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
238
+ model.provider,
239
+ limits?.defaultFirstEventTimeoutMs,
240
+ );
241
+ watchedSource = iterateWithIdleTimeout(source, {
242
+ idleTimeoutMs,
243
+ firstItemTimeoutMs:
244
+ options.streamFirstEventTimeoutMs ??
245
+ getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
246
+ errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
247
+ firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
248
+ onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
249
+ onFirstItemTimeout: () =>
250
+ abortTracker.abortLocally(new FirstEventTimeoutError(LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR)),
251
+ abortSignal: options.signal,
252
+ // Synthetic starts and tool-capability negotiation are control-plane events,
253
+ // not model progress. Keep the first-event window active until assistant output
254
+ // arrives instead of switching early to the shorter idle timeout.
255
+ isProgressItem: event => {
256
+ if (!event || typeof event !== "object") return true;
257
+ const eventType = (event as { type?: unknown }).type;
258
+ return typeof eventType !== "string" || !LAZY_STREAM_NON_PROGRESS_EVENT_TYPES.has(eventType);
259
+ },
260
+ });
261
+ }
249
262
 
250
263
  for await (const event of watchedSource) {
251
264
  target.push(event);
@@ -423,17 +436,29 @@ function loadBedrockProviderModule(): Promise<LazyProviderModule<"bedrock-conver
423
436
  // are loaded on first use instead of during package initialization.
424
437
  // ---------------------------------------------------------------------------
425
438
 
426
- export const streamAnthropic = createLazyStream(loadAnthropicProviderModule);
427
- export const streamAzureOpenAIResponses = createLazyStream(loadAzureOpenAIResponsesProviderModule);
439
+ export const streamAnthropic = createLazyStream(loadAnthropicProviderModule, PROVIDER_OWNED_STREAM_WATCHDOG);
440
+ export const streamAzureOpenAIResponses = createLazyStream(
441
+ loadAzureOpenAIResponsesProviderModule,
442
+ PROVIDER_OWNED_STREAM_WATCHDOG,
443
+ );
428
444
  export const streamGoogle = createLazyStream(loadGoogleProviderModule);
429
445
  export const streamGoogleGeminiCli = createLazyStream(
430
446
  loadGoogleGeminiCliProviderModule,
431
447
  GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS,
432
448
  );
433
449
  export const streamGoogleVertex = createLazyStream(loadGoogleVertexProviderModule);
434
- export const streamOpenAICodexResponses = createLazyStream(loadOpenAICodexResponsesProviderModule);
435
- export const streamOpenAICompletions = createLazyStream(loadOpenAICompletionsProviderModule);
436
- export const streamOpenAIResponses = createLazyStream(loadOpenAIResponsesProviderModule);
450
+ export const streamOpenAICodexResponses = createLazyStream(
451
+ loadOpenAICodexResponsesProviderModule,
452
+ PROVIDER_OWNED_STREAM_WATCHDOG,
453
+ );
454
+ export const streamOpenAICompletions = createLazyStream(
455
+ loadOpenAICompletionsProviderModule,
456
+ PROVIDER_OWNED_STREAM_WATCHDOG,
457
+ );
458
+ export const streamOpenAIResponses = createLazyStream(
459
+ loadOpenAIResponsesProviderModule,
460
+ PROVIDER_OWNED_STREAM_WATCHDOG,
461
+ );
437
462
  export const streamCursor = createLazyStream(loadCursorProviderModule);
438
463
  export const streamOllama = createLazyStream(loadOllamaProviderModule);
439
464
 
package/src/types.ts CHANGED
@@ -743,7 +743,11 @@ export type Static<S> = S extends ZodType ? z.infer<S> : S extends { static: inf
743
743
  export type RawArgumentRejectionCode =
744
744
  | "ask-intent-review-requires-positive-round"
745
745
  | "ask-intent-contract-requires-non-empty-authority"
746
- | "ask-deep-interview-metadata-requires-deep-interview-gate";
746
+ | "ask-deep-interview-metadata-requires-deep-interview-gate"
747
+ | "todo-write-unknown-root-key"
748
+ | "todo-write-unknown-op-entry-key"
749
+ | "todo-write-done-drop-requires-target"
750
+ | "todo-write-unknown-init-entry-key";
747
751
 
748
752
  export type RawArgumentValidationResult =
749
753
  | { outcome: "passthrough" }
@@ -947,8 +951,10 @@ export interface AnthropicCompat extends ToolChoiceCompat {
947
951
  supportsLongCacheRetention?: boolean;
948
952
  /**
949
953
  * Prompt-cache transport accepted by this Anthropic-compatible endpoint.
950
- * Canonical Anthropic defaults to `"automatic"`; noncanonical endpoints default
951
- * to `"none"` and must explicitly opt into generated `"explicit"` markers.
954
+ * Canonical Anthropic and Claude-family models default to `"automatic"`;
955
+ * noncanonical non-Claude endpoints default to `"none"`. Set `"automatic"` to
956
+ * opt an otherwise unknown compatible endpoint into top-level caching, `"none"`
957
+ * to opt out, or `"explicit"` for endpoints that require block-level markers.
952
958
  */
953
959
  promptCacheMode?: "none" | "explicit" | "automatic";
954
960
  }
@@ -68,6 +68,37 @@ export function getStreamFirstEventTimeoutMs(
68
68
  return normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallback);
69
69
  }
70
70
 
71
+ /**
72
+ * Resolves the OpenAI SDK client `timeout` so stalled-before-headers requests are
73
+ * bounded by the same first-event window the transport watchdog uses after
74
+ * `create()` returns. Without this, providers that only arm
75
+ * `iterateWithIdleTimeout` post-setup can wait the full SDK default (10 minutes
76
+ * per attempt) before any provider-owned watchdog exists.
77
+ *
78
+ * - Explicit `0` disables the request timeout (the SDK treats `timeout: 0` as an
79
+ * immediate failure, so callers that disable the first-event watchdog must not
80
+ * pass a timeout).
81
+ * - Providers with a first-event fallback (Alibaba, Kimi) honor an explicit
82
+ * nonzero override as-is, even when shorter than the fallback.
83
+ * - Other providers floor an explicit override at the env/default first-event
84
+ * window so a short post-connect first-event budget cannot kill legitimate
85
+ * slow setup.
86
+ */
87
+ export function resolveOpenAISdkRequestTimeoutMs(
88
+ provider: string,
89
+ streamFirstEventTimeoutOverride?: number,
90
+ ): number | undefined {
91
+ const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(provider);
92
+ const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
93
+ if (streamFirstEventTimeoutOverride === 0) return undefined;
94
+ if (streamFirstEventTimeoutOverride !== undefined) {
95
+ return providerFirstEventFallbackMs !== undefined
96
+ ? streamFirstEventTimeoutOverride
97
+ : Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride);
98
+ }
99
+ return envSdkTimeoutMs;
100
+ }
101
+
71
102
  export type Watchdog = NodeJS.Timeout | undefined;
72
103
  export class FirstEventTimeoutError extends Error {
73
104
  readonly providerCode = STREAM_FIRST_EVENT_TIMEOUT_PROVIDER_CODE;