@gajae-code/ai 0.12.11 → 0.12.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/dist/types/providers/anthropic.d.ts +13 -0
- package/dist/types/providers/openai-responses-shared.d.ts +1 -0
- package/dist/types/types.d.ts +5 -3
- package/dist/types/utils/idle-iterator.d.ts +17 -0
- package/dist/types/utils/tool-choice-capability.d.ts +5 -0
- package/dist/types/utils.d.ts +28 -0
- package/package.json +2 -2
- package/src/model-thinking.ts +34 -11
- package/src/models.json +52 -118
- package/src/provider-models/descriptors.ts +3 -3
- package/src/providers/anthropic.ts +57 -9
- package/src/providers/azure-openai-responses.ts +20 -4
- package/src/providers/openai-codex-responses.ts +31 -6
- package/src/providers/openai-completions.ts +3 -21
- package/src/providers/openai-responses-shared.ts +26 -0
- package/src/providers/openai-responses.ts +8 -26
- package/src/providers/register-builtins.ts +52 -27
- package/src/types.ts +9 -3
- package/src/utils/idle-iterator.ts +31 -0
- package/src/utils/tool-choice-capability.ts +33 -1
- package/src/utils/validation.ts +5 -0
- package/src/utils.ts +56 -0
|
@@ -395,8 +395,20 @@ export function isAnthropicFastModeUnsupportedError(error: unknown): boolean {
|
|
|
395
395
|
return false;
|
|
396
396
|
}
|
|
397
397
|
|
|
398
|
+
/**
|
|
399
|
+
* Proxies (e.g. CLIProxyAPI) can deliver Anthropic's 400 body as an in-stream
|
|
400
|
+
* SSE `error` event on an HTTP 200 response; the thrown error then carries no
|
|
401
|
+
* HTTP status at all (issue #3900). Accept both the direct 400 and the
|
|
402
|
+
* statusless SSE shape — the strict `invalid_request_error` message checks in
|
|
403
|
+
* each matcher keep the statusless branch from claiming unrelated failures.
|
|
404
|
+
*/
|
|
405
|
+
function isAnthropicInvalidRequestStatus(error: unknown): boolean {
|
|
406
|
+
const status = extractHttpStatusFromError(error);
|
|
407
|
+
return status === 400 || status === undefined;
|
|
408
|
+
}
|
|
409
|
+
|
|
398
410
|
export function isAnthropicThinkingBlockMutationError(error: unknown): boolean {
|
|
399
|
-
if (
|
|
411
|
+
if (!isAnthropicInvalidRequestStatus(error)) return false;
|
|
400
412
|
const message = error instanceof Error ? error.message : String(error);
|
|
401
413
|
return (
|
|
402
414
|
/invalid_request_error/i.test(message) &&
|
|
@@ -414,7 +426,7 @@ export function isAnthropicThinkingBlockMutationError(error: unknown): boolean {
|
|
|
414
426
|
* than only the latest one.
|
|
415
427
|
*/
|
|
416
428
|
export function isAnthropicThinkingSignatureInvalidError(error: unknown): boolean {
|
|
417
|
-
if (
|
|
429
|
+
if (!isAnthropicInvalidRequestStatus(error)) return false;
|
|
418
430
|
const message = error instanceof Error ? error.message : String(error);
|
|
419
431
|
return (
|
|
420
432
|
/invalid_request_error/i.test(message) &&
|
|
@@ -423,6 +435,27 @@ export function isAnthropicThinkingSignatureInvalidError(error: unknown): boolea
|
|
|
423
435
|
);
|
|
424
436
|
}
|
|
425
437
|
|
|
438
|
+
/**
|
|
439
|
+
* CLIProxyAPI replaces Anthropic's rejection body wholesale instead of forwarding
|
|
440
|
+
* it: the client only ever sees
|
|
441
|
+
* `{"type":"error","error":{"type":"api_error","message":"An error occurred while
|
|
442
|
+
* processing the request."}}`, delivered as an in-stream SSE `error` event on an
|
|
443
|
+
* HTTP 200 response, so neither the status nor the message survives. Captured CPA
|
|
444
|
+
* traces for that masked shape carry the thinking-integrity 400 upstream (issue
|
|
445
|
+
* #3900), and the generic body matches no transient phrase either, so the turn
|
|
446
|
+
* dies unrecoverably. Nothing in the payload names the cause; callers must pair
|
|
447
|
+
* this with a request that actually replays signed thinking blocks before
|
|
448
|
+
* treating it as a thinking-replay rejection.
|
|
449
|
+
*/
|
|
450
|
+
export function isAnthropicMaskedProxyRejection(error: unknown): boolean {
|
|
451
|
+
const status = extractHttpStatusFromError(error);
|
|
452
|
+
if (status !== undefined && status !== 400) return false;
|
|
453
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
454
|
+
// A body that still names its error type is classified by the strict matchers.
|
|
455
|
+
if (/invalid_request_error/i.test(message)) return false;
|
|
456
|
+
return /"type"\s*:\s*"api_error"/.test(message) && /an error occurred while processing/i.test(message);
|
|
457
|
+
}
|
|
458
|
+
|
|
426
459
|
function hasStrictAnthropicTools(params: MessageCreateParamsStreaming): boolean {
|
|
427
460
|
const tools = params.tools as Array<{ strict?: unknown }> | undefined;
|
|
428
461
|
return tools?.some(tool => tool.strict === true) ?? false;
|
|
@@ -447,12 +480,21 @@ function dropAnthropicStrictTools(params: MessageCreateParamsStreaming): void {
|
|
|
447
480
|
}
|
|
448
481
|
}
|
|
449
482
|
|
|
483
|
+
function isClaudeFamilyModel(model: Model<"anthropic-messages">): boolean {
|
|
484
|
+
// Classify the same identifier the request body serializes (`params.model =
|
|
485
|
+
// model.id` in buildParams); a differing `wireModelId` is not dispatched by
|
|
486
|
+
// this transport, so it must not drive the cache decision either.
|
|
487
|
+
const id = model.id;
|
|
488
|
+
const shortId = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
|
|
489
|
+
return shortId.toLowerCase().startsWith("claude-");
|
|
490
|
+
}
|
|
491
|
+
|
|
450
492
|
function getCacheControl(
|
|
451
493
|
model: Model<"anthropic-messages">,
|
|
452
494
|
baseUrl: string,
|
|
453
495
|
cacheRetention?: CacheRetention,
|
|
454
496
|
): { mode: AnthropicCacheMode; cacheControl?: AnthropicCacheControl } {
|
|
455
|
-
const retention = resolveCacheRetention(cacheRetention, "long");
|
|
497
|
+
const retention = resolveCacheRetention(cacheRetention ?? model.cacheRetention, "long");
|
|
456
498
|
if (retention === "none") return { mode: "none" };
|
|
457
499
|
|
|
458
500
|
const isCanonicalApi = isAnthropicApiBaseUrl(baseUrl);
|
|
@@ -462,7 +504,7 @@ function getCacheControl(
|
|
|
462
504
|
? "none"
|
|
463
505
|
: promptCacheMode === "explicit"
|
|
464
506
|
? "explicit"
|
|
465
|
-
: isCanonicalApi
|
|
507
|
+
: promptCacheMode === "automatic" || isCanonicalApi || isClaudeFamilyModel(model)
|
|
466
508
|
? "automatic"
|
|
467
509
|
: "none";
|
|
468
510
|
if (mode === "none") return { mode };
|
|
@@ -1847,7 +1889,12 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1847
1889
|
!options?.fallbackManaged &&
|
|
1848
1890
|
!repairAllAssistantThinking &&
|
|
1849
1891
|
firstTokenTime === undefined &&
|
|
1850
|
-
(thinkingSignatureInvalid ||
|
|
1892
|
+
(thinkingSignatureInvalid ||
|
|
1893
|
+
isAnthropicThinkingBlockMutationError(streamFailure) ||
|
|
1894
|
+
// Masked proxy rejection: unclassifiable on its own, so the replayed
|
|
1895
|
+
// request shape is the evidence. Without signed thinking blocks in
|
|
1896
|
+
// flight there is nothing to repair and the error must surface.
|
|
1897
|
+
(isAnthropicMaskedProxyRejection(streamFailure) && hasNativeThinkingBlocks(params.messages)))
|
|
1851
1898
|
) {
|
|
1852
1899
|
// The mutation 400 blames the "latest assistant message", but its cited
|
|
1853
1900
|
// `messages.N.content.M` path can point at an EARLIER replayed turn, so the
|
|
@@ -2285,10 +2332,11 @@ function applyExplicitPromptCaching(params: AnthropicCacheParams, cacheControl:
|
|
|
2285
2332
|
const currentUser = params.messages[currentUserIndex];
|
|
2286
2333
|
if (!currentUser) return;
|
|
2287
2334
|
|
|
2288
|
-
//
|
|
2289
|
-
//
|
|
2290
|
-
// so
|
|
2291
|
-
|
|
2335
|
+
// Tool results are encoded as role "user" on the wire but belong to the
|
|
2336
|
+
// assistant tool-use turn immediately before them. Anchor the latest completed
|
|
2337
|
+
// assistant turn so the reusable prefix advances during an agent tool loop,
|
|
2338
|
+
// while keeping the newest tool output outside the cache boundary.
|
|
2339
|
+
for (let index = params.messages.length - 1; index >= 0; index--) {
|
|
2292
2340
|
const message = params.messages[index];
|
|
2293
2341
|
if (message?.role !== "assistant" || !Array.isArray(message.content)) continue;
|
|
2294
2342
|
if (
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
|
|
2
|
-
import { AzureOpenAI } from "openai";
|
|
2
|
+
import { APIConnectionTimeoutError, AzureOpenAI } from "openai";
|
|
3
3
|
import type {
|
|
4
4
|
Tool as OpenAITool,
|
|
5
5
|
ResponseCreateParamsStreaming,
|
|
@@ -22,9 +22,11 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
22
22
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
23
23
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
24
24
|
import {
|
|
25
|
+
FirstEventTimeoutError,
|
|
25
26
|
getOpenAIStreamIdleTimeoutMs,
|
|
26
27
|
getStreamFirstEventTimeoutMs,
|
|
27
28
|
iterateWithIdleTimeout,
|
|
29
|
+
resolveOpenAISdkRequestTimeoutMs,
|
|
28
30
|
} from "../utils/idle-iterator";
|
|
29
31
|
import { resolveRetryBudget } from "../utils/retry-budget";
|
|
30
32
|
import { flattenToolRootCombinators, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
|
|
@@ -44,6 +46,7 @@ import {
|
|
|
44
46
|
convertResponsesAssistantMessage,
|
|
45
47
|
convertResponsesInputContent,
|
|
46
48
|
createInitialResponsesAssistantMessage,
|
|
49
|
+
isOpenAIResponsesProgressEvent,
|
|
47
50
|
normalizeResponsesToolCallIdForTransform,
|
|
48
51
|
processResponsesStream,
|
|
49
52
|
} from "./openai-responses-shared";
|
|
@@ -108,6 +111,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
108
111
|
(async () => {
|
|
109
112
|
const startTime = Date.now();
|
|
110
113
|
let firstTokenTime: number | undefined;
|
|
114
|
+
let streamConnected = false;
|
|
111
115
|
const deploymentName = resolveDeploymentName(model, options);
|
|
112
116
|
|
|
113
117
|
const output: AssistantMessage = createInitialResponsesAssistantMessage(
|
|
@@ -162,6 +166,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
162
166
|
rawRequestDump = { ...rawRequestDump, body: params };
|
|
163
167
|
openaiStream = await client.responses.create(params, { signal: requestSignal });
|
|
164
168
|
}
|
|
169
|
+
streamConnected = true;
|
|
165
170
|
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
|
|
166
171
|
stream.push({ type: "start", partial: output });
|
|
167
172
|
|
|
@@ -173,6 +178,8 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
173
178
|
errorMessage: "Azure OpenAI responses stream stalled while waiting for the next event",
|
|
174
179
|
onIdle: () => requestAbortController.abort(),
|
|
175
180
|
onFirstItemTimeout: () => requestAbortController.abort(),
|
|
181
|
+
isProgressItem: isOpenAIResponsesProgressEvent,
|
|
182
|
+
abortSignal: options?.signal,
|
|
176
183
|
}),
|
|
177
184
|
output,
|
|
178
185
|
stream,
|
|
@@ -203,10 +210,15 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
203
210
|
} catch (error) {
|
|
204
211
|
for (const block of output.content) delete (block as { index?: number }).index;
|
|
205
212
|
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
|
213
|
+
const normalizedError =
|
|
214
|
+
!streamConnected && error instanceof APIConnectionTimeoutError
|
|
215
|
+
? new FirstEventTimeoutError(AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE)
|
|
216
|
+
: error;
|
|
206
217
|
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
207
|
-
output.errorStatus = extractHttpStatusFromError(
|
|
208
|
-
output.transportFailure = transportFailureFacts(
|
|
209
|
-
output.errorMessage =
|
|
218
|
+
output.errorStatus = extractHttpStatusFromError(firstEventTimeoutError ?? normalizedError);
|
|
219
|
+
output.transportFailure = transportFailureFacts(firstEventTimeoutError ?? normalizedError);
|
|
220
|
+
output.errorMessage =
|
|
221
|
+
firstEventTimeoutError?.message ?? (await finalizeErrorMessage(normalizedError, rawRequestDump));
|
|
210
222
|
output.duration = Date.now() - startTime;
|
|
211
223
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
212
224
|
stream.push({ type: "error", reason: output.stopReason, error: output });
|
|
@@ -305,6 +317,9 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
|
|
|
305
317
|
|
|
306
318
|
const baseFetch = wrapOpenAIFetchForBoundedRateLimits(options?.fetch ?? fetch, options?.maxRetryDelayMs);
|
|
307
319
|
const onSseEvent = options?.onSseEvent;
|
|
320
|
+
// Bound HTTP request timeout to the first-event window so a stalled-before-headers
|
|
321
|
+
// fetch cannot wait the SDK's 10-minute default before the transport watchdog arms.
|
|
322
|
+
const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, options?.streamFirstEventTimeoutMs);
|
|
308
323
|
return new AzureOpenAI({
|
|
309
324
|
apiKey,
|
|
310
325
|
apiVersion,
|
|
@@ -315,6 +330,7 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
|
|
|
315
330
|
fetch: onSseEvent
|
|
316
331
|
? wrapFetchForSseDebug(baseFetch, event => onSseEvent(event, model, options?.attemptScope))
|
|
317
332
|
: baseFetch,
|
|
333
|
+
...(sdkTimeoutMs !== undefined ? { timeout: sdkTimeoutMs } : {}),
|
|
318
334
|
});
|
|
319
335
|
}
|
|
320
336
|
|
|
@@ -66,6 +66,7 @@ import {
|
|
|
66
66
|
toolWireSchema,
|
|
67
67
|
} from "../utils/schema";
|
|
68
68
|
import {
|
|
69
|
+
isCodexStatuslessNamedToolChoiceNotFoundError,
|
|
69
70
|
isForcedToolChoiceUnsupportedError,
|
|
70
71
|
markToolChoiceIncapability,
|
|
71
72
|
resolveToolChoice,
|
|
@@ -247,9 +248,7 @@ async function retryCodexInitialTransportWithoutToolChoice(
|
|
|
247
248
|
requestBodyForState: RequestBody;
|
|
248
249
|
transport: CodexTransport;
|
|
249
250
|
}> {
|
|
250
|
-
if (
|
|
251
|
-
!isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(requestContext.transformedBody.tool_choice))
|
|
252
|
-
) {
|
|
251
|
+
if (!isCodexForcedToolChoiceUnsupportedError(error, requestContext.transformedBody)) {
|
|
253
252
|
throw error;
|
|
254
253
|
}
|
|
255
254
|
const reason = await finalizeErrorMessage(error, requestContext.rawRequestDump);
|
|
@@ -555,10 +554,12 @@ function getCodexServiceTierCostMultiplier(
|
|
|
555
554
|
|
|
556
555
|
function resolveCodexCostServiceTier(res: unknown, req?: unknown): ServiceTier | "default" | undefined {
|
|
557
556
|
switch (res) {
|
|
557
|
+
case "auto":
|
|
558
|
+
case "default":
|
|
558
559
|
case "flex":
|
|
559
|
-
|
|
560
|
+
case "scale":
|
|
560
561
|
case "priority":
|
|
561
|
-
return
|
|
562
|
+
return res;
|
|
562
563
|
default:
|
|
563
564
|
if (req === "flex" || req === "priority") {
|
|
564
565
|
return req;
|
|
@@ -1525,7 +1526,7 @@ async function tryRetryWithoutForcedToolChoice(
|
|
|
1525
1526
|
context.output.content.length > 0 ||
|
|
1526
1527
|
context.firstTokenTime !== undefined ||
|
|
1527
1528
|
context.options?.signal?.aborted ||
|
|
1528
|
-
!
|
|
1529
|
+
!isCodexForcedToolChoiceUnsupportedError(error, runtime.requestBodyForState)
|
|
1529
1530
|
) {
|
|
1530
1531
|
return false;
|
|
1531
1532
|
}
|
|
@@ -1577,6 +1578,30 @@ async function tryRetryWithoutForcedToolChoice(
|
|
|
1577
1578
|
function isForcedCodexToolChoice(choice: RequestBody["tool_choice"]): boolean {
|
|
1578
1579
|
return !!choice && choice !== "none" && choice !== "auto";
|
|
1579
1580
|
}
|
|
1581
|
+
function isCodexForcedToolChoiceUnsupportedError(error: unknown, body: RequestBody): boolean {
|
|
1582
|
+
if (isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(body.tool_choice))) {
|
|
1583
|
+
return true;
|
|
1584
|
+
}
|
|
1585
|
+
return isCodexStatuslessNamedToolChoiceNotFoundError(
|
|
1586
|
+
error,
|
|
1587
|
+
codexNamedFunctionToolChoiceName(body.tool_choice),
|
|
1588
|
+
codexSerializedToolNames(body.tools),
|
|
1589
|
+
);
|
|
1590
|
+
}
|
|
1591
|
+
|
|
1592
|
+
function codexNamedFunctionToolChoiceName(choice: RequestBody["tool_choice"]): string | undefined {
|
|
1593
|
+
if (!choice || typeof choice !== "object") return undefined;
|
|
1594
|
+
const namedChoice = choice as { type?: unknown; name?: unknown };
|
|
1595
|
+
return namedChoice.type === "function" && typeof namedChoice.name === "string" ? namedChoice.name : undefined;
|
|
1596
|
+
}
|
|
1597
|
+
|
|
1598
|
+
function codexSerializedToolNames(tools: RequestBody["tools"]): string[] {
|
|
1599
|
+
if (!Array.isArray(tools)) return [];
|
|
1600
|
+
return tools.flatMap(tool => {
|
|
1601
|
+
const name = (tool as { name?: unknown }).name;
|
|
1602
|
+
return typeof name === "string" ? [name] : [];
|
|
1603
|
+
});
|
|
1604
|
+
}
|
|
1580
1605
|
|
|
1581
1606
|
/**
|
|
1582
1607
|
* Handles `websocket_connection_limit_reached` errors by closing the stale connection
|
|
@@ -52,6 +52,7 @@ import {
|
|
|
52
52
|
getProviderFirstEventTimeoutFallbackMs,
|
|
53
53
|
getStreamFirstEventTimeoutMs,
|
|
54
54
|
iterateWithIdleTimeout,
|
|
55
|
+
resolveOpenAISdkRequestTimeoutMs,
|
|
55
56
|
} from "../utils/idle-iterator";
|
|
56
57
|
import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
|
|
57
58
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
@@ -524,7 +525,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
524
525
|
|
|
525
526
|
try {
|
|
526
527
|
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
|
|
527
|
-
const idleTimeoutMs = getOpenAIStreamIdleTimeoutMs();
|
|
528
|
+
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
|
528
529
|
const {
|
|
529
530
|
client,
|
|
530
531
|
copilotPremiumRequests,
|
|
@@ -1233,26 +1234,7 @@ async function createClient(
|
|
|
1233
1234
|
// The OpenAI SDK's default is 10 minutes per attempt × `maxRetries`, which
|
|
1234
1235
|
// turns a stalled-before-headers fetch into a multi-minute hang invisible
|
|
1235
1236
|
// to the agent loop (the iterator watchdog only arms AFTER `create()` returns).
|
|
1236
|
-
|
|
1237
|
-
// before the agent watchdog would have, surfacing a real error to the catch
|
|
1238
|
-
// in the IIFE.
|
|
1239
|
-
// A caller may raise `StreamOptions.streamFirstEventTimeoutMs` for a slow-
|
|
1240
|
-
// before-headers provider; respect it so the SDK doesn't give up before the
|
|
1241
|
-
// wrapping watchdog arms. Provider-specific fallbacks apply only when the
|
|
1242
|
-
// caller does not pin a value, so an explicit nonzero override must beat that
|
|
1243
|
-
// fallback even when it is shorter. An explicit `0` disables the watchdog,
|
|
1244
|
-
// and the SDK treats `timeout: 0` as an immediate timeout, so do not pass a
|
|
1245
|
-
// request timeout in that case.
|
|
1246
|
-
const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
1247
|
-
const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
|
|
1248
|
-
const sdkTimeoutMs =
|
|
1249
|
-
streamFirstEventTimeoutOverride === 0
|
|
1250
|
-
? undefined
|
|
1251
|
-
: streamFirstEventTimeoutOverride !== undefined
|
|
1252
|
-
? providerFirstEventFallbackMs !== undefined
|
|
1253
|
-
? streamFirstEventTimeoutOverride
|
|
1254
|
-
: Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride)
|
|
1255
|
-
: envSdkTimeoutMs;
|
|
1237
|
+
const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, streamFirstEventTimeoutOverride);
|
|
1256
1238
|
return {
|
|
1257
1239
|
client: new OpenAI({
|
|
1258
1240
|
apiKey,
|
|
@@ -33,6 +33,32 @@ import type { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
33
33
|
import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
|
|
34
34
|
import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
|
|
35
35
|
|
|
36
|
+
const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
|
|
37
|
+
"response.created",
|
|
38
|
+
"response.output_item.added",
|
|
39
|
+
"response.reasoning_summary_part.added",
|
|
40
|
+
"response.reasoning_summary_text.delta",
|
|
41
|
+
"response.reasoning_summary_part.done",
|
|
42
|
+
"response.reasoning_text.delta",
|
|
43
|
+
"response.content_part.added",
|
|
44
|
+
"response.output_text.delta",
|
|
45
|
+
"response.refusal.delta",
|
|
46
|
+
"response.function_call_arguments.delta",
|
|
47
|
+
"response.function_call_arguments.done",
|
|
48
|
+
"response.custom_tool_call_input.delta",
|
|
49
|
+
"response.custom_tool_call_input.done",
|
|
50
|
+
"response.output_item.done",
|
|
51
|
+
"response.completed",
|
|
52
|
+
"response.failed",
|
|
53
|
+
"error",
|
|
54
|
+
]);
|
|
55
|
+
|
|
56
|
+
export function isOpenAIResponsesProgressEvent(event: unknown): boolean {
|
|
57
|
+
if (!event || typeof event !== "object") return false;
|
|
58
|
+
const type = (event as { type?: unknown }).type;
|
|
59
|
+
return typeof type === "string" && OPENAI_RESPONSES_PROGRESS_EVENT_TYPES.has(type);
|
|
60
|
+
}
|
|
61
|
+
|
|
36
62
|
export function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string {
|
|
37
63
|
const payload: TextSignatureV1 = { v: 1, id };
|
|
38
64
|
if (phase) payload.phase = phase;
|
|
@@ -43,6 +43,7 @@ import {
|
|
|
43
43
|
getProviderFirstEventTimeoutFallbackMs,
|
|
44
44
|
getStreamFirstEventTimeoutMs,
|
|
45
45
|
iterateWithIdleTimeout,
|
|
46
|
+
resolveOpenAISdkRequestTimeoutMs,
|
|
46
47
|
} from "../utils/idle-iterator";
|
|
47
48
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
48
49
|
import { notifyProviderResponse } from "../utils/provider-response";
|
|
@@ -85,6 +86,7 @@ import {
|
|
|
85
86
|
convertResponsesAssistantMessage,
|
|
86
87
|
convertResponsesInputContent,
|
|
87
88
|
createInitialResponsesAssistantMessage,
|
|
89
|
+
isOpenAIResponsesProgressEvent,
|
|
88
90
|
normalizeResponsesToolCallIdForTransform,
|
|
89
91
|
processResponsesStream,
|
|
90
92
|
repairOrphanResponsesToolOutputs,
|
|
@@ -229,32 +231,6 @@ function appendQueryToRequest(input: string | URL | Request, query?: OpenAIRespo
|
|
|
229
231
|
return url;
|
|
230
232
|
}
|
|
231
233
|
|
|
232
|
-
const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
|
|
233
|
-
"response.created",
|
|
234
|
-
"response.output_item.added",
|
|
235
|
-
"response.reasoning_summary_part.added",
|
|
236
|
-
"response.reasoning_summary_text.delta",
|
|
237
|
-
"response.reasoning_summary_part.done",
|
|
238
|
-
"response.reasoning_text.delta",
|
|
239
|
-
"response.content_part.added",
|
|
240
|
-
"response.output_text.delta",
|
|
241
|
-
"response.refusal.delta",
|
|
242
|
-
"response.function_call_arguments.delta",
|
|
243
|
-
"response.function_call_arguments.done",
|
|
244
|
-
"response.custom_tool_call_input.delta",
|
|
245
|
-
"response.custom_tool_call_input.done",
|
|
246
|
-
"response.output_item.done",
|
|
247
|
-
"response.completed",
|
|
248
|
-
"response.failed",
|
|
249
|
-
"error",
|
|
250
|
-
]);
|
|
251
|
-
|
|
252
|
-
function isOpenAIResponsesProgressEvent(event: unknown): boolean {
|
|
253
|
-
if (!event || typeof event !== "object") return false;
|
|
254
|
-
const type = (event as { type?: unknown }).type;
|
|
255
|
-
return typeof type === "string" && OPENAI_RESPONSES_PROGRESS_EVENT_TYPES.has(type);
|
|
256
|
-
}
|
|
257
|
-
|
|
258
234
|
interface OpenAIResponsesProviderSessionState extends ProviderSessionState {
|
|
259
235
|
nativeHistoryReplayWarmed: boolean;
|
|
260
236
|
}
|
|
@@ -343,6 +319,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
343
319
|
options?.requestMaxRetries,
|
|
344
320
|
options?.maxRetryDelayMs,
|
|
345
321
|
options?.attemptScope,
|
|
322
|
+
options?.streamFirstEventTimeoutMs,
|
|
346
323
|
);
|
|
347
324
|
const premiumRequestsTotal = copilotPremiumRequests;
|
|
348
325
|
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
|
|
@@ -498,6 +475,7 @@ function createClient(
|
|
|
498
475
|
requestMaxRetries?: number,
|
|
499
476
|
maxRetryDelayMs?: number,
|
|
500
477
|
attemptScope?: import("../types.js").AttemptScopeRef,
|
|
478
|
+
streamFirstEventTimeoutOverride?: number,
|
|
501
479
|
): {
|
|
502
480
|
client: OpenAI;
|
|
503
481
|
copilotPremiumRequests: number | undefined;
|
|
@@ -568,6 +546,9 @@ function createClient(
|
|
|
568
546
|
model.requestTransform,
|
|
569
547
|
`Gajae-Code/${packageJson.version}`,
|
|
570
548
|
);
|
|
549
|
+
// Bound HTTP request timeout to the first-event window so a stalled-before-headers
|
|
550
|
+
// fetch cannot wait the SDK's 10-minute default before the transport watchdog arms.
|
|
551
|
+
const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, streamFirstEventTimeoutOverride);
|
|
571
552
|
return {
|
|
572
553
|
client: new OpenAI({
|
|
573
554
|
apiKey,
|
|
@@ -578,6 +559,7 @@ function createClient(
|
|
|
578
559
|
fetch: onSseEvent
|
|
579
560
|
? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model, attemptScope))
|
|
580
561
|
: transformedFetch,
|
|
562
|
+
...(sdkTimeoutMs !== undefined ? { timeout: sdkTimeoutMs } : {}),
|
|
581
563
|
}),
|
|
582
564
|
copilotPremiumRequests,
|
|
583
565
|
baseUrl,
|
|
@@ -167,6 +167,7 @@ export function setBedrockProviderModule(module: BedrockProviderModule): void {
|
|
|
167
167
|
|
|
168
168
|
const LAZY_STREAM_IDLE_TIMEOUT_ERROR = "Provider stream stalled while waiting for the next event";
|
|
169
169
|
const LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR = "Provider stream timed out while waiting for the first event";
|
|
170
|
+
const LAZY_STREAM_NON_PROGRESS_EVENT_TYPES = new Set(["start", "toolChoiceIncapability"]);
|
|
170
171
|
|
|
171
172
|
function hasFinalResult(
|
|
172
173
|
source: AsyncIterable<AssistantMessageEvent>,
|
|
@@ -184,8 +185,14 @@ function hasFinalResult(
|
|
|
184
185
|
interface LazyStreamLimits {
|
|
185
186
|
defaultFirstEventTimeoutMs?: number;
|
|
186
187
|
defaultIdleTimeoutMs?: number;
|
|
188
|
+
/** The provider already watches raw transport events, which this normalized wrapper cannot observe. */
|
|
189
|
+
providerOwnsWatchdog?: boolean;
|
|
187
190
|
}
|
|
188
191
|
|
|
192
|
+
const PROVIDER_OWNED_STREAM_WATCHDOG: LazyStreamLimits = {
|
|
193
|
+
providerOwnsWatchdog: true,
|
|
194
|
+
};
|
|
195
|
+
|
|
189
196
|
/**
|
|
190
197
|
* Cloud Code Assist (google-gemini-cli / google-antigravity) routinely takes
|
|
191
198
|
* longer than the global 100s default to emit its first SSE event when serving
|
|
@@ -224,28 +231,34 @@ function forwardStream<TApi extends Api>(
|
|
|
224
231
|
): void {
|
|
225
232
|
(async () => {
|
|
226
233
|
try {
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
234
|
+
let watchedSource = source;
|
|
235
|
+
if (!limits?.providerOwnsWatchdog) {
|
|
236
|
+
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
|
|
237
|
+
const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
|
|
238
|
+
model.provider,
|
|
239
|
+
limits?.defaultFirstEventTimeoutMs,
|
|
240
|
+
);
|
|
241
|
+
watchedSource = iterateWithIdleTimeout(source, {
|
|
242
|
+
idleTimeoutMs,
|
|
243
|
+
firstItemTimeoutMs:
|
|
244
|
+
options.streamFirstEventTimeoutMs ??
|
|
245
|
+
getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
246
|
+
errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
|
|
247
|
+
firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
|
|
248
|
+
onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
|
|
249
|
+
onFirstItemTimeout: () =>
|
|
250
|
+
abortTracker.abortLocally(new FirstEventTimeoutError(LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR)),
|
|
251
|
+
abortSignal: options.signal,
|
|
252
|
+
// Synthetic starts and tool-capability negotiation are control-plane events,
|
|
253
|
+
// not model progress. Keep the first-event window active until assistant output
|
|
254
|
+
// arrives instead of switching early to the shorter idle timeout.
|
|
255
|
+
isProgressItem: event => {
|
|
256
|
+
if (!event || typeof event !== "object") return true;
|
|
257
|
+
const eventType = (event as { type?: unknown }).type;
|
|
258
|
+
return typeof eventType !== "string" || !LAZY_STREAM_NON_PROGRESS_EVENT_TYPES.has(eventType);
|
|
259
|
+
},
|
|
260
|
+
});
|
|
261
|
+
}
|
|
249
262
|
|
|
250
263
|
for await (const event of watchedSource) {
|
|
251
264
|
target.push(event);
|
|
@@ -423,17 +436,29 @@ function loadBedrockProviderModule(): Promise<LazyProviderModule<"bedrock-conver
|
|
|
423
436
|
// are loaded on first use instead of during package initialization.
|
|
424
437
|
// ---------------------------------------------------------------------------
|
|
425
438
|
|
|
426
|
-
export const streamAnthropic = createLazyStream(loadAnthropicProviderModule);
|
|
427
|
-
export const streamAzureOpenAIResponses = createLazyStream(
|
|
439
|
+
export const streamAnthropic = createLazyStream(loadAnthropicProviderModule, PROVIDER_OWNED_STREAM_WATCHDOG);
|
|
440
|
+
export const streamAzureOpenAIResponses = createLazyStream(
|
|
441
|
+
loadAzureOpenAIResponsesProviderModule,
|
|
442
|
+
PROVIDER_OWNED_STREAM_WATCHDOG,
|
|
443
|
+
);
|
|
428
444
|
export const streamGoogle = createLazyStream(loadGoogleProviderModule);
|
|
429
445
|
export const streamGoogleGeminiCli = createLazyStream(
|
|
430
446
|
loadGoogleGeminiCliProviderModule,
|
|
431
447
|
GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS,
|
|
432
448
|
);
|
|
433
449
|
export const streamGoogleVertex = createLazyStream(loadGoogleVertexProviderModule);
|
|
434
|
-
export const streamOpenAICodexResponses = createLazyStream(
|
|
435
|
-
|
|
436
|
-
|
|
450
|
+
export const streamOpenAICodexResponses = createLazyStream(
|
|
451
|
+
loadOpenAICodexResponsesProviderModule,
|
|
452
|
+
PROVIDER_OWNED_STREAM_WATCHDOG,
|
|
453
|
+
);
|
|
454
|
+
export const streamOpenAICompletions = createLazyStream(
|
|
455
|
+
loadOpenAICompletionsProviderModule,
|
|
456
|
+
PROVIDER_OWNED_STREAM_WATCHDOG,
|
|
457
|
+
);
|
|
458
|
+
export const streamOpenAIResponses = createLazyStream(
|
|
459
|
+
loadOpenAIResponsesProviderModule,
|
|
460
|
+
PROVIDER_OWNED_STREAM_WATCHDOG,
|
|
461
|
+
);
|
|
437
462
|
export const streamCursor = createLazyStream(loadCursorProviderModule);
|
|
438
463
|
export const streamOllama = createLazyStream(loadOllamaProviderModule);
|
|
439
464
|
|
package/src/types.ts
CHANGED
|
@@ -743,7 +743,11 @@ export type Static<S> = S extends ZodType ? z.infer<S> : S extends { static: inf
|
|
|
743
743
|
export type RawArgumentRejectionCode =
|
|
744
744
|
| "ask-intent-review-requires-positive-round"
|
|
745
745
|
| "ask-intent-contract-requires-non-empty-authority"
|
|
746
|
-
| "ask-deep-interview-metadata-requires-deep-interview-gate"
|
|
746
|
+
| "ask-deep-interview-metadata-requires-deep-interview-gate"
|
|
747
|
+
| "todo-write-unknown-root-key"
|
|
748
|
+
| "todo-write-unknown-op-entry-key"
|
|
749
|
+
| "todo-write-done-drop-requires-target"
|
|
750
|
+
| "todo-write-unknown-init-entry-key";
|
|
747
751
|
|
|
748
752
|
export type RawArgumentValidationResult =
|
|
749
753
|
| { outcome: "passthrough" }
|
|
@@ -947,8 +951,10 @@ export interface AnthropicCompat extends ToolChoiceCompat {
|
|
|
947
951
|
supportsLongCacheRetention?: boolean;
|
|
948
952
|
/**
|
|
949
953
|
* Prompt-cache transport accepted by this Anthropic-compatible endpoint.
|
|
950
|
-
* Canonical Anthropic
|
|
951
|
-
* to `"none"
|
|
954
|
+
* Canonical Anthropic and Claude-family models default to `"automatic"`;
|
|
955
|
+
* noncanonical non-Claude endpoints default to `"none"`. Set `"automatic"` to
|
|
956
|
+
* opt an otherwise unknown compatible endpoint into top-level caching, `"none"`
|
|
957
|
+
* to opt out, or `"explicit"` for endpoints that require block-level markers.
|
|
952
958
|
*/
|
|
953
959
|
promptCacheMode?: "none" | "explicit" | "automatic";
|
|
954
960
|
}
|
|
@@ -68,6 +68,37 @@ export function getStreamFirstEventTimeoutMs(
|
|
|
68
68
|
return normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallback);
|
|
69
69
|
}
|
|
70
70
|
|
|
71
|
+
/**
|
|
72
|
+
* Resolves the OpenAI SDK client `timeout` so stalled-before-headers requests are
|
|
73
|
+
* bounded by the same first-event window the transport watchdog uses after
|
|
74
|
+
* `create()` returns. Without this, providers that only arm
|
|
75
|
+
* `iterateWithIdleTimeout` post-setup can wait the full SDK default (10 minutes
|
|
76
|
+
* per attempt) before any provider-owned watchdog exists.
|
|
77
|
+
*
|
|
78
|
+
* - Explicit `0` disables the request timeout (the SDK treats `timeout: 0` as an
|
|
79
|
+
* immediate failure, so callers that disable the first-event watchdog must not
|
|
80
|
+
* pass a timeout).
|
|
81
|
+
* - Providers with a first-event fallback (Alibaba, Kimi) honor an explicit
|
|
82
|
+
* nonzero override as-is, even when shorter than the fallback.
|
|
83
|
+
* - Other providers floor an explicit override at the env/default first-event
|
|
84
|
+
* window so a short post-connect first-event budget cannot kill legitimate
|
|
85
|
+
* slow setup.
|
|
86
|
+
*/
|
|
87
|
+
export function resolveOpenAISdkRequestTimeoutMs(
|
|
88
|
+
provider: string,
|
|
89
|
+
streamFirstEventTimeoutOverride?: number,
|
|
90
|
+
): number | undefined {
|
|
91
|
+
const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(provider);
|
|
92
|
+
const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
|
|
93
|
+
if (streamFirstEventTimeoutOverride === 0) return undefined;
|
|
94
|
+
if (streamFirstEventTimeoutOverride !== undefined) {
|
|
95
|
+
return providerFirstEventFallbackMs !== undefined
|
|
96
|
+
? streamFirstEventTimeoutOverride
|
|
97
|
+
: Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride);
|
|
98
|
+
}
|
|
99
|
+
return envSdkTimeoutMs;
|
|
100
|
+
}
|
|
101
|
+
|
|
71
102
|
export type Watchdog = NodeJS.Timeout | undefined;
|
|
72
103
|
export class FirstEventTimeoutError extends Error {
|
|
73
104
|
readonly providerCode = STREAM_FIRST_EVENT_TIMEOUT_PROVIDER_CODE;
|