@gajae-code/ai 0.12.10 → 0.12.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -2
- package/dist/types/providers/openai-responses-shared.d.ts +1 -0
- package/dist/types/utils/idle-iterator.d.ts +17 -0
- package/dist/types/utils.d.ts +28 -0
- package/package.json +2 -2
- package/src/model-thinking.ts +11 -5
- package/src/models.json +27 -0
- package/src/providers/azure-openai-responses.ts +20 -4
- package/src/providers/openai-codex-responses.ts +4 -2
- package/src/providers/openai-completions.ts +3 -21
- package/src/providers/openai-responses-shared.ts +26 -0
- package/src/providers/openai-responses.ts +8 -26
- package/src/providers/register-builtins.ts +52 -27
- package/src/utils/idle-iterator.ts +31 -0
- package/src/utils.ts +56 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,9 +2,17 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
-
## [0.12.
|
|
5
|
+
## [0.12.12] - 2026-08-05
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
6
8
|
|
|
7
|
-
|
|
9
|
+
- OpenAI Responses and Azure OpenAI Responses now map the first-event timeout into the SDK request/setup timeout the same way Completions does, so a never-resolving pre-headers fetch on a provider-owned lazy stream cannot wait the SDK's 10-minute default before any transport watchdog exists. Alibaba Responses honors an explicit shorter first-event override before headers; Azure/env-pinned setup timeouts normalize to the typed `stream_first_event_timeout` failure.
|
|
10
|
+
- OpenAI Codex cost estimates now treat an explicit response `service_tier` as authoritative, so a request for priority processing that the provider serves at the default tier is no longer charged the priority multiplier; the requested tier remains the fallback when the terminal response omits the field.
|
|
11
|
+
- Added shared `isReasoningContentReplayError` classifier and `stripUnusableReasoningItems` repair for the DeepSeek-family reasoning-content replay rejection ("The `reasoning_content` in the thinking mode must be passed back to the API"). The classifier detects the error across message carrier shapes; the repair removes only `reasoning` items whose `encrypted_content` a proxy stripped to empty, preserving all non-reasoning history (text, tool calls, tool outputs). The agent loop consumes both for a bounded repair-and-resend circuit breaker.
|
|
12
|
+
|
|
13
|
+
## [0.12.11] - 2026-08-03
|
|
14
|
+
|
|
15
|
+
## [0.12.10] - 2026-08-03
|
|
8
16
|
### Added
|
|
9
17
|
|
|
10
18
|
- Anthropic OAuth can now pair by pasting the authorization code Anthropic displays (`https://platform.claude.com/oauth/code/callback`) instead of waiting on `http://localhost:54545/callback`, so a browser with no network route back to the machine running gjc can complete the login. Opt in per login with `OAuthLoginOptions.manualCode`; the loopback flow stays the default and is unchanged. Callback flows can now opt out of binding a local listener entirely (`OAuthCallbackFlowOptions.skipCallbackServer`), which fails fast when no manual code handler is supplied instead of idling until the five-minute timeout. The hosted redirect is a hard-coded constant with no env or config override, so it cannot be repointed at an attacker-controlled collector.
|
|
@@ -29,6 +37,7 @@
|
|
|
29
37
|
### Fixed
|
|
30
38
|
|
|
31
39
|
- Closed the two remaining ingress holes behind bare `Request Blocked` failures on OpenAI codex models. (1) The chatgpt.com/backend-api pre-model gate rejects with an HTTP 400 bare-`detail` body (`{"detail": "Request blocked."}`) carrying no `error.*` envelope and no `code=invalid_prompt`, so `parseCodexError` surfaced an unexplained message, `isInvalidPromptError` and the codex non-retryable classification missed it, and the session-level `invalid_prompt` circuit breaker never attempted a repaired resend. `parseCodexError` now reads top-level `detail` (string or `{message}`) bodies and classifies a leading `Request blocked` message without an explicit provider code as `invalid_prompt`, surfacing `Request blocked (code=invalid_prompt)` so every existing invalid_prompt contract engages. (2) Outgoing tool definitions (descriptions and JSON-schema strings) bypassed every request-boundary sanitizer on both the OpenAI Responses and OpenAI-codex-responses transports, so a `<|channel|>`-quoting MCP/skill tool description poisoned every request on the session in a way no history repair could fix. Both `convertTools` paths now neutralize reserved control tokens across the whole tool payload via the shared idempotent zero-width-space insertion (ref openai/codex#35838).
|
|
40
|
+
- Lazy built-in streams no longer place a normalized-event watchdog in front of providers that already monitor raw transport progress. This prevents active Anthropic, Azure OpenAI, and OpenAI-family streams from being replaced by a blank `Provider stream stalled while waiting for the next event` error when transport-only events refresh the provider watchdog; providers without their own watchdog keep the shared lazy-stream protection.
|
|
32
41
|
|
|
33
42
|
### Fixed
|
|
34
43
|
|
|
@@ -2,6 +2,7 @@ import type OpenAI from "openai";
|
|
|
2
2
|
import type { ResponseInput, ResponseInputContent, ResponseOutputItem } from "openai/resources/responses/responses";
|
|
3
3
|
import { type Api, type AssistantMessage, type ImageContent, type Model, type ServiceTier, type StopReason, type StreamOptions, type TextContent, type TextSignatureV1, type ToolCall, type ToolResultMessage } from "../types";
|
|
4
4
|
import type { AssistantMessageEventStream } from "../utils/event-stream";
|
|
5
|
+
export declare function isOpenAIResponsesProgressEvent(event: unknown): boolean;
|
|
5
6
|
export declare function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string;
|
|
6
7
|
export declare function parseTextSignature(signature: string | undefined): {
|
|
7
8
|
id: string;
|
|
@@ -29,6 +29,23 @@ export declare function getOpenAIStreamIdleTimeoutMs(): number | undefined;
|
|
|
29
29
|
* env overrides still trump the fallback.
|
|
30
30
|
*/
|
|
31
31
|
export declare function getStreamFirstEventTimeoutMs(idleTimeoutMs?: number, fallbackMs?: number): number | undefined;
|
|
32
|
+
/**
|
|
33
|
+
* Resolves the OpenAI SDK client `timeout` so stalled-before-headers requests are
|
|
34
|
+
* bounded by the same first-event window the transport watchdog uses after
|
|
35
|
+
* `create()` returns. Without this, providers that only arm
|
|
36
|
+
* `iterateWithIdleTimeout` post-setup can wait the full SDK default (10 minutes
|
|
37
|
+
* per attempt) before any provider-owned watchdog exists.
|
|
38
|
+
*
|
|
39
|
+
* - Explicit `0` disables the request timeout (the SDK treats `timeout: 0` as an
|
|
40
|
+
* immediate failure, so callers that disable the first-event watchdog must not
|
|
41
|
+
* pass a timeout).
|
|
42
|
+
* - Providers with a first-event fallback (Alibaba, Kimi) honor an explicit
|
|
43
|
+
* nonzero override as-is, even when shorter than the fallback.
|
|
44
|
+
* - Other providers floor an explicit override at the env/default first-event
|
|
45
|
+
* window so a short post-connect first-event budget cannot kill legitimate
|
|
46
|
+
* slow setup.
|
|
47
|
+
*/
|
|
48
|
+
export declare function resolveOpenAISdkRequestTimeoutMs(provider: string, streamFirstEventTimeoutOverride?: number): number | undefined;
|
|
32
49
|
export type Watchdog = NodeJS.Timeout | undefined;
|
|
33
50
|
export declare class FirstEventTimeoutError extends Error {
|
|
34
51
|
readonly providerCode = "stream_first_event_timeout";
|
package/dist/types/utils.d.ts
CHANGED
|
@@ -59,6 +59,34 @@ export declare function neutralizeReservedControlTokens(text: string): string;
|
|
|
59
59
|
* uniformly testable across transports.
|
|
60
60
|
*/
|
|
61
61
|
export declare function isInvalidPromptError(input: unknown): boolean;
|
|
62
|
+
/**
|
|
63
|
+
* Shape-tolerant classifier for the DeepSeek-family reasoning-content replay
|
|
64
|
+
* rejection: "The `reasoning_content` in the thinking mode must be passed back
|
|
65
|
+
* to the API." DeepSeek V4 (and reasoning-capable siblings reached through any
|
|
66
|
+
* OpenAI-compatible proxy) 400 every follow-up turn once a prior assistant turn
|
|
67
|
+
* carried reasoning the proxy stripped to an empty `encrypted_content` /
|
|
68
|
+
* `reasoning_content`. Resending the identical history re-triggers it, so naive
|
|
69
|
+
* session auto-retry just burns the budget — it needs the same bounded
|
|
70
|
+
* repair-and-resend contract as the `invalid_prompt` poisoned-history breaker.
|
|
71
|
+
*
|
|
72
|
+
* Accepts a raw provider error, an assistant message, or any object carrying an
|
|
73
|
+
* `errorMessage` field (the agent-loop circuit breaker keys on this).
|
|
74
|
+
*/
|
|
75
|
+
export declare function isReasoningContentReplayError(input: unknown): boolean;
|
|
76
|
+
/**
|
|
77
|
+
* Remove Responses-API `reasoning` items whose `encrypted_content` is missing or
|
|
78
|
+
* empty from an outgoing history payload. DeepSeek rejects replay of reasoning
|
|
79
|
+
* whose encrypted blob a proxy stripped to `""`; dropping those items lets the
|
|
80
|
+
* model re-reason on the next turn instead of re-triggering a deterministic 400.
|
|
81
|
+
* Non-reasoning items (text, function_call, function_call_output, ...) are kept
|
|
82
|
+
* verbatim so tool-use pairing and message order are preserved. Returns whether
|
|
83
|
+
* any item was actually removed — the circuit breaker uses this to decide
|
|
84
|
+
* between a single repaired resend (removed) and immediate fail-fast (unchanged).
|
|
85
|
+
*/
|
|
86
|
+
export declare function stripUnusableReasoningItems(items: Array<Record<string, unknown>>): {
|
|
87
|
+
result: Array<Record<string, unknown>>;
|
|
88
|
+
removed: number;
|
|
89
|
+
};
|
|
62
90
|
/**
|
|
63
91
|
* Neutralize leaked reserved control tokens across every string in an outgoing
|
|
64
92
|
* Responses `input` array. This is the request-boundary complement to the
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.12.
|
|
4
|
+
"version": "0.12.12",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.12.
|
|
43
|
+
"@gajae-code/utils": "0.12.12",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/model-thinking.ts
CHANGED
|
@@ -456,11 +456,17 @@ function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
|
|
|
456
456
|
requiresReasoningContentForToolCalls: true,
|
|
457
457
|
};
|
|
458
458
|
}
|
|
459
|
-
// MiniMax-M3
|
|
460
|
-
//
|
|
461
|
-
//
|
|
462
|
-
if (
|
|
463
|
-
model.
|
|
459
|
+
// MiniMax-M3's official Token Plan routes expose a 1M context window.
|
|
460
|
+
// Scope the correction to the four first-class regional MiniMax routes;
|
|
461
|
+
// unrelated catalog aliases and providers keep their own contracts.
|
|
462
|
+
if (
|
|
463
|
+
model.id === "minimax-m3" &&
|
|
464
|
+
(model.provider === "minimax" ||
|
|
465
|
+
model.provider === "minimax-cn" ||
|
|
466
|
+
model.provider === "minimax-code" ||
|
|
467
|
+
model.provider === "minimax-code-cn")
|
|
468
|
+
) {
|
|
469
|
+
model.contextWindow = 1_000_000;
|
|
464
470
|
}
|
|
465
471
|
}
|
|
466
472
|
|
package/src/models.json
CHANGED
|
@@ -89,6 +89,33 @@
|
|
|
89
89
|
"maxLevel": "xhigh"
|
|
90
90
|
}
|
|
91
91
|
},
|
|
92
|
+
"qwen-3.8-max": {
|
|
93
|
+
"id": "qwen-3.8-max",
|
|
94
|
+
"name": "Qwen3.8 Max",
|
|
95
|
+
"api": "openai-responses",
|
|
96
|
+
"provider": "alibaba-token-plan",
|
|
97
|
+
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
|
|
98
|
+
"reasoning": true,
|
|
99
|
+
"input": [
|
|
100
|
+
"text"
|
|
101
|
+
],
|
|
102
|
+
"cost": {
|
|
103
|
+
"input": 0,
|
|
104
|
+
"output": 0,
|
|
105
|
+
"cacheRead": 0,
|
|
106
|
+
"cacheWrite": 0
|
|
107
|
+
},
|
|
108
|
+
"contextWindow": 1000000,
|
|
109
|
+
"maxTokens": 65536,
|
|
110
|
+
"compat": {
|
|
111
|
+
"supportsDeveloperRole": false
|
|
112
|
+
},
|
|
113
|
+
"thinking": {
|
|
114
|
+
"mode": "effort",
|
|
115
|
+
"minLevel": "minimal",
|
|
116
|
+
"maxLevel": "xhigh"
|
|
117
|
+
}
|
|
118
|
+
},
|
|
92
119
|
"qwen3.8-max-preview": {
|
|
93
120
|
"id": "qwen3.8-max-preview",
|
|
94
121
|
"name": "Qwen3.8 Max Preview",
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
|
|
2
|
-
import { AzureOpenAI } from "openai";
|
|
2
|
+
import { APIConnectionTimeoutError, AzureOpenAI } from "openai";
|
|
3
3
|
import type {
|
|
4
4
|
Tool as OpenAITool,
|
|
5
5
|
ResponseCreateParamsStreaming,
|
|
@@ -22,9 +22,11 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
22
22
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
23
23
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
24
24
|
import {
|
|
25
|
+
FirstEventTimeoutError,
|
|
25
26
|
getOpenAIStreamIdleTimeoutMs,
|
|
26
27
|
getStreamFirstEventTimeoutMs,
|
|
27
28
|
iterateWithIdleTimeout,
|
|
29
|
+
resolveOpenAISdkRequestTimeoutMs,
|
|
28
30
|
} from "../utils/idle-iterator";
|
|
29
31
|
import { resolveRetryBudget } from "../utils/retry-budget";
|
|
30
32
|
import { flattenToolRootCombinators, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
|
|
@@ -44,6 +46,7 @@ import {
|
|
|
44
46
|
convertResponsesAssistantMessage,
|
|
45
47
|
convertResponsesInputContent,
|
|
46
48
|
createInitialResponsesAssistantMessage,
|
|
49
|
+
isOpenAIResponsesProgressEvent,
|
|
47
50
|
normalizeResponsesToolCallIdForTransform,
|
|
48
51
|
processResponsesStream,
|
|
49
52
|
} from "./openai-responses-shared";
|
|
@@ -108,6 +111,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
108
111
|
(async () => {
|
|
109
112
|
const startTime = Date.now();
|
|
110
113
|
let firstTokenTime: number | undefined;
|
|
114
|
+
let streamConnected = false;
|
|
111
115
|
const deploymentName = resolveDeploymentName(model, options);
|
|
112
116
|
|
|
113
117
|
const output: AssistantMessage = createInitialResponsesAssistantMessage(
|
|
@@ -162,6 +166,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
162
166
|
rawRequestDump = { ...rawRequestDump, body: params };
|
|
163
167
|
openaiStream = await client.responses.create(params, { signal: requestSignal });
|
|
164
168
|
}
|
|
169
|
+
streamConnected = true;
|
|
165
170
|
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
|
|
166
171
|
stream.push({ type: "start", partial: output });
|
|
167
172
|
|
|
@@ -173,6 +178,8 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
173
178
|
errorMessage: "Azure OpenAI responses stream stalled while waiting for the next event",
|
|
174
179
|
onIdle: () => requestAbortController.abort(),
|
|
175
180
|
onFirstItemTimeout: () => requestAbortController.abort(),
|
|
181
|
+
isProgressItem: isOpenAIResponsesProgressEvent,
|
|
182
|
+
abortSignal: options?.signal,
|
|
176
183
|
}),
|
|
177
184
|
output,
|
|
178
185
|
stream,
|
|
@@ -203,10 +210,15 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
203
210
|
} catch (error) {
|
|
204
211
|
for (const block of output.content) delete (block as { index?: number }).index;
|
|
205
212
|
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
|
213
|
+
const normalizedError =
|
|
214
|
+
!streamConnected && error instanceof APIConnectionTimeoutError
|
|
215
|
+
? new FirstEventTimeoutError(AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE)
|
|
216
|
+
: error;
|
|
206
217
|
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
207
|
-
output.errorStatus = extractHttpStatusFromError(
|
|
208
|
-
output.transportFailure = transportFailureFacts(
|
|
209
|
-
output.errorMessage =
|
|
218
|
+
output.errorStatus = extractHttpStatusFromError(firstEventTimeoutError ?? normalizedError);
|
|
219
|
+
output.transportFailure = transportFailureFacts(firstEventTimeoutError ?? normalizedError);
|
|
220
|
+
output.errorMessage =
|
|
221
|
+
firstEventTimeoutError?.message ?? (await finalizeErrorMessage(normalizedError, rawRequestDump));
|
|
210
222
|
output.duration = Date.now() - startTime;
|
|
211
223
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
212
224
|
stream.push({ type: "error", reason: output.stopReason, error: output });
|
|
@@ -305,6 +317,9 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
|
|
|
305
317
|
|
|
306
318
|
const baseFetch = wrapOpenAIFetchForBoundedRateLimits(options?.fetch ?? fetch, options?.maxRetryDelayMs);
|
|
307
319
|
const onSseEvent = options?.onSseEvent;
|
|
320
|
+
// Bound HTTP request timeout to the first-event window so a stalled-before-headers
|
|
321
|
+
// fetch cannot wait the SDK's 10-minute default before the transport watchdog arms.
|
|
322
|
+
const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, options?.streamFirstEventTimeoutMs);
|
|
308
323
|
return new AzureOpenAI({
|
|
309
324
|
apiKey,
|
|
310
325
|
apiVersion,
|
|
@@ -315,6 +330,7 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
|
|
|
315
330
|
fetch: onSseEvent
|
|
316
331
|
? wrapFetchForSseDebug(baseFetch, event => onSseEvent(event, model, options?.attemptScope))
|
|
317
332
|
: baseFetch,
|
|
333
|
+
...(sdkTimeoutMs !== undefined ? { timeout: sdkTimeoutMs } : {}),
|
|
318
334
|
});
|
|
319
335
|
}
|
|
320
336
|
|
|
@@ -555,10 +555,12 @@ function getCodexServiceTierCostMultiplier(
|
|
|
555
555
|
|
|
556
556
|
function resolveCodexCostServiceTier(res: unknown, req?: unknown): ServiceTier | "default" | undefined {
|
|
557
557
|
switch (res) {
|
|
558
|
+
case "auto":
|
|
559
|
+
case "default":
|
|
558
560
|
case "flex":
|
|
559
|
-
|
|
561
|
+
case "scale":
|
|
560
562
|
case "priority":
|
|
561
|
-
return
|
|
563
|
+
return res;
|
|
562
564
|
default:
|
|
563
565
|
if (req === "flex" || req === "priority") {
|
|
564
566
|
return req;
|
|
@@ -52,6 +52,7 @@ import {
|
|
|
52
52
|
getProviderFirstEventTimeoutFallbackMs,
|
|
53
53
|
getStreamFirstEventTimeoutMs,
|
|
54
54
|
iterateWithIdleTimeout,
|
|
55
|
+
resolveOpenAISdkRequestTimeoutMs,
|
|
55
56
|
} from "../utils/idle-iterator";
|
|
56
57
|
import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
|
|
57
58
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
@@ -524,7 +525,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
524
525
|
|
|
525
526
|
try {
|
|
526
527
|
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
|
|
527
|
-
const idleTimeoutMs = getOpenAIStreamIdleTimeoutMs();
|
|
528
|
+
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
|
528
529
|
const {
|
|
529
530
|
client,
|
|
530
531
|
copilotPremiumRequests,
|
|
@@ -1233,26 +1234,7 @@ async function createClient(
|
|
|
1233
1234
|
// The OpenAI SDK's default is 10 minutes per attempt × `maxRetries`, which
|
|
1234
1235
|
// turns a stalled-before-headers fetch into a multi-minute hang invisible
|
|
1235
1236
|
// to the agent loop (the iterator watchdog only arms AFTER `create()` returns).
|
|
1236
|
-
|
|
1237
|
-
// before the agent watchdog would have, surfacing a real error to the catch
|
|
1238
|
-
// in the IIFE.
|
|
1239
|
-
// A caller may raise `StreamOptions.streamFirstEventTimeoutMs` for a slow-
|
|
1240
|
-
// before-headers provider; respect it so the SDK doesn't give up before the
|
|
1241
|
-
// wrapping watchdog arms. Provider-specific fallbacks apply only when the
|
|
1242
|
-
// caller does not pin a value, so an explicit nonzero override must beat that
|
|
1243
|
-
// fallback even when it is shorter. An explicit `0` disables the watchdog,
|
|
1244
|
-
// and the SDK treats `timeout: 0` as an immediate timeout, so do not pass a
|
|
1245
|
-
// request timeout in that case.
|
|
1246
|
-
const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
1247
|
-
const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
|
|
1248
|
-
const sdkTimeoutMs =
|
|
1249
|
-
streamFirstEventTimeoutOverride === 0
|
|
1250
|
-
? undefined
|
|
1251
|
-
: streamFirstEventTimeoutOverride !== undefined
|
|
1252
|
-
? providerFirstEventFallbackMs !== undefined
|
|
1253
|
-
? streamFirstEventTimeoutOverride
|
|
1254
|
-
: Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride)
|
|
1255
|
-
: envSdkTimeoutMs;
|
|
1237
|
+
const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, streamFirstEventTimeoutOverride);
|
|
1256
1238
|
return {
|
|
1257
1239
|
client: new OpenAI({
|
|
1258
1240
|
apiKey,
|
|
@@ -33,6 +33,32 @@ import type { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
33
33
|
import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
|
|
34
34
|
import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
|
|
35
35
|
|
|
36
|
+
const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
|
|
37
|
+
"response.created",
|
|
38
|
+
"response.output_item.added",
|
|
39
|
+
"response.reasoning_summary_part.added",
|
|
40
|
+
"response.reasoning_summary_text.delta",
|
|
41
|
+
"response.reasoning_summary_part.done",
|
|
42
|
+
"response.reasoning_text.delta",
|
|
43
|
+
"response.content_part.added",
|
|
44
|
+
"response.output_text.delta",
|
|
45
|
+
"response.refusal.delta",
|
|
46
|
+
"response.function_call_arguments.delta",
|
|
47
|
+
"response.function_call_arguments.done",
|
|
48
|
+
"response.custom_tool_call_input.delta",
|
|
49
|
+
"response.custom_tool_call_input.done",
|
|
50
|
+
"response.output_item.done",
|
|
51
|
+
"response.completed",
|
|
52
|
+
"response.failed",
|
|
53
|
+
"error",
|
|
54
|
+
]);
|
|
55
|
+
|
|
56
|
+
export function isOpenAIResponsesProgressEvent(event: unknown): boolean {
|
|
57
|
+
if (!event || typeof event !== "object") return false;
|
|
58
|
+
const type = (event as { type?: unknown }).type;
|
|
59
|
+
return typeof type === "string" && OPENAI_RESPONSES_PROGRESS_EVENT_TYPES.has(type);
|
|
60
|
+
}
|
|
61
|
+
|
|
36
62
|
export function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string {
|
|
37
63
|
const payload: TextSignatureV1 = { v: 1, id };
|
|
38
64
|
if (phase) payload.phase = phase;
|
|
@@ -43,6 +43,7 @@ import {
|
|
|
43
43
|
getProviderFirstEventTimeoutFallbackMs,
|
|
44
44
|
getStreamFirstEventTimeoutMs,
|
|
45
45
|
iterateWithIdleTimeout,
|
|
46
|
+
resolveOpenAISdkRequestTimeoutMs,
|
|
46
47
|
} from "../utils/idle-iterator";
|
|
47
48
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
48
49
|
import { notifyProviderResponse } from "../utils/provider-response";
|
|
@@ -85,6 +86,7 @@ import {
|
|
|
85
86
|
convertResponsesAssistantMessage,
|
|
86
87
|
convertResponsesInputContent,
|
|
87
88
|
createInitialResponsesAssistantMessage,
|
|
89
|
+
isOpenAIResponsesProgressEvent,
|
|
88
90
|
normalizeResponsesToolCallIdForTransform,
|
|
89
91
|
processResponsesStream,
|
|
90
92
|
repairOrphanResponsesToolOutputs,
|
|
@@ -229,32 +231,6 @@ function appendQueryToRequest(input: string | URL | Request, query?: OpenAIRespo
|
|
|
229
231
|
return url;
|
|
230
232
|
}
|
|
231
233
|
|
|
232
|
-
const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
|
|
233
|
-
"response.created",
|
|
234
|
-
"response.output_item.added",
|
|
235
|
-
"response.reasoning_summary_part.added",
|
|
236
|
-
"response.reasoning_summary_text.delta",
|
|
237
|
-
"response.reasoning_summary_part.done",
|
|
238
|
-
"response.reasoning_text.delta",
|
|
239
|
-
"response.content_part.added",
|
|
240
|
-
"response.output_text.delta",
|
|
241
|
-
"response.refusal.delta",
|
|
242
|
-
"response.function_call_arguments.delta",
|
|
243
|
-
"response.function_call_arguments.done",
|
|
244
|
-
"response.custom_tool_call_input.delta",
|
|
245
|
-
"response.custom_tool_call_input.done",
|
|
246
|
-
"response.output_item.done",
|
|
247
|
-
"response.completed",
|
|
248
|
-
"response.failed",
|
|
249
|
-
"error",
|
|
250
|
-
]);
|
|
251
|
-
|
|
252
|
-
function isOpenAIResponsesProgressEvent(event: unknown): boolean {
|
|
253
|
-
if (!event || typeof event !== "object") return false;
|
|
254
|
-
const type = (event as { type?: unknown }).type;
|
|
255
|
-
return typeof type === "string" && OPENAI_RESPONSES_PROGRESS_EVENT_TYPES.has(type);
|
|
256
|
-
}
|
|
257
|
-
|
|
258
234
|
interface OpenAIResponsesProviderSessionState extends ProviderSessionState {
|
|
259
235
|
nativeHistoryReplayWarmed: boolean;
|
|
260
236
|
}
|
|
@@ -343,6 +319,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
343
319
|
options?.requestMaxRetries,
|
|
344
320
|
options?.maxRetryDelayMs,
|
|
345
321
|
options?.attemptScope,
|
|
322
|
+
options?.streamFirstEventTimeoutMs,
|
|
346
323
|
);
|
|
347
324
|
const premiumRequestsTotal = copilotPremiumRequests;
|
|
348
325
|
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
|
|
@@ -498,6 +475,7 @@ function createClient(
|
|
|
498
475
|
requestMaxRetries?: number,
|
|
499
476
|
maxRetryDelayMs?: number,
|
|
500
477
|
attemptScope?: import("../types.js").AttemptScopeRef,
|
|
478
|
+
streamFirstEventTimeoutOverride?: number,
|
|
501
479
|
): {
|
|
502
480
|
client: OpenAI;
|
|
503
481
|
copilotPremiumRequests: number | undefined;
|
|
@@ -568,6 +546,9 @@ function createClient(
|
|
|
568
546
|
model.requestTransform,
|
|
569
547
|
`Gajae-Code/${packageJson.version}`,
|
|
570
548
|
);
|
|
549
|
+
// Bound HTTP request timeout to the first-event window so a stalled-before-headers
|
|
550
|
+
// fetch cannot wait the SDK's 10-minute default before the transport watchdog arms.
|
|
551
|
+
const sdkTimeoutMs = resolveOpenAISdkRequestTimeoutMs(model.provider, streamFirstEventTimeoutOverride);
|
|
571
552
|
return {
|
|
572
553
|
client: new OpenAI({
|
|
573
554
|
apiKey,
|
|
@@ -578,6 +559,7 @@ function createClient(
|
|
|
578
559
|
fetch: onSseEvent
|
|
579
560
|
? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model, attemptScope))
|
|
580
561
|
: transformedFetch,
|
|
562
|
+
...(sdkTimeoutMs !== undefined ? { timeout: sdkTimeoutMs } : {}),
|
|
581
563
|
}),
|
|
582
564
|
copilotPremiumRequests,
|
|
583
565
|
baseUrl,
|
|
@@ -167,6 +167,7 @@ export function setBedrockProviderModule(module: BedrockProviderModule): void {
|
|
|
167
167
|
|
|
168
168
|
const LAZY_STREAM_IDLE_TIMEOUT_ERROR = "Provider stream stalled while waiting for the next event";
|
|
169
169
|
const LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR = "Provider stream timed out while waiting for the first event";
|
|
170
|
+
const LAZY_STREAM_NON_PROGRESS_EVENT_TYPES = new Set(["start", "toolChoiceIncapability"]);
|
|
170
171
|
|
|
171
172
|
function hasFinalResult(
|
|
172
173
|
source: AsyncIterable<AssistantMessageEvent>,
|
|
@@ -184,8 +185,14 @@ function hasFinalResult(
|
|
|
184
185
|
interface LazyStreamLimits {
|
|
185
186
|
defaultFirstEventTimeoutMs?: number;
|
|
186
187
|
defaultIdleTimeoutMs?: number;
|
|
188
|
+
/** The provider already watches raw transport events, which this normalized wrapper cannot observe. */
|
|
189
|
+
providerOwnsWatchdog?: boolean;
|
|
187
190
|
}
|
|
188
191
|
|
|
192
|
+
const PROVIDER_OWNED_STREAM_WATCHDOG: LazyStreamLimits = {
|
|
193
|
+
providerOwnsWatchdog: true,
|
|
194
|
+
};
|
|
195
|
+
|
|
189
196
|
/**
|
|
190
197
|
* Cloud Code Assist (google-gemini-cli / google-antigravity) routinely takes
|
|
191
198
|
* longer than the global 100s default to emit its first SSE event when serving
|
|
@@ -224,28 +231,34 @@ function forwardStream<TApi extends Api>(
|
|
|
224
231
|
): void {
|
|
225
232
|
(async () => {
|
|
226
233
|
try {
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
234
|
+
let watchedSource = source;
|
|
235
|
+
if (!limits?.providerOwnsWatchdog) {
|
|
236
|
+
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
|
|
237
|
+
const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
|
|
238
|
+
model.provider,
|
|
239
|
+
limits?.defaultFirstEventTimeoutMs,
|
|
240
|
+
);
|
|
241
|
+
watchedSource = iterateWithIdleTimeout(source, {
|
|
242
|
+
idleTimeoutMs,
|
|
243
|
+
firstItemTimeoutMs:
|
|
244
|
+
options.streamFirstEventTimeoutMs ??
|
|
245
|
+
getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
246
|
+
errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
|
|
247
|
+
firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
|
|
248
|
+
onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
|
|
249
|
+
onFirstItemTimeout: () =>
|
|
250
|
+
abortTracker.abortLocally(new FirstEventTimeoutError(LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR)),
|
|
251
|
+
abortSignal: options.signal,
|
|
252
|
+
// Synthetic starts and tool-capability negotiation are control-plane events,
|
|
253
|
+
// not model progress. Keep the first-event window active until assistant output
|
|
254
|
+
// arrives instead of switching early to the shorter idle timeout.
|
|
255
|
+
isProgressItem: event => {
|
|
256
|
+
if (!event || typeof event !== "object") return true;
|
|
257
|
+
const eventType = (event as { type?: unknown }).type;
|
|
258
|
+
return typeof eventType !== "string" || !LAZY_STREAM_NON_PROGRESS_EVENT_TYPES.has(eventType);
|
|
259
|
+
},
|
|
260
|
+
});
|
|
261
|
+
}
|
|
249
262
|
|
|
250
263
|
for await (const event of watchedSource) {
|
|
251
264
|
target.push(event);
|
|
@@ -423,17 +436,29 @@ function loadBedrockProviderModule(): Promise<LazyProviderModule<"bedrock-conver
|
|
|
423
436
|
// are loaded on first use instead of during package initialization.
|
|
424
437
|
// ---------------------------------------------------------------------------
|
|
425
438
|
|
|
426
|
-
export const streamAnthropic = createLazyStream(loadAnthropicProviderModule);
|
|
427
|
-
export const streamAzureOpenAIResponses = createLazyStream(
|
|
439
|
+
export const streamAnthropic = createLazyStream(loadAnthropicProviderModule, PROVIDER_OWNED_STREAM_WATCHDOG);
|
|
440
|
+
export const streamAzureOpenAIResponses = createLazyStream(
|
|
441
|
+
loadAzureOpenAIResponsesProviderModule,
|
|
442
|
+
PROVIDER_OWNED_STREAM_WATCHDOG,
|
|
443
|
+
);
|
|
428
444
|
export const streamGoogle = createLazyStream(loadGoogleProviderModule);
|
|
429
445
|
export const streamGoogleGeminiCli = createLazyStream(
|
|
430
446
|
loadGoogleGeminiCliProviderModule,
|
|
431
447
|
GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS,
|
|
432
448
|
);
|
|
433
449
|
export const streamGoogleVertex = createLazyStream(loadGoogleVertexProviderModule);
|
|
434
|
-
export const streamOpenAICodexResponses = createLazyStream(
|
|
435
|
-
|
|
436
|
-
|
|
450
|
+
export const streamOpenAICodexResponses = createLazyStream(
|
|
451
|
+
loadOpenAICodexResponsesProviderModule,
|
|
452
|
+
PROVIDER_OWNED_STREAM_WATCHDOG,
|
|
453
|
+
);
|
|
454
|
+
export const streamOpenAICompletions = createLazyStream(
|
|
455
|
+
loadOpenAICompletionsProviderModule,
|
|
456
|
+
PROVIDER_OWNED_STREAM_WATCHDOG,
|
|
457
|
+
);
|
|
458
|
+
export const streamOpenAIResponses = createLazyStream(
|
|
459
|
+
loadOpenAIResponsesProviderModule,
|
|
460
|
+
PROVIDER_OWNED_STREAM_WATCHDOG,
|
|
461
|
+
);
|
|
437
462
|
export const streamCursor = createLazyStream(loadCursorProviderModule);
|
|
438
463
|
export const streamOllama = createLazyStream(loadOllamaProviderModule);
|
|
439
464
|
|
|
@@ -68,6 +68,37 @@ export function getStreamFirstEventTimeoutMs(
|
|
|
68
68
|
return normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallback);
|
|
69
69
|
}
|
|
70
70
|
|
|
71
|
+
/**
|
|
72
|
+
* Resolves the OpenAI SDK client `timeout` so stalled-before-headers requests are
|
|
73
|
+
* bounded by the same first-event window the transport watchdog uses after
|
|
74
|
+
* `create()` returns. Without this, providers that only arm
|
|
75
|
+
* `iterateWithIdleTimeout` post-setup can wait the full SDK default (10 minutes
|
|
76
|
+
* per attempt) before any provider-owned watchdog exists.
|
|
77
|
+
*
|
|
78
|
+
* - Explicit `0` disables the request timeout (the SDK treats `timeout: 0` as an
|
|
79
|
+
* immediate failure, so callers that disable the first-event watchdog must not
|
|
80
|
+
* pass a timeout).
|
|
81
|
+
* - Providers with a first-event fallback (Alibaba, Kimi) honor an explicit
|
|
82
|
+
* nonzero override as-is, even when shorter than the fallback.
|
|
83
|
+
* - Other providers floor an explicit override at the env/default first-event
|
|
84
|
+
* window so a short post-connect first-event budget cannot kill legitimate
|
|
85
|
+
* slow setup.
|
|
86
|
+
*/
|
|
87
|
+
export function resolveOpenAISdkRequestTimeoutMs(
|
|
88
|
+
provider: string,
|
|
89
|
+
streamFirstEventTimeoutOverride?: number,
|
|
90
|
+
): number | undefined {
|
|
91
|
+
const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(provider);
|
|
92
|
+
const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
|
|
93
|
+
if (streamFirstEventTimeoutOverride === 0) return undefined;
|
|
94
|
+
if (streamFirstEventTimeoutOverride !== undefined) {
|
|
95
|
+
return providerFirstEventFallbackMs !== undefined
|
|
96
|
+
? streamFirstEventTimeoutOverride
|
|
97
|
+
: Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride);
|
|
98
|
+
}
|
|
99
|
+
return envSdkTimeoutMs;
|
|
100
|
+
}
|
|
101
|
+
|
|
71
102
|
export type Watchdog = NodeJS.Timeout | undefined;
|
|
72
103
|
export class FirstEventTimeoutError extends Error {
|
|
73
104
|
readonly providerCode = STREAM_FIRST_EVENT_TIMEOUT_PROVIDER_CODE;
|
package/src/utils.ts
CHANGED
|
@@ -199,6 +199,62 @@ function asLowerString(value: unknown): string | undefined {
|
|
|
199
199
|
return typeof value === "string" ? value.toLowerCase() : undefined;
|
|
200
200
|
}
|
|
201
201
|
|
|
202
|
+
/**
|
|
203
|
+
* Shape-tolerant classifier for the DeepSeek-family reasoning-content replay
|
|
204
|
+
* rejection: "The `reasoning_content` in the thinking mode must be passed back
|
|
205
|
+
* to the API." DeepSeek V4 (and reasoning-capable siblings reached through any
|
|
206
|
+
* OpenAI-compatible proxy) 400 every follow-up turn once a prior assistant turn
|
|
207
|
+
* carried reasoning the proxy stripped to an empty `encrypted_content` /
|
|
208
|
+
* `reasoning_content`. Resending the identical history re-triggers it, so naive
|
|
209
|
+
* session auto-retry just burns the budget — it needs the same bounded
|
|
210
|
+
* repair-and-resend contract as the `invalid_prompt` poisoned-history breaker.
|
|
211
|
+
*
|
|
212
|
+
* Accepts a raw provider error, an assistant message, or any object carrying an
|
|
213
|
+
* `errorMessage` field (the agent-loop circuit breaker keys on this).
|
|
214
|
+
*/
|
|
215
|
+
export function isReasoningContentReplayError(input: unknown): boolean {
|
|
216
|
+
if (!input) return false;
|
|
217
|
+
const message =
|
|
218
|
+
typeof input === "string"
|
|
219
|
+
? input
|
|
220
|
+
: input && typeof input === "object"
|
|
221
|
+
? ((input as { errorMessage?: unknown; message?: unknown }).errorMessage ??
|
|
222
|
+
(input as { message?: unknown }).message)
|
|
223
|
+
: undefined;
|
|
224
|
+
return typeof message === "string" && REASONING_CONTENT_REPLAY_MESSAGE_RE.test(message);
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
const REASONING_CONTENT_REPLAY_MESSAGE_RE =
|
|
228
|
+
/reasoning_content[\s\S]*must be passed back to the API|reasoning content[\s\S]*must be passed back to the API/i;
|
|
229
|
+
|
|
230
|
+
/**
|
|
231
|
+
* Remove Responses-API `reasoning` items whose `encrypted_content` is missing or
|
|
232
|
+
* empty from an outgoing history payload. DeepSeek rejects replay of reasoning
|
|
233
|
+
* whose encrypted blob a proxy stripped to `""`; dropping those items lets the
|
|
234
|
+
* model re-reason on the next turn instead of re-triggering a deterministic 400.
|
|
235
|
+
* Non-reasoning items (text, function_call, function_call_output, ...) are kept
|
|
236
|
+
* verbatim so tool-use pairing and message order are preserved. Returns whether
|
|
237
|
+
* any item was actually removed — the circuit breaker uses this to decide
|
|
238
|
+
* between a single repaired resend (removed) and immediate fail-fast (unchanged).
|
|
239
|
+
*/
|
|
240
|
+
export function stripUnusableReasoningItems(items: Array<Record<string, unknown>>): {
|
|
241
|
+
result: Array<Record<string, unknown>>;
|
|
242
|
+
removed: number;
|
|
243
|
+
} {
|
|
244
|
+
let removed = 0;
|
|
245
|
+
const result: Array<Record<string, unknown>> = [];
|
|
246
|
+
for (const item of items) {
|
|
247
|
+
if (item?.type === "reasoning") {
|
|
248
|
+
const encrypted = item.encrypted_content;
|
|
249
|
+
if (encrypted === undefined || encrypted === null || encrypted === "") {
|
|
250
|
+
removed++;
|
|
251
|
+
continue;
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
result.push(item);
|
|
255
|
+
}
|
|
256
|
+
return { result, removed };
|
|
257
|
+
}
|
|
202
258
|
/**
|
|
203
259
|
* Neutralize leaked reserved control tokens across every string in an outgoing
|
|
204
260
|
* Responses `input` array. This is the request-boundary complement to the
|