@gajae-code/ai 0.10.2 → 0.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -2
- package/dist/types/auth-gateway/server.d.ts +19 -0
- package/dist/types/auth-storage.d.ts +2 -0
- package/dist/types/context-cap-policy.d.ts +10 -0
- package/dist/types/index.d.ts +2 -0
- package/dist/types/model-cache.d.ts +2 -0
- package/dist/types/providers/openai-codex/response-handler.d.ts +1 -0
- package/dist/types/providers/openai-codex-responses.d.ts +3 -2
- package/dist/types/providers/pi-native-server.d.ts +3 -3
- package/dist/types/types.d.ts +12 -4
- package/dist/types/utils/event-stream.d.ts +9 -3
- package/dist/types/utils/fallback-transport.d.ts +55 -0
- package/dist/types/utils/retry.d.ts +1 -0
- package/dist/types/utils.d.ts +29 -0
- package/package.json +2 -2
- package/src/auth-gateway/server.ts +164 -22
- package/src/auth-storage.ts +62 -40
- package/src/context-cap-policy.ts +59 -0
- package/src/index.ts +2 -0
- package/src/model-cache.ts +10 -0
- package/src/model-manager.ts +12 -7
- package/src/model-thinking.ts +7 -8
- package/src/providers/amazon-bedrock.ts +9 -1
- package/src/providers/anthropic.ts +6 -0
- package/src/providers/azure-openai-responses.ts +6 -1
- package/src/providers/google-gemini-cli.ts +26 -13
- package/src/providers/google-shared.ts +7 -1
- package/src/providers/ollama.ts +7 -1
- package/src/providers/openai-codex/response-handler.ts +11 -3
- package/src/providers/openai-codex-responses.ts +29 -6
- package/src/providers/openai-completions.ts +13 -2
- package/src/providers/openai-responses.ts +24 -2
- package/src/providers/pi-native-client.ts +24 -12
- package/src/providers/pi-native-server.ts +4 -3
- package/src/stream.ts +31 -2
- package/src/types.ts +27 -4
- package/src/utils/discovery/codex.ts +3 -12
- package/src/utils/event-stream.ts +138 -54
- package/src/utils/fallback-transport.ts +185 -0
- package/src/utils/retry.ts +2 -2
- package/src/utils.ts +64 -1
|
@@ -48,6 +48,7 @@ import {
|
|
|
48
48
|
sanitizeOpenAIResponsesHistoryItemsForReplay,
|
|
49
49
|
} from "../utils";
|
|
50
50
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
51
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
51
52
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
52
53
|
import { getOpenAIStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
|
|
53
54
|
import { parseStreamingJson } from "../utils/json-parse";
|
|
@@ -112,9 +113,15 @@ const CODEX_NON_RETRYABLE_EVENT_CODES = new Set([
|
|
|
112
113
|
"invalid_request_error",
|
|
113
114
|
"invalid_schema",
|
|
114
115
|
"invalid_tool_schema",
|
|
116
|
+
// A poisoned-history rejection (`Request blocked (code=invalid_prompt)`) is a
|
|
117
|
+
// deterministic content fault, not a transient upstream failure: retrying the
|
|
118
|
+
// same request re-sends the same offending item and re-triggers the block, so
|
|
119
|
+
// classify it as explicitly non-retryable instead of relying on omission
|
|
120
|
+
// (the request-boundary sanitizer, not a provider retry, is the recovery path).
|
|
121
|
+
"invalid_prompt",
|
|
115
122
|
]);
|
|
116
123
|
const CODEX_NON_RETRYABLE_EVENT_MESSAGE =
|
|
117
|
-
/invalid[_ -]function[_ -]parameters|invalid schema for function|invalid[_ -]tool[_ -]schema|schema must have type ["']?object["']
|
|
124
|
+
/invalid[_ -]function[_ -]parameters|invalid schema for function|invalid[_ -]tool[_ -]schema|schema must have type ["']?object["']?|request blocked[^\n]*invalid[_ -]prompt|code=invalid[_ -]prompt/i;
|
|
118
125
|
const CODEX_RETRYABLE_EVENT_MESSAGE =
|
|
119
126
|
/processing your request|retry your request|temporar(?:y|ily)|overloaded|service.?unavailable|internal error|server error/i;
|
|
120
127
|
const CODEX_PROVIDER_SESSION_STATE_KEY = "openai-codex-responses";
|
|
@@ -1373,6 +1380,7 @@ async function recoverCodexStreamError(
|
|
|
1373
1380
|
runtime: CodexStreamRuntime,
|
|
1374
1381
|
error: unknown,
|
|
1375
1382
|
): Promise<boolean> {
|
|
1383
|
+
if (context.options?.fallbackManaged) return false;
|
|
1376
1384
|
if (await tryRetryWithoutForcedToolChoice(context, runtime, error)) {
|
|
1377
1385
|
return true;
|
|
1378
1386
|
}
|
|
@@ -1397,6 +1405,7 @@ async function tryRetryWithoutForcedToolChoice(
|
|
|
1397
1405
|
error: unknown,
|
|
1398
1406
|
): Promise<boolean> {
|
|
1399
1407
|
if (
|
|
1408
|
+
context.options?.fallbackManaged ||
|
|
1400
1409
|
runtime.providerRetryAttempt > 0 ||
|
|
1401
1410
|
context.output.content.length > 0 ||
|
|
1402
1411
|
context.firstTokenTime !== undefined ||
|
|
@@ -1469,7 +1478,12 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
|
|
|
1469
1478
|
return false;
|
|
1470
1479
|
}
|
|
1471
1480
|
const websocketState = context.requestContext.websocketState;
|
|
1472
|
-
if (
|
|
1481
|
+
if (
|
|
1482
|
+
!websocketState ||
|
|
1483
|
+
runtime.transport !== "websocket" ||
|
|
1484
|
+
context.options?.signal?.aborted ||
|
|
1485
|
+
context.options?.fallbackManaged
|
|
1486
|
+
) {
|
|
1473
1487
|
return false;
|
|
1474
1488
|
}
|
|
1475
1489
|
|
|
@@ -1520,6 +1534,7 @@ async function tryRecoverCodexPreviousResponseNotFound(
|
|
|
1520
1534
|
if (
|
|
1521
1535
|
!isCodexPreviousResponseNotFound(error) ||
|
|
1522
1536
|
!websocketState ||
|
|
1537
|
+
context.options?.fallbackManaged ||
|
|
1523
1538
|
runtime.transport !== "websocket" ||
|
|
1524
1539
|
context.output.content.length > 0 ||
|
|
1525
1540
|
context.options?.signal?.aborted ||
|
|
@@ -1557,7 +1572,8 @@ async function tryReplayWebsocketFailureOverSse(
|
|
|
1557
1572
|
isCodexWebSocketRetryableStreamError(error) &&
|
|
1558
1573
|
runtime.canSafelyReplayWebsocketOverSse &&
|
|
1559
1574
|
!runtime.sawTerminalEvent &&
|
|
1560
|
-
!context.options?.signal?.aborted
|
|
1575
|
+
!context.options?.signal?.aborted &&
|
|
1576
|
+
!context.options?.fallbackManaged;
|
|
1561
1577
|
if (!canReplay) return false;
|
|
1562
1578
|
|
|
1563
1579
|
const state = websocketState;
|
|
@@ -1609,7 +1625,8 @@ async function tryRetryCodexProviderError(
|
|
|
1609
1625
|
!isRetryableCodexProviderError(error) ||
|
|
1610
1626
|
context.output.content.length > 0 ||
|
|
1611
1627
|
runtime.providerRetryAttempt >= resolveRetryBudget(context.options?.streamMaxRetries, CODEX_MAX_RETRIES) ||
|
|
1612
|
-
context.options?.signal?.aborted
|
|
1628
|
+
context.options?.signal?.aborted ||
|
|
1629
|
+
context.options?.fallbackManaged
|
|
1613
1630
|
) {
|
|
1614
1631
|
return false;
|
|
1615
1632
|
}
|
|
@@ -1693,6 +1710,7 @@ async function handleCodexStreamFailure(
|
|
|
1693
1710
|
}
|
|
1694
1711
|
output.stopReason = context.options?.signal?.aborted ? "aborted" : "error";
|
|
1695
1712
|
output.errorStatus = extractHttpStatusFromError(error);
|
|
1713
|
+
output.transportFailure = transportFailureFacts(error);
|
|
1696
1714
|
output.errorMessage = await finalizeErrorMessage(error, context.requestContext.rawRequestDump);
|
|
1697
1715
|
output.duration = Date.now() - context.startTime;
|
|
1698
1716
|
if (context.firstTokenTime) {
|
|
@@ -1720,6 +1738,7 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
|
|
|
1720
1738
|
try {
|
|
1721
1739
|
initialTransport = await openInitialCodexEventStream(model, options, requestSetup, requestContext);
|
|
1722
1740
|
} catch (error) {
|
|
1741
|
+
if (options?.fallbackManaged) throw error;
|
|
1723
1742
|
initialTransport = await retryCodexInitialTransportWithoutToolChoice(
|
|
1724
1743
|
model,
|
|
1725
1744
|
options,
|
|
@@ -2409,6 +2428,7 @@ async function openCodexSseEventStream(
|
|
|
2409
2428
|
const error = new Error(info.friendlyMessage || info.message);
|
|
2410
2429
|
(error as { headers?: Headers; status?: number }).headers = response.headers;
|
|
2411
2430
|
(error as { headers?: Headers; status?: number }).status = response.status;
|
|
2431
|
+
(error as { code?: string }).code = info.code;
|
|
2412
2432
|
throw error;
|
|
2413
2433
|
}
|
|
2414
2434
|
if (!response.body) {
|
|
@@ -2636,8 +2656,11 @@ function normalizeInputMessageContent(
|
|
|
2636
2656
|
return convertResponsesInputContent(content, model.input.includes("image")) ?? [];
|
|
2637
2657
|
}
|
|
2638
2658
|
|
|
2639
|
-
/** @internal Exported for tests. */
|
|
2640
|
-
export {
|
|
2659
|
+
/** @internal Exported for tests. `classifyCodexFailureEventRetryable` is the retry classification of a Codex failure event. */
|
|
2660
|
+
export {
|
|
2661
|
+
convertMessages as convertCodexResponsesMessages,
|
|
2662
|
+
isRetryableCodexFailureEvent as classifyCodexFailureEventRetryable,
|
|
2663
|
+
};
|
|
2641
2664
|
|
|
2642
2665
|
/**
|
|
2643
2666
|
* Whether this OpenAI code backend-backend model should get the custom-tool grammar
|
|
@@ -38,6 +38,7 @@ import {
|
|
|
38
38
|
import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
|
|
39
39
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
40
40
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
41
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
41
42
|
import { toFirepassWireModelId, toFireworksWireModelId } from "../utils/fireworks-model-id";
|
|
42
43
|
import {
|
|
43
44
|
type CapturedHttpErrorResponse,
|
|
@@ -512,13 +513,18 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
512
513
|
openaiStream = await callWithCopilotModelRetry(() => createCompletionsStream(), {
|
|
513
514
|
provider: model.provider,
|
|
514
515
|
signal: requestSignal,
|
|
516
|
+
fallbackManaged: options?.fallbackManaged,
|
|
515
517
|
});
|
|
516
518
|
} catch (error) {
|
|
517
519
|
const capturedErrorResponse = getCapturedErrorResponse();
|
|
518
520
|
const sentForcedToolChoice = isForcedToolChoice(
|
|
519
521
|
(rawRequestDump?.body as { tool_choice?: unknown } | undefined)?.tool_choice,
|
|
520
522
|
);
|
|
521
|
-
if (
|
|
523
|
+
if (
|
|
524
|
+
!options?.fallbackManaged &&
|
|
525
|
+
firstTokenTime === undefined &&
|
|
526
|
+
isForcedToolChoiceUnsupportedError(error, sentForcedToolChoice)
|
|
527
|
+
) {
|
|
522
528
|
const reason = await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse);
|
|
523
529
|
markToolChoiceIncapability(model, "auto", reason);
|
|
524
530
|
const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
|
|
@@ -534,6 +540,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
534
540
|
});
|
|
535
541
|
openaiStream = await createCompletionsStream();
|
|
536
542
|
} else if (
|
|
543
|
+
!options?.fallbackManaged &&
|
|
537
544
|
isOpenRouterAnthropicModel(model) &&
|
|
538
545
|
!disableStrictTools &&
|
|
539
546
|
isCompiledGrammarTooLargeStrictError(error, capturedErrorResponse)
|
|
@@ -546,7 +553,10 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
546
553
|
disableStrictTools = true;
|
|
547
554
|
openaiStream = await createCompletionsStream("none");
|
|
548
555
|
} else {
|
|
549
|
-
if (
|
|
556
|
+
if (
|
|
557
|
+
options?.fallbackManaged ||
|
|
558
|
+
!shouldRetryWithoutStrictTools(error, capturedErrorResponse, appliedToolStrictMode, context.tools)
|
|
559
|
+
) {
|
|
550
560
|
throw error;
|
|
551
561
|
}
|
|
552
562
|
openaiStream = await createCompletionsStream("none");
|
|
@@ -960,6 +970,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
960
970
|
const capturedErrorResponse = getCapturedErrorResponse?.();
|
|
961
971
|
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
962
972
|
output.errorStatus = extractHttpStatusFromError(error) ?? capturedErrorResponse?.status;
|
|
973
|
+
output.transportFailure = transportFailureFacts(error, capturedErrorResponse);
|
|
963
974
|
output.errorMessage =
|
|
964
975
|
firstEventTimeoutError?.message ??
|
|
965
976
|
(await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse));
|
|
@@ -33,6 +33,7 @@ import {
|
|
|
33
33
|
createOpenAIResponsesHistoryPayload,
|
|
34
34
|
getOpenAIResponsesHistoryItems,
|
|
35
35
|
getOpenAIResponsesHistoryPayload,
|
|
36
|
+
isInvalidPromptError,
|
|
36
37
|
neutralizeResponsesInputControlTokens,
|
|
37
38
|
normalizeSystemPrompts,
|
|
38
39
|
resolveCacheRetention,
|
|
@@ -40,6 +41,7 @@ import {
|
|
|
40
41
|
} from "../utils";
|
|
41
42
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
42
43
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
44
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
43
45
|
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
|
|
44
46
|
import {
|
|
45
47
|
createWatchdog,
|
|
@@ -299,9 +301,12 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
299
301
|
await notifyProviderResponse(options, response, model, request_id);
|
|
300
302
|
return data;
|
|
301
303
|
},
|
|
302
|
-
{ provider: model.provider, signal: requestSignal },
|
|
304
|
+
{ provider: model.provider, signal: requestSignal, fallbackManaged: options?.fallbackManaged },
|
|
303
305
|
).catch(async error => {
|
|
304
|
-
if (
|
|
306
|
+
if (
|
|
307
|
+
options?.fallbackManaged ||
|
|
308
|
+
!isForcedToolChoiceUnsupportedError(error, isForcedOpenAIResponsesToolChoice(params.tool_choice))
|
|
309
|
+
) {
|
|
305
310
|
throw error;
|
|
306
311
|
}
|
|
307
312
|
const reason = await finalizeErrorMessage(error, rawRequestDump);
|
|
@@ -380,8 +385,25 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
380
385
|
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
|
381
386
|
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
382
387
|
output.errorStatus = extractHttpStatusFromError(error);
|
|
388
|
+
output.transportFailure = transportFailureFacts(error);
|
|
383
389
|
output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
384
390
|
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
|
|
391
|
+
// Explicitly mark the poisoned-history rejection so the shared
|
|
392
|
+
// `invalid_prompt` contract is present even when the SDK error surfaces
|
|
393
|
+
// only a message (no structured code). This keeps the responses
|
|
394
|
+
// transport's classification uniform with the codex transport's
|
|
395
|
+
// non-retryable event set and lets the session-level circuit breaker
|
|
396
|
+
// key on one durable marker instead of per-transport string matching.
|
|
397
|
+
if (
|
|
398
|
+
output.stopReason === "error" &&
|
|
399
|
+
!output.transportFailure?.providerCode &&
|
|
400
|
+
(isInvalidPromptError(error) || isInvalidPromptError(output.errorMessage))
|
|
401
|
+
) {
|
|
402
|
+
output.transportFailure = {
|
|
403
|
+
...(output.transportFailure ?? { kind: "transport" }),
|
|
404
|
+
providerCode: "invalid_prompt",
|
|
405
|
+
};
|
|
406
|
+
}
|
|
385
407
|
output.duration = Date.now() - startTime;
|
|
386
408
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
387
409
|
stream.push({ type: "error", reason: output.stopReason, error: output });
|
|
@@ -11,8 +11,8 @@
|
|
|
11
11
|
*
|
|
12
12
|
* Activated when a {@link Model} has `transport: "pi-native"` set; the
|
|
13
13
|
* dispatch hook lives in `streamSimple()` (see `../stream.ts`). Used by
|
|
14
|
-
* containerized
|
|
15
|
-
*
|
|
14
|
+
* containerized GJC deployments that route every LLM call through a
|
|
15
|
+
* credential-holding sidecar so the container stays credential-free.
|
|
16
16
|
*/
|
|
17
17
|
import { readSseJson } from "@gajae-code/utils";
|
|
18
18
|
import type {
|
|
@@ -44,6 +44,7 @@ const NON_WIRE_KEYS = new Set<keyof SimpleStreamOptions>([
|
|
|
44
44
|
"cursorExecHandlers",
|
|
45
45
|
"cursorOnToolResult",
|
|
46
46
|
"providerSessionState",
|
|
47
|
+
"fallbackAttempt",
|
|
47
48
|
]);
|
|
48
49
|
|
|
49
50
|
function buildWireOptions(options: SimpleStreamOptions | undefined): Record<string, unknown> {
|
|
@@ -70,15 +71,27 @@ async function decodeGatewayError(response: Response): Promise<Error> {
|
|
|
70
71
|
if (typeof err === "object" && err !== null) {
|
|
71
72
|
const message = (err as { message?: unknown }).message;
|
|
72
73
|
const type = (err as { type?: unknown }).type;
|
|
74
|
+
const code = (err as { code?: unknown }).code;
|
|
73
75
|
const out = new Error(typeof message === "string" ? message : `auth-gateway ${status}`);
|
|
74
|
-
|
|
75
|
-
|
|
76
|
+
const transportError = out as Error & {
|
|
77
|
+
status?: number;
|
|
78
|
+
type?: string;
|
|
79
|
+
providerCode?: string;
|
|
80
|
+
headers?: Headers;
|
|
81
|
+
};
|
|
82
|
+
transportError.status = status;
|
|
83
|
+
transportError.headers = response.headers;
|
|
84
|
+
if (typeof type === "string") transportError.type = type;
|
|
85
|
+
if (typeof code === "string") transportError.providerCode = code;
|
|
86
|
+
else if (typeof type === "string") transportError.providerCode = type;
|
|
76
87
|
return out;
|
|
77
88
|
}
|
|
78
89
|
}
|
|
79
90
|
const text = typeof body === "string" ? body : JSON.stringify(body);
|
|
80
91
|
const err = new Error(`auth-gateway ${status}: ${text || response.statusText}`);
|
|
81
|
-
|
|
92
|
+
const transportError = err as Error & { status?: number; headers?: Headers };
|
|
93
|
+
transportError.status = status;
|
|
94
|
+
transportError.headers = response.headers;
|
|
82
95
|
return err;
|
|
83
96
|
}
|
|
84
97
|
|
|
@@ -180,19 +193,18 @@ export function streamPiNative<TApi extends Api>(
|
|
|
180
193
|
}
|
|
181
194
|
|
|
182
195
|
if (!sawTerminal) {
|
|
183
|
-
// SSE closed before a terminal event reached us — synthesize one
|
|
184
|
-
// so awaiters of `.result()` resolve instead of hanging forever.
|
|
185
|
-
// Matches the gateway's own defensive fallback in
|
|
186
|
-
// `pi-native-server.encodeStream`.
|
|
187
196
|
const aborted = signal?.aborted === true;
|
|
188
|
-
const partial = makeSyntheticAssistant(model as Model<Api>);
|
|
189
197
|
if (aborted) {
|
|
198
|
+
const partial = makeSyntheticAssistant(model as Model<Api>);
|
|
190
199
|
partial.stopReason = "aborted";
|
|
191
200
|
partial.errorMessage = "stream closed without terminal event";
|
|
192
201
|
stream.push({ type: "error", reason: "aborted", error: partial });
|
|
193
202
|
} else {
|
|
194
|
-
|
|
195
|
-
|
|
203
|
+
const error = Object.assign(new Error("pi-native SSE stream closed without terminal event"), {
|
|
204
|
+
status: 502,
|
|
205
|
+
headers: response.headers,
|
|
206
|
+
});
|
|
207
|
+
stream.fail(error);
|
|
196
208
|
}
|
|
197
209
|
}
|
|
198
210
|
stream.end();
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* Where the OpenAI / Anthropic / Responses route modules translate foreign
|
|
5
5
|
* wire shapes through pi-ai's canonical {@link Context}, this module accepts
|
|
6
6
|
* the canonical shape *directly* — for clients that already speak pi-ai
|
|
7
|
-
* (containerized
|
|
7
|
+
* (containerized GJC deployments and sidecar auth gateways).
|
|
8
8
|
* Skipping the wire-format → Context → wire-format round-trip cuts
|
|
9
9
|
* per-request CPU but, more importantly, avoids the quantization that those
|
|
10
10
|
* translations impose on first-class pi-ai fields (service tier, cache
|
|
@@ -56,6 +56,7 @@ const ALLOWED_OPTION_KEYS: ReadonlySet<keyof SimpleStreamOptions> = new Set([
|
|
|
56
56
|
"headers",
|
|
57
57
|
"initiatorOverride",
|
|
58
58
|
"maxRetryDelayMs",
|
|
59
|
+
"fallbackManaged",
|
|
59
60
|
"metadata",
|
|
60
61
|
"sessionId",
|
|
61
62
|
"streamFirstEventTimeoutMs",
|
|
@@ -154,8 +155,8 @@ const SSE_DONE = SSE_ENCODER.encode("data: [DONE]\n\n");
|
|
|
154
155
|
* canonical event type IS the wire type. Including the rolling
|
|
155
156
|
* `partial: AssistantMessage` on every delta is quadratic in turn length
|
|
156
157
|
* on the wire, but for the loopback / sidecar topology this transport
|
|
157
|
-
* targets (containerized
|
|
158
|
-
*
|
|
158
|
+
* targets (containerized GJC → host gateway) the bandwidth cost is negligible
|
|
159
|
+
* compared to provider latency —
|
|
159
160
|
* and the client gets to feed the events straight into its existing
|
|
160
161
|
* `AssistantMessageEventStream.push()` plumbing with zero translation.
|
|
161
162
|
*/
|
package/src/stream.ts
CHANGED
|
@@ -2,6 +2,18 @@ import * as fs from "node:fs";
|
|
|
2
2
|
import * as os from "node:os";
|
|
3
3
|
import * as path from "node:path";
|
|
4
4
|
import { $credentialEnv, $env, $pickCredentialEnv, extractHttpStatusFromError } from "@gajae-code/utils";
|
|
5
|
+
import { assertManagedAttempt } from "./utils/fallback-transport";
|
|
6
|
+
|
|
7
|
+
const managedAttemptValidated = Symbol("managedAttemptValidated");
|
|
8
|
+
|
|
9
|
+
function hasValidatedManagedAttempt(options: object | undefined): boolean {
|
|
10
|
+
return (options as Record<symbol, unknown> | undefined)?.[managedAttemptValidated] === true;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
function markManagedAttemptValidated<T extends object>(options: T): T {
|
|
14
|
+
return Object.assign(options, { [managedAttemptValidated]: true });
|
|
15
|
+
}
|
|
16
|
+
|
|
5
17
|
import { getCustomApi } from "./api-registry";
|
|
6
18
|
import type { Effort } from "./model-thinking";
|
|
7
19
|
import {
|
|
@@ -276,6 +288,10 @@ export function stream<TApi extends Api>(
|
|
|
276
288
|
context: Context,
|
|
277
289
|
options?: OptionsForApi<TApi>,
|
|
278
290
|
): AssistantMessageEventStream {
|
|
291
|
+
if (!hasValidatedManagedAttempt(options)) assertManagedAttempt(options);
|
|
292
|
+
if (options?.fallbackManaged) {
|
|
293
|
+
options = { ...options, requestMaxRetries: 0, streamMaxRetries: 0 } as OptionsForApi<TApi>;
|
|
294
|
+
}
|
|
279
295
|
// Check custom API registry first (extension-provided APIs like "vertex-Anthropic model-api")
|
|
280
296
|
const customApiProvider = getCustomApi(model.api);
|
|
281
297
|
if (customApiProvider) {
|
|
@@ -395,6 +411,16 @@ export function streamSimple<TApi extends Api>(
|
|
|
395
411
|
context: Context,
|
|
396
412
|
options?: SimpleStreamOptions,
|
|
397
413
|
): AssistantMessageEventStream {
|
|
414
|
+
assertManagedAttempt(options);
|
|
415
|
+
if (options?.fallbackManaged) {
|
|
416
|
+
options = {
|
|
417
|
+
...options,
|
|
418
|
+
requestMaxRetries: 0,
|
|
419
|
+
streamMaxRetries: 0,
|
|
420
|
+
onAuthError: undefined,
|
|
421
|
+
};
|
|
422
|
+
options = markManagedAttemptValidated(options);
|
|
423
|
+
}
|
|
398
424
|
const retryApiKey = options?.onAuthError ? (options.apiKey ?? getEnvApiKey(model.provider)) : undefined;
|
|
399
425
|
if (retryApiKey) {
|
|
400
426
|
const outer = new AssistantMessageEventStream();
|
|
@@ -662,12 +688,14 @@ function mapOptionsForApi<TApi extends Api>(
|
|
|
662
688
|
maxTokens: options?.maxTokens || Math.min(model.maxTokens, 32000),
|
|
663
689
|
signal: options?.signal,
|
|
664
690
|
apiKey: apiKey || options?.apiKey,
|
|
691
|
+
fallbackManaged: options?.fallbackManaged,
|
|
692
|
+
fallbackAttempt: options?.fallbackAttempt,
|
|
665
693
|
cacheRetention: options?.cacheRetention ?? model.cacheRetention,
|
|
666
694
|
headers: options?.headers,
|
|
667
695
|
initiatorOverride: options?.initiatorOverride,
|
|
668
696
|
maxRetryDelayMs: options?.maxRetryDelayMs,
|
|
669
|
-
requestMaxRetries: options?.requestMaxRetries,
|
|
670
|
-
streamMaxRetries: options?.streamMaxRetries,
|
|
697
|
+
requestMaxRetries: options?.fallbackManaged ? 0 : options?.requestMaxRetries,
|
|
698
|
+
streamMaxRetries: options?.fallbackManaged ? 0 : options?.streamMaxRetries,
|
|
671
699
|
metadata: options?.metadata,
|
|
672
700
|
sessionId: options?.sessionId,
|
|
673
701
|
providerSessionState: options?.providerSessionState,
|
|
@@ -675,6 +703,7 @@ function mapOptionsForApi<TApi extends Api>(
|
|
|
675
703
|
onResponse: options?.onResponse,
|
|
676
704
|
onSseEvent: options?.onSseEvent,
|
|
677
705
|
execHandlers: options?.execHandlers,
|
|
706
|
+
[managedAttemptValidated]: hasValidatedManagedAttempt(options),
|
|
678
707
|
};
|
|
679
708
|
|
|
680
709
|
switch (model.api) {
|
package/src/types.ts
CHANGED
|
@@ -28,6 +28,7 @@ import type { OpenAICodexResponsesOptions } from "./providers/openai-codex-respo
|
|
|
28
28
|
import type { OpenAICompletionsOptions } from "./providers/openai-completions";
|
|
29
29
|
import type { OpenAIResponsesOptions } from "./providers/openai-responses";
|
|
30
30
|
import type { AssistantMessageEventStream } from "./utils/event-stream";
|
|
31
|
+
import type { FallbackAttemptToken, TransportFailureFacts } from "./utils/fallback-transport";
|
|
31
32
|
|
|
32
33
|
export type { AssistantMessageEventStream } from "./utils/event-stream";
|
|
33
34
|
|
|
@@ -77,6 +78,23 @@ export type ThinkingControlMode =
|
|
|
77
78
|
| "anthropic-adaptive"
|
|
78
79
|
| "anthropic-budget-effort";
|
|
79
80
|
|
|
81
|
+
/** Canonical runtime vocabulary for provider thinking transports. */
|
|
82
|
+
export const THINKING_CONTROL_MODES = [
|
|
83
|
+
"effort",
|
|
84
|
+
"budget",
|
|
85
|
+
"google-level",
|
|
86
|
+
"anthropic-adaptive",
|
|
87
|
+
"anthropic-budget-effort",
|
|
88
|
+
] as const satisfies readonly ThinkingControlMode[];
|
|
89
|
+
|
|
90
|
+
type _CheckThinkingControlModes = [
|
|
91
|
+
Exclude<ThinkingControlMode, (typeof THINKING_CONTROL_MODES)[number]>,
|
|
92
|
+
Exclude<(typeof THINKING_CONTROL_MODES)[number], ThinkingControlMode>,
|
|
93
|
+
] extends [never, never]
|
|
94
|
+
? true
|
|
95
|
+
: false;
|
|
96
|
+
true satisfies _CheckThinkingControlModes;
|
|
97
|
+
|
|
80
98
|
/** Per-model thinking capabilities used to clamp and map user-facing effort levels. */
|
|
81
99
|
export interface ThinkingConfig {
|
|
82
100
|
/** Least intensive supported user-facing effort level. */
|
|
@@ -306,6 +324,10 @@ export interface StreamOptions {
|
|
|
306
324
|
maxTokens?: number;
|
|
307
325
|
signal?: AbortSignal;
|
|
308
326
|
apiKey?: string;
|
|
327
|
+
/** Disables all transport-level replay; the fallback controller owns retries. */
|
|
328
|
+
fallbackManaged?: boolean;
|
|
329
|
+
/** Opaque token returned by beginAttempt for a managed transport invocation. */
|
|
330
|
+
fallbackAttempt?: FallbackAttemptToken;
|
|
309
331
|
/**
|
|
310
332
|
* Called when a provider returns 401 before any replay-unsafe assistant
|
|
311
333
|
* event has been emitted. Returning a different key retries the provider
|
|
@@ -598,6 +620,8 @@ export interface AssistantMessage {
|
|
|
598
620
|
errorKind?: AssistantErrorKind;
|
|
599
621
|
/** HTTP status surfaced by the provider when the request failed. Populated by every provider's catch block alongside `errorMessage` so consumers (auth retry, telemetry, UI) can branch without regex-scraping the message. */
|
|
600
622
|
errorStatus?: number;
|
|
623
|
+
/** Typed upstream failure facts retained for retry classification without parsing errorMessage. */
|
|
624
|
+
transportFailure?: TransportFailureFacts;
|
|
601
625
|
/**
|
|
602
626
|
* Stable identifiers for request features the provider silently dropped
|
|
603
627
|
* during this turn (e.g. `"priority"`). Set when a server-side rejection
|
|
@@ -939,10 +963,9 @@ export interface Model<TApi extends Api = any> {
|
|
|
939
963
|
* (or compatible) host; `headers.Authorization` (or `apiKey` resolved by
|
|
940
964
|
* the registry) carries the gateway bearer.
|
|
941
965
|
*
|
|
942
|
-
* Used by containerized
|
|
943
|
-
*
|
|
944
|
-
*
|
|
945
|
-
* thinking config, …) still resolves locally; only the streaming
|
|
966
|
+
* Used by containerized GJC installs to route every LLM call through a
|
|
967
|
+
* sidecar gateway that holds the real provider credentials. The model's other
|
|
968
|
+
* metadata (pricing, context window, thinking config, …) still resolves locally; only the streaming
|
|
946
969
|
* dispatch is redirected.
|
|
947
970
|
*/
|
|
948
971
|
transport?: "pi-native";
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import * as z from "zod/v4";
|
|
2
|
+
import { resolveCodexGpt56DiscoveryContext } from "../../context-cap-policy";
|
|
2
3
|
import { CODEX_BASE_URL, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "../../providers/openai-codex/constants";
|
|
3
4
|
import type { Model } from "../../types";
|
|
4
5
|
import { isRecord } from "../../utils";
|
|
5
6
|
|
|
6
7
|
const DEFAULT_MODEL_LIST_PATHS = ["/codex/models", "/models"] as const;
|
|
7
|
-
const DEFAULT_CONTEXT_WINDOW = 272_000;
|
|
8
8
|
const DEFAULT_MAX_TOKENS = 128_000;
|
|
9
9
|
const DEFAULT_CODEX_CLIENT_VERSION = "0.99.0";
|
|
10
10
|
const NPM_CODEX_LATEST_URL = "https://registry.npmjs.org/@openai%2Fcodex/latest";
|
|
@@ -258,7 +258,8 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo
|
|
|
258
258
|
}
|
|
259
259
|
|
|
260
260
|
const name = toNonEmptyString(payload.display_name) ?? slug;
|
|
261
|
-
const
|
|
261
|
+
const modelIdentity = { id: slug, api: "openai-codex-responses", provider: "openai-codex" } as const;
|
|
262
|
+
const contextWindow = resolveCodexGpt56DiscoveryContext(modelIdentity, payload.context_window);
|
|
262
263
|
const maxTokens = Math.min(DEFAULT_MAX_TOKENS, contextWindow);
|
|
263
264
|
const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels);
|
|
264
265
|
const input = normalizeInputModalities(payload.input_modalities);
|
|
@@ -346,16 +347,6 @@ function toNonEmptyString(value: unknown): string | null {
|
|
|
346
347
|
return trimmed.length > 0 ? trimmed : null;
|
|
347
348
|
}
|
|
348
349
|
|
|
349
|
-
function toPositiveInt(value: unknown): number | null {
|
|
350
|
-
if (typeof value !== "number" || !Number.isFinite(value)) {
|
|
351
|
-
return null;
|
|
352
|
-
}
|
|
353
|
-
if (value <= 0) {
|
|
354
|
-
return null;
|
|
355
|
-
}
|
|
356
|
-
return Math.trunc(value);
|
|
357
|
-
}
|
|
358
|
-
|
|
359
350
|
function toFiniteNumber(value: unknown): number | null {
|
|
360
351
|
if (typeof value !== "number" || !Number.isFinite(value)) {
|
|
361
352
|
return null;
|