@gajae-code/ai 0.10.2 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/CHANGELOG.md +13 -2
  2. package/dist/types/auth-gateway/server.d.ts +19 -0
  3. package/dist/types/auth-storage.d.ts +2 -0
  4. package/dist/types/context-cap-policy.d.ts +10 -0
  5. package/dist/types/index.d.ts +2 -0
  6. package/dist/types/providers/openai-codex/response-handler.d.ts +1 -0
  7. package/dist/types/providers/pi-native-server.d.ts +3 -3
  8. package/dist/types/types.d.ts +12 -4
  9. package/dist/types/utils/event-stream.d.ts +9 -3
  10. package/dist/types/utils/fallback-transport.d.ts +55 -0
  11. package/dist/types/utils/retry.d.ts +1 -0
  12. package/dist/types/utils.d.ts +17 -0
  13. package/package.json +2 -2
  14. package/src/auth-gateway/server.ts +164 -22
  15. package/src/auth-storage.ts +62 -40
  16. package/src/context-cap-policy.ts +59 -0
  17. package/src/index.ts +2 -0
  18. package/src/model-manager.ts +12 -7
  19. package/src/model-thinking.ts +7 -8
  20. package/src/providers/amazon-bedrock.ts +9 -1
  21. package/src/providers/anthropic.ts +6 -0
  22. package/src/providers/azure-openai-responses.ts +6 -1
  23. package/src/providers/google-gemini-cli.ts +26 -13
  24. package/src/providers/google-shared.ts +7 -1
  25. package/src/providers/ollama.ts +7 -1
  26. package/src/providers/openai-codex/response-handler.ts +11 -3
  27. package/src/providers/openai-codex-responses.ts +17 -3
  28. package/src/providers/openai-completions.ts +13 -2
  29. package/src/providers/openai-responses.ts +7 -2
  30. package/src/providers/pi-native-client.ts +24 -12
  31. package/src/providers/pi-native-server.ts +4 -3
  32. package/src/stream.ts +31 -2
  33. package/src/types.ts +27 -4
  34. package/src/utils/discovery/codex.ts +3 -12
  35. package/src/utils/event-stream.ts +138 -54
  36. package/src/utils/fallback-transport.ts +185 -0
  37. package/src/utils/retry.ts +2 -2
  38. package/src/utils.ts +19 -1
@@ -48,6 +48,7 @@ import {
48
48
  sanitizeOpenAIResponsesHistoryItemsForReplay,
49
49
  } from "../utils";
50
50
  import { AssistantMessageEventStream } from "../utils/event-stream";
51
+ import { transportFailureFacts } from "../utils/fallback-transport";
51
52
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
52
53
  import { getOpenAIStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
53
54
  import { parseStreamingJson } from "../utils/json-parse";
@@ -1373,6 +1374,7 @@ async function recoverCodexStreamError(
1373
1374
  runtime: CodexStreamRuntime,
1374
1375
  error: unknown,
1375
1376
  ): Promise<boolean> {
1377
+ if (context.options?.fallbackManaged) return false;
1376
1378
  if (await tryRetryWithoutForcedToolChoice(context, runtime, error)) {
1377
1379
  return true;
1378
1380
  }
@@ -1397,6 +1399,7 @@ async function tryRetryWithoutForcedToolChoice(
1397
1399
  error: unknown,
1398
1400
  ): Promise<boolean> {
1399
1401
  if (
1402
+ context.options?.fallbackManaged ||
1400
1403
  runtime.providerRetryAttempt > 0 ||
1401
1404
  context.output.content.length > 0 ||
1402
1405
  context.firstTokenTime !== undefined ||
@@ -1469,7 +1472,12 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
1469
1472
  return false;
1470
1473
  }
1471
1474
  const websocketState = context.requestContext.websocketState;
1472
- if (!websocketState || runtime.transport !== "websocket" || context.options?.signal?.aborted) {
1475
+ if (
1476
+ !websocketState ||
1477
+ runtime.transport !== "websocket" ||
1478
+ context.options?.signal?.aborted ||
1479
+ context.options?.fallbackManaged
1480
+ ) {
1473
1481
  return false;
1474
1482
  }
1475
1483
 
@@ -1520,6 +1528,7 @@ async function tryRecoverCodexPreviousResponseNotFound(
1520
1528
  if (
1521
1529
  !isCodexPreviousResponseNotFound(error) ||
1522
1530
  !websocketState ||
1531
+ context.options?.fallbackManaged ||
1523
1532
  runtime.transport !== "websocket" ||
1524
1533
  context.output.content.length > 0 ||
1525
1534
  context.options?.signal?.aborted ||
@@ -1557,7 +1566,8 @@ async function tryReplayWebsocketFailureOverSse(
1557
1566
  isCodexWebSocketRetryableStreamError(error) &&
1558
1567
  runtime.canSafelyReplayWebsocketOverSse &&
1559
1568
  !runtime.sawTerminalEvent &&
1560
- !context.options?.signal?.aborted;
1569
+ !context.options?.signal?.aborted &&
1570
+ !context.options?.fallbackManaged;
1561
1571
  if (!canReplay) return false;
1562
1572
 
1563
1573
  const state = websocketState;
@@ -1609,7 +1619,8 @@ async function tryRetryCodexProviderError(
1609
1619
  !isRetryableCodexProviderError(error) ||
1610
1620
  context.output.content.length > 0 ||
1611
1621
  runtime.providerRetryAttempt >= resolveRetryBudget(context.options?.streamMaxRetries, CODEX_MAX_RETRIES) ||
1612
- context.options?.signal?.aborted
1622
+ context.options?.signal?.aborted ||
1623
+ context.options?.fallbackManaged
1613
1624
  ) {
1614
1625
  return false;
1615
1626
  }
@@ -1693,6 +1704,7 @@ async function handleCodexStreamFailure(
1693
1704
  }
1694
1705
  output.stopReason = context.options?.signal?.aborted ? "aborted" : "error";
1695
1706
  output.errorStatus = extractHttpStatusFromError(error);
1707
+ output.transportFailure = transportFailureFacts(error);
1696
1708
  output.errorMessage = await finalizeErrorMessage(error, context.requestContext.rawRequestDump);
1697
1709
  output.duration = Date.now() - context.startTime;
1698
1710
  if (context.firstTokenTime) {
@@ -1720,6 +1732,7 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
1720
1732
  try {
1721
1733
  initialTransport = await openInitialCodexEventStream(model, options, requestSetup, requestContext);
1722
1734
  } catch (error) {
1735
+ if (options?.fallbackManaged) throw error;
1723
1736
  initialTransport = await retryCodexInitialTransportWithoutToolChoice(
1724
1737
  model,
1725
1738
  options,
@@ -2409,6 +2422,7 @@ async function openCodexSseEventStream(
2409
2422
  const error = new Error(info.friendlyMessage || info.message);
2410
2423
  (error as { headers?: Headers; status?: number }).headers = response.headers;
2411
2424
  (error as { headers?: Headers; status?: number }).status = response.status;
2425
+ (error as { code?: string }).code = info.code;
2412
2426
  throw error;
2413
2427
  }
2414
2428
  if (!response.body) {
@@ -38,6 +38,7 @@ import {
38
38
  import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
39
39
  import { createAbortSourceTracker } from "../utils/abort";
40
40
  import { AssistantMessageEventStream } from "../utils/event-stream";
41
+ import { transportFailureFacts } from "../utils/fallback-transport";
41
42
  import { toFirepassWireModelId, toFireworksWireModelId } from "../utils/fireworks-model-id";
42
43
  import {
43
44
  type CapturedHttpErrorResponse,
@@ -512,13 +513,18 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
512
513
  openaiStream = await callWithCopilotModelRetry(() => createCompletionsStream(), {
513
514
  provider: model.provider,
514
515
  signal: requestSignal,
516
+ fallbackManaged: options?.fallbackManaged,
515
517
  });
516
518
  } catch (error) {
517
519
  const capturedErrorResponse = getCapturedErrorResponse();
518
520
  const sentForcedToolChoice = isForcedToolChoice(
519
521
  (rawRequestDump?.body as { tool_choice?: unknown } | undefined)?.tool_choice,
520
522
  );
521
- if (firstTokenTime === undefined && isForcedToolChoiceUnsupportedError(error, sentForcedToolChoice)) {
523
+ if (
524
+ !options?.fallbackManaged &&
525
+ firstTokenTime === undefined &&
526
+ isForcedToolChoiceUnsupportedError(error, sentForcedToolChoice)
527
+ ) {
522
528
  const reason = await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse);
523
529
  markToolChoiceIncapability(model, "auto", reason);
524
530
  const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
@@ -534,6 +540,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
534
540
  });
535
541
  openaiStream = await createCompletionsStream();
536
542
  } else if (
543
+ !options?.fallbackManaged &&
537
544
  isOpenRouterAnthropicModel(model) &&
538
545
  !disableStrictTools &&
539
546
  isCompiledGrammarTooLargeStrictError(error, capturedErrorResponse)
@@ -546,7 +553,10 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
546
553
  disableStrictTools = true;
547
554
  openaiStream = await createCompletionsStream("none");
548
555
  } else {
549
- if (!shouldRetryWithoutStrictTools(error, capturedErrorResponse, appliedToolStrictMode, context.tools)) {
556
+ if (
557
+ options?.fallbackManaged ||
558
+ !shouldRetryWithoutStrictTools(error, capturedErrorResponse, appliedToolStrictMode, context.tools)
559
+ ) {
550
560
  throw error;
551
561
  }
552
562
  openaiStream = await createCompletionsStream("none");
@@ -960,6 +970,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
960
970
  const capturedErrorResponse = getCapturedErrorResponse?.();
961
971
  output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
962
972
  output.errorStatus = extractHttpStatusFromError(error) ?? capturedErrorResponse?.status;
973
+ output.transportFailure = transportFailureFacts(error, capturedErrorResponse);
963
974
  output.errorMessage =
964
975
  firstEventTimeoutError?.message ??
965
976
  (await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse));
@@ -40,6 +40,7 @@ import {
40
40
  } from "../utils";
41
41
  import { createAbortSourceTracker } from "../utils/abort";
42
42
  import { AssistantMessageEventStream } from "../utils/event-stream";
43
+ import { transportFailureFacts } from "../utils/fallback-transport";
43
44
  import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
44
45
  import {
45
46
  createWatchdog,
@@ -299,9 +300,12 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
299
300
  await notifyProviderResponse(options, response, model, request_id);
300
301
  return data;
301
302
  },
302
- { provider: model.provider, signal: requestSignal },
303
+ { provider: model.provider, signal: requestSignal, fallbackManaged: options?.fallbackManaged },
303
304
  ).catch(async error => {
304
- if (!isForcedToolChoiceUnsupportedError(error, isForcedOpenAIResponsesToolChoice(params.tool_choice))) {
305
+ if (
306
+ options?.fallbackManaged ||
307
+ !isForcedToolChoiceUnsupportedError(error, isForcedOpenAIResponsesToolChoice(params.tool_choice))
308
+ ) {
305
309
  throw error;
306
310
  }
307
311
  const reason = await finalizeErrorMessage(error, rawRequestDump);
@@ -380,6 +384,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
380
384
  const firstEventTimeoutError = abortTracker.getLocalAbortReason();
381
385
  output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
382
386
  output.errorStatus = extractHttpStatusFromError(error);
387
+ output.transportFailure = transportFailureFacts(error);
383
388
  output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
384
389
  output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
385
390
  output.duration = Date.now() - startTime;
@@ -11,8 +11,8 @@
11
11
  *
12
12
  * Activated when a {@link Model} has `transport: "pi-native"` set; the
13
13
  * dispatch hook lives in `streamSimple()` (see `../stream.ts`). Used by
14
- * containerized gjc deployments (e.g. robogjc slots) that route every LLM call
15
- * through a credential-holding sidecar so the slot itself stays credential-free.
14
+ * containerized GJC deployments that route every LLM call through a
15
+ * credential-holding sidecar so the container stays credential-free.
16
16
  */
17
17
  import { readSseJson } from "@gajae-code/utils";
18
18
  import type {
@@ -44,6 +44,7 @@ const NON_WIRE_KEYS = new Set<keyof SimpleStreamOptions>([
44
44
  "cursorExecHandlers",
45
45
  "cursorOnToolResult",
46
46
  "providerSessionState",
47
+ "fallbackAttempt",
47
48
  ]);
48
49
 
49
50
  function buildWireOptions(options: SimpleStreamOptions | undefined): Record<string, unknown> {
@@ -70,15 +71,27 @@ async function decodeGatewayError(response: Response): Promise<Error> {
70
71
  if (typeof err === "object" && err !== null) {
71
72
  const message = (err as { message?: unknown }).message;
72
73
  const type = (err as { type?: unknown }).type;
74
+ const code = (err as { code?: unknown }).code;
73
75
  const out = new Error(typeof message === "string" ? message : `auth-gateway ${status}`);
74
- (out as { status?: number; type?: string }).status = status;
75
- if (typeof type === "string") (out as { type?: string }).type = type;
76
+ const transportError = out as Error & {
77
+ status?: number;
78
+ type?: string;
79
+ providerCode?: string;
80
+ headers?: Headers;
81
+ };
82
+ transportError.status = status;
83
+ transportError.headers = response.headers;
84
+ if (typeof type === "string") transportError.type = type;
85
+ if (typeof code === "string") transportError.providerCode = code;
86
+ else if (typeof type === "string") transportError.providerCode = type;
76
87
  return out;
77
88
  }
78
89
  }
79
90
  const text = typeof body === "string" ? body : JSON.stringify(body);
80
91
  const err = new Error(`auth-gateway ${status}: ${text || response.statusText}`);
81
- (err as { status?: number }).status = status;
92
+ const transportError = err as Error & { status?: number; headers?: Headers };
93
+ transportError.status = status;
94
+ transportError.headers = response.headers;
82
95
  return err;
83
96
  }
84
97
 
@@ -180,19 +193,18 @@ export function streamPiNative<TApi extends Api>(
180
193
  }
181
194
 
182
195
  if (!sawTerminal) {
183
- // SSE closed before a terminal event reached us — synthesize one
184
- // so awaiters of `.result()` resolve instead of hanging forever.
185
- // Matches the gateway's own defensive fallback in
186
- // `pi-native-server.encodeStream`.
187
196
  const aborted = signal?.aborted === true;
188
- const partial = makeSyntheticAssistant(model as Model<Api>);
189
197
  if (aborted) {
198
+ const partial = makeSyntheticAssistant(model as Model<Api>);
190
199
  partial.stopReason = "aborted";
191
200
  partial.errorMessage = "stream closed without terminal event";
192
201
  stream.push({ type: "error", reason: "aborted", error: partial });
193
202
  } else {
194
- partial.stopReason = "stop";
195
- stream.push({ type: "done", reason: "stop", message: partial });
203
+ const error = Object.assign(new Error("pi-native SSE stream closed without terminal event"), {
204
+ status: 502,
205
+ headers: response.headers,
206
+ });
207
+ stream.fail(error);
196
208
  }
197
209
  }
198
210
  stream.end();
@@ -4,7 +4,7 @@
4
4
  * Where the OpenAI / Anthropic / Responses route modules translate foreign
5
5
  * wire shapes through pi-ai's canonical {@link Context}, this module accepts
6
6
  * the canonical shape *directly* — for clients that already speak pi-ai
7
- * (containerized gjc and robogjc's sidecar auth-gateway).
7
+ * (containerized GJC deployments and sidecar auth gateways).
8
8
  * Skipping the wire-format → Context → wire-format round-trip cuts
9
9
  * per-request CPU but, more importantly, avoids the quantization that those
10
10
  * translations impose on first-class pi-ai fields (service tier, cache
@@ -56,6 +56,7 @@ const ALLOWED_OPTION_KEYS: ReadonlySet<keyof SimpleStreamOptions> = new Set([
56
56
  "headers",
57
57
  "initiatorOverride",
58
58
  "maxRetryDelayMs",
59
+ "fallbackManaged",
59
60
  "metadata",
60
61
  "sessionId",
61
62
  "streamFirstEventTimeoutMs",
@@ -154,8 +155,8 @@ const SSE_DONE = SSE_ENCODER.encode("data: [DONE]\n\n");
154
155
  * canonical event type IS the wire type. Including the rolling
155
156
  * `partial: AssistantMessage` on every delta is quadratic in turn length
156
157
  * on the wire, but for the loopback / sidecar topology this transport
157
- * targets (containerized gjc → host gateway, robogjc slot → gjc-auth-gateway
158
- * sidecar) the bandwidth cost is negligible compared to provider latency —
158
+ * targets (containerized GJC → host gateway) the bandwidth cost is negligible
159
+ * compared to provider latency —
159
160
  * and the client gets to feed the events straight into its existing
160
161
  * `AssistantMessageEventStream.push()` plumbing with zero translation.
161
162
  */
package/src/stream.ts CHANGED
@@ -2,6 +2,18 @@ import * as fs from "node:fs";
2
2
  import * as os from "node:os";
3
3
  import * as path from "node:path";
4
4
  import { $credentialEnv, $env, $pickCredentialEnv, extractHttpStatusFromError } from "@gajae-code/utils";
5
+ import { assertManagedAttempt } from "./utils/fallback-transport";
6
+
7
+ const managedAttemptValidated = Symbol("managedAttemptValidated");
8
+
9
+ function hasValidatedManagedAttempt(options: object | undefined): boolean {
10
+ return (options as Record<symbol, unknown> | undefined)?.[managedAttemptValidated] === true;
11
+ }
12
+
13
+ function markManagedAttemptValidated<T extends object>(options: T): T {
14
+ return Object.assign(options, { [managedAttemptValidated]: true });
15
+ }
16
+
5
17
  import { getCustomApi } from "./api-registry";
6
18
  import type { Effort } from "./model-thinking";
7
19
  import {
@@ -276,6 +288,10 @@ export function stream<TApi extends Api>(
276
288
  context: Context,
277
289
  options?: OptionsForApi<TApi>,
278
290
  ): AssistantMessageEventStream {
291
+ if (!hasValidatedManagedAttempt(options)) assertManagedAttempt(options);
292
+ if (options?.fallbackManaged) {
293
+ options = { ...options, requestMaxRetries: 0, streamMaxRetries: 0 } as OptionsForApi<TApi>;
294
+ }
279
295
  // Check custom API registry first (extension-provided APIs like "vertex-Anthropic model-api")
280
296
  const customApiProvider = getCustomApi(model.api);
281
297
  if (customApiProvider) {
@@ -395,6 +411,16 @@ export function streamSimple<TApi extends Api>(
395
411
  context: Context,
396
412
  options?: SimpleStreamOptions,
397
413
  ): AssistantMessageEventStream {
414
+ assertManagedAttempt(options);
415
+ if (options?.fallbackManaged) {
416
+ options = {
417
+ ...options,
418
+ requestMaxRetries: 0,
419
+ streamMaxRetries: 0,
420
+ onAuthError: undefined,
421
+ };
422
+ options = markManagedAttemptValidated(options);
423
+ }
398
424
  const retryApiKey = options?.onAuthError ? (options.apiKey ?? getEnvApiKey(model.provider)) : undefined;
399
425
  if (retryApiKey) {
400
426
  const outer = new AssistantMessageEventStream();
@@ -662,12 +688,14 @@ function mapOptionsForApi<TApi extends Api>(
662
688
  maxTokens: options?.maxTokens || Math.min(model.maxTokens, 32000),
663
689
  signal: options?.signal,
664
690
  apiKey: apiKey || options?.apiKey,
691
+ fallbackManaged: options?.fallbackManaged,
692
+ fallbackAttempt: options?.fallbackAttempt,
665
693
  cacheRetention: options?.cacheRetention ?? model.cacheRetention,
666
694
  headers: options?.headers,
667
695
  initiatorOverride: options?.initiatorOverride,
668
696
  maxRetryDelayMs: options?.maxRetryDelayMs,
669
- requestMaxRetries: options?.requestMaxRetries,
670
- streamMaxRetries: options?.streamMaxRetries,
697
+ requestMaxRetries: options?.fallbackManaged ? 0 : options?.requestMaxRetries,
698
+ streamMaxRetries: options?.fallbackManaged ? 0 : options?.streamMaxRetries,
671
699
  metadata: options?.metadata,
672
700
  sessionId: options?.sessionId,
673
701
  providerSessionState: options?.providerSessionState,
@@ -675,6 +703,7 @@ function mapOptionsForApi<TApi extends Api>(
675
703
  onResponse: options?.onResponse,
676
704
  onSseEvent: options?.onSseEvent,
677
705
  execHandlers: options?.execHandlers,
706
+ [managedAttemptValidated]: hasValidatedManagedAttempt(options),
678
707
  };
679
708
 
680
709
  switch (model.api) {
package/src/types.ts CHANGED
@@ -28,6 +28,7 @@ import type { OpenAICodexResponsesOptions } from "./providers/openai-codex-respo
28
28
  import type { OpenAICompletionsOptions } from "./providers/openai-completions";
29
29
  import type { OpenAIResponsesOptions } from "./providers/openai-responses";
30
30
  import type { AssistantMessageEventStream } from "./utils/event-stream";
31
+ import type { FallbackAttemptToken, TransportFailureFacts } from "./utils/fallback-transport";
31
32
 
32
33
  export type { AssistantMessageEventStream } from "./utils/event-stream";
33
34
 
@@ -77,6 +78,23 @@ export type ThinkingControlMode =
77
78
  | "anthropic-adaptive"
78
79
  | "anthropic-budget-effort";
79
80
 
81
+ /** Canonical runtime vocabulary for provider thinking transports. */
82
+ export const THINKING_CONTROL_MODES = [
83
+ "effort",
84
+ "budget",
85
+ "google-level",
86
+ "anthropic-adaptive",
87
+ "anthropic-budget-effort",
88
+ ] as const satisfies readonly ThinkingControlMode[];
89
+
90
+ type _CheckThinkingControlModes = [
91
+ Exclude<ThinkingControlMode, (typeof THINKING_CONTROL_MODES)[number]>,
92
+ Exclude<(typeof THINKING_CONTROL_MODES)[number], ThinkingControlMode>,
93
+ ] extends [never, never]
94
+ ? true
95
+ : false;
96
+ true satisfies _CheckThinkingControlModes;
97
+
80
98
  /** Per-model thinking capabilities used to clamp and map user-facing effort levels. */
81
99
  export interface ThinkingConfig {
82
100
  /** Least intensive supported user-facing effort level. */
@@ -306,6 +324,10 @@ export interface StreamOptions {
306
324
  maxTokens?: number;
307
325
  signal?: AbortSignal;
308
326
  apiKey?: string;
327
+ /** Disables all transport-level replay; the fallback controller owns retries. */
328
+ fallbackManaged?: boolean;
329
+ /** Opaque token returned by beginAttempt for a managed transport invocation. */
330
+ fallbackAttempt?: FallbackAttemptToken;
309
331
  /**
310
332
  * Called when a provider returns 401 before any replay-unsafe assistant
311
333
  * event has been emitted. Returning a different key retries the provider
@@ -598,6 +620,8 @@ export interface AssistantMessage {
598
620
  errorKind?: AssistantErrorKind;
599
621
  /** HTTP status surfaced by the provider when the request failed. Populated by every provider's catch block alongside `errorMessage` so consumers (auth retry, telemetry, UI) can branch without regex-scraping the message. */
600
622
  errorStatus?: number;
623
+ /** Typed upstream failure facts retained for retry classification without parsing errorMessage. */
624
+ transportFailure?: TransportFailureFacts;
601
625
  /**
602
626
  * Stable identifiers for request features the provider silently dropped
603
627
  * during this turn (e.g. `"priority"`). Set when a server-side rejection
@@ -939,10 +963,9 @@ export interface Model<TApi extends Api = any> {
939
963
  * (or compatible) host; `headers.Authorization` (or `apiKey` resolved by
940
964
  * the registry) carries the gateway bearer.
941
965
  *
942
- * Used by containerized gjc installs (e.g. robogjc slots) to route every
943
- * LLM call through a sidecar gateway that holds the real provider
944
- * credentials. The model's other metadata (pricing, context window,
945
- * thinking config, …) still resolves locally; only the streaming
966
+ * Used by containerized GJC installs to route every LLM call through a
967
+ * sidecar gateway that holds the real provider credentials. The model's other
968
+ * metadata (pricing, context window, thinking config, …) still resolves locally; only the streaming
946
969
  * dispatch is redirected.
947
970
  */
948
971
  transport?: "pi-native";
@@ -1,10 +1,10 @@
1
1
  import * as z from "zod/v4";
2
+ import { resolveCodexGpt56DiscoveryContext } from "../../context-cap-policy";
2
3
  import { CODEX_BASE_URL, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "../../providers/openai-codex/constants";
3
4
  import type { Model } from "../../types";
4
5
  import { isRecord } from "../../utils";
5
6
 
6
7
  const DEFAULT_MODEL_LIST_PATHS = ["/codex/models", "/models"] as const;
7
- const DEFAULT_CONTEXT_WINDOW = 272_000;
8
8
  const DEFAULT_MAX_TOKENS = 128_000;
9
9
  const DEFAULT_CODEX_CLIENT_VERSION = "0.99.0";
10
10
  const NPM_CODEX_LATEST_URL = "https://registry.npmjs.org/@openai%2Fcodex/latest";
@@ -258,7 +258,8 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo
258
258
  }
259
259
 
260
260
  const name = toNonEmptyString(payload.display_name) ?? slug;
261
- const contextWindow = toPositiveInt(payload.context_window) ?? DEFAULT_CONTEXT_WINDOW;
261
+ const modelIdentity = { id: slug, api: "openai-codex-responses", provider: "openai-codex" } as const;
262
+ const contextWindow = resolveCodexGpt56DiscoveryContext(modelIdentity, payload.context_window);
262
263
  const maxTokens = Math.min(DEFAULT_MAX_TOKENS, contextWindow);
263
264
  const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels);
264
265
  const input = normalizeInputModalities(payload.input_modalities);
@@ -346,16 +347,6 @@ function toNonEmptyString(value: unknown): string | null {
346
347
  return trimmed.length > 0 ? trimmed : null;
347
348
  }
348
349
 
349
- function toPositiveInt(value: unknown): number | null {
350
- if (typeof value !== "number" || !Number.isFinite(value)) {
351
- return null;
352
- }
353
- if (value <= 0) {
354
- return null;
355
- }
356
- return Math.trunc(value);
357
- }
358
-
359
350
  function toFiniteNumber(value: unknown): number | null {
360
351
  if (typeof value !== "number" || !Number.isFinite(value)) {
361
352
  return null;