@gajae-code/ai 0.10.2 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/CHANGELOG.md +13 -2
  2. package/dist/types/auth-gateway/server.d.ts +19 -0
  3. package/dist/types/auth-storage.d.ts +2 -0
  4. package/dist/types/context-cap-policy.d.ts +10 -0
  5. package/dist/types/index.d.ts +2 -0
  6. package/dist/types/providers/openai-codex/response-handler.d.ts +1 -0
  7. package/dist/types/providers/pi-native-server.d.ts +3 -3
  8. package/dist/types/types.d.ts +12 -4
  9. package/dist/types/utils/event-stream.d.ts +9 -3
  10. package/dist/types/utils/fallback-transport.d.ts +55 -0
  11. package/dist/types/utils/retry.d.ts +1 -0
  12. package/dist/types/utils.d.ts +17 -0
  13. package/package.json +2 -2
  14. package/src/auth-gateway/server.ts +164 -22
  15. package/src/auth-storage.ts +62 -40
  16. package/src/context-cap-policy.ts +59 -0
  17. package/src/index.ts +2 -0
  18. package/src/model-manager.ts +12 -7
  19. package/src/model-thinking.ts +7 -8
  20. package/src/providers/amazon-bedrock.ts +9 -1
  21. package/src/providers/anthropic.ts +6 -0
  22. package/src/providers/azure-openai-responses.ts +6 -1
  23. package/src/providers/google-gemini-cli.ts +26 -13
  24. package/src/providers/google-shared.ts +7 -1
  25. package/src/providers/ollama.ts +7 -1
  26. package/src/providers/openai-codex/response-handler.ts +11 -3
  27. package/src/providers/openai-codex-responses.ts +17 -3
  28. package/src/providers/openai-completions.ts +13 -2
  29. package/src/providers/openai-responses.ts +7 -2
  30. package/src/providers/pi-native-client.ts +24 -12
  31. package/src/providers/pi-native-server.ts +4 -3
  32. package/src/stream.ts +31 -2
  33. package/src/types.ts +27 -4
  34. package/src/utils/discovery/codex.ts +3 -12
  35. package/src/utils/event-stream.ts +138 -54
  36. package/src/utils/fallback-transport.ts +185 -0
  37. package/src/utils/retry.ts +2 -2
  38. package/src/utils.ts +19 -1
@@ -2067,7 +2067,11 @@ export class AuthStorage {
2067
2067
  });
2068
2068
  }
2069
2069
 
2070
- async #fetchUsageUncached(request: UsageRequestDescriptor, timeoutMs?: number): Promise<UsageReport | null> {
2070
+ async #fetchUsageUncached(
2071
+ request: UsageRequestDescriptor,
2072
+ timeoutMs?: number,
2073
+ logDetails: boolean = true,
2074
+ ): Promise<UsageReport | null> {
2071
2075
  const resolver = this.#usageProviderResolver;
2072
2076
  if (!resolver) return null;
2073
2077
 
@@ -2105,10 +2109,12 @@ export class AuthStorage {
2105
2109
  credential: refreshedCredential,
2106
2110
  };
2107
2111
  } catch (error) {
2108
- this.#usageLogger?.debug("Usage credential refresh failed, using original credential", {
2109
- provider: request.provider,
2110
- error: String(error),
2111
- });
2112
+ if (logDetails) {
2113
+ this.#usageLogger?.debug("Usage credential refresh failed, using original credential", {
2114
+ provider: request.provider,
2115
+ error: String(error),
2116
+ });
2117
+ }
2112
2118
  }
2113
2119
  }
2114
2120
  }
@@ -2118,18 +2124,24 @@ export class AuthStorage {
2118
2124
  try {
2119
2125
  return await providerImpl.fetchUsage(params, {
2120
2126
  fetch: this.#usageFetch,
2121
- logger: this.#usageLogger,
2127
+ logger: logDetails ? this.#usageLogger : undefined,
2122
2128
  });
2123
2129
  } catch (error) {
2124
- logger.debug("AuthStorage usage fetch failed", {
2125
- provider: request.provider,
2126
- error: String(error),
2127
- });
2130
+ if (logDetails) {
2131
+ logger.debug("AuthStorage usage fetch failed", {
2132
+ provider: request.provider,
2133
+ error: String(error),
2134
+ });
2135
+ }
2128
2136
  return null;
2129
2137
  }
2130
2138
  }
2131
2139
 
2132
- async #fetchUsageCached(request: UsageRequestDescriptor, timeoutMs?: number): Promise<UsageReport | null> {
2140
+ async #fetchUsageCached(
2141
+ request: UsageRequestDescriptor,
2142
+ timeoutMs?: number,
2143
+ logDetails: boolean = true,
2144
+ ): Promise<UsageReport | null> {
2133
2145
  const cacheKey = this.#buildUsageReportCacheKey(request);
2134
2146
  const now = Date.now();
2135
2147
  const cached = this.#usageCache.get<UsageReport | null>(cacheKey);
@@ -2142,7 +2154,7 @@ export class AuthStorage {
2142
2154
  if (inFlight) return inFlight;
2143
2155
 
2144
2156
  const promise = (async () => {
2145
- const report = await this.#fetchUsageUncached(request, timeoutMs);
2157
+ const report = await this.#fetchUsageUncached(request, timeoutMs, logDetails);
2146
2158
  const ttlJitter = USAGE_REPORT_TTL_MS * (Math.random() * 0.5 - 0.25);
2147
2159
  if (report !== null) {
2148
2160
  // Success: stagger per-credential cache expiry so all accounts don't
@@ -2385,6 +2397,8 @@ export class AuthStorage {
2385
2397
  baseUrlResolver?: (provider: Provider) => string | undefined;
2386
2398
  /** Caller's cancel signal; only rejects this caller, never the shared upstream fetch. */
2387
2399
  signal?: AbortSignal;
2400
+ /** Disable provider/account/error logging for secret-safe control surfaces. */
2401
+ logDetails?: boolean;
2388
2402
  }): Promise<UsageReport[] | null> {
2389
2403
  // Caller override > store-level hook > local per-credential fan-out.
2390
2404
  // `RemoteAuthCredentialStore` implements the store hook so a gateway
@@ -2413,9 +2427,11 @@ export class AuthStorage {
2413
2427
  const requests = this.#collectUsageRequests(options);
2414
2428
  if (requests.length === 0) return [];
2415
2429
 
2416
- this.#usageLogger?.debug("Usage fetch requested", {
2417
- providers: [...new Set(requests.map(request => request.provider))].sort(),
2418
- });
2430
+ if (options?.logDetails !== false) {
2431
+ this.#usageLogger?.debug("Usage fetch requested", {
2432
+ providers: [...new Set(requests.map(request => request.provider))].sort(),
2433
+ });
2434
+ }
2419
2435
 
2420
2436
  // Per-credential caching with jitter lives in #fetchUsageCached, so we
2421
2437
  // don't store the aggregated result here — doing so locks the widget to
@@ -2428,39 +2444,45 @@ export class AuthStorage {
2428
2444
  if (inFlight) return inFlight;
2429
2445
 
2430
2446
  const promise = (async () => {
2431
- for (const request of requests) {
2432
- this.#usageLogger?.debug("Usage fetch queued", {
2433
- provider: request.provider,
2434
- credentialType: request.credential.type,
2435
- baseUrl: request.baseUrl,
2436
- accountId: request.credential.accountId,
2437
- email: request.credential.email,
2438
- });
2447
+ if (options?.logDetails !== false) {
2448
+ for (const request of requests) {
2449
+ this.#usageLogger?.debug("Usage fetch queued", {
2450
+ provider: request.provider,
2451
+ credentialType: request.credential.type,
2452
+ baseUrl: request.baseUrl,
2453
+ accountId: request.credential.accountId,
2454
+ email: request.credential.email,
2455
+ });
2456
+ }
2439
2457
  }
2440
2458
 
2441
2459
  const results = await Promise.all(
2442
- requests.map(request => this.#fetchUsageCached(request, this.#usageRequestTimeoutMs)),
2460
+ requests.map(request =>
2461
+ this.#fetchUsageCached(request, this.#usageRequestTimeoutMs, options?.logDetails !== false),
2462
+ ),
2443
2463
  );
2444
2464
  const reports = results.filter((report): report is UsageReport => report !== null);
2445
2465
  const deduped = this.#dedupeUsageReports(reports);
2446
2466
  // no outer cache write — see comment above.
2447
2467
  const resolved = deduped;
2448
- this.#usageLogger?.debug("Usage fetch resolved", {
2449
- reports: resolved.map(report => {
2450
- const accountLabel =
2451
- this.#getUsageReportMetadataValue(report, "email") ??
2452
- this.#getUsageReportMetadataValue(report, "accountId") ??
2453
- this.#getUsageReportMetadataValue(report, "account") ??
2454
- this.#getUsageReportMetadataValue(report, "user") ??
2455
- this.#getUsageReportMetadataValue(report, "username") ??
2456
- this.#getUsageReportScopeAccountId(report);
2457
- return {
2458
- provider: report.provider,
2459
- limits: report.limits.length,
2460
- account: accountLabel,
2461
- };
2462
- }),
2463
- });
2468
+ if (options?.logDetails !== false) {
2469
+ this.#usageLogger?.debug("Usage fetch resolved", {
2470
+ reports: resolved.map(report => {
2471
+ const accountLabel =
2472
+ this.#getUsageReportMetadataValue(report, "email") ??
2473
+ this.#getUsageReportMetadataValue(report, "accountId") ??
2474
+ this.#getUsageReportMetadataValue(report, "account") ??
2475
+ this.#getUsageReportMetadataValue(report, "user") ??
2476
+ this.#getUsageReportMetadataValue(report, "username") ??
2477
+ this.#getUsageReportScopeAccountId(report);
2478
+ return {
2479
+ provider: report.provider,
2480
+ limits: report.limits.length,
2481
+ account: accountLabel,
2482
+ };
2483
+ }),
2484
+ });
2485
+ }
2464
2486
  return resolved;
2465
2487
  })().finally(() => {
2466
2488
  this.#usageReportsInFlight.delete(cacheKey);
@@ -0,0 +1,59 @@
1
+ import type { Api, Model } from "./types";
2
+
3
+ export interface CodexGpt56ContextCapPolicy {
4
+ fallback: number;
5
+ ceiling: number;
6
+ }
7
+
8
+ export const CODEX_GPT_5_6_CONTEXT_CAP: CodexGpt56ContextCapPolicy = {
9
+ fallback: 272_000,
10
+ ceiling: 272_000,
11
+ };
12
+
13
+ const CODEX_GPT_5_6_MODEL_IDS: ReadonlySet<string> = new Set([
14
+ "gpt-5.6",
15
+ "gpt-5.6-sol",
16
+ "gpt-5.6-terra",
17
+ "gpt-5.6-luna",
18
+ ]);
19
+
20
+ export function isCodexProductTransport(model: Pick<Model<Api>, "api" | "provider">): boolean {
21
+ return model.provider === "openai-codex" || model.api === "openai-codex-responses";
22
+ }
23
+
24
+ export function isCodexGpt56Tier(model: Pick<Model<Api>, "id">): boolean {
25
+ return CODEX_GPT_5_6_MODEL_IDS.has(model.id.toLowerCase());
26
+ }
27
+
28
+ export function resolveCodexGpt56DiscoveryContext(
29
+ model: Pick<Model<Api>, "api" | "id" | "provider">,
30
+ rawContextWindow: unknown,
31
+ policy: CodexGpt56ContextCapPolicy = CODEX_GPT_5_6_CONTEXT_CAP,
32
+ ): number {
33
+ const observed = isPositiveFiniteNumber(rawContextWindow) ? rawContextWindow : policy.fallback;
34
+ if (!isCodexGpt56Tier(model) || !isCodexProductTransport(model)) {
35
+ return observed;
36
+ }
37
+ return Math.min(observed, policy.ceiling);
38
+ }
39
+
40
+ export function applyFinalCodexGpt56ContextCap<TApi extends Api>(
41
+ models: readonly Model<TApi>[],
42
+ policy: CodexGpt56ContextCapPolicy = CODEX_GPT_5_6_CONTEXT_CAP,
43
+ ): Model<TApi>[] {
44
+ return models.map(model => {
45
+ if (
46
+ !isCodexGpt56Tier(model as Model<Api>) ||
47
+ !isCodexProductTransport(model as Model<Api>) ||
48
+ !isPositiveFiniteNumber(model.contextWindow) ||
49
+ model.contextWindow <= policy.ceiling
50
+ ) {
51
+ return model;
52
+ }
53
+ return { ...model, contextWindow: policy.ceiling };
54
+ });
55
+ }
56
+
57
+ function isPositiveFiniteNumber(value: unknown): value is number {
58
+ return typeof value === "number" && Number.isFinite(value) && value > 0;
59
+ }
package/src/index.ts CHANGED
@@ -4,6 +4,7 @@ export * from "./auth-broker";
4
4
  export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server";
5
5
  export * from "./auth-gateway/types";
6
6
  export * from "./auth-storage";
7
+ export * from "./context-cap-policy";
7
8
  export * from "./model-cache";
8
9
  export * from "./model-manager";
9
10
  export * from "./model-thinking";
@@ -41,6 +42,7 @@ export * from "./usage/zai";
41
42
  export * from "./utils/anthropic-auth";
42
43
  export * from "./utils/discovery";
43
44
  export * from "./utils/event-stream";
45
+ export * from "./utils/fallback-transport";
44
46
  export * from "./utils/h2-fetch";
45
47
  export * from "./utils/oauth";
46
48
  export type {
@@ -1,3 +1,4 @@
1
+ import { applyFinalCodexGpt56ContextCap } from "./context-cap-policy";
1
2
  import { readModelCache, writeModelCache } from "./model-cache";
2
3
  import { isRetiredModel, isRetiredModelKey } from "./model-retirements";
3
4
  import { applyGeneratedModelPolicies, enrichModelThinking } from "./model-thinking";
@@ -98,7 +99,7 @@ function passModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
98
99
  out.push(enrichModelThinking(item as Model<TApi>));
99
100
  }
100
101
  applyGeneratedModelPolicies(out as Model<Api>[]);
101
- return out;
102
+ return applyFinalCodexGpt56ContextCap(out);
102
103
  }
103
104
 
104
105
  /**
@@ -161,11 +162,13 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
161
162
  const cacheModels = dynamicFetchSucceeded ? [] : normalizeModelList<TApi>(cache?.models ?? []);
162
163
  const dynamicModels = fetchedDynamicModels ?? [];
163
164
  const mergedWithCache = mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), cacheModels);
164
- const models = mergeDynamicModels(mergedWithCache, dynamicModels);
165
+ const models = applyFinalCodexGpt56ContextCap(mergeDynamicModels(mergedWithCache, dynamicModels));
165
166
  const dynamicAuthoritative = !hasDynamicFetcher || dynamicFetchSucceeded || shouldUseFreshCacheAsAuthoritative;
166
167
  if (shouldFetchFromNetwork) {
167
168
  if (dynamicFetchSucceeded) {
168
- const snapshotModels = mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), dynamicModels);
169
+ const snapshotModels = applyFinalCodexGpt56ContextCap(
170
+ mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), dynamicModels),
171
+ );
169
172
  writeModelCache(options.providerId, now(), snapshotModels, true, staticFingerprint, dbPath);
170
173
  } else {
171
174
  // Dynamic fetch failed — update cache with a non-authoritative snapshot so
@@ -174,9 +177,11 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
174
177
  writeModelCache(
175
178
  options.providerId,
176
179
  now(),
177
- mergeDynamicModels(
178
- mergeModelSources(staticModels, modelsDevModels),
179
- normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
180
+ applyFinalCodexGpt56ContextCap(
181
+ mergeDynamicModels(
182
+ mergeModelSources(staticModels, modelsDevModels),
183
+ normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
184
+ ),
180
185
  ),
181
186
  false,
182
187
  staticFingerprint,
@@ -393,7 +398,7 @@ function normalizeModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
393
398
  models.push(enrichModelThinking(item as Model<TApi>));
394
399
  }
395
400
  }
396
- return models;
401
+ return applyFinalCodexGpt56ContextCap(models);
397
402
  }
398
403
 
399
404
  function isModelLike(value: unknown): value is Model<Api> {
@@ -1,3 +1,4 @@
1
+ import { CODEX_GPT_5_6_CONTEXT_CAP, isCodexGpt56Tier, isCodexProductTransport } from "./context-cap-policy";
1
2
  import { resolveOpenAICompat } from "./providers/openai-completions-compat";
2
3
  import type { Api, Model as ApiModel, ThinkingConfig } from "./types";
3
4
  import { isClaudeForcedToolChoiceIncapableModelId } from "./utils/tool-choice-capability";
@@ -477,15 +478,13 @@ function applyGpt55ContextWindow(model: ApiModel<Api>, parsedModel: OpenAIModel)
477
478
  }
478
479
  return false;
479
480
  }
480
- const GPT_5_6_TIER_VARIANTS: ReadonlySet<OpenAIVariant> = new Set(["base", "sol", "terra", "luna"]);
481
-
482
- function applyGpt56ContextWindow(model: ApiModel<Api>, parsedModel: OpenAIModel): boolean {
483
- if (!semverGte(parsedModel.version, "5.6") || !GPT_5_6_TIER_VARIANTS.has(parsedModel.variant)) {
481
+ function applyGpt56ContextWindow(model: ApiModel<Api>): boolean {
482
+ if (!isCodexGpt56Tier(model) || !isCodexProductTransport(model)) {
484
483
  return false;
485
484
  }
486
- // GPT-5.6 tiers enforce a ~373K usable prompt budget on both transports
487
- // (matches the openai-codex live catalog), despite the 1M+ marketing window.
488
- model.contextWindow = 373_000;
485
+ // Codex product metadata is bounded by the currently enforced prompt cap.
486
+ // Smaller observed limits remain authoritative; first-party OpenAI is untouched.
487
+ model.contextWindow = Math.min(model.contextWindow, CODEX_GPT_5_6_CONTEXT_CAP.ceiling);
489
488
  return true;
490
489
  }
491
490
 
@@ -493,7 +492,7 @@ function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel
493
492
  if (applyGpt55ContextWindow(model, parsedModel)) {
494
493
  return;
495
494
  }
496
- if (applyGpt56ContextWindow(model, parsedModel)) {
495
+ if (applyGpt56ContextWindow(model)) {
497
496
  return;
498
497
  }
499
498
  // OpenAI code backend models: 400K figure includes output budget; input window is 272K.
@@ -30,6 +30,7 @@ import type {
30
30
  } from "../types";
31
31
  import { normalizeToolCallId, resolveCacheRetention, sanitizeJsonStrings } from "../utils";
32
32
  import { AssistantMessageEventStream } from "../utils/event-stream";
33
+ import { transportFailureFacts } from "../utils/fallback-transport";
33
34
  import { appendRawHttpRequestDumpFor400, type RawHttpRequestDump, withHttpStatus } from "../utils/http-inspector";
34
35
  import { parseStreamingJson } from "../utils/json-parse";
35
36
  import { resolveRetryBudget } from "../utils/retry-budget";
@@ -317,7 +318,12 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
317
318
  new Error(`Bedrock HTTP ${response.status}: ${errBody.slice(0, 1000)}`),
318
319
  response.status,
319
320
  );
320
- if (firstTokenTime === undefined && !fallbackRan && isForcedToolChoiceUnsupportedError(error, true)) {
321
+ if (
322
+ firstTokenTime === undefined &&
323
+ !fallbackRan &&
324
+ !options.fallbackManaged &&
325
+ isForcedToolChoiceUnsupportedError(error, true)
326
+ ) {
321
327
  response = await retryWithoutForcedToolChoice(error.message);
322
328
  } else {
323
329
  throw error;
@@ -349,6 +355,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
349
355
  firstTokenTime === undefined &&
350
356
  sentForcedToolChoice &&
351
357
  !fallbackRan &&
358
+ !options.fallbackManaged &&
352
359
  isForcedToolChoiceUnsupportedError(error, true)
353
360
  ) {
354
361
  response = await retryWithoutForcedToolChoice(error.message);
@@ -432,6 +439,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
432
439
  }
433
440
  output.stopReason = options.signal?.aborted ? "aborted" : "error";
434
441
  output.errorStatus = extractHttpStatusFromError(error);
442
+ output.transportFailure = transportFailureFacts(error);
435
443
  const baseMessage = error instanceof Error ? error.message : JSON.stringify(error);
436
444
  // Enrich error with thinking block diagnostics for signature-related failures
437
445
  let diagnostics = "";
@@ -57,6 +57,7 @@ import {
57
57
  } from "../utils";
58
58
  import { createAbortSourceTracker } from "../utils/abort";
59
59
  import { AssistantMessageEventStream } from "../utils/event-stream";
60
+ import { transportFailureFacts } from "../utils/fallback-transport";
60
61
  import { isFoundryEnabled } from "../utils/foundry";
61
62
  import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
62
63
  import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
@@ -1637,6 +1638,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1637
1638
  throw streamFailure;
1638
1639
  }
1639
1640
  if (
1641
+ !options?.fallbackManaged &&
1640
1642
  !disableStrictTools &&
1641
1643
  firstTokenTime === undefined &&
1642
1644
  hasStrictAnthropicTools(params) &&
@@ -1655,6 +1657,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1655
1657
  if (
1656
1658
  !droppedForcedToolChoice &&
1657
1659
  firstTokenTime === undefined &&
1660
+ !options?.fallbackManaged &&
1658
1661
  isSentForcedAnthropicToolChoice(params.tool_choice) &&
1659
1662
  isForcedToolChoiceUnsupportedError(streamFailure, true)
1660
1663
  ) {
@@ -1681,6 +1684,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1681
1684
  continue;
1682
1685
  }
1683
1686
  if (
1687
+ !options?.fallbackManaged &&
1684
1688
  !thinkingRepairAttempted &&
1685
1689
  firstTokenTime === undefined &&
1686
1690
  isAnthropicThinkingBlockMutationError(streamFailure)
@@ -1696,6 +1700,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1696
1700
  continue;
1697
1701
  }
1698
1702
  if (
1703
+ !options?.fallbackManaged &&
1699
1704
  !dropFastMode &&
1700
1705
  resolveServiceTier(options?.serviceTier, model.provider) === "priority" &&
1701
1706
  firstTokenTime === undefined &&
@@ -1752,6 +1757,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1752
1757
  const firstEventTimeoutError = activeAbortTracker.getLocalAbortReason();
1753
1758
  output.stopReason = activeAbortTracker.wasCallerAbort() ? "aborted" : "error";
1754
1759
  output.errorStatus = extractHttpStatusFromError(error);
1760
+ output.transportFailure = transportFailureFacts(error);
1755
1761
  if (output.errorKind !== "provider_safety_stop" || !output.errorMessage) {
1756
1762
  output.errorMessage =
1757
1763
  firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
@@ -19,6 +19,7 @@ import type {
19
19
  import { normalizeSystemPrompts } from "../utils";
20
20
  import { createAbortSourceTracker } from "../utils/abort";
21
21
  import { AssistantMessageEventStream } from "../utils/event-stream";
22
+ import { transportFailureFacts } from "../utils/fallback-transport";
22
23
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
23
24
  import {
24
25
  createWatchdog,
@@ -140,7 +141,10 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
140
141
  try {
141
142
  openaiStream = await client.responses.create(params, { signal: requestSignal });
142
143
  } catch (error) {
143
- if (!isForcedToolChoiceUnsupportedError(error, isForcedAzureResponsesToolChoice(params.tool_choice))) {
144
+ if (
145
+ !isForcedToolChoiceUnsupportedError(error, isForcedAzureResponsesToolChoice(params.tool_choice)) ||
146
+ options?.fallbackManaged
147
+ ) {
144
148
  throw error;
145
149
  }
146
150
  const reason = await finalizeErrorMessage(error, rawRequestDump);
@@ -204,6 +208,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
204
208
  const firstEventTimeoutError = abortTracker.getLocalAbortReason();
205
209
  output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
206
210
  output.errorStatus = extractHttpStatusFromError(error);
211
+ output.transportFailure = transportFailureFacts(error);
207
212
  output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
208
213
  output.duration = Date.now() - startTime;
209
214
  if (firstTokenTime) output.ttft = firstTokenTime - startTime;
@@ -20,6 +20,7 @@ import type {
20
20
  } from "../types";
21
21
  import { normalizeSystemPrompts } from "../utils";
22
22
  import { AssistantMessageEventStream } from "../utils/event-stream";
23
+ import { transportFailureFacts } from "../utils/fallback-transport";
23
24
  import { appendRawHttpRequestDumpFor400, type RawHttpRequestDump, withHttpStatus } from "../utils/http-inspector";
24
25
  import { resolveRetryBudget } from "../utils/retry-budget";
25
26
  // Refresh is the sole responsibility of AuthStorage (broker-aware, single-flighted);
@@ -130,6 +131,22 @@ function extractErrorMessage(errorText: string): string {
130
131
  return errorText;
131
132
  }
132
133
 
134
+ function createGeminiCliHttpError(response: Response, errorText: string, formatErrorMessage = true): Error {
135
+ const message = formatErrorMessage ? extractErrorMessage(errorText) : errorText;
136
+ const error = withHttpStatus(
137
+ new Error(`Cloud Code Assist API error (${response.status}): ${message}`),
138
+ response.status,
139
+ ) as Error & { code?: string; headers?: Headers };
140
+ error.headers = response.headers;
141
+ try {
142
+ const code = (JSON.parse(errorText) as { error?: { code?: unknown } }).error?.code;
143
+ if (typeof code === "string") error.code = code;
144
+ } catch {
145
+ // The response body is not JSON.
146
+ }
147
+ return error;
148
+ }
149
+
133
150
  interface GeminiCliApiKeyPayload {
134
151
  token?: unknown;
135
152
  projectId?: unknown;
@@ -380,11 +397,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
380
397
  );
381
398
  if (!response.ok && sentForcedToolChoice) {
382
399
  const errorText = await response.text();
383
- const error = withHttpStatus(
384
- new Error(`Cloud Code Assist API error (${response.status}): ${extractErrorMessage(errorText)}`),
385
- response.status,
386
- );
387
- if (firstTokenTime === undefined && isForcedToolChoiceUnsupportedError(error, true)) {
400
+ const error = createGeminiCliHttpError(response, errorText);
401
+ if (
402
+ !options?.fallbackManaged &&
403
+ firstTokenTime === undefined &&
404
+ isForcedToolChoiceUnsupportedError(error, true)
405
+ ) {
388
406
  const beforeMark = resolveToolChoice(model, options?.toolChoice);
389
407
  markToolChoiceIncapability(model, "auto", error.message);
390
408
  stream.push({
@@ -423,10 +441,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
423
441
  }
424
442
  if (!response.ok) {
425
443
  const errorText = await response.text();
426
- throw withHttpStatus(
427
- new Error(`Cloud Code Assist API error (${response.status}): ${extractErrorMessage(errorText)}`),
428
- response.status,
429
- );
444
+ throw createGeminiCliHttpError(response, errorText);
430
445
  }
431
446
  const requestUrl = response.url;
432
447
 
@@ -629,10 +644,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
629
644
 
630
645
  if (!currentResponse.ok) {
631
646
  const retryErrorText = await currentResponse.text();
632
- throw withHttpStatus(
633
- new Error(`Cloud Code Assist API error (${currentResponse.status}): ${retryErrorText}`),
634
- currentResponse.status,
635
- );
647
+ throw createGeminiCliHttpError(currentResponse, retryErrorText, false);
636
648
  }
637
649
  }
638
650
 
@@ -671,6 +683,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
671
683
  }
672
684
  output.stopReason = options?.signal?.aborted ? "aborted" : "error";
673
685
  output.errorStatus = extractHttpStatusFromError(error);
686
+ output.transportFailure = transportFailureFacts(error);
674
687
  output.errorMessage = await appendRawHttpRequestDumpFor400(
675
688
  error instanceof Error ? error.message : JSON.stringify(error),
676
689
  error,
@@ -20,6 +20,7 @@ import type {
20
20
  } from "../types";
21
21
  import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
22
22
  import { AssistantMessageEventStream } from "../utils/event-stream";
23
+ import { transportFailureFacts } from "../utils/fallback-transport";
23
24
  import { finalizeErrorMessage, type RawHttpRequestDump, withHttpStatus } from "../utils/http-inspector";
24
25
  import { normalizeSchemaForCCA, normalizeSchemaForGoogle, toolWireSchema } from "../utils/schema";
25
26
  import {
@@ -868,7 +869,11 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
868
869
  new Error(`Google API error (${response.status}): ${extractGoogleErrorMessage(errorText)}`),
869
870
  response.status,
870
871
  );
871
- if (firstTokenTime === undefined && isForcedToolChoiceUnsupportedError(error, true)) {
872
+ if (
873
+ !options?.fallbackManaged &&
874
+ firstTokenTime === undefined &&
875
+ isForcedToolChoiceUnsupportedError(error, true)
876
+ ) {
872
877
  const beforeMark = resolveToolChoice(model, options?.toolChoice);
873
878
  markToolChoiceIncapability(model, "auto", error.message);
874
879
  stream.push({
@@ -938,6 +943,7 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
938
943
  }
939
944
  output.stopReason = options?.signal?.aborted ? "aborted" : "error";
940
945
  output.errorStatus = extractHttpStatusFromError(error);
946
+ output.transportFailure = transportFailureFacts(error);
941
947
  output.errorMessage = await finalizeErrorMessage(error, rawRequestDump);
942
948
  output.duration = Date.now() - startTime;
943
949
  if (firstTokenTime) output.ttft = firstTokenTime - startTime;
@@ -16,6 +16,7 @@ import type {
16
16
  } from "../types";
17
17
  import { normalizeSystemPrompts } from "../utils";
18
18
  import { AssistantMessageEventStream } from "../utils/event-stream";
19
+ import { transportFailureFacts } from "../utils/fallback-transport";
19
20
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
20
21
  import { parseStreamingJson } from "../utils/json-parse";
21
22
  import { resolveRetryBudget } from "../utils/retry-budget";
@@ -431,7 +432,11 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
431
432
  `HTTP ${response.status} from ${baseUrl}/api/chat: ${await response.text().catch(() => "")}`,
432
433
  );
433
434
  (error as Error & { status?: number }).status = response.status;
434
- if (firstTokenTime === undefined && isForcedToolChoiceUnsupportedError(error, true)) {
435
+ if (
436
+ firstTokenTime === undefined &&
437
+ !options.fallbackManaged &&
438
+ isForcedToolChoiceUnsupportedError(error, true)
439
+ ) {
435
440
  markToolChoiceIncapability(model, "auto", error.message);
436
441
  stream.push({
437
442
  type: "toolChoiceIncapability",
@@ -590,6 +595,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
590
595
  }
591
596
  output.stopReason = options.signal?.aborted ? "aborted" : "error";
592
597
  output.errorStatus = extractHttpStatusFromError(error);
598
+ output.transportFailure = transportFailureFacts(error);
593
599
  output.errorMessage = await finalizeErrorMessage(error, rawRequestDump);
594
600
  output.duration = Date.now() - startTime;
595
601
  if (firstTokenTime) {
@@ -14,6 +14,7 @@ export type CodexRateLimits = {
14
14
  export type CodexErrorInfo = {
15
15
  message: string;
16
16
  status: number;
17
+ code?: string;
17
18
  friendlyMessage?: string;
18
19
  rateLimits?: CodexRateLimits;
19
20
  raw?: string;
@@ -24,6 +25,7 @@ export async function parseCodexError(response: Response): Promise<CodexErrorInf
24
25
  let message = raw || response.statusText || "Request failed";
25
26
  let friendlyMessage: string | undefined;
26
27
  let rateLimits: CodexRateLimits | undefined;
28
+ let code: string | undefined;
27
29
 
28
30
  try {
29
31
  const parsed = JSON.parse(raw) as { error?: Record<string, unknown> };
@@ -45,16 +47,21 @@ export async function parseCodexError(response: Response): Promise<CodexErrorInf
45
47
  ? { primary, secondary }
46
48
  : undefined;
47
49
 
48
- const code = String((err as { code?: string; type?: string }).code ?? (err as { type?: string }).type ?? "");
50
+ code =
51
+ typeof (err as { code?: unknown }).code === "string"
52
+ ? (err as { code: string }).code
53
+ : typeof (err as { type?: unknown }).type === "string"
54
+ ? (err as { type: string }).type
55
+ : undefined;
49
56
  const resetsAt = (err as { resets_at?: number }).resets_at ?? primary.resets_at ?? secondary.resets_at;
50
57
  const mins = resetsAt ? Math.max(0, Math.round((resetsAt * 1000 - Date.now()) / 60000)) : undefined;
51
58
 
52
- if (/usage_limit_reached|usage_not_included/i.test(code)) {
59
+ if (/usage_limit_reached|usage_not_included/i.test(code ?? "")) {
53
60
  const planType = (err as { plan_type?: string }).plan_type;
54
61
  const plan = planType ? ` (${String(planType).toLowerCase()} plan)` : "";
55
62
  const when = mins !== undefined ? ` Try again in ~${mins} min.` : "";
56
63
  friendlyMessage = `You have hit your ChatGPT usage limit${plan}.${when}`.trim();
57
- } else if (/rate_limit_exceeded/i.test(code) || response.status === 429) {
64
+ } else if (/rate_limit_exceeded/i.test(code ?? "") || response.status === 429) {
58
65
  const when = mins !== undefined ? ` Try again in ~${mins} min.` : "";
59
66
  friendlyMessage = `ChatGPT rate limit exceeded.${when}`.trim();
60
67
  }
@@ -69,6 +76,7 @@ export async function parseCodexError(response: Response): Promise<CodexErrorInf
69
76
  message,
70
77
  status: response.status,
71
78
  friendlyMessage,
79
+ code,
72
80
  rateLimits,
73
81
  raw: raw,
74
82
  };