@gajae-code/ai 0.10.2 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -2
- package/dist/types/auth-gateway/server.d.ts +19 -0
- package/dist/types/auth-storage.d.ts +2 -0
- package/dist/types/context-cap-policy.d.ts +10 -0
- package/dist/types/index.d.ts +2 -0
- package/dist/types/providers/openai-codex/response-handler.d.ts +1 -0
- package/dist/types/providers/pi-native-server.d.ts +3 -3
- package/dist/types/types.d.ts +12 -4
- package/dist/types/utils/event-stream.d.ts +9 -3
- package/dist/types/utils/fallback-transport.d.ts +55 -0
- package/dist/types/utils/retry.d.ts +1 -0
- package/dist/types/utils.d.ts +17 -0
- package/package.json +2 -2
- package/src/auth-gateway/server.ts +164 -22
- package/src/auth-storage.ts +62 -40
- package/src/context-cap-policy.ts +59 -0
- package/src/index.ts +2 -0
- package/src/model-manager.ts +12 -7
- package/src/model-thinking.ts +7 -8
- package/src/providers/amazon-bedrock.ts +9 -1
- package/src/providers/anthropic.ts +6 -0
- package/src/providers/azure-openai-responses.ts +6 -1
- package/src/providers/google-gemini-cli.ts +26 -13
- package/src/providers/google-shared.ts +7 -1
- package/src/providers/ollama.ts +7 -1
- package/src/providers/openai-codex/response-handler.ts +11 -3
- package/src/providers/openai-codex-responses.ts +17 -3
- package/src/providers/openai-completions.ts +13 -2
- package/src/providers/openai-responses.ts +7 -2
- package/src/providers/pi-native-client.ts +24 -12
- package/src/providers/pi-native-server.ts +4 -3
- package/src/stream.ts +31 -2
- package/src/types.ts +27 -4
- package/src/utils/discovery/codex.ts +3 -12
- package/src/utils/event-stream.ts +138 -54
- package/src/utils/fallback-transport.ts +185 -0
- package/src/utils/retry.ts +2 -2
- package/src/utils.ts +19 -1
package/src/auth-storage.ts
CHANGED
|
@@ -2067,7 +2067,11 @@ export class AuthStorage {
|
|
|
2067
2067
|
});
|
|
2068
2068
|
}
|
|
2069
2069
|
|
|
2070
|
-
async #fetchUsageUncached(
|
|
2070
|
+
async #fetchUsageUncached(
|
|
2071
|
+
request: UsageRequestDescriptor,
|
|
2072
|
+
timeoutMs?: number,
|
|
2073
|
+
logDetails: boolean = true,
|
|
2074
|
+
): Promise<UsageReport | null> {
|
|
2071
2075
|
const resolver = this.#usageProviderResolver;
|
|
2072
2076
|
if (!resolver) return null;
|
|
2073
2077
|
|
|
@@ -2105,10 +2109,12 @@ export class AuthStorage {
|
|
|
2105
2109
|
credential: refreshedCredential,
|
|
2106
2110
|
};
|
|
2107
2111
|
} catch (error) {
|
|
2108
|
-
|
|
2109
|
-
|
|
2110
|
-
|
|
2111
|
-
|
|
2112
|
+
if (logDetails) {
|
|
2113
|
+
this.#usageLogger?.debug("Usage credential refresh failed, using original credential", {
|
|
2114
|
+
provider: request.provider,
|
|
2115
|
+
error: String(error),
|
|
2116
|
+
});
|
|
2117
|
+
}
|
|
2112
2118
|
}
|
|
2113
2119
|
}
|
|
2114
2120
|
}
|
|
@@ -2118,18 +2124,24 @@ export class AuthStorage {
|
|
|
2118
2124
|
try {
|
|
2119
2125
|
return await providerImpl.fetchUsage(params, {
|
|
2120
2126
|
fetch: this.#usageFetch,
|
|
2121
|
-
logger: this.#usageLogger,
|
|
2127
|
+
logger: logDetails ? this.#usageLogger : undefined,
|
|
2122
2128
|
});
|
|
2123
2129
|
} catch (error) {
|
|
2124
|
-
|
|
2125
|
-
|
|
2126
|
-
|
|
2127
|
-
|
|
2130
|
+
if (logDetails) {
|
|
2131
|
+
logger.debug("AuthStorage usage fetch failed", {
|
|
2132
|
+
provider: request.provider,
|
|
2133
|
+
error: String(error),
|
|
2134
|
+
});
|
|
2135
|
+
}
|
|
2128
2136
|
return null;
|
|
2129
2137
|
}
|
|
2130
2138
|
}
|
|
2131
2139
|
|
|
2132
|
-
async #fetchUsageCached(
|
|
2140
|
+
async #fetchUsageCached(
|
|
2141
|
+
request: UsageRequestDescriptor,
|
|
2142
|
+
timeoutMs?: number,
|
|
2143
|
+
logDetails: boolean = true,
|
|
2144
|
+
): Promise<UsageReport | null> {
|
|
2133
2145
|
const cacheKey = this.#buildUsageReportCacheKey(request);
|
|
2134
2146
|
const now = Date.now();
|
|
2135
2147
|
const cached = this.#usageCache.get<UsageReport | null>(cacheKey);
|
|
@@ -2142,7 +2154,7 @@ export class AuthStorage {
|
|
|
2142
2154
|
if (inFlight) return inFlight;
|
|
2143
2155
|
|
|
2144
2156
|
const promise = (async () => {
|
|
2145
|
-
const report = await this.#fetchUsageUncached(request, timeoutMs);
|
|
2157
|
+
const report = await this.#fetchUsageUncached(request, timeoutMs, logDetails);
|
|
2146
2158
|
const ttlJitter = USAGE_REPORT_TTL_MS * (Math.random() * 0.5 - 0.25);
|
|
2147
2159
|
if (report !== null) {
|
|
2148
2160
|
// Success: stagger per-credential cache expiry so all accounts don't
|
|
@@ -2385,6 +2397,8 @@ export class AuthStorage {
|
|
|
2385
2397
|
baseUrlResolver?: (provider: Provider) => string | undefined;
|
|
2386
2398
|
/** Caller's cancel signal; only rejects this caller, never the shared upstream fetch. */
|
|
2387
2399
|
signal?: AbortSignal;
|
|
2400
|
+
/** Disable provider/account/error logging for secret-safe control surfaces. */
|
|
2401
|
+
logDetails?: boolean;
|
|
2388
2402
|
}): Promise<UsageReport[] | null> {
|
|
2389
2403
|
// Caller override > store-level hook > local per-credential fan-out.
|
|
2390
2404
|
// `RemoteAuthCredentialStore` implements the store hook so a gateway
|
|
@@ -2413,9 +2427,11 @@ export class AuthStorage {
|
|
|
2413
2427
|
const requests = this.#collectUsageRequests(options);
|
|
2414
2428
|
if (requests.length === 0) return [];
|
|
2415
2429
|
|
|
2416
|
-
|
|
2417
|
-
|
|
2418
|
-
|
|
2430
|
+
if (options?.logDetails !== false) {
|
|
2431
|
+
this.#usageLogger?.debug("Usage fetch requested", {
|
|
2432
|
+
providers: [...new Set(requests.map(request => request.provider))].sort(),
|
|
2433
|
+
});
|
|
2434
|
+
}
|
|
2419
2435
|
|
|
2420
2436
|
// Per-credential caching with jitter lives in #fetchUsageCached, so we
|
|
2421
2437
|
// don't store the aggregated result here — doing so locks the widget to
|
|
@@ -2428,39 +2444,45 @@ export class AuthStorage {
|
|
|
2428
2444
|
if (inFlight) return inFlight;
|
|
2429
2445
|
|
|
2430
2446
|
const promise = (async () => {
|
|
2431
|
-
|
|
2432
|
-
|
|
2433
|
-
|
|
2434
|
-
|
|
2435
|
-
|
|
2436
|
-
|
|
2437
|
-
|
|
2438
|
-
|
|
2447
|
+
if (options?.logDetails !== false) {
|
|
2448
|
+
for (const request of requests) {
|
|
2449
|
+
this.#usageLogger?.debug("Usage fetch queued", {
|
|
2450
|
+
provider: request.provider,
|
|
2451
|
+
credentialType: request.credential.type,
|
|
2452
|
+
baseUrl: request.baseUrl,
|
|
2453
|
+
accountId: request.credential.accountId,
|
|
2454
|
+
email: request.credential.email,
|
|
2455
|
+
});
|
|
2456
|
+
}
|
|
2439
2457
|
}
|
|
2440
2458
|
|
|
2441
2459
|
const results = await Promise.all(
|
|
2442
|
-
requests.map(request =>
|
|
2460
|
+
requests.map(request =>
|
|
2461
|
+
this.#fetchUsageCached(request, this.#usageRequestTimeoutMs, options?.logDetails !== false),
|
|
2462
|
+
),
|
|
2443
2463
|
);
|
|
2444
2464
|
const reports = results.filter((report): report is UsageReport => report !== null);
|
|
2445
2465
|
const deduped = this.#dedupeUsageReports(reports);
|
|
2446
2466
|
// no outer cache write — see comment above.
|
|
2447
2467
|
const resolved = deduped;
|
|
2448
|
-
|
|
2449
|
-
|
|
2450
|
-
|
|
2451
|
-
|
|
2452
|
-
|
|
2453
|
-
|
|
2454
|
-
|
|
2455
|
-
|
|
2456
|
-
|
|
2457
|
-
|
|
2458
|
-
|
|
2459
|
-
|
|
2460
|
-
|
|
2461
|
-
|
|
2462
|
-
|
|
2463
|
-
|
|
2468
|
+
if (options?.logDetails !== false) {
|
|
2469
|
+
this.#usageLogger?.debug("Usage fetch resolved", {
|
|
2470
|
+
reports: resolved.map(report => {
|
|
2471
|
+
const accountLabel =
|
|
2472
|
+
this.#getUsageReportMetadataValue(report, "email") ??
|
|
2473
|
+
this.#getUsageReportMetadataValue(report, "accountId") ??
|
|
2474
|
+
this.#getUsageReportMetadataValue(report, "account") ??
|
|
2475
|
+
this.#getUsageReportMetadataValue(report, "user") ??
|
|
2476
|
+
this.#getUsageReportMetadataValue(report, "username") ??
|
|
2477
|
+
this.#getUsageReportScopeAccountId(report);
|
|
2478
|
+
return {
|
|
2479
|
+
provider: report.provider,
|
|
2480
|
+
limits: report.limits.length,
|
|
2481
|
+
account: accountLabel,
|
|
2482
|
+
};
|
|
2483
|
+
}),
|
|
2484
|
+
});
|
|
2485
|
+
}
|
|
2464
2486
|
return resolved;
|
|
2465
2487
|
})().finally(() => {
|
|
2466
2488
|
this.#usageReportsInFlight.delete(cacheKey);
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
import type { Api, Model } from "./types";
|
|
2
|
+
|
|
3
|
+
export interface CodexGpt56ContextCapPolicy {
|
|
4
|
+
fallback: number;
|
|
5
|
+
ceiling: number;
|
|
6
|
+
}
|
|
7
|
+
|
|
8
|
+
export const CODEX_GPT_5_6_CONTEXT_CAP: CodexGpt56ContextCapPolicy = {
|
|
9
|
+
fallback: 272_000,
|
|
10
|
+
ceiling: 272_000,
|
|
11
|
+
};
|
|
12
|
+
|
|
13
|
+
const CODEX_GPT_5_6_MODEL_IDS: ReadonlySet<string> = new Set([
|
|
14
|
+
"gpt-5.6",
|
|
15
|
+
"gpt-5.6-sol",
|
|
16
|
+
"gpt-5.6-terra",
|
|
17
|
+
"gpt-5.6-luna",
|
|
18
|
+
]);
|
|
19
|
+
|
|
20
|
+
export function isCodexProductTransport(model: Pick<Model<Api>, "api" | "provider">): boolean {
|
|
21
|
+
return model.provider === "openai-codex" || model.api === "openai-codex-responses";
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export function isCodexGpt56Tier(model: Pick<Model<Api>, "id">): boolean {
|
|
25
|
+
return CODEX_GPT_5_6_MODEL_IDS.has(model.id.toLowerCase());
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export function resolveCodexGpt56DiscoveryContext(
|
|
29
|
+
model: Pick<Model<Api>, "api" | "id" | "provider">,
|
|
30
|
+
rawContextWindow: unknown,
|
|
31
|
+
policy: CodexGpt56ContextCapPolicy = CODEX_GPT_5_6_CONTEXT_CAP,
|
|
32
|
+
): number {
|
|
33
|
+
const observed = isPositiveFiniteNumber(rawContextWindow) ? rawContextWindow : policy.fallback;
|
|
34
|
+
if (!isCodexGpt56Tier(model) || !isCodexProductTransport(model)) {
|
|
35
|
+
return observed;
|
|
36
|
+
}
|
|
37
|
+
return Math.min(observed, policy.ceiling);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function applyFinalCodexGpt56ContextCap<TApi extends Api>(
|
|
41
|
+
models: readonly Model<TApi>[],
|
|
42
|
+
policy: CodexGpt56ContextCapPolicy = CODEX_GPT_5_6_CONTEXT_CAP,
|
|
43
|
+
): Model<TApi>[] {
|
|
44
|
+
return models.map(model => {
|
|
45
|
+
if (
|
|
46
|
+
!isCodexGpt56Tier(model as Model<Api>) ||
|
|
47
|
+
!isCodexProductTransport(model as Model<Api>) ||
|
|
48
|
+
!isPositiveFiniteNumber(model.contextWindow) ||
|
|
49
|
+
model.contextWindow <= policy.ceiling
|
|
50
|
+
) {
|
|
51
|
+
return model;
|
|
52
|
+
}
|
|
53
|
+
return { ...model, contextWindow: policy.ceiling };
|
|
54
|
+
});
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function isPositiveFiniteNumber(value: unknown): value is number {
|
|
58
|
+
return typeof value === "number" && Number.isFinite(value) && value > 0;
|
|
59
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -4,6 +4,7 @@ export * from "./auth-broker";
|
|
|
4
4
|
export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server";
|
|
5
5
|
export * from "./auth-gateway/types";
|
|
6
6
|
export * from "./auth-storage";
|
|
7
|
+
export * from "./context-cap-policy";
|
|
7
8
|
export * from "./model-cache";
|
|
8
9
|
export * from "./model-manager";
|
|
9
10
|
export * from "./model-thinking";
|
|
@@ -41,6 +42,7 @@ export * from "./usage/zai";
|
|
|
41
42
|
export * from "./utils/anthropic-auth";
|
|
42
43
|
export * from "./utils/discovery";
|
|
43
44
|
export * from "./utils/event-stream";
|
|
45
|
+
export * from "./utils/fallback-transport";
|
|
44
46
|
export * from "./utils/h2-fetch";
|
|
45
47
|
export * from "./utils/oauth";
|
|
46
48
|
export type {
|
package/src/model-manager.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { applyFinalCodexGpt56ContextCap } from "./context-cap-policy";
|
|
1
2
|
import { readModelCache, writeModelCache } from "./model-cache";
|
|
2
3
|
import { isRetiredModel, isRetiredModelKey } from "./model-retirements";
|
|
3
4
|
import { applyGeneratedModelPolicies, enrichModelThinking } from "./model-thinking";
|
|
@@ -98,7 +99,7 @@ function passModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
|
|
|
98
99
|
out.push(enrichModelThinking(item as Model<TApi>));
|
|
99
100
|
}
|
|
100
101
|
applyGeneratedModelPolicies(out as Model<Api>[]);
|
|
101
|
-
return out;
|
|
102
|
+
return applyFinalCodexGpt56ContextCap(out);
|
|
102
103
|
}
|
|
103
104
|
|
|
104
105
|
/**
|
|
@@ -161,11 +162,13 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
|
|
|
161
162
|
const cacheModels = dynamicFetchSucceeded ? [] : normalizeModelList<TApi>(cache?.models ?? []);
|
|
162
163
|
const dynamicModels = fetchedDynamicModels ?? [];
|
|
163
164
|
const mergedWithCache = mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), cacheModels);
|
|
164
|
-
const models = mergeDynamicModels(mergedWithCache, dynamicModels);
|
|
165
|
+
const models = applyFinalCodexGpt56ContextCap(mergeDynamicModels(mergedWithCache, dynamicModels));
|
|
165
166
|
const dynamicAuthoritative = !hasDynamicFetcher || dynamicFetchSucceeded || shouldUseFreshCacheAsAuthoritative;
|
|
166
167
|
if (shouldFetchFromNetwork) {
|
|
167
168
|
if (dynamicFetchSucceeded) {
|
|
168
|
-
const snapshotModels =
|
|
169
|
+
const snapshotModels = applyFinalCodexGpt56ContextCap(
|
|
170
|
+
mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), dynamicModels),
|
|
171
|
+
);
|
|
169
172
|
writeModelCache(options.providerId, now(), snapshotModels, true, staticFingerprint, dbPath);
|
|
170
173
|
} else {
|
|
171
174
|
// Dynamic fetch failed — update cache with a non-authoritative snapshot so
|
|
@@ -174,9 +177,11 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
|
|
|
174
177
|
writeModelCache(
|
|
175
178
|
options.providerId,
|
|
176
179
|
now(),
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
+
applyFinalCodexGpt56ContextCap(
|
|
181
|
+
mergeDynamicModels(
|
|
182
|
+
mergeModelSources(staticModels, modelsDevModels),
|
|
183
|
+
normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
|
|
184
|
+
),
|
|
180
185
|
),
|
|
181
186
|
false,
|
|
182
187
|
staticFingerprint,
|
|
@@ -393,7 +398,7 @@ function normalizeModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
|
|
|
393
398
|
models.push(enrichModelThinking(item as Model<TApi>));
|
|
394
399
|
}
|
|
395
400
|
}
|
|
396
|
-
return models;
|
|
401
|
+
return applyFinalCodexGpt56ContextCap(models);
|
|
397
402
|
}
|
|
398
403
|
|
|
399
404
|
function isModelLike(value: unknown): value is Model<Api> {
|
package/src/model-thinking.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { CODEX_GPT_5_6_CONTEXT_CAP, isCodexGpt56Tier, isCodexProductTransport } from "./context-cap-policy";
|
|
1
2
|
import { resolveOpenAICompat } from "./providers/openai-completions-compat";
|
|
2
3
|
import type { Api, Model as ApiModel, ThinkingConfig } from "./types";
|
|
3
4
|
import { isClaudeForcedToolChoiceIncapableModelId } from "./utils/tool-choice-capability";
|
|
@@ -477,15 +478,13 @@ function applyGpt55ContextWindow(model: ApiModel<Api>, parsedModel: OpenAIModel)
|
|
|
477
478
|
}
|
|
478
479
|
return false;
|
|
479
480
|
}
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
function applyGpt56ContextWindow(model: ApiModel<Api>, parsedModel: OpenAIModel): boolean {
|
|
483
|
-
if (!semverGte(parsedModel.version, "5.6") || !GPT_5_6_TIER_VARIANTS.has(parsedModel.variant)) {
|
|
481
|
+
function applyGpt56ContextWindow(model: ApiModel<Api>): boolean {
|
|
482
|
+
if (!isCodexGpt56Tier(model) || !isCodexProductTransport(model)) {
|
|
484
483
|
return false;
|
|
485
484
|
}
|
|
486
|
-
//
|
|
487
|
-
//
|
|
488
|
-
model.contextWindow =
|
|
485
|
+
// Codex product metadata is bounded by the currently enforced prompt cap.
|
|
486
|
+
// Smaller observed limits remain authoritative; first-party OpenAI is untouched.
|
|
487
|
+
model.contextWindow = Math.min(model.contextWindow, CODEX_GPT_5_6_CONTEXT_CAP.ceiling);
|
|
489
488
|
return true;
|
|
490
489
|
}
|
|
491
490
|
|
|
@@ -493,7 +492,7 @@ function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel
|
|
|
493
492
|
if (applyGpt55ContextWindow(model, parsedModel)) {
|
|
494
493
|
return;
|
|
495
494
|
}
|
|
496
|
-
if (applyGpt56ContextWindow(model
|
|
495
|
+
if (applyGpt56ContextWindow(model)) {
|
|
497
496
|
return;
|
|
498
497
|
}
|
|
499
498
|
// OpenAI code backend models: 400K figure includes output budget; input window is 272K.
|
|
@@ -30,6 +30,7 @@ import type {
|
|
|
30
30
|
} from "../types";
|
|
31
31
|
import { normalizeToolCallId, resolveCacheRetention, sanitizeJsonStrings } from "../utils";
|
|
32
32
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
33
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
33
34
|
import { appendRawHttpRequestDumpFor400, type RawHttpRequestDump, withHttpStatus } from "../utils/http-inspector";
|
|
34
35
|
import { parseStreamingJson } from "../utils/json-parse";
|
|
35
36
|
import { resolveRetryBudget } from "../utils/retry-budget";
|
|
@@ -317,7 +318,12 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
|
317
318
|
new Error(`Bedrock HTTP ${response.status}: ${errBody.slice(0, 1000)}`),
|
|
318
319
|
response.status,
|
|
319
320
|
);
|
|
320
|
-
if (
|
|
321
|
+
if (
|
|
322
|
+
firstTokenTime === undefined &&
|
|
323
|
+
!fallbackRan &&
|
|
324
|
+
!options.fallbackManaged &&
|
|
325
|
+
isForcedToolChoiceUnsupportedError(error, true)
|
|
326
|
+
) {
|
|
321
327
|
response = await retryWithoutForcedToolChoice(error.message);
|
|
322
328
|
} else {
|
|
323
329
|
throw error;
|
|
@@ -349,6 +355,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
|
349
355
|
firstTokenTime === undefined &&
|
|
350
356
|
sentForcedToolChoice &&
|
|
351
357
|
!fallbackRan &&
|
|
358
|
+
!options.fallbackManaged &&
|
|
352
359
|
isForcedToolChoiceUnsupportedError(error, true)
|
|
353
360
|
) {
|
|
354
361
|
response = await retryWithoutForcedToolChoice(error.message);
|
|
@@ -432,6 +439,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
|
432
439
|
}
|
|
433
440
|
output.stopReason = options.signal?.aborted ? "aborted" : "error";
|
|
434
441
|
output.errorStatus = extractHttpStatusFromError(error);
|
|
442
|
+
output.transportFailure = transportFailureFacts(error);
|
|
435
443
|
const baseMessage = error instanceof Error ? error.message : JSON.stringify(error);
|
|
436
444
|
// Enrich error with thinking block diagnostics for signature-related failures
|
|
437
445
|
let diagnostics = "";
|
|
@@ -57,6 +57,7 @@ import {
|
|
|
57
57
|
} from "../utils";
|
|
58
58
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
59
59
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
60
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
60
61
|
import { isFoundryEnabled } from "../utils/foundry";
|
|
61
62
|
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
|
|
62
63
|
import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
|
|
@@ -1637,6 +1638,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1637
1638
|
throw streamFailure;
|
|
1638
1639
|
}
|
|
1639
1640
|
if (
|
|
1641
|
+
!options?.fallbackManaged &&
|
|
1640
1642
|
!disableStrictTools &&
|
|
1641
1643
|
firstTokenTime === undefined &&
|
|
1642
1644
|
hasStrictAnthropicTools(params) &&
|
|
@@ -1655,6 +1657,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1655
1657
|
if (
|
|
1656
1658
|
!droppedForcedToolChoice &&
|
|
1657
1659
|
firstTokenTime === undefined &&
|
|
1660
|
+
!options?.fallbackManaged &&
|
|
1658
1661
|
isSentForcedAnthropicToolChoice(params.tool_choice) &&
|
|
1659
1662
|
isForcedToolChoiceUnsupportedError(streamFailure, true)
|
|
1660
1663
|
) {
|
|
@@ -1681,6 +1684,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1681
1684
|
continue;
|
|
1682
1685
|
}
|
|
1683
1686
|
if (
|
|
1687
|
+
!options?.fallbackManaged &&
|
|
1684
1688
|
!thinkingRepairAttempted &&
|
|
1685
1689
|
firstTokenTime === undefined &&
|
|
1686
1690
|
isAnthropicThinkingBlockMutationError(streamFailure)
|
|
@@ -1696,6 +1700,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1696
1700
|
continue;
|
|
1697
1701
|
}
|
|
1698
1702
|
if (
|
|
1703
|
+
!options?.fallbackManaged &&
|
|
1699
1704
|
!dropFastMode &&
|
|
1700
1705
|
resolveServiceTier(options?.serviceTier, model.provider) === "priority" &&
|
|
1701
1706
|
firstTokenTime === undefined &&
|
|
@@ -1752,6 +1757,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1752
1757
|
const firstEventTimeoutError = activeAbortTracker.getLocalAbortReason();
|
|
1753
1758
|
output.stopReason = activeAbortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
1754
1759
|
output.errorStatus = extractHttpStatusFromError(error);
|
|
1760
|
+
output.transportFailure = transportFailureFacts(error);
|
|
1755
1761
|
if (output.errorKind !== "provider_safety_stop" || !output.errorMessage) {
|
|
1756
1762
|
output.errorMessage =
|
|
1757
1763
|
firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
@@ -19,6 +19,7 @@ import type {
|
|
|
19
19
|
import { normalizeSystemPrompts } from "../utils";
|
|
20
20
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
21
21
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
22
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
22
23
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
23
24
|
import {
|
|
24
25
|
createWatchdog,
|
|
@@ -140,7 +141,10 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
140
141
|
try {
|
|
141
142
|
openaiStream = await client.responses.create(params, { signal: requestSignal });
|
|
142
143
|
} catch (error) {
|
|
143
|
-
if (
|
|
144
|
+
if (
|
|
145
|
+
!isForcedToolChoiceUnsupportedError(error, isForcedAzureResponsesToolChoice(params.tool_choice)) ||
|
|
146
|
+
options?.fallbackManaged
|
|
147
|
+
) {
|
|
144
148
|
throw error;
|
|
145
149
|
}
|
|
146
150
|
const reason = await finalizeErrorMessage(error, rawRequestDump);
|
|
@@ -204,6 +208,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
204
208
|
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
|
205
209
|
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
206
210
|
output.errorStatus = extractHttpStatusFromError(error);
|
|
211
|
+
output.transportFailure = transportFailureFacts(error);
|
|
207
212
|
output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
208
213
|
output.duration = Date.now() - startTime;
|
|
209
214
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
@@ -20,6 +20,7 @@ import type {
|
|
|
20
20
|
} from "../types";
|
|
21
21
|
import { normalizeSystemPrompts } from "../utils";
|
|
22
22
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
23
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
23
24
|
import { appendRawHttpRequestDumpFor400, type RawHttpRequestDump, withHttpStatus } from "../utils/http-inspector";
|
|
24
25
|
import { resolveRetryBudget } from "../utils/retry-budget";
|
|
25
26
|
// Refresh is the sole responsibility of AuthStorage (broker-aware, single-flighted);
|
|
@@ -130,6 +131,22 @@ function extractErrorMessage(errorText: string): string {
|
|
|
130
131
|
return errorText;
|
|
131
132
|
}
|
|
132
133
|
|
|
134
|
+
function createGeminiCliHttpError(response: Response, errorText: string, formatErrorMessage = true): Error {
|
|
135
|
+
const message = formatErrorMessage ? extractErrorMessage(errorText) : errorText;
|
|
136
|
+
const error = withHttpStatus(
|
|
137
|
+
new Error(`Cloud Code Assist API error (${response.status}): ${message}`),
|
|
138
|
+
response.status,
|
|
139
|
+
) as Error & { code?: string; headers?: Headers };
|
|
140
|
+
error.headers = response.headers;
|
|
141
|
+
try {
|
|
142
|
+
const code = (JSON.parse(errorText) as { error?: { code?: unknown } }).error?.code;
|
|
143
|
+
if (typeof code === "string") error.code = code;
|
|
144
|
+
} catch {
|
|
145
|
+
// The response body is not JSON.
|
|
146
|
+
}
|
|
147
|
+
return error;
|
|
148
|
+
}
|
|
149
|
+
|
|
133
150
|
interface GeminiCliApiKeyPayload {
|
|
134
151
|
token?: unknown;
|
|
135
152
|
projectId?: unknown;
|
|
@@ -380,11 +397,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
|
|
380
397
|
);
|
|
381
398
|
if (!response.ok && sentForcedToolChoice) {
|
|
382
399
|
const errorText = await response.text();
|
|
383
|
-
const error =
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
400
|
+
const error = createGeminiCliHttpError(response, errorText);
|
|
401
|
+
if (
|
|
402
|
+
!options?.fallbackManaged &&
|
|
403
|
+
firstTokenTime === undefined &&
|
|
404
|
+
isForcedToolChoiceUnsupportedError(error, true)
|
|
405
|
+
) {
|
|
388
406
|
const beforeMark = resolveToolChoice(model, options?.toolChoice);
|
|
389
407
|
markToolChoiceIncapability(model, "auto", error.message);
|
|
390
408
|
stream.push({
|
|
@@ -423,10 +441,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
|
|
423
441
|
}
|
|
424
442
|
if (!response.ok) {
|
|
425
443
|
const errorText = await response.text();
|
|
426
|
-
throw
|
|
427
|
-
new Error(`Cloud Code Assist API error (${response.status}): ${extractErrorMessage(errorText)}`),
|
|
428
|
-
response.status,
|
|
429
|
-
);
|
|
444
|
+
throw createGeminiCliHttpError(response, errorText);
|
|
430
445
|
}
|
|
431
446
|
const requestUrl = response.url;
|
|
432
447
|
|
|
@@ -629,10 +644,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
|
|
629
644
|
|
|
630
645
|
if (!currentResponse.ok) {
|
|
631
646
|
const retryErrorText = await currentResponse.text();
|
|
632
|
-
throw
|
|
633
|
-
new Error(`Cloud Code Assist API error (${currentResponse.status}): ${retryErrorText}`),
|
|
634
|
-
currentResponse.status,
|
|
635
|
-
);
|
|
647
|
+
throw createGeminiCliHttpError(currentResponse, retryErrorText, false);
|
|
636
648
|
}
|
|
637
649
|
}
|
|
638
650
|
|
|
@@ -671,6 +683,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
|
|
671
683
|
}
|
|
672
684
|
output.stopReason = options?.signal?.aborted ? "aborted" : "error";
|
|
673
685
|
output.errorStatus = extractHttpStatusFromError(error);
|
|
686
|
+
output.transportFailure = transportFailureFacts(error);
|
|
674
687
|
output.errorMessage = await appendRawHttpRequestDumpFor400(
|
|
675
688
|
error instanceof Error ? error.message : JSON.stringify(error),
|
|
676
689
|
error,
|
|
@@ -20,6 +20,7 @@ import type {
|
|
|
20
20
|
} from "../types";
|
|
21
21
|
import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
|
|
22
22
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
23
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
23
24
|
import { finalizeErrorMessage, type RawHttpRequestDump, withHttpStatus } from "../utils/http-inspector";
|
|
24
25
|
import { normalizeSchemaForCCA, normalizeSchemaForGoogle, toolWireSchema } from "../utils/schema";
|
|
25
26
|
import {
|
|
@@ -868,7 +869,11 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
|
|
|
868
869
|
new Error(`Google API error (${response.status}): ${extractGoogleErrorMessage(errorText)}`),
|
|
869
870
|
response.status,
|
|
870
871
|
);
|
|
871
|
-
if (
|
|
872
|
+
if (
|
|
873
|
+
!options?.fallbackManaged &&
|
|
874
|
+
firstTokenTime === undefined &&
|
|
875
|
+
isForcedToolChoiceUnsupportedError(error, true)
|
|
876
|
+
) {
|
|
872
877
|
const beforeMark = resolveToolChoice(model, options?.toolChoice);
|
|
873
878
|
markToolChoiceIncapability(model, "auto", error.message);
|
|
874
879
|
stream.push({
|
|
@@ -938,6 +943,7 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
|
|
|
938
943
|
}
|
|
939
944
|
output.stopReason = options?.signal?.aborted ? "aborted" : "error";
|
|
940
945
|
output.errorStatus = extractHttpStatusFromError(error);
|
|
946
|
+
output.transportFailure = transportFailureFacts(error);
|
|
941
947
|
output.errorMessage = await finalizeErrorMessage(error, rawRequestDump);
|
|
942
948
|
output.duration = Date.now() - startTime;
|
|
943
949
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
package/src/providers/ollama.ts
CHANGED
|
@@ -16,6 +16,7 @@ import type {
|
|
|
16
16
|
} from "../types";
|
|
17
17
|
import { normalizeSystemPrompts } from "../utils";
|
|
18
18
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
19
|
+
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
19
20
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
20
21
|
import { parseStreamingJson } from "../utils/json-parse";
|
|
21
22
|
import { resolveRetryBudget } from "../utils/retry-budget";
|
|
@@ -431,7 +432,11 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
|
|
431
432
|
`HTTP ${response.status} from ${baseUrl}/api/chat: ${await response.text().catch(() => "")}`,
|
|
432
433
|
);
|
|
433
434
|
(error as Error & { status?: number }).status = response.status;
|
|
434
|
-
if (
|
|
435
|
+
if (
|
|
436
|
+
firstTokenTime === undefined &&
|
|
437
|
+
!options.fallbackManaged &&
|
|
438
|
+
isForcedToolChoiceUnsupportedError(error, true)
|
|
439
|
+
) {
|
|
435
440
|
markToolChoiceIncapability(model, "auto", error.message);
|
|
436
441
|
stream.push({
|
|
437
442
|
type: "toolChoiceIncapability",
|
|
@@ -590,6 +595,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
|
|
590
595
|
}
|
|
591
596
|
output.stopReason = options.signal?.aborted ? "aborted" : "error";
|
|
592
597
|
output.errorStatus = extractHttpStatusFromError(error);
|
|
598
|
+
output.transportFailure = transportFailureFacts(error);
|
|
593
599
|
output.errorMessage = await finalizeErrorMessage(error, rawRequestDump);
|
|
594
600
|
output.duration = Date.now() - startTime;
|
|
595
601
|
if (firstTokenTime) {
|
|
@@ -14,6 +14,7 @@ export type CodexRateLimits = {
|
|
|
14
14
|
export type CodexErrorInfo = {
|
|
15
15
|
message: string;
|
|
16
16
|
status: number;
|
|
17
|
+
code?: string;
|
|
17
18
|
friendlyMessage?: string;
|
|
18
19
|
rateLimits?: CodexRateLimits;
|
|
19
20
|
raw?: string;
|
|
@@ -24,6 +25,7 @@ export async function parseCodexError(response: Response): Promise<CodexErrorInf
|
|
|
24
25
|
let message = raw || response.statusText || "Request failed";
|
|
25
26
|
let friendlyMessage: string | undefined;
|
|
26
27
|
let rateLimits: CodexRateLimits | undefined;
|
|
28
|
+
let code: string | undefined;
|
|
27
29
|
|
|
28
30
|
try {
|
|
29
31
|
const parsed = JSON.parse(raw) as { error?: Record<string, unknown> };
|
|
@@ -45,16 +47,21 @@ export async function parseCodexError(response: Response): Promise<CodexErrorInf
|
|
|
45
47
|
? { primary, secondary }
|
|
46
48
|
: undefined;
|
|
47
49
|
|
|
48
|
-
|
|
50
|
+
code =
|
|
51
|
+
typeof (err as { code?: unknown }).code === "string"
|
|
52
|
+
? (err as { code: string }).code
|
|
53
|
+
: typeof (err as { type?: unknown }).type === "string"
|
|
54
|
+
? (err as { type: string }).type
|
|
55
|
+
: undefined;
|
|
49
56
|
const resetsAt = (err as { resets_at?: number }).resets_at ?? primary.resets_at ?? secondary.resets_at;
|
|
50
57
|
const mins = resetsAt ? Math.max(0, Math.round((resetsAt * 1000 - Date.now()) / 60000)) : undefined;
|
|
51
58
|
|
|
52
|
-
if (/usage_limit_reached|usage_not_included/i.test(code)) {
|
|
59
|
+
if (/usage_limit_reached|usage_not_included/i.test(code ?? "")) {
|
|
53
60
|
const planType = (err as { plan_type?: string }).plan_type;
|
|
54
61
|
const plan = planType ? ` (${String(planType).toLowerCase()} plan)` : "";
|
|
55
62
|
const when = mins !== undefined ? ` Try again in ~${mins} min.` : "";
|
|
56
63
|
friendlyMessage = `You have hit your ChatGPT usage limit${plan}.${when}`.trim();
|
|
57
|
-
} else if (/rate_limit_exceeded/i.test(code) || response.status === 429) {
|
|
64
|
+
} else if (/rate_limit_exceeded/i.test(code ?? "") || response.status === 429) {
|
|
58
65
|
const when = mins !== undefined ? ` Try again in ~${mins} min.` : "";
|
|
59
66
|
friendlyMessage = `ChatGPT rate limit exceeded.${when}`.trim();
|
|
60
67
|
}
|
|
@@ -69,6 +76,7 @@ export async function parseCodexError(response: Response): Promise<CodexErrorInf
|
|
|
69
76
|
message,
|
|
70
77
|
status: response.status,
|
|
71
78
|
friendlyMessage,
|
|
79
|
+
code,
|
|
72
80
|
rateLimits,
|
|
73
81
|
raw: raw,
|
|
74
82
|
};
|