@gajae-code/ai 0.13.3 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +45 -2
- package/dist/types/auth-broker/client.d.ts +9 -1
- package/dist/types/auth-broker/redact.d.ts +7 -0
- package/dist/types/auth-broker/remote-store.d.ts +50 -9
- package/dist/types/auth-broker/types.d.ts +14 -0
- package/dist/types/auth-broker/wire-schemas.d.ts +25 -0
- package/dist/types/auth-storage.d.ts +200 -6
- package/dist/types/core.d.ts +1 -0
- package/dist/types/model-cache.d.ts +4 -1
- package/dist/types/model-manager.d.ts +11 -0
- package/dist/types/provider-models/openai-compat.d.ts +5 -0
- package/dist/types/providers/anthropic.d.ts +31 -0
- package/dist/types/providers/cursor.d.ts +9 -1
- package/dist/types/providers/mock.d.ts +7 -1
- package/dist/types/providers/transform-messages.d.ts +18 -0
- package/dist/types/types.d.ts +28 -14
- package/dist/types/usage/grok-cli.d.ts +5 -0
- package/dist/types/usage.d.ts +6 -0
- package/dist/types/utils/discovery/openai-compatible.d.ts +5 -0
- package/dist/types/utils/event-stream.d.ts +4 -2
- package/dist/types/utils/fallback-transport.d.ts +10 -0
- package/dist/types/utils/http-inspector.d.ts +1 -0
- package/dist/types/utils/idle-iterator.d.ts +13 -1
- package/dist/types/utils/oauth/callback-server.d.ts +13 -0
- package/dist/types/utils/parse-bind.d.ts +8 -5
- package/dist/types/utils/tool-call-healing.d.ts +7 -0
- package/dist/types/utils/tool-choice-capability.d.ts +11 -0
- package/package.json +3 -2
- package/src/auth-broker/client.ts +30 -0
- package/src/auth-broker/redact.ts +15 -0
- package/src/auth-broker/refresher.ts +4 -2
- package/src/auth-broker/remote-store.ts +693 -70
- package/src/auth-broker/server.ts +57 -12
- package/src/auth-broker/types.ts +16 -0
- package/src/auth-broker/wire-schemas.ts +21 -0
- package/src/auth-gateway/server.ts +84 -19
- package/src/auth-storage.ts +985 -41
- package/src/core.ts +1 -0
- package/src/model-cache.ts +23 -4
- package/src/model-manager.ts +70 -11
- package/src/model-thinking.ts +21 -1
- package/src/models.json +1733 -392
- package/src/provider-models/descriptors.ts +5 -1
- package/src/provider-models/openai-compat.ts +52 -28
- package/src/providers/amazon-bedrock.ts +2 -1
- package/src/providers/anthropic.ts +824 -29
- package/src/providers/cursor.ts +83 -3
- package/src/providers/mock.ts +13 -3
- package/src/providers/ollama.ts +9 -2
- package/src/providers/openai-codex-responses.ts +16 -9
- package/src/providers/openai-completions.ts +5 -3
- package/src/providers/openai-responses-shared.ts +175 -21
- package/src/providers/register-builtins.ts +5 -2
- package/src/providers/transform-messages.ts +64 -1
- package/src/stream.ts +12 -2
- package/src/types.ts +28 -13
- package/src/usage/grok-cli.ts +86 -1
- package/src/usage.ts +7 -0
- package/src/utils/discovery/openai-compatible.ts +89 -4
- package/src/utils/event-stream.ts +11 -2
- package/src/utils/fallback-transport.ts +44 -2
- package/src/utils/http-inspector.ts +1 -0
- package/src/utils/idle-iterator.ts +29 -6
- package/src/utils/oauth/callback-server.ts +31 -1
- package/src/utils/parse-bind.ts +27 -0
- package/src/utils/tool-call-healing.ts +13 -2
- package/src/utils/tool-choice-capability.ts +386 -6
|
@@ -27,6 +27,64 @@ const enum ToolCallStatus {
|
|
|
27
27
|
* - Injects synthetic "aborted" tool results
|
|
28
28
|
* - Adds a <turn-aborted> guidance marker for the model
|
|
29
29
|
*/
|
|
30
|
+
/**
|
|
31
|
+
* Detect directly adjacent private thinking blocks inside one assistant message's
|
|
32
|
+
* content. `thinking` and `redacted_thinking` are one adjacency class: the
|
|
33
|
+
* Anthropic wire contract rejects a replayed assistant turn where two such blocks
|
|
34
|
+
* sit next to each other with no intervening `tool_use`/`text` block (#4416).
|
|
35
|
+
*
|
|
36
|
+
* This is a pure, allocation-free predicate used by defense-in-depth diagnostics
|
|
37
|
+
* (issue #4443): the write-time transcript assertion (coding-agent persistence)
|
|
38
|
+
* and the stream-assembler SSE diagnostic (anthropic stream completion). It never
|
|
39
|
+
* inspects block payloads — only the block-type sequence — so it cannot leak
|
|
40
|
+
* thinking text, signatures, or credentials.
|
|
41
|
+
*
|
|
42
|
+
* Blocks separated by any non-private block (`tool_use`, `text`, …) are ordinary
|
|
43
|
+
* interleaved-thinking shape and return `false`.
|
|
44
|
+
*/
|
|
45
|
+
export function hasAdjacentPrivateThinkingBlocks(content: { type: string }[]): boolean {
|
|
46
|
+
let previousWasPrivate = false;
|
|
47
|
+
for (const block of content) {
|
|
48
|
+
const isPrivate = block.type === "thinking" || block.type === "redactedThinking";
|
|
49
|
+
if (isPrivate && previousWasPrivate) return true;
|
|
50
|
+
previousWasPrivate = isPrivate;
|
|
51
|
+
}
|
|
52
|
+
return false;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Collapse a run of directly adjacent `thinking` blocks inside one assistant message down
|
|
57
|
+
* to its first block.
|
|
58
|
+
*
|
|
59
|
+
* Anthropic accepts a replayed assistant turn carrying a single thinking block, and accepts
|
|
60
|
+
* thinking blocks separated by a `tool_use` (ordinary interleaved-thinking shape), but
|
|
61
|
+
* rejects two directly adjacent `thinking` blocks with
|
|
62
|
+
* `messages.N.content.M: thinking or redacted_thinking blocks in the latest assistant
|
|
63
|
+
* message cannot be modified`, citing the *second* block of the pair. Because the offending
|
|
64
|
+
* message keeps its index as history grows, a single such turn makes every later request in
|
|
65
|
+
* that session fail, and the mutation repair - scoped to the latest assistant message -
|
|
66
|
+
* can never reach it (#4416).
|
|
67
|
+
*
|
|
68
|
+
* `redactedThinking` is not folded in this phase, but the final send-boundary
|
|
69
|
+
* collapse in `convertAnthropicMessages` (#4425) treats `thinking` and
|
|
70
|
+
* `redacted_thinking` as one adjacency class per the API contract.
|
|
71
|
+
*/
|
|
72
|
+
function collapseAdjacentThinking<T extends { type: string }>(content: T[]): T[] {
|
|
73
|
+
let previousWasThinking = false;
|
|
74
|
+
let dropped = false;
|
|
75
|
+
const collapsed: T[] = [];
|
|
76
|
+
for (const block of content) {
|
|
77
|
+
const thinking = block.type === "thinking";
|
|
78
|
+
if (thinking && previousWasThinking) {
|
|
79
|
+
dropped = true;
|
|
80
|
+
continue;
|
|
81
|
+
}
|
|
82
|
+
previousWasThinking = thinking;
|
|
83
|
+
collapsed.push(block);
|
|
84
|
+
}
|
|
85
|
+
return dropped ? collapsed : content;
|
|
86
|
+
}
|
|
87
|
+
|
|
30
88
|
export function transformMessages<TApi extends Api>(
|
|
31
89
|
messages: Message[],
|
|
32
90
|
model: Model<TApi>,
|
|
@@ -160,9 +218,14 @@ export function transformMessages<TApi extends Api>(
|
|
|
160
218
|
return block;
|
|
161
219
|
});
|
|
162
220
|
|
|
221
|
+
// Only the Anthropic wire shape rejects adjacent private blocks; other targets
|
|
222
|
+
// either degrade reasoning to text above or carry their own encoding rules.
|
|
223
|
+
const replayableContent =
|
|
224
|
+
model.api === "anthropic-messages" ? collapseAdjacentThinking(transformedContent) : transformedContent;
|
|
225
|
+
|
|
163
226
|
return {
|
|
164
227
|
...assistantMsg,
|
|
165
|
-
content:
|
|
228
|
+
content: replayableContent,
|
|
166
229
|
};
|
|
167
230
|
}
|
|
168
231
|
return msg;
|
package/src/stream.ts
CHANGED
|
@@ -163,6 +163,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
163
163
|
nvidia: "NVIDIA_API_KEY",
|
|
164
164
|
nanogpt: "NANO_GPT_API_KEY",
|
|
165
165
|
"lm-studio": "LM_STUDIO_API_KEY",
|
|
166
|
+
omlx: "OMLX_API_KEY",
|
|
166
167
|
ollama: "OLLAMA_API_KEY",
|
|
167
168
|
"ollama-cloud": "OLLAMA_CLOUD_API_KEY",
|
|
168
169
|
"llama.cpp": "LLAMA_CPP_API_KEY",
|
|
@@ -493,7 +494,11 @@ export function streamSimple<TApi extends Api>(
|
|
|
493
494
|
}
|
|
494
495
|
const retryApiKey = options?.onAuthError ? (options.apiKey ?? getEnvApiKey(model.provider)) : undefined;
|
|
495
496
|
if (retryApiKey) {
|
|
496
|
-
const
|
|
497
|
+
const consumerAbortController = new AbortController();
|
|
498
|
+
const outer = new AssistantMessageEventStream(() => consumerAbortController.abort());
|
|
499
|
+
const requestSignal = options?.signal
|
|
500
|
+
? AbortSignal.any([options.signal, consumerAbortController.signal])
|
|
501
|
+
: consumerAbortController.signal;
|
|
497
502
|
const onAuthError = options!.onAuthError!;
|
|
498
503
|
const runAttempt = async (apiKey: string, captureAuthFailure: boolean): Promise<AuthRetryFailure | undefined> => {
|
|
499
504
|
const bufferedEvents: AssistantMessageEvent[] = [];
|
|
@@ -504,7 +509,12 @@ export function streamSimple<TApi extends Api>(
|
|
|
504
509
|
};
|
|
505
510
|
|
|
506
511
|
try {
|
|
507
|
-
const inner = streamSimple(model, context, {
|
|
512
|
+
const inner = streamSimple(model, context, {
|
|
513
|
+
...options,
|
|
514
|
+
apiKey,
|
|
515
|
+
onAuthError: undefined,
|
|
516
|
+
signal: requestSignal,
|
|
517
|
+
});
|
|
508
518
|
for await (const event of inner) {
|
|
509
519
|
if (!emittedReplayUnsafeEvent && event.type === "start") {
|
|
510
520
|
bufferedEvents.push(event);
|
package/src/types.ts
CHANGED
|
@@ -176,6 +176,7 @@ export const KNOWN_PROVIDERS = [
|
|
|
176
176
|
"xiaomi-token-plan-cn",
|
|
177
177
|
"zenmux",
|
|
178
178
|
"lm-studio",
|
|
179
|
+
"omlx",
|
|
179
180
|
] as const;
|
|
180
181
|
|
|
181
182
|
export type KnownProvider = (typeof KNOWN_PROVIDERS)[number];
|
|
@@ -581,21 +582,35 @@ export interface ToolCall {
|
|
|
581
582
|
*/
|
|
582
583
|
customWireName?: string;
|
|
583
584
|
/**
|
|
584
|
-
* Set when the provider detected the argument JSON was
|
|
585
|
-
* hit its output-token limit (or the response was otherwise cut short)
|
|
586
|
-
* emitting a complete arguments object
|
|
587
|
-
*
|
|
588
|
-
*
|
|
585
|
+
* Set when the provider detected the argument JSON was not safely executable —
|
|
586
|
+
* the model hit its output-token limit (or the response was otherwise cut short)
|
|
587
|
+
* before emitting a complete arguments object, the terminal payload was malformed,
|
|
588
|
+
* the streamed and terminal payloads conflicted, or the tool-call identity was
|
|
589
|
+
* ambiguous on the wire. The `arguments` field then holds a best-effort partial
|
|
590
|
+
* parse and must not be executed as-is; the agent loop rejects the call with a
|
|
591
|
+
* retryable, reason-specific error instead.
|
|
589
592
|
*/
|
|
590
593
|
incompleteArguments?: boolean;
|
|
591
594
|
/**
|
|
592
|
-
*
|
|
593
|
-
*
|
|
594
|
-
*
|
|
595
|
-
*
|
|
596
|
-
*
|
|
597
|
-
*
|
|
598
|
-
*
|
|
595
|
+
* When `incompleteArguments` is set, the typed cause so the agent loop can give
|
|
596
|
+
* reason-specific recovery guidance:
|
|
597
|
+
* - `"truncated"`: the response was cut short mid-arguments (output-token limit).
|
|
598
|
+
* - `"malformed"`: the terminal arguments did not decode to a valid JSON object.
|
|
599
|
+
* - `"conflicting"`: the streamed and terminal argument payloads disagree.
|
|
600
|
+
* - `"ambiguous"`: the tool-call identity could not be unambiguously resolved
|
|
601
|
+
* (duplicate `call_id`, id/call_id collision, etc.), so attribution is unsafe.
|
|
602
|
+
* Absent when `incompleteArguments` is not set. Existing callers that read only
|
|
603
|
+
* `incompleteArguments` continue to work.
|
|
604
|
+
*/
|
|
605
|
+
incompleteArgumentsReason?: "truncated" | "malformed" | "conflicting" | "ambiguous";
|
|
606
|
+
/**
|
|
607
|
+
* Set when the raw argument JSON spelled a printable non-ASCII character as a
|
|
608
|
+
* `\uXXXX` escape instead of literal UTF-8. Such a payload parses cleanly but
|
|
609
|
+
* is unverifiable: one mistyped hex digit decodes to a different, equally
|
|
610
|
+
* valid character, so the text can be silently wrong with no in-band evidence.
|
|
611
|
+
* The agent loop rejects the call with a retryable error instead of executing
|
|
612
|
+
* it. Escapes that are required (control characters) or unavoidable (lone
|
|
613
|
+
* surrogates) never set this.
|
|
599
614
|
*/
|
|
600
615
|
escapedNonAsciiArguments?: boolean;
|
|
601
616
|
}
|
|
@@ -648,7 +663,7 @@ export interface Usage {
|
|
|
648
663
|
}
|
|
649
664
|
|
|
650
665
|
export type StopReason = "stop" | "length" | "toolUse" | "error" | "aborted";
|
|
651
|
-
export type AssistantErrorKind = "provider_safety_stop";
|
|
666
|
+
export type AssistantErrorKind = "provider_safety_stop" | "local_snapshot_failure" | "local_buffer_overflow";
|
|
652
667
|
|
|
653
668
|
export interface OpenAIResponsesHistoryPayload {
|
|
654
669
|
type: "openaiResponsesHistory";
|
package/src/usage/grok-cli.ts
CHANGED
|
@@ -13,8 +13,15 @@ interface BillingUsage {
|
|
|
13
13
|
used: number;
|
|
14
14
|
billingPeriodEnd: string;
|
|
15
15
|
}
|
|
16
|
+
|
|
17
|
+
interface WeeklyBillingUsage {
|
|
18
|
+
creditUsagePercent: number;
|
|
19
|
+
billingPeriodEnd: string;
|
|
20
|
+
}
|
|
21
|
+
|
|
16
22
|
const DEFAULT_GROK_BUILD_BASE_URL = "https://cli-chat-proxy.grok.com/v1";
|
|
17
23
|
const ALLOWED_GROK_BUILD_HOSTS = new Set(["cli-chat-proxy.grok.com"]);
|
|
24
|
+
const GROK_WEEKLY_WINDOW_MS = 7 * 24 * 60 * 60 * 1000;
|
|
18
25
|
|
|
19
26
|
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
20
27
|
return !!value && typeof value === "object";
|
|
@@ -46,6 +53,23 @@ export function parseGrokCliBillingUsage(payload: unknown): BillingUsage {
|
|
|
46
53
|
return { monthlyLimit, used, billingPeriodEnd };
|
|
47
54
|
}
|
|
48
55
|
|
|
56
|
+
export function parseGrokCliWeeklyBillingUsage(payload: unknown): WeeklyBillingUsage | undefined {
|
|
57
|
+
if (!isRecord(payload) || !isRecord(payload.config)) return undefined;
|
|
58
|
+
const currentPeriod = isRecord(payload.config.currentPeriod) ? payload.config.currentPeriod : undefined;
|
|
59
|
+
if (currentPeriod?.type !== "USAGE_PERIOD_TYPE_WEEKLY") return undefined;
|
|
60
|
+
|
|
61
|
+
const billingPeriodEnd = payload.config.billingPeriodEnd;
|
|
62
|
+
if (typeof billingPeriodEnd !== "string" || !Number.isFinite(new Date(billingPeriodEnd).getTime())) {
|
|
63
|
+
return undefined;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const rawCreditUsagePercent = payload.config.creditUsagePercent;
|
|
67
|
+
const creditUsagePercent = rawCreditUsagePercent === undefined ? 0 : finiteNumber(rawCreditUsagePercent);
|
|
68
|
+
if (creditUsagePercent === undefined) return undefined;
|
|
69
|
+
|
|
70
|
+
return { creditUsagePercent, billingPeriodEnd };
|
|
71
|
+
}
|
|
72
|
+
|
|
49
73
|
function isAllowedGrokCredentialHost(baseUrl: string): boolean {
|
|
50
74
|
try {
|
|
51
75
|
const url = new URL(baseUrl);
|
|
@@ -108,6 +132,33 @@ function buildMonthlyUsageLimit(usage: BillingUsage, nowMs: number): UsageLimit
|
|
|
108
132
|
};
|
|
109
133
|
}
|
|
110
134
|
|
|
135
|
+
function buildWeeklyUsageLimit(usage: WeeklyBillingUsage): UsageLimit {
|
|
136
|
+
const percent = Math.min(Math.max(usage.creditUsagePercent, 0), 100);
|
|
137
|
+
const usedFraction = percent / 100;
|
|
138
|
+
const resetsAt = new Date(usage.billingPeriodEnd).getTime();
|
|
139
|
+
return {
|
|
140
|
+
id: "grok-build:weekly",
|
|
141
|
+
label: "SuperGrok weekly credits",
|
|
142
|
+
scope: { provider: "grok-build", shared: true, windowId: "weekly" },
|
|
143
|
+
window: {
|
|
144
|
+
id: "weekly",
|
|
145
|
+
label: "Weekly",
|
|
146
|
+
durationMs: GROK_WEEKLY_WINDOW_MS,
|
|
147
|
+
resetsAt,
|
|
148
|
+
},
|
|
149
|
+
amount: {
|
|
150
|
+
unit: "percent",
|
|
151
|
+
used: percent,
|
|
152
|
+
limit: 100,
|
|
153
|
+
remaining: 100 - percent,
|
|
154
|
+
usedFraction,
|
|
155
|
+
remainingFraction: 1 - usedFraction,
|
|
156
|
+
},
|
|
157
|
+
status: percent >= 95 ? "exhausted" : percent >= 80 ? "warning" : "ok",
|
|
158
|
+
notes: [`${percent}% used`],
|
|
159
|
+
};
|
|
160
|
+
}
|
|
161
|
+
|
|
111
162
|
export const grokCliUsageProvider: UsageProvider = {
|
|
112
163
|
id: "grok-build",
|
|
113
164
|
|
|
@@ -151,10 +202,42 @@ export const grokCliUsageProvider: UsageProvider = {
|
|
|
151
202
|
}
|
|
152
203
|
|
|
153
204
|
const nowMs = Date.now();
|
|
205
|
+
let weekly: WeeklyBillingUsage | undefined;
|
|
206
|
+
try {
|
|
207
|
+
const weeklyResponse = await ctx.fetch(`${billingBaseUrl}/billing?format=credits`, {
|
|
208
|
+
headers: {
|
|
209
|
+
Authorization: `Bearer ${accessToken}`,
|
|
210
|
+
"x-xai-token-auth": "xai-grok-cli",
|
|
211
|
+
accept: "application/json",
|
|
212
|
+
},
|
|
213
|
+
signal: params.signal,
|
|
214
|
+
});
|
|
215
|
+
if (weeklyResponse.ok) {
|
|
216
|
+
weekly = parseGrokCliWeeklyBillingUsage((await weeklyResponse.json()) as unknown);
|
|
217
|
+
}
|
|
218
|
+
} catch (error) {
|
|
219
|
+
if (params.signal?.aborted && billing.monthlyLimit <= 0) throw error;
|
|
220
|
+
ctx.logger?.debug("Grok Build weekly billing request failed", { error: String(error) });
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
const limits: UsageLimit[] = [];
|
|
224
|
+
if (billing.monthlyLimit > 0) {
|
|
225
|
+
limits.push(buildMonthlyUsageLimit(billing, nowMs));
|
|
226
|
+
}
|
|
227
|
+
if (weekly) {
|
|
228
|
+
limits.push(buildWeeklyUsageLimit(weekly));
|
|
229
|
+
}
|
|
230
|
+
if (limits.length === 0) {
|
|
231
|
+
ctx.logger?.warn("Grok Build billing response contained no usable quota", {
|
|
232
|
+
provider: params.provider,
|
|
233
|
+
});
|
|
234
|
+
return null;
|
|
235
|
+
}
|
|
236
|
+
|
|
154
237
|
return {
|
|
155
238
|
provider: "grok-build",
|
|
156
239
|
fetchedAt: nowMs,
|
|
157
|
-
limits
|
|
240
|
+
limits,
|
|
158
241
|
metadata: {
|
|
159
242
|
email: params.credential.email,
|
|
160
243
|
accountId: params.credential.accountId,
|
|
@@ -167,6 +250,8 @@ export const grokCliUsageProvider: UsageProvider = {
|
|
|
167
250
|
|
|
168
251
|
export const grokCliRankingStrategy: CredentialRankingStrategy = {
|
|
169
252
|
findWindowLimits(report) {
|
|
253
|
+
const weekly = report.limits.find(limit => limit.id === "grok-build:weekly");
|
|
254
|
+
if (weekly) return { secondary: weekly };
|
|
170
255
|
const monthly = report.limits.find(limit => limit.id === "grok-build:7d");
|
|
171
256
|
return { secondary: monthly };
|
|
172
257
|
},
|
package/src/usage.ts
CHANGED
|
@@ -131,6 +131,12 @@ export interface UsageLogger {
|
|
|
131
131
|
}
|
|
132
132
|
|
|
133
133
|
/** Credential bundle for usage endpoints. */
|
|
134
|
+
/** MCP OAuth authority carried through usage-triggered refresh. */
|
|
135
|
+
export interface UsageMCPOAuthBinding {
|
|
136
|
+
resourceOrigin: string;
|
|
137
|
+
tokenEndpoint: string;
|
|
138
|
+
}
|
|
139
|
+
|
|
134
140
|
export interface UsageCredential {
|
|
135
141
|
type: "api_key" | "oauth";
|
|
136
142
|
apiKey?: string;
|
|
@@ -141,6 +147,7 @@ export interface UsageCredential {
|
|
|
141
147
|
projectId?: string;
|
|
142
148
|
email?: string;
|
|
143
149
|
enterpriseUrl?: string;
|
|
150
|
+
mcpBinding?: UsageMCPOAuthBinding;
|
|
144
151
|
metadata?: Record<string, unknown>;
|
|
145
152
|
}
|
|
146
153
|
|
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "@gajae-code/ai";
|
|
2
2
|
import * as z from "zod/v4";
|
|
3
3
|
import type { Api, FetchImpl, Model, Provider } from "../../types";
|
|
4
|
+
import { toNumber } from "../../utils";
|
|
4
5
|
|
|
5
6
|
const MODELS_PATH = "/models";
|
|
7
|
+
const MAX_MODELS_RESPONSE_BYTES = 1_000_000;
|
|
6
8
|
|
|
7
9
|
/**
|
|
8
10
|
* Minimal OpenAI-style model entry shape consumed by discovery.
|
|
@@ -100,6 +102,26 @@ export interface FetchOpenAICompatibleModelsOptions<TApi extends Api> {
|
|
|
100
102
|
) => Model<TApi> | null;
|
|
101
103
|
}
|
|
102
104
|
|
|
105
|
+
/**
|
|
106
|
+
* Resolves an endpoint for an implicit local provider without allowing an
|
|
107
|
+
* environment override to turn its keyless discovery into a remote request.
|
|
108
|
+
*/
|
|
109
|
+
export function resolveLoopbackOpenAIBaseUrl(value: string | undefined, fallback: string): string {
|
|
110
|
+
const candidate = value?.trim();
|
|
111
|
+
if (!candidate) return fallback;
|
|
112
|
+
try {
|
|
113
|
+
const parsed = new URL(candidate);
|
|
114
|
+
const host = parsed.hostname.toLowerCase();
|
|
115
|
+
const isLoopback = host === "localhost" || host === "::1" || /^127(?:\.\d{1,3}){3}$/.test(host);
|
|
116
|
+
if ((parsed.protocol === "http:" || parsed.protocol === "https:") && isLoopback) {
|
|
117
|
+
return candidate;
|
|
118
|
+
}
|
|
119
|
+
} catch {
|
|
120
|
+
// Fall back to the fixed loopback endpoint below.
|
|
121
|
+
}
|
|
122
|
+
return fallback;
|
|
123
|
+
}
|
|
124
|
+
|
|
103
125
|
/**
|
|
104
126
|
* Fetches and normalizes an OpenAI-compatible `/models` catalog.
|
|
105
127
|
*
|
|
@@ -128,7 +150,9 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
|
|
|
128
150
|
response = await fetchImpl(buildModelsUrl(baseUrl), {
|
|
129
151
|
method: "GET",
|
|
130
152
|
headers: requestHeaders,
|
|
131
|
-
signal: options.signal
|
|
153
|
+
signal: options.signal
|
|
154
|
+
? AbortSignal.any([options.signal, AbortSignal.timeout(5_000)])
|
|
155
|
+
: AbortSignal.timeout(5_000),
|
|
132
156
|
});
|
|
133
157
|
} catch {
|
|
134
158
|
return null;
|
|
@@ -144,7 +168,7 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
|
|
|
144
168
|
|
|
145
169
|
let payload: unknown;
|
|
146
170
|
try {
|
|
147
|
-
payload = await response
|
|
171
|
+
payload = JSON.parse(await readModelsResponse(response));
|
|
148
172
|
} catch {
|
|
149
173
|
return null;
|
|
150
174
|
}
|
|
@@ -162,6 +186,17 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
|
|
|
162
186
|
|
|
163
187
|
const deduped = new Map<string, Model<TApi>>();
|
|
164
188
|
for (const entry of entries) {
|
|
189
|
+
const rawContextWindow = firstPositiveModelNumber(
|
|
190
|
+
UNK_CONTEXT_WINDOW,
|
|
191
|
+
entry.max_model_len,
|
|
192
|
+
entry.context_length,
|
|
193
|
+
entry.context_window,
|
|
194
|
+
entry.max_context_length,
|
|
195
|
+
entry.max_position_embeddings,
|
|
196
|
+
);
|
|
197
|
+
|
|
198
|
+
const rawMaxTokens = firstPositiveModelNumber(UNK_MAX_TOKENS, entry.max_tokens, entry.max_output_tokens);
|
|
199
|
+
|
|
165
200
|
const defaults: Model<TApi> = {
|
|
166
201
|
id: entry.id,
|
|
167
202
|
name: typeof entry.name === "string" && entry.name.length > 0 ? entry.name : entry.id,
|
|
@@ -171,8 +206,8 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
|
|
|
171
206
|
reasoning: false,
|
|
172
207
|
input: ["text"],
|
|
173
208
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
174
|
-
contextWindow:
|
|
175
|
-
maxTokens:
|
|
209
|
+
contextWindow: rawContextWindow,
|
|
210
|
+
maxTokens: rawMaxTokens,
|
|
176
211
|
};
|
|
177
212
|
|
|
178
213
|
const mapped = options.mapModel?.(entry, defaults, context) ?? defaults;
|
|
@@ -188,6 +223,39 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
|
|
|
188
223
|
return Array.from(deduped.values()).sort((left, right) => left.id.localeCompare(right.id));
|
|
189
224
|
}
|
|
190
225
|
|
|
226
|
+
async function readModelsResponse(response: Response): Promise<string> {
|
|
227
|
+
const contentLength = Number(response.headers.get("content-length"));
|
|
228
|
+
if (Number.isFinite(contentLength) && contentLength > MAX_MODELS_RESPONSE_BYTES) {
|
|
229
|
+
throw new Error("OpenAI-compatible models response exceeds the size limit");
|
|
230
|
+
}
|
|
231
|
+
if (!response.body) return "";
|
|
232
|
+
const reader = response.body.getReader();
|
|
233
|
+
const chunks: Uint8Array[] = [];
|
|
234
|
+
let total = 0;
|
|
235
|
+
try {
|
|
236
|
+
while (true) {
|
|
237
|
+
const { done, value } = await reader.read();
|
|
238
|
+
if (done) break;
|
|
239
|
+
if (!value) continue;
|
|
240
|
+
total += value.byteLength;
|
|
241
|
+
if (total > MAX_MODELS_RESPONSE_BYTES) {
|
|
242
|
+
await reader.cancel();
|
|
243
|
+
throw new Error("OpenAI-compatible models response exceeds the size limit");
|
|
244
|
+
}
|
|
245
|
+
chunks.push(value);
|
|
246
|
+
}
|
|
247
|
+
} finally {
|
|
248
|
+
reader.releaseLock();
|
|
249
|
+
}
|
|
250
|
+
const body = new Uint8Array(total);
|
|
251
|
+
let offset = 0;
|
|
252
|
+
for (const chunk of chunks) {
|
|
253
|
+
body.set(chunk, offset);
|
|
254
|
+
offset += chunk.byteLength;
|
|
255
|
+
}
|
|
256
|
+
return new TextDecoder().decode(body);
|
|
257
|
+
}
|
|
258
|
+
|
|
191
259
|
function normalizeBaseUrl(baseUrl: string): string {
|
|
192
260
|
const trimmed = baseUrl.trim();
|
|
193
261
|
if (!trimmed) {
|
|
@@ -244,3 +312,20 @@ function extractModelEntriesFromNode(node: unknown): ParsedOpenAICompatibleModel
|
|
|
244
312
|
|
|
245
313
|
return null;
|
|
246
314
|
}
|
|
315
|
+
|
|
316
|
+
/**
|
|
317
|
+
* First finite positive number among candidates, else the fallback.
|
|
318
|
+
*
|
|
319
|
+
* Rejects non-numbers, non-finite values (JSON `1e400` parses to
|
|
320
|
+
* `Infinity`), zero, and negatives so a malformed catalog field can never
|
|
321
|
+
* poison compaction thresholds or output budgets with `Infinity`.
|
|
322
|
+
*/
|
|
323
|
+
function firstPositiveModelNumber(fallback: number, ...candidates: readonly unknown[]): number {
|
|
324
|
+
for (const candidate of candidates) {
|
|
325
|
+
const value = toNumber(candidate);
|
|
326
|
+
if (value !== undefined && value > 0 && Number.isFinite(value)) {
|
|
327
|
+
return value;
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
return fallback;
|
|
331
|
+
}
|
|
@@ -35,8 +35,9 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
|
|
35
35
|
rejectFinalResult!: (err: unknown) => void;
|
|
36
36
|
isComplete: (event: T) => boolean;
|
|
37
37
|
extractResult: (event: T) => R;
|
|
38
|
+
#onConsumerClose?: () => void;
|
|
38
39
|
|
|
39
|
-
constructor(isComplete: (event: T) => boolean, extractResult: (event: T) => R) {
|
|
40
|
+
constructor(isComplete: (event: T) => boolean, extractResult: (event: T) => R, onConsumerClose?: () => void) {
|
|
40
41
|
const { promise, resolve, reject } = Promise.withResolvers<R>();
|
|
41
42
|
// Prevent an unhandled rejection when fail() is called but nobody awaits result().
|
|
42
43
|
// Callers who do await result() still receive the rejection normally.
|
|
@@ -46,6 +47,7 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
|
|
46
47
|
this.rejectFinalResult = reject;
|
|
47
48
|
this.isComplete = isComplete;
|
|
48
49
|
this.extractResult = extractResult;
|
|
50
|
+
this.#onConsumerClose = onConsumerClose;
|
|
49
51
|
}
|
|
50
52
|
|
|
51
53
|
#enqueue(node: QueueNode<T>): void {
|
|
@@ -82,6 +84,11 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
|
|
82
84
|
return this.#pendingConsumerDrains.size;
|
|
83
85
|
}
|
|
84
86
|
|
|
87
|
+
/** Whether an async iterator is currently consuming this stream. */
|
|
88
|
+
get hasActiveConsumer(): boolean {
|
|
89
|
+
return this.#activeConsumerCount > 0;
|
|
90
|
+
}
|
|
91
|
+
|
|
85
92
|
#settleConsumerDrain(drain: ConsumerDrain, status: "resolve" | "reject", reason?: unknown): void {
|
|
86
93
|
if (drain.settled) return;
|
|
87
94
|
drain.settled = true;
|
|
@@ -235,6 +242,7 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
|
|
235
242
|
} finally {
|
|
236
243
|
this.#activeConsumerCount -= 1;
|
|
237
244
|
this.#settleAllConsumerDrains("reject", new Error("Event stream consumer stopped before drain completed"));
|
|
245
|
+
if (!this.done) this.#onConsumerClose?.();
|
|
238
246
|
}
|
|
239
247
|
}
|
|
240
248
|
|
|
@@ -244,7 +252,7 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
|
|
244
252
|
}
|
|
245
253
|
|
|
246
254
|
export class AssistantMessageEventStream extends EventStream<AssistantMessageEvent, AssistantMessage> {
|
|
247
|
-
constructor() {
|
|
255
|
+
constructor(onConsumerClose?: () => void) {
|
|
248
256
|
super(
|
|
249
257
|
event => event.type === "done" || event.type === "error",
|
|
250
258
|
event => {
|
|
@@ -255,6 +263,7 @@ export class AssistantMessageEventStream extends EventStream<AssistantMessageEve
|
|
|
255
263
|
}
|
|
256
264
|
throw new Error("Unexpected event type for final result");
|
|
257
265
|
},
|
|
266
|
+
onConsumerClose,
|
|
258
267
|
);
|
|
259
268
|
}
|
|
260
269
|
}
|
|
@@ -49,6 +49,16 @@ export interface TransportFailureFacts {
|
|
|
49
49
|
/** OpenAI's typed `error.code`, preserved separately at the transport boundary. */
|
|
50
50
|
openaiErrorCode?: string;
|
|
51
51
|
headers?: Record<string, string>;
|
|
52
|
+
/** Safe request-size observation for retry amplification policy. Never contains body content. */
|
|
53
|
+
requestBytes?: number;
|
|
54
|
+
/** Time spent waiting for the first semantic stream event on the failed request. */
|
|
55
|
+
firstEventElapsedMs?: number;
|
|
56
|
+
/** Configured first-event window before any bounded endpoint grace. */
|
|
57
|
+
firstEventTimeoutMs?: number;
|
|
58
|
+
/** Coarse endpoint class; deliberately excludes host, path, credentials, and query parameters. */
|
|
59
|
+
endpointClass?: "canonical" | "custom";
|
|
60
|
+
/** Provider-supplied ceiling for total attempts, including the initial request. */
|
|
61
|
+
retryMaxAttempts?: number;
|
|
52
62
|
}
|
|
53
63
|
|
|
54
64
|
/** Opaque per-invocation marker required by managed fallback transport calls. */
|
|
@@ -118,6 +128,14 @@ function finiteStatus(value: unknown): number | undefined {
|
|
|
118
128
|
return typeof value === "number" && Number.isFinite(value) ? value : undefined;
|
|
119
129
|
}
|
|
120
130
|
|
|
131
|
+
function finiteNonNegativeInteger(value: unknown): number | undefined {
|
|
132
|
+
return typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : undefined;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
function finitePositiveInteger(value: unknown): number | undefined {
|
|
136
|
+
return typeof value === "number" && Number.isInteger(value) && value > 0 ? value : undefined;
|
|
137
|
+
}
|
|
138
|
+
|
|
121
139
|
function stringValue(value: unknown): string | undefined {
|
|
122
140
|
return typeof value === "string" ? value : undefined;
|
|
123
141
|
}
|
|
@@ -207,6 +225,13 @@ export function transportFailureFacts(
|
|
|
207
225
|
// (consumers deliberately re-run transportFailureFacts on embedded facts).
|
|
208
226
|
const headers = retainedHeaderRecord(rawHeaders);
|
|
209
227
|
const normalizedCode = providerCode?.toLowerCase();
|
|
228
|
+
const requestBytes = finiteNonNegativeInteger(propertyOf(value, "requestBytes"));
|
|
229
|
+
const firstEventElapsedMs = finiteNonNegativeInteger(propertyOf(value, "firstEventElapsedMs"));
|
|
230
|
+
const firstEventTimeoutMs = finiteNonNegativeInteger(propertyOf(value, "firstEventTimeoutMs"));
|
|
231
|
+
const retryMaxAttempts = finitePositiveInteger(propertyOf(value, "retryMaxAttempts"));
|
|
232
|
+
const endpointClassValue = propertyOf(value, "endpointClass");
|
|
233
|
+
const endpointClass =
|
|
234
|
+
endpointClassValue === "canonical" || endpointClassValue === "custom" ? endpointClassValue : undefined;
|
|
210
235
|
if (
|
|
211
236
|
status === undefined &&
|
|
212
237
|
headers === undefined &&
|
|
@@ -215,11 +240,28 @@ export function transportFailureFacts(
|
|
|
215
240
|
!isRateLimitCode(normalizedCode) &&
|
|
216
241
|
!isContextOverflowCode(normalizedCode) &&
|
|
217
242
|
normalizedCode !== STREAM_FIRST_EVENT_TIMEOUT_PROVIDER_CODE &&
|
|
218
|
-
normalizedCode !== EMPTY_RESPONSE_PROVIDER_CODE
|
|
243
|
+
normalizedCode !== EMPTY_RESPONSE_PROVIDER_CODE &&
|
|
244
|
+
requestBytes === undefined &&
|
|
245
|
+
firstEventElapsedMs === undefined &&
|
|
246
|
+
firstEventTimeoutMs === undefined &&
|
|
247
|
+
endpointClass === undefined &&
|
|
248
|
+
retryMaxAttempts === undefined
|
|
219
249
|
) {
|
|
220
250
|
return undefined;
|
|
221
251
|
}
|
|
222
|
-
return {
|
|
252
|
+
return {
|
|
253
|
+
kind: "transport",
|
|
254
|
+
status,
|
|
255
|
+
providerCode,
|
|
256
|
+
anthropicErrorType,
|
|
257
|
+
openaiErrorCode,
|
|
258
|
+
headers,
|
|
259
|
+
...(requestBytes === undefined ? {} : { requestBytes }),
|
|
260
|
+
...(firstEventElapsedMs === undefined ? {} : { firstEventElapsedMs }),
|
|
261
|
+
...(firstEventTimeoutMs === undefined ? {} : { firstEventTimeoutMs }),
|
|
262
|
+
...(endpointClass === undefined ? {} : { endpointClass }),
|
|
263
|
+
...(retryMaxAttempts === undefined ? {} : { retryMaxAttempts }),
|
|
264
|
+
};
|
|
223
265
|
}
|
|
224
266
|
|
|
225
267
|
function headersOf(headers: TransportHeaders | undefined): Headers | undefined {
|