@gajae-code/ai 0.13.3 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/CHANGELOG.md +45 -2
  2. package/dist/types/auth-broker/client.d.ts +9 -1
  3. package/dist/types/auth-broker/redact.d.ts +7 -0
  4. package/dist/types/auth-broker/remote-store.d.ts +50 -9
  5. package/dist/types/auth-broker/types.d.ts +14 -0
  6. package/dist/types/auth-broker/wire-schemas.d.ts +25 -0
  7. package/dist/types/auth-storage.d.ts +200 -6
  8. package/dist/types/core.d.ts +1 -0
  9. package/dist/types/model-cache.d.ts +4 -1
  10. package/dist/types/model-manager.d.ts +11 -0
  11. package/dist/types/provider-models/openai-compat.d.ts +5 -0
  12. package/dist/types/providers/anthropic.d.ts +31 -0
  13. package/dist/types/providers/cursor.d.ts +9 -1
  14. package/dist/types/providers/mock.d.ts +7 -1
  15. package/dist/types/providers/transform-messages.d.ts +18 -0
  16. package/dist/types/types.d.ts +28 -14
  17. package/dist/types/usage/grok-cli.d.ts +5 -0
  18. package/dist/types/usage.d.ts +6 -0
  19. package/dist/types/utils/discovery/openai-compatible.d.ts +5 -0
  20. package/dist/types/utils/event-stream.d.ts +4 -2
  21. package/dist/types/utils/fallback-transport.d.ts +10 -0
  22. package/dist/types/utils/http-inspector.d.ts +1 -0
  23. package/dist/types/utils/idle-iterator.d.ts +13 -1
  24. package/dist/types/utils/oauth/callback-server.d.ts +13 -0
  25. package/dist/types/utils/parse-bind.d.ts +8 -5
  26. package/dist/types/utils/tool-call-healing.d.ts +7 -0
  27. package/dist/types/utils/tool-choice-capability.d.ts +11 -0
  28. package/package.json +3 -2
  29. package/src/auth-broker/client.ts +30 -0
  30. package/src/auth-broker/redact.ts +15 -0
  31. package/src/auth-broker/refresher.ts +4 -2
  32. package/src/auth-broker/remote-store.ts +693 -70
  33. package/src/auth-broker/server.ts +57 -12
  34. package/src/auth-broker/types.ts +16 -0
  35. package/src/auth-broker/wire-schemas.ts +21 -0
  36. package/src/auth-gateway/server.ts +84 -19
  37. package/src/auth-storage.ts +985 -41
  38. package/src/core.ts +1 -0
  39. package/src/model-cache.ts +23 -4
  40. package/src/model-manager.ts +70 -11
  41. package/src/model-thinking.ts +21 -1
  42. package/src/models.json +1733 -392
  43. package/src/provider-models/descriptors.ts +5 -1
  44. package/src/provider-models/openai-compat.ts +52 -28
  45. package/src/providers/amazon-bedrock.ts +2 -1
  46. package/src/providers/anthropic.ts +824 -29
  47. package/src/providers/cursor.ts +83 -3
  48. package/src/providers/mock.ts +13 -3
  49. package/src/providers/ollama.ts +9 -2
  50. package/src/providers/openai-codex-responses.ts +16 -9
  51. package/src/providers/openai-completions.ts +5 -3
  52. package/src/providers/openai-responses-shared.ts +175 -21
  53. package/src/providers/register-builtins.ts +5 -2
  54. package/src/providers/transform-messages.ts +64 -1
  55. package/src/stream.ts +12 -2
  56. package/src/types.ts +28 -13
  57. package/src/usage/grok-cli.ts +86 -1
  58. package/src/usage.ts +7 -0
  59. package/src/utils/discovery/openai-compatible.ts +89 -4
  60. package/src/utils/event-stream.ts +11 -2
  61. package/src/utils/fallback-transport.ts +44 -2
  62. package/src/utils/http-inspector.ts +1 -0
  63. package/src/utils/idle-iterator.ts +29 -6
  64. package/src/utils/oauth/callback-server.ts +31 -1
  65. package/src/utils/parse-bind.ts +27 -0
  66. package/src/utils/tool-call-healing.ts +13 -2
  67. package/src/utils/tool-choice-capability.ts +386 -6
@@ -27,6 +27,64 @@ const enum ToolCallStatus {
27
27
  * - Injects synthetic "aborted" tool results
28
28
  * - Adds a <turn-aborted> guidance marker for the model
29
29
  */
30
+ /**
31
+ * Detect directly adjacent private thinking blocks inside one assistant message's
32
+ * content. `thinking` and `redacted_thinking` are one adjacency class: the
33
+ * Anthropic wire contract rejects a replayed assistant turn where two such blocks
34
+ * sit next to each other with no intervening `tool_use`/`text` block (#4416).
35
+ *
36
+ * This is a pure, allocation-free predicate used by defense-in-depth diagnostics
37
+ * (issue #4443): the write-time transcript assertion (coding-agent persistence)
38
+ * and the stream-assembler SSE diagnostic (anthropic stream completion). It never
39
+ * inspects block payloads — only the block-type sequence — so it cannot leak
40
+ * thinking text, signatures, or credentials.
41
+ *
42
+ * Blocks separated by any non-private block (`tool_use`, `text`, …) are ordinary
43
+ * interleaved-thinking shape and return `false`.
44
+ */
45
+ export function hasAdjacentPrivateThinkingBlocks(content: { type: string }[]): boolean {
46
+ let previousWasPrivate = false;
47
+ for (const block of content) {
48
+ const isPrivate = block.type === "thinking" || block.type === "redactedThinking";
49
+ if (isPrivate && previousWasPrivate) return true;
50
+ previousWasPrivate = isPrivate;
51
+ }
52
+ return false;
53
+ }
54
+
55
+ /**
56
+ * Collapse a run of directly adjacent `thinking` blocks inside one assistant message down
57
+ * to its first block.
58
+ *
59
+ * Anthropic accepts a replayed assistant turn carrying a single thinking block, and accepts
60
+ * thinking blocks separated by a `tool_use` (ordinary interleaved-thinking shape), but
61
+ * rejects two directly adjacent `thinking` blocks with
62
+ * `messages.N.content.M: thinking or redacted_thinking blocks in the latest assistant
63
+ * message cannot be modified`, citing the *second* block of the pair. Because the offending
64
+ * message keeps its index as history grows, a single such turn makes every later request in
65
+ * that session fail, and the mutation repair - scoped to the latest assistant message -
66
+ * can never reach it (#4416).
67
+ *
68
+ * `redactedThinking` is not folded in this phase, but the final send-boundary
69
+ * collapse in `convertAnthropicMessages` (#4425) treats `thinking` and
70
+ * `redacted_thinking` as one adjacency class per the API contract.
71
+ */
72
+ function collapseAdjacentThinking<T extends { type: string }>(content: T[]): T[] {
73
+ let previousWasThinking = false;
74
+ let dropped = false;
75
+ const collapsed: T[] = [];
76
+ for (const block of content) {
77
+ const thinking = block.type === "thinking";
78
+ if (thinking && previousWasThinking) {
79
+ dropped = true;
80
+ continue;
81
+ }
82
+ previousWasThinking = thinking;
83
+ collapsed.push(block);
84
+ }
85
+ return dropped ? collapsed : content;
86
+ }
87
+
30
88
  export function transformMessages<TApi extends Api>(
31
89
  messages: Message[],
32
90
  model: Model<TApi>,
@@ -160,9 +218,14 @@ export function transformMessages<TApi extends Api>(
160
218
  return block;
161
219
  });
162
220
 
221
+ // Only the Anthropic wire shape rejects adjacent private blocks; other targets
222
+ // either degrade reasoning to text above or carry their own encoding rules.
223
+ const replayableContent =
224
+ model.api === "anthropic-messages" ? collapseAdjacentThinking(transformedContent) : transformedContent;
225
+
163
226
  return {
164
227
  ...assistantMsg,
165
- content: transformedContent,
228
+ content: replayableContent,
166
229
  };
167
230
  }
168
231
  return msg;
package/src/stream.ts CHANGED
@@ -163,6 +163,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
163
163
  nvidia: "NVIDIA_API_KEY",
164
164
  nanogpt: "NANO_GPT_API_KEY",
165
165
  "lm-studio": "LM_STUDIO_API_KEY",
166
+ omlx: "OMLX_API_KEY",
166
167
  ollama: "OLLAMA_API_KEY",
167
168
  "ollama-cloud": "OLLAMA_CLOUD_API_KEY",
168
169
  "llama.cpp": "LLAMA_CPP_API_KEY",
@@ -493,7 +494,11 @@ export function streamSimple<TApi extends Api>(
493
494
  }
494
495
  const retryApiKey = options?.onAuthError ? (options.apiKey ?? getEnvApiKey(model.provider)) : undefined;
495
496
  if (retryApiKey) {
496
- const outer = new AssistantMessageEventStream();
497
+ const consumerAbortController = new AbortController();
498
+ const outer = new AssistantMessageEventStream(() => consumerAbortController.abort());
499
+ const requestSignal = options?.signal
500
+ ? AbortSignal.any([options.signal, consumerAbortController.signal])
501
+ : consumerAbortController.signal;
497
502
  const onAuthError = options!.onAuthError!;
498
503
  const runAttempt = async (apiKey: string, captureAuthFailure: boolean): Promise<AuthRetryFailure | undefined> => {
499
504
  const bufferedEvents: AssistantMessageEvent[] = [];
@@ -504,7 +509,12 @@ export function streamSimple<TApi extends Api>(
504
509
  };
505
510
 
506
511
  try {
507
- const inner = streamSimple(model, context, { ...options, apiKey, onAuthError: undefined });
512
+ const inner = streamSimple(model, context, {
513
+ ...options,
514
+ apiKey,
515
+ onAuthError: undefined,
516
+ signal: requestSignal,
517
+ });
508
518
  for await (const event of inner) {
509
519
  if (!emittedReplayUnsafeEvent && event.type === "start") {
510
520
  bufferedEvents.push(event);
package/src/types.ts CHANGED
@@ -176,6 +176,7 @@ export const KNOWN_PROVIDERS = [
176
176
  "xiaomi-token-plan-cn",
177
177
  "zenmux",
178
178
  "lm-studio",
179
+ "omlx",
179
180
  ] as const;
180
181
 
181
182
  export type KnownProvider = (typeof KNOWN_PROVIDERS)[number];
@@ -581,21 +582,35 @@ export interface ToolCall {
581
582
  */
582
583
  customWireName?: string;
583
584
  /**
584
- * Set when the provider detected the argument JSON was truncated — the model
585
- * hit its output-token limit (or the response was otherwise cut short) before
586
- * emitting a complete arguments object. The `arguments` field then holds a
587
- * best-effort partial parse and must not be executed as-is; the agent loop
588
- * rejects the call with a retryable error instead.
585
+ * Set when the provider detected the argument JSON was not safely executable —
586
+ * the model hit its output-token limit (or the response was otherwise cut short)
587
+ * before emitting a complete arguments object, the terminal payload was malformed,
588
+ * the streamed and terminal payloads conflicted, or the tool-call identity was
589
+ * ambiguous on the wire. The `arguments` field then holds a best-effort partial
590
+ * parse and must not be executed as-is; the agent loop rejects the call with a
591
+ * retryable, reason-specific error instead.
589
592
  */
590
593
  incompleteArguments?: boolean;
591
594
  /**
592
- * Set when the provider saw the argument JSON spell a printable non-ASCII
593
- * character as a `\uXXXX` escape instead of literal UTF-8. Hand-written hex
594
- * is where models mistype digits, and a mistyped nibble decodes to a
595
- * different but equally valid character, so the decoded arguments cannot be
596
- * verified or repaired after parsing. The agent loop treats such a turn as a
597
- * sampling accident: managed runs discard and re-request it, and execution
598
- * rejects the call rather than running on silently corrupted text.
595
+ * When `incompleteArguments` is set, the typed cause so the agent loop can give
596
+ * reason-specific recovery guidance:
597
+ * - `"truncated"`: the response was cut short mid-arguments (output-token limit).
598
+ * - `"malformed"`: the terminal arguments did not decode to a valid JSON object.
599
+ * - `"conflicting"`: the streamed and terminal argument payloads disagree.
600
+ * - `"ambiguous"`: the tool-call identity could not be unambiguously resolved
601
+ * (duplicate `call_id`, id/call_id collision, etc.), so attribution is unsafe.
602
+ * Absent when `incompleteArguments` is not set. Existing callers that read only
603
+ * `incompleteArguments` continue to work.
604
+ */
605
+ incompleteArgumentsReason?: "truncated" | "malformed" | "conflicting" | "ambiguous";
606
+ /**
607
+ * Set when the raw argument JSON spelled a printable non-ASCII character as a
608
+ * `\uXXXX` escape instead of literal UTF-8. Such a payload parses cleanly but
609
+ * is unverifiable: one mistyped hex digit decodes to a different, equally
610
+ * valid character, so the text can be silently wrong with no in-band evidence.
611
+ * The agent loop rejects the call with a retryable error instead of executing
612
+ * it. Escapes that are required (control characters) or unavoidable (lone
613
+ * surrogates) never set this.
599
614
  */
600
615
  escapedNonAsciiArguments?: boolean;
601
616
  }
@@ -648,7 +663,7 @@ export interface Usage {
648
663
  }
649
664
 
650
665
  export type StopReason = "stop" | "length" | "toolUse" | "error" | "aborted";
651
- export type AssistantErrorKind = "provider_safety_stop";
666
+ export type AssistantErrorKind = "provider_safety_stop" | "local_snapshot_failure" | "local_buffer_overflow";
652
667
 
653
668
  export interface OpenAIResponsesHistoryPayload {
654
669
  type: "openaiResponsesHistory";
@@ -13,8 +13,15 @@ interface BillingUsage {
13
13
  used: number;
14
14
  billingPeriodEnd: string;
15
15
  }
16
+
17
+ interface WeeklyBillingUsage {
18
+ creditUsagePercent: number;
19
+ billingPeriodEnd: string;
20
+ }
21
+
16
22
  const DEFAULT_GROK_BUILD_BASE_URL = "https://cli-chat-proxy.grok.com/v1";
17
23
  const ALLOWED_GROK_BUILD_HOSTS = new Set(["cli-chat-proxy.grok.com"]);
24
+ const GROK_WEEKLY_WINDOW_MS = 7 * 24 * 60 * 60 * 1000;
18
25
 
19
26
  function isRecord(value: unknown): value is Record<string, unknown> {
20
27
  return !!value && typeof value === "object";
@@ -46,6 +53,23 @@ export function parseGrokCliBillingUsage(payload: unknown): BillingUsage {
46
53
  return { monthlyLimit, used, billingPeriodEnd };
47
54
  }
48
55
 
56
+ export function parseGrokCliWeeklyBillingUsage(payload: unknown): WeeklyBillingUsage | undefined {
57
+ if (!isRecord(payload) || !isRecord(payload.config)) return undefined;
58
+ const currentPeriod = isRecord(payload.config.currentPeriod) ? payload.config.currentPeriod : undefined;
59
+ if (currentPeriod?.type !== "USAGE_PERIOD_TYPE_WEEKLY") return undefined;
60
+
61
+ const billingPeriodEnd = payload.config.billingPeriodEnd;
62
+ if (typeof billingPeriodEnd !== "string" || !Number.isFinite(new Date(billingPeriodEnd).getTime())) {
63
+ return undefined;
64
+ }
65
+
66
+ const rawCreditUsagePercent = payload.config.creditUsagePercent;
67
+ const creditUsagePercent = rawCreditUsagePercent === undefined ? 0 : finiteNumber(rawCreditUsagePercent);
68
+ if (creditUsagePercent === undefined) return undefined;
69
+
70
+ return { creditUsagePercent, billingPeriodEnd };
71
+ }
72
+
49
73
  function isAllowedGrokCredentialHost(baseUrl: string): boolean {
50
74
  try {
51
75
  const url = new URL(baseUrl);
@@ -108,6 +132,33 @@ function buildMonthlyUsageLimit(usage: BillingUsage, nowMs: number): UsageLimit
108
132
  };
109
133
  }
110
134
 
135
+ function buildWeeklyUsageLimit(usage: WeeklyBillingUsage): UsageLimit {
136
+ const percent = Math.min(Math.max(usage.creditUsagePercent, 0), 100);
137
+ const usedFraction = percent / 100;
138
+ const resetsAt = new Date(usage.billingPeriodEnd).getTime();
139
+ return {
140
+ id: "grok-build:weekly",
141
+ label: "SuperGrok weekly credits",
142
+ scope: { provider: "grok-build", shared: true, windowId: "weekly" },
143
+ window: {
144
+ id: "weekly",
145
+ label: "Weekly",
146
+ durationMs: GROK_WEEKLY_WINDOW_MS,
147
+ resetsAt,
148
+ },
149
+ amount: {
150
+ unit: "percent",
151
+ used: percent,
152
+ limit: 100,
153
+ remaining: 100 - percent,
154
+ usedFraction,
155
+ remainingFraction: 1 - usedFraction,
156
+ },
157
+ status: percent >= 95 ? "exhausted" : percent >= 80 ? "warning" : "ok",
158
+ notes: [`${percent}% used`],
159
+ };
160
+ }
161
+
111
162
  export const grokCliUsageProvider: UsageProvider = {
112
163
  id: "grok-build",
113
164
 
@@ -151,10 +202,42 @@ export const grokCliUsageProvider: UsageProvider = {
151
202
  }
152
203
 
153
204
  const nowMs = Date.now();
205
+ let weekly: WeeklyBillingUsage | undefined;
206
+ try {
207
+ const weeklyResponse = await ctx.fetch(`${billingBaseUrl}/billing?format=credits`, {
208
+ headers: {
209
+ Authorization: `Bearer ${accessToken}`,
210
+ "x-xai-token-auth": "xai-grok-cli",
211
+ accept: "application/json",
212
+ },
213
+ signal: params.signal,
214
+ });
215
+ if (weeklyResponse.ok) {
216
+ weekly = parseGrokCliWeeklyBillingUsage((await weeklyResponse.json()) as unknown);
217
+ }
218
+ } catch (error) {
219
+ if (params.signal?.aborted && billing.monthlyLimit <= 0) throw error;
220
+ ctx.logger?.debug("Grok Build weekly billing request failed", { error: String(error) });
221
+ }
222
+
223
+ const limits: UsageLimit[] = [];
224
+ if (billing.monthlyLimit > 0) {
225
+ limits.push(buildMonthlyUsageLimit(billing, nowMs));
226
+ }
227
+ if (weekly) {
228
+ limits.push(buildWeeklyUsageLimit(weekly));
229
+ }
230
+ if (limits.length === 0) {
231
+ ctx.logger?.warn("Grok Build billing response contained no usable quota", {
232
+ provider: params.provider,
233
+ });
234
+ return null;
235
+ }
236
+
154
237
  return {
155
238
  provider: "grok-build",
156
239
  fetchedAt: nowMs,
157
- limits: [buildMonthlyUsageLimit(billing, nowMs)],
240
+ limits,
158
241
  metadata: {
159
242
  email: params.credential.email,
160
243
  accountId: params.credential.accountId,
@@ -167,6 +250,8 @@ export const grokCliUsageProvider: UsageProvider = {
167
250
 
168
251
  export const grokCliRankingStrategy: CredentialRankingStrategy = {
169
252
  findWindowLimits(report) {
253
+ const weekly = report.limits.find(limit => limit.id === "grok-build:weekly");
254
+ if (weekly) return { secondary: weekly };
170
255
  const monthly = report.limits.find(limit => limit.id === "grok-build:7d");
171
256
  return { secondary: monthly };
172
257
  },
package/src/usage.ts CHANGED
@@ -131,6 +131,12 @@ export interface UsageLogger {
131
131
  }
132
132
 
133
133
  /** Credential bundle for usage endpoints. */
134
+ /** MCP OAuth authority carried through usage-triggered refresh. */
135
+ export interface UsageMCPOAuthBinding {
136
+ resourceOrigin: string;
137
+ tokenEndpoint: string;
138
+ }
139
+
134
140
  export interface UsageCredential {
135
141
  type: "api_key" | "oauth";
136
142
  apiKey?: string;
@@ -141,6 +147,7 @@ export interface UsageCredential {
141
147
  projectId?: string;
142
148
  email?: string;
143
149
  enterpriseUrl?: string;
150
+ mcpBinding?: UsageMCPOAuthBinding;
144
151
  metadata?: Record<string, unknown>;
145
152
  }
146
153
 
@@ -1,8 +1,10 @@
1
1
  import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "@gajae-code/ai";
2
2
  import * as z from "zod/v4";
3
3
  import type { Api, FetchImpl, Model, Provider } from "../../types";
4
+ import { toNumber } from "../../utils";
4
5
 
5
6
  const MODELS_PATH = "/models";
7
+ const MAX_MODELS_RESPONSE_BYTES = 1_000_000;
6
8
 
7
9
  /**
8
10
  * Minimal OpenAI-style model entry shape consumed by discovery.
@@ -100,6 +102,26 @@ export interface FetchOpenAICompatibleModelsOptions<TApi extends Api> {
100
102
  ) => Model<TApi> | null;
101
103
  }
102
104
 
105
+ /**
106
+ * Resolves an endpoint for an implicit local provider without allowing an
107
+ * environment override to turn its keyless discovery into a remote request.
108
+ */
109
+ export function resolveLoopbackOpenAIBaseUrl(value: string | undefined, fallback: string): string {
110
+ const candidate = value?.trim();
111
+ if (!candidate) return fallback;
112
+ try {
113
+ const parsed = new URL(candidate);
114
+ const host = parsed.hostname.toLowerCase();
115
+ const isLoopback = host === "localhost" || host === "::1" || /^127(?:\.\d{1,3}){3}$/.test(host);
116
+ if ((parsed.protocol === "http:" || parsed.protocol === "https:") && isLoopback) {
117
+ return candidate;
118
+ }
119
+ } catch {
120
+ // Fall back to the fixed loopback endpoint below.
121
+ }
122
+ return fallback;
123
+ }
124
+
103
125
  /**
104
126
  * Fetches and normalizes an OpenAI-compatible `/models` catalog.
105
127
  *
@@ -128,7 +150,9 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
128
150
  response = await fetchImpl(buildModelsUrl(baseUrl), {
129
151
  method: "GET",
130
152
  headers: requestHeaders,
131
- signal: options.signal,
153
+ signal: options.signal
154
+ ? AbortSignal.any([options.signal, AbortSignal.timeout(5_000)])
155
+ : AbortSignal.timeout(5_000),
132
156
  });
133
157
  } catch {
134
158
  return null;
@@ -144,7 +168,7 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
144
168
 
145
169
  let payload: unknown;
146
170
  try {
147
- payload = await response.json();
171
+ payload = JSON.parse(await readModelsResponse(response));
148
172
  } catch {
149
173
  return null;
150
174
  }
@@ -162,6 +186,17 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
162
186
 
163
187
  const deduped = new Map<string, Model<TApi>>();
164
188
  for (const entry of entries) {
189
+ const rawContextWindow = firstPositiveModelNumber(
190
+ UNK_CONTEXT_WINDOW,
191
+ entry.max_model_len,
192
+ entry.context_length,
193
+ entry.context_window,
194
+ entry.max_context_length,
195
+ entry.max_position_embeddings,
196
+ );
197
+
198
+ const rawMaxTokens = firstPositiveModelNumber(UNK_MAX_TOKENS, entry.max_tokens, entry.max_output_tokens);
199
+
165
200
  const defaults: Model<TApi> = {
166
201
  id: entry.id,
167
202
  name: typeof entry.name === "string" && entry.name.length > 0 ? entry.name : entry.id,
@@ -171,8 +206,8 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
171
206
  reasoning: false,
172
207
  input: ["text"],
173
208
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
174
- contextWindow: UNK_CONTEXT_WINDOW,
175
- maxTokens: UNK_MAX_TOKENS,
209
+ contextWindow: rawContextWindow,
210
+ maxTokens: rawMaxTokens,
176
211
  };
177
212
 
178
213
  const mapped = options.mapModel?.(entry, defaults, context) ?? defaults;
@@ -188,6 +223,39 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
188
223
  return Array.from(deduped.values()).sort((left, right) => left.id.localeCompare(right.id));
189
224
  }
190
225
 
226
+ async function readModelsResponse(response: Response): Promise<string> {
227
+ const contentLength = Number(response.headers.get("content-length"));
228
+ if (Number.isFinite(contentLength) && contentLength > MAX_MODELS_RESPONSE_BYTES) {
229
+ throw new Error("OpenAI-compatible models response exceeds the size limit");
230
+ }
231
+ if (!response.body) return "";
232
+ const reader = response.body.getReader();
233
+ const chunks: Uint8Array[] = [];
234
+ let total = 0;
235
+ try {
236
+ while (true) {
237
+ const { done, value } = await reader.read();
238
+ if (done) break;
239
+ if (!value) continue;
240
+ total += value.byteLength;
241
+ if (total > MAX_MODELS_RESPONSE_BYTES) {
242
+ await reader.cancel();
243
+ throw new Error("OpenAI-compatible models response exceeds the size limit");
244
+ }
245
+ chunks.push(value);
246
+ }
247
+ } finally {
248
+ reader.releaseLock();
249
+ }
250
+ const body = new Uint8Array(total);
251
+ let offset = 0;
252
+ for (const chunk of chunks) {
253
+ body.set(chunk, offset);
254
+ offset += chunk.byteLength;
255
+ }
256
+ return new TextDecoder().decode(body);
257
+ }
258
+
191
259
  function normalizeBaseUrl(baseUrl: string): string {
192
260
  const trimmed = baseUrl.trim();
193
261
  if (!trimmed) {
@@ -244,3 +312,20 @@ function extractModelEntriesFromNode(node: unknown): ParsedOpenAICompatibleModel
244
312
 
245
313
  return null;
246
314
  }
315
+
316
+ /**
317
+ * First finite positive number among candidates, else the fallback.
318
+ *
319
+ * Rejects non-numbers, non-finite values (JSON `1e400` parses to
320
+ * `Infinity`), zero, and negatives so a malformed catalog field can never
321
+ * poison compaction thresholds or output budgets with `Infinity`.
322
+ */
323
+ function firstPositiveModelNumber(fallback: number, ...candidates: readonly unknown[]): number {
324
+ for (const candidate of candidates) {
325
+ const value = toNumber(candidate);
326
+ if (value !== undefined && value > 0 && Number.isFinite(value)) {
327
+ return value;
328
+ }
329
+ }
330
+ return fallback;
331
+ }
@@ -35,8 +35,9 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
35
35
  rejectFinalResult!: (err: unknown) => void;
36
36
  isComplete: (event: T) => boolean;
37
37
  extractResult: (event: T) => R;
38
+ #onConsumerClose?: () => void;
38
39
 
39
- constructor(isComplete: (event: T) => boolean, extractResult: (event: T) => R) {
40
+ constructor(isComplete: (event: T) => boolean, extractResult: (event: T) => R, onConsumerClose?: () => void) {
40
41
  const { promise, resolve, reject } = Promise.withResolvers<R>();
41
42
  // Prevent an unhandled rejection when fail() is called but nobody awaits result().
42
43
  // Callers who do await result() still receive the rejection normally.
@@ -46,6 +47,7 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
46
47
  this.rejectFinalResult = reject;
47
48
  this.isComplete = isComplete;
48
49
  this.extractResult = extractResult;
50
+ this.#onConsumerClose = onConsumerClose;
49
51
  }
50
52
 
51
53
  #enqueue(node: QueueNode<T>): void {
@@ -82,6 +84,11 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
82
84
  return this.#pendingConsumerDrains.size;
83
85
  }
84
86
 
87
+ /** Whether an async iterator is currently consuming this stream. */
88
+ get hasActiveConsumer(): boolean {
89
+ return this.#activeConsumerCount > 0;
90
+ }
91
+
85
92
  #settleConsumerDrain(drain: ConsumerDrain, status: "resolve" | "reject", reason?: unknown): void {
86
93
  if (drain.settled) return;
87
94
  drain.settled = true;
@@ -235,6 +242,7 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
235
242
  } finally {
236
243
  this.#activeConsumerCount -= 1;
237
244
  this.#settleAllConsumerDrains("reject", new Error("Event stream consumer stopped before drain completed"));
245
+ if (!this.done) this.#onConsumerClose?.();
238
246
  }
239
247
  }
240
248
 
@@ -244,7 +252,7 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
244
252
  }
245
253
 
246
254
  export class AssistantMessageEventStream extends EventStream<AssistantMessageEvent, AssistantMessage> {
247
- constructor() {
255
+ constructor(onConsumerClose?: () => void) {
248
256
  super(
249
257
  event => event.type === "done" || event.type === "error",
250
258
  event => {
@@ -255,6 +263,7 @@ export class AssistantMessageEventStream extends EventStream<AssistantMessageEve
255
263
  }
256
264
  throw new Error("Unexpected event type for final result");
257
265
  },
266
+ onConsumerClose,
258
267
  );
259
268
  }
260
269
  }
@@ -49,6 +49,16 @@ export interface TransportFailureFacts {
49
49
  /** OpenAI's typed `error.code`, preserved separately at the transport boundary. */
50
50
  openaiErrorCode?: string;
51
51
  headers?: Record<string, string>;
52
+ /** Safe request-size observation for retry amplification policy. Never contains body content. */
53
+ requestBytes?: number;
54
+ /** Time spent waiting for the first semantic stream event on the failed request. */
55
+ firstEventElapsedMs?: number;
56
+ /** Configured first-event window before any bounded endpoint grace. */
57
+ firstEventTimeoutMs?: number;
58
+ /** Coarse endpoint class; deliberately excludes host, path, credentials, and query parameters. */
59
+ endpointClass?: "canonical" | "custom";
60
+ /** Provider-supplied ceiling for total attempts, including the initial request. */
61
+ retryMaxAttempts?: number;
52
62
  }
53
63
 
54
64
  /** Opaque per-invocation marker required by managed fallback transport calls. */
@@ -118,6 +128,14 @@ function finiteStatus(value: unknown): number | undefined {
118
128
  return typeof value === "number" && Number.isFinite(value) ? value : undefined;
119
129
  }
120
130
 
131
+ function finiteNonNegativeInteger(value: unknown): number | undefined {
132
+ return typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : undefined;
133
+ }
134
+
135
+ function finitePositiveInteger(value: unknown): number | undefined {
136
+ return typeof value === "number" && Number.isInteger(value) && value > 0 ? value : undefined;
137
+ }
138
+
121
139
  function stringValue(value: unknown): string | undefined {
122
140
  return typeof value === "string" ? value : undefined;
123
141
  }
@@ -207,6 +225,13 @@ export function transportFailureFacts(
207
225
  // (consumers deliberately re-run transportFailureFacts on embedded facts).
208
226
  const headers = retainedHeaderRecord(rawHeaders);
209
227
  const normalizedCode = providerCode?.toLowerCase();
228
+ const requestBytes = finiteNonNegativeInteger(propertyOf(value, "requestBytes"));
229
+ const firstEventElapsedMs = finiteNonNegativeInteger(propertyOf(value, "firstEventElapsedMs"));
230
+ const firstEventTimeoutMs = finiteNonNegativeInteger(propertyOf(value, "firstEventTimeoutMs"));
231
+ const retryMaxAttempts = finitePositiveInteger(propertyOf(value, "retryMaxAttempts"));
232
+ const endpointClassValue = propertyOf(value, "endpointClass");
233
+ const endpointClass =
234
+ endpointClassValue === "canonical" || endpointClassValue === "custom" ? endpointClassValue : undefined;
210
235
  if (
211
236
  status === undefined &&
212
237
  headers === undefined &&
@@ -215,11 +240,28 @@ export function transportFailureFacts(
215
240
  !isRateLimitCode(normalizedCode) &&
216
241
  !isContextOverflowCode(normalizedCode) &&
217
242
  normalizedCode !== STREAM_FIRST_EVENT_TIMEOUT_PROVIDER_CODE &&
218
- normalizedCode !== EMPTY_RESPONSE_PROVIDER_CODE
243
+ normalizedCode !== EMPTY_RESPONSE_PROVIDER_CODE &&
244
+ requestBytes === undefined &&
245
+ firstEventElapsedMs === undefined &&
246
+ firstEventTimeoutMs === undefined &&
247
+ endpointClass === undefined &&
248
+ retryMaxAttempts === undefined
219
249
  ) {
220
250
  return undefined;
221
251
  }
222
- return { kind: "transport", status, providerCode, anthropicErrorType, openaiErrorCode, headers };
252
+ return {
253
+ kind: "transport",
254
+ status,
255
+ providerCode,
256
+ anthropicErrorType,
257
+ openaiErrorCode,
258
+ headers,
259
+ ...(requestBytes === undefined ? {} : { requestBytes }),
260
+ ...(firstEventElapsedMs === undefined ? {} : { firstEventElapsedMs }),
261
+ ...(firstEventTimeoutMs === undefined ? {} : { firstEventTimeoutMs }),
262
+ ...(endpointClass === undefined ? {} : { endpointClass }),
263
+ ...(retryMaxAttempts === undefined ? {} : { retryMaxAttempts }),
264
+ };
223
265
  }
224
266
 
225
267
  function headersOf(headers: TransportHeaders | undefined): Headers | undefined {
@@ -12,6 +12,7 @@ export type RawHttpRequestDump = {
12
12
  url?: string;
13
13
  headers?: Record<string, string>;
14
14
  body?: unknown;
15
+ diagnostics?: Record<string, unknown>;
15
16
  };
16
17
 
17
18
  export type CapturedHttpErrorResponse = {