@gajae-code/ai 0.5.2 → 0.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@ import { $env, $inheritedEnv } from "@gajae-code/utils";
2
2
  import type { ModelManagerOptions } from "../model-manager";
3
3
  import { Effort } from "../model-thinking";
4
4
  import { getBundledModels } from "../models";
5
- import type { Api, Model, ThinkingConfig } from "../types";
5
+ import type { Api, FetchImpl, Model, Provider, ThinkingConfig } from "../types";
6
6
  import { isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
7
7
  import {
8
8
  fetchOpenAICompatibleModels,
@@ -1414,64 +1414,99 @@ export function cloudflareAiGatewayModelManagerOptions(
1414
1414
  // ---------------------------------------------------------------------------
1415
1415
  // 20. Xiaomi
1416
1416
  // ---------------------------------------------------------------------------
1417
+ /** Region codes for Xiaomi Token Plan clusters exposed as separate login providers. */
1418
+ export type XiaomiTokenPlanRegion = "sgp" | "ams" | "cn";
1417
1419
 
1420
+ /** Configures Xiaomi standard or regional Token Plan OpenAI-compatible model discovery. */
1418
1421
  export interface XiaomiModelManagerConfig {
1419
1422
  apiKey?: string;
1420
1423
  baseUrl?: string;
1424
+ fetch?: FetchImpl;
1425
+ providerId?: Provider;
1426
+ tokenPlanRegion?: XiaomiTokenPlanRegion;
1427
+ }
1428
+ const XIAOMI_TOKEN_PLAN_BASE_URLS: Record<XiaomiTokenPlanRegion, string> = {
1429
+ sgp: "https://token-plan-sgp.xiaomimimo.com/v1",
1430
+ ams: "https://token-plan-ams.xiaomimimo.com/v1",
1431
+ cn: "https://token-plan-cn.xiaomimimo.com/v1",
1432
+ };
1433
+
1434
+ const XIAOMI_TOKEN_PLAN_FALLBACK_BASE_URLS = [
1435
+ XIAOMI_TOKEN_PLAN_BASE_URLS.sgp,
1436
+ XIAOMI_TOKEN_PLAN_BASE_URLS.ams,
1437
+ XIAOMI_TOKEN_PLAN_BASE_URLS.cn,
1438
+ ];
1439
+
1440
+ function inferXiaomiTokenPlanRegion(providerId: Provider): XiaomiTokenPlanRegion | undefined {
1441
+ if (providerId === "xiaomi-token-plan-sgp") return "sgp";
1442
+ if (providerId === "xiaomi-token-plan-ams") return "ams";
1443
+ if (providerId === "xiaomi-token-plan-cn") return "cn";
1444
+ return undefined;
1421
1445
  }
1422
1446
 
1447
+ /** Builds a Xiaomi model manager, preserving Token Plan region provider ids during discovery. */
1423
1448
  export function xiaomiModelManagerOptions(
1424
1449
  config?: XiaomiModelManagerConfig,
1425
1450
  ): ModelManagerOptions<"openai-completions"> {
1426
1451
  const apiKey = config?.apiKey;
1427
- // Xiaomi splits API keys across two backends: standard `sk-` keys hit
1428
- // api.xiaomimimo.com; "token plan" `tp-` keys hit either the SG or EU
1429
- // token-plan host. Try SGP first; if discovery fails, retry AMS.
1430
- const TOKEN_PLAN_SGP_BASE_URL = "https://token-plan-sgp.xiaomimimo.com/v1";
1431
- const TOKEN_PLAN_AMS_BASE_URL = "https://token-plan-ams.xiaomimimo.com/v1";
1432
- const defaultBaseUrl = apiKey?.startsWith("tp-") ? TOKEN_PLAN_SGP_BASE_URL : "https://api.xiaomimimo.com/v1";
1433
- // Token-plan keys always use the TP baseUrl; config?.baseUrl (from catalog)
1452
+ const providerId = config?.providerId ?? "xiaomi";
1453
+ const tokenPlanRegion = config?.tokenPlanRegion ?? inferXiaomiTokenPlanRegion(providerId);
1454
+ const tokenPlanBaseUrls = tokenPlanRegion
1455
+ ? [XIAOMI_TOKEN_PLAN_BASE_URLS[tokenPlanRegion]]
1456
+ : XIAOMI_TOKEN_PLAN_FALLBACK_BASE_URLS;
1457
+ const XIAOMI_STANDARD_BASE_URL = "https://api.xiaomimimo.com/v1";
1458
+ const isTokenPlanProvider = tokenPlanRegion !== undefined || providerId.startsWith("xiaomi-token-plan-");
1459
+ const isTokenPlanKey = isTokenPlanProvider || apiKey?.startsWith("tp-");
1460
+ // Token-plan keys always use a TP cluster; config?.baseUrl (from catalog)
1434
1461
  // would incorrectly pin to the standard endpoint (api.xiaomimimo.com).
1435
- const baseUrl = apiKey?.startsWith("tp-") ? defaultBaseUrl : (config?.baseUrl ?? defaultBaseUrl);
1462
+ const baseUrl = isTokenPlanKey ? tokenPlanBaseUrls[0] : (config?.baseUrl ?? XIAOMI_STANDARD_BASE_URL);
1436
1463
  const references = createBundledReferenceMap<"openai-completions">("xiaomi");
1464
+ const throwOnAuthStatus = (response: Response): Error | undefined => {
1465
+ if (response.status === 401 || response.status === 403) {
1466
+ return new Error(`Authentication failed (${response.status}) for ${response.url}`);
1467
+ }
1468
+ return undefined;
1469
+ };
1470
+ const fetchModels = (url: string) =>
1471
+ fetchOpenAICompatibleModels({
1472
+ api: "openai-completions",
1473
+ provider: providerId,
1474
+ baseUrl: url,
1475
+ apiKey,
1476
+ filterModel: (_entry, model) => !model.id.includes("-tts"),
1477
+ mapModel: (entry, defaults) => {
1478
+ const reference = references.get(defaults.id);
1479
+ const model = mapWithBundledReference(entry, defaults, reference);
1480
+ return {
1481
+ ...model,
1482
+ api: "openai-completions",
1483
+ provider: providerId,
1484
+ baseUrl: defaults.baseUrl,
1485
+ name: toModelName(entry.display_name, model.name),
1486
+ };
1487
+ },
1488
+ fetch: config?.fetch,
1489
+ throwOnStatus: throwOnAuthStatus,
1490
+ });
1437
1491
  return {
1438
- providerId: "xiaomi",
1492
+ providerId,
1439
1493
  ...(apiKey && {
1440
1494
  fetchDynamicModels: async () => {
1441
- const sgpResult = await fetchOpenAICompatibleModels({
1442
- api: "openai-completions",
1443
- provider: "xiaomi",
1444
- baseUrl,
1445
- apiKey,
1446
- filterModel: (_entry, model) => !model.id.includes("-tts"),
1447
- mapModel: (entry, defaults) => {
1448
- const reference = references.get(defaults.id);
1449
- const model = mapWithBundledReference(entry, defaults, reference);
1450
- return {
1451
- ...model,
1452
- name: toModelName(entry.display_name, model.name),
1453
- };
1454
- },
1455
- });
1456
- if (sgpResult || !apiKey?.startsWith("tp-")) {
1457
- return sgpResult;
1495
+ if (!isTokenPlanKey) {
1496
+ return fetchModels(baseUrl);
1458
1497
  }
1459
- // Token-plan discovery failed with SGP; retry with AMS
1460
- return fetchOpenAICompatibleModels({
1461
- api: "openai-completions",
1462
- provider: "xiaomi",
1463
- baseUrl: TOKEN_PLAN_AMS_BASE_URL,
1464
- apiKey,
1465
- filterModel: (_entry, model) => !model.id.includes("-tts"),
1466
- mapModel: (entry, defaults) => {
1467
- const reference = references.get(defaults.id);
1468
- const model = mapWithBundledReference(entry, defaults, reference);
1469
- return {
1470
- ...model,
1471
- name: toModelName(entry.display_name, model.name),
1472
- };
1473
- },
1474
- });
1498
+ for (const url of tokenPlanBaseUrls) {
1499
+ try {
1500
+ const result = await fetchModels(url);
1501
+ if (result) return result;
1502
+ } catch (error) {
1503
+ // Auth errors (401/403) should fail fast, not retry other regions
1504
+ const message = error instanceof Error ? error.message : String(error);
1505
+ if (message.includes("Authentication failed")) throw error;
1506
+ // Network/timeout errors: try next URL
1507
+ }
1508
+ }
1509
+ return null;
1475
1510
  },
1476
1511
  }),
1477
1512
  };
@@ -2166,9 +2201,37 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDe
2166
2201
  // --- zAI ---
2167
2202
  anthropicMessagesDescriptor("zai-coding-plan", "zai", "https://api.z.ai/api/anthropic"),
2168
2203
  // --- Xiaomi ---
2169
- anthropicMessagesDescriptor("xiaomi", "xiaomi", "https://api.xiaomimimo.com/anthropic", {
2204
+ openAiCompletionsDescriptor("xiaomi", "xiaomi", "https://api.xiaomimimo.com/v1", {
2205
+ defaultContextWindow: 262144,
2206
+ defaultMaxTokens: 8192,
2207
+ compat: {
2208
+ supportsStore: false,
2209
+ thinkingFormat: "zai",
2210
+ },
2211
+ }),
2212
+ openAiCompletionsDescriptor("xiaomi", "xiaomi-token-plan-sgp", "https://token-plan-sgp.xiaomimimo.com/v1", {
2170
2213
  defaultContextWindow: 262144,
2171
2214
  defaultMaxTokens: 8192,
2215
+ compat: {
2216
+ supportsStore: false,
2217
+ thinkingFormat: "zai",
2218
+ },
2219
+ }),
2220
+ openAiCompletionsDescriptor("xiaomi", "xiaomi-token-plan-ams", "https://token-plan-ams.xiaomimimo.com/v1", {
2221
+ defaultContextWindow: 262144,
2222
+ defaultMaxTokens: 8192,
2223
+ compat: {
2224
+ supportsStore: false,
2225
+ thinkingFormat: "zai",
2226
+ },
2227
+ }),
2228
+ openAiCompletionsDescriptor("xiaomi", "xiaomi-token-plan-cn", "https://token-plan-cn.xiaomimimo.com/v1", {
2229
+ defaultContextWindow: 262144,
2230
+ defaultMaxTokens: 8192,
2231
+ compat: {
2232
+ supportsStore: false,
2233
+ thinkingFormat: "zai",
2234
+ },
2172
2235
  }),
2173
2236
  // --- MiniMax Coding Plan ---
2174
2237
  openAiCompletionsDescriptor("minimax-coding-plan", "minimax-code", "https://api.minimax.io/v1", {
@@ -136,6 +136,38 @@ export const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
136
136
  const conversationStateCache = new Map<string, ConversationStateStructure>();
137
137
  const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
138
138
 
139
+ // F15: bound the module-global conversation caches so long-lived / many-session use cannot
140
+ // grow them without limit. LRU by conversation count + TTL on idle conversations.
141
+ const CURSOR_MAX_CONVERSATIONS = 64;
142
+ const CURSOR_CONVERSATION_TTL_MS = 60 * 60 * 1000;
143
+ const conversationLastAccess = new Map<string, number>();
144
+
145
+ /** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
146
+ export function disposeCursorConversation(conversationId: string): void {
147
+ conversationStateCache.delete(conversationId);
148
+ conversationBlobStores.delete(conversationId);
149
+ conversationLastAccess.delete(conversationId);
150
+ }
151
+
152
+ /** Refresh recency for a conversation and evict TTL-stale / LRU-overflow entries (F15). */
153
+ function touchCursorConversation(conversationId: string): void {
154
+ const now = Date.now();
155
+ for (const [id, ts] of conversationLastAccess) {
156
+ if (id !== conversationId && now - ts > CURSOR_CONVERSATION_TTL_MS) disposeCursorConversation(id);
157
+ }
158
+ conversationLastAccess.set(conversationId, now);
159
+ const state = conversationStateCache.get(conversationId);
160
+ if (state !== undefined) {
161
+ conversationStateCache.delete(conversationId);
162
+ conversationStateCache.set(conversationId, state);
163
+ }
164
+ while (conversationStateCache.size > CURSOR_MAX_CONVERSATIONS) {
165
+ const oldest = conversationStateCache.keys().next().value;
166
+ if (oldest === undefined || oldest === conversationId) break;
167
+ disposeCursorConversation(oldest);
168
+ }
169
+ }
170
+
139
171
  export interface CursorOptions extends StreamOptions {
140
172
  customSystemPrompt?: string;
141
173
  conversationId?: string;
@@ -349,6 +381,7 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
349
381
  conversationState: cachedState,
350
382
  });
351
383
  conversationStateCache.set(conversationId, conversationState);
384
+ touchCursorConversation(conversationId);
352
385
  const requestContextTools = buildMcpToolDefinitions(context.tools);
353
386
 
354
387
  const baseUrl = model.baseUrl || CURSOR_API_URL;
@@ -405,6 +438,7 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
405
438
 
406
439
  const onConversationCheckpoint = (checkpoint: ConversationStateStructure) => {
407
440
  conversationStateCache.set(conversationId, checkpoint);
441
+ touchCursorConversation(conversationId);
408
442
  };
409
443
 
410
444
  let resolveH2: (() => void) | undefined;
@@ -96,6 +96,7 @@ const CODEX_WEBSOCKET_IDLE_TIMEOUT_MS = 300000;
96
96
  const CODEX_WEBSOCKET_FIRST_EVENT_TIMEOUT_MS = 15000;
97
97
  const CODEX_WEBSOCKET_RETRY_BUDGET = CODEX_MAX_RETRIES;
98
98
  const CODEX_WEBSOCKET_TRANSPORT_ERROR_PREFIX = "Codex websocket transport error";
99
+ const CODEX_PREVIOUS_RESPONSE_STALE_CODES = new Set(["previous_response_not_found", "codex_previous_response_stale"]);
99
100
  const CODEX_RETRYABLE_EVENT_CODES = new Set(["model_error", "server_error", "internal_error"]);
100
101
  const CODEX_RETRYABLE_EVENT_MESSAGE =
101
102
  /processing your request|retry your request|temporar(?:y|ily)|overloaded|service.?unavailable|internal error|server error/i;
@@ -170,7 +171,7 @@ interface CodexProviderSessionState extends ProviderSessionState {
170
171
 
171
172
  interface CodexRequestContext {
172
173
  apiKey: string;
173
- accountId: string;
174
+ accountId: string | undefined;
174
175
  baseUrl: string;
175
176
  url: string;
176
177
  requestHeaders: Record<string, string>;
@@ -1475,7 +1476,11 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
1475
1476
  }
1476
1477
 
1477
1478
  function isCodexPreviousResponseNotFound(error: unknown): boolean {
1478
- return error instanceof CodexProviderStreamError && error.code === "previous_response_not_found";
1479
+ return (
1480
+ error instanceof CodexProviderStreamError &&
1481
+ typeof error.code === "string" &&
1482
+ CODEX_PREVIOUS_RESPONSE_STALE_CODES.has(error.code)
1483
+ );
1479
1484
  }
1480
1485
 
1481
1486
  async function tryRecoverCodexPreviousResponseNotFound(
@@ -1801,12 +1806,12 @@ export async function prewarmOpenAICodexResponses(
1801
1806
  function getCodexWebSocketSessionKey(
1802
1807
  sessionId: string | undefined,
1803
1808
  model: Model<"openai-codex-responses">,
1804
- accountId: string,
1809
+ accountId: string | undefined,
1805
1810
  baseUrl: string,
1806
1811
  ): string | undefined {
1807
1812
  const promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(sessionId);
1808
1813
  if (!promptCacheKey) return undefined;
1809
- return `${accountId}:${baseUrl}:${model.id}:${promptCacheKey}`;
1814
+ return `${accountId ?? "opaque"}:${baseUrl}:${model.id}:${promptCacheKey}`;
1810
1815
  }
1811
1816
 
1812
1817
  function getCodexPublicSessionKey(
@@ -2335,7 +2340,7 @@ async function getOrCreateCodexWebSocketConnection(
2335
2340
  async function openCodexSseEventStream(
2336
2341
  url: string,
2337
2342
  requestHeaders: Record<string, string> | undefined,
2338
- accountId: string,
2343
+ accountId: string | undefined,
2339
2344
  apiKey: string,
2340
2345
  sessionId: string | undefined,
2341
2346
  body: RequestBody,
@@ -2400,7 +2405,7 @@ async function openCodexWebSocketEventStream(
2400
2405
 
2401
2406
  function createCodexHeaders(
2402
2407
  initHeaders: Record<string, string> | undefined,
2403
- accountId: string,
2408
+ accountId: string | undefined,
2404
2409
  accessToken: string,
2405
2410
  promptCacheKey?: string,
2406
2411
  transport: CodexTransport = "sse",
@@ -2409,7 +2414,11 @@ function createCodexHeaders(
2409
2414
  const headers = new Headers(initHeaders ?? {});
2410
2415
  headers.delete("x-api-key");
2411
2416
  headers.set("Authorization", `Bearer ${accessToken}`);
2412
- headers.set(OPENAI_HEADERS.ACCOUNT_ID, accountId);
2417
+ if (accountId) {
2418
+ headers.set(OPENAI_HEADERS.ACCOUNT_ID, accountId);
2419
+ } else {
2420
+ headers.delete(OPENAI_HEADERS.ACCOUNT_ID);
2421
+ }
2413
2422
  const betaHeader =
2414
2423
  transport === "websocket"
2415
2424
  ? OPENAI_HEADER_VALUES.BETA_RESPONSES_WEBSOCKETS_V2
@@ -2482,12 +2491,8 @@ function resolveCodexResponsesUrl(baseUrl: string | undefined): string {
2482
2491
  return `${normalized}/codex/responses`;
2483
2492
  }
2484
2493
 
2485
- function getAccountId(accessToken: string): string {
2486
- const accountId = getCodexAccountId(accessToken);
2487
- if (!accountId) {
2488
- throw new Error("Failed to extract accountId from token");
2489
- }
2490
- return accountId;
2494
+ function getAccountId(accessToken: string): string | undefined {
2495
+ return getCodexAccountId(accessToken);
2491
2496
  }
2492
2497
 
2493
2498
  function convertMessages(model: Model<"openai-codex-responses">, context: Context): ResponseInput {
@@ -2669,6 +2674,22 @@ function getString(value: unknown): string | undefined {
2669
2674
  return typeof value === "string" ? value : undefined;
2670
2675
  }
2671
2676
 
2677
+ function getCodexEventError(rawEvent: Record<string, unknown>): Record<string, unknown> | null {
2678
+ const response = asRecord(rawEvent.response);
2679
+ return asRecord(rawEvent.error) ?? (response ? asRecord(response.error) : null);
2680
+ }
2681
+
2682
+ function getCodexEventErrorCode(rawEvent: Record<string, unknown>): string {
2683
+ const error = getCodexEventError(rawEvent);
2684
+ return getString(error?.code) ?? getString(error?.type) ?? getString(rawEvent.code) ?? "";
2685
+ }
2686
+
2687
+ function getCodexEventErrorMessage(rawEvent: Record<string, unknown>): string {
2688
+ const response = asRecord(rawEvent.response);
2689
+ const error = getCodexEventError(rawEvent);
2690
+ return getString(error?.message) ?? getString(rawEvent.message) ?? getString(response?.message) ?? "";
2691
+ }
2692
+
2672
2693
  class CodexProviderStreamError extends Error {
2673
2694
  readonly retryable: boolean;
2674
2695
  readonly code?: string;
@@ -2682,19 +2703,17 @@ class CodexProviderStreamError extends Error {
2682
2703
  }
2683
2704
 
2684
2705
  function isRetryableCodexFailureEvent(rawEvent: Record<string, unknown>): boolean {
2685
- const response = asRecord(rawEvent.response);
2686
- const error = asRecord(rawEvent.error) ?? (response ? asRecord(response.error) : null);
2687
- const code = getString(error?.code) ?? getString(error?.type) ?? getString(rawEvent.code);
2706
+ const code = getCodexEventErrorCode(rawEvent);
2688
2707
  if (code && CODEX_RETRYABLE_EVENT_CODES.has(code.toLowerCase())) {
2689
2708
  return true;
2690
2709
  }
2691
- const message = getString(error?.message) ?? getString(rawEvent.message) ?? getString(response?.message);
2710
+ const message = getCodexEventErrorMessage(rawEvent);
2692
2711
  return !!message && CODEX_RETRYABLE_EVENT_MESSAGE.test(message);
2693
2712
  }
2694
2713
 
2695
2714
  function createCodexProviderStreamError(rawEvent: Record<string, unknown>): CodexProviderStreamError {
2696
- const code = getString(rawEvent.code) ?? "";
2697
- const message = getString(rawEvent.message) ?? "";
2715
+ const code = getCodexEventErrorCode(rawEvent);
2716
+ const message = getCodexEventErrorMessage(rawEvent);
2698
2717
  const formattedMessage =
2699
2718
  typeof rawEvent.type === "string" && rawEvent.type === "error"
2700
2719
  ? formatCodexErrorEvent(rawEvent, code, message)
package/src/stream.ts CHANGED
@@ -194,6 +194,60 @@ export function listProvidersWithEnvKey(): string[] {
194
194
  return Object.keys(serviceProviderMap);
195
195
  }
196
196
 
197
+ /**
198
+ * Subscription-style providers whose "subscription" is delivered as an API key
199
+ * (created at https://opencode.ai/auth), not a separate OAuth/session token.
200
+ * Used to give OpenCode users an accurate headless auth diagnostic (#755).
201
+ */
202
+ const OPENCODE_SUBSCRIPTION_PROVIDERS = new Set(["opencode-go", "opencode-zen"]);
203
+
204
+ /**
205
+ * Provider-specific credential guidance appended to "no credential" errors.
206
+ *
207
+ * Headless GJC has no interactive `/login` TUI, so a bare "No API key" /
208
+ * "No credentials" error left users — OpenCode Go subscribers especially
209
+ * (#755) — unsure what signal GJC actually reads. OpenCode subscriptions are
210
+ * themselves API keys, so this names the env var GJC reads for the provider,
211
+ * warns that a project `.env` is intentionally ignored for provider
212
+ * credentials, and points OpenCode users at one-time interactive CLI credential capture.
213
+ *
214
+ * Returns an empty string when the provider has no env-var key and no special
215
+ * handling, so callers can append it unconditionally.
216
+ */
217
+ export function formatProviderCredentialHint(provider: string): string {
218
+ const resolver = serviceProviderMap[provider];
219
+ const envVar = typeof resolver === "string" ? resolver : undefined;
220
+ const isOpenCodeSubscription = OPENCODE_SUBSCRIPTION_PROVIDERS.has(provider);
221
+ const parts: string[] = [];
222
+ if (isOpenCodeSubscription) {
223
+ parts.push(
224
+ "OpenCode subscriptions authenticate with an API key (created at https://opencode.ai/auth), not a separate session/OAuth token.",
225
+ );
226
+ }
227
+ if (envVar) {
228
+ parts.push(
229
+ `Headless GJC reads this provider's key from ${envVar} (exported in your shell or set in ~/.gjc/.env).`,
230
+ );
231
+ parts.push("A value set only in a project .env is intentionally ignored for provider credentials.");
232
+ }
233
+ if (isOpenCodeSubscription) {
234
+ parts.push(
235
+ `Or run \`gjc auth-broker login ${provider}\` once before headless/print mode to store the key interactively.`,
236
+ );
237
+ }
238
+ return parts.join(" ");
239
+ }
240
+
241
+ /**
242
+ * Build an actionable "missing API key" error for a provider, used by the
243
+ * low-level `stream`/`complete` entry points (#755).
244
+ */
245
+ export function formatMissingApiKeyError(provider: string): string {
246
+ const base = `No API key for provider: ${provider}.`;
247
+ const hint = formatProviderCredentialHint(provider);
248
+ return hint ? `${base} ${hint}` : base;
249
+ }
250
+
197
251
  export function stream<TApi extends Api>(
198
252
  model: Model<TApi>,
199
253
  context: Context,
@@ -208,7 +262,7 @@ export function stream<TApi extends Api>(
208
262
  if (isGitLabDuoModel(model)) {
209
263
  const apiKey = (options as StreamOptions | undefined)?.apiKey || getEnvApiKey(model.provider);
210
264
  if (!apiKey) {
211
- throw new Error(`No API key for provider: ${model.provider}`);
265
+ throw new Error(formatMissingApiKeyError(model.provider));
212
266
  }
213
267
  return streamGitLabDuo(model, context, {
214
268
  ...(options as SimpleStreamOptions | undefined),
@@ -226,7 +280,7 @@ export function stream<TApi extends Api>(
226
280
 
227
281
  const apiKey = options?.apiKey || getEnvApiKey(model.provider);
228
282
  if (!apiKey) {
229
- throw new Error(`No API key for provider: ${model.provider}`);
283
+ throw new Error(formatMissingApiKeyError(model.provider));
230
284
  }
231
285
  const providerOptions = { ...options, apiKey };
232
286
 
@@ -410,7 +464,7 @@ export function streamSimple<TApi extends Api>(
410
464
 
411
465
  const apiKey = options?.apiKey || getEnvApiKey(model.provider);
412
466
  if (!apiKey) {
413
- throw new Error(`No API key for provider: ${model.provider}`);
467
+ throw new Error(formatMissingApiKeyError(model.provider));
414
468
  }
415
469
 
416
470
  // GitLab Duo - wraps Anthropic/OpenAI behind GitLab AI Gateway direct access tokens
package/src/types.ts CHANGED
@@ -141,6 +141,9 @@ export type KnownProvider =
141
141
  | "venice"
142
142
  | "vllm"
143
143
  | "xiaomi"
144
+ | "xiaomi-token-plan-sgp"
145
+ | "xiaomi-token-plan-ams"
146
+ | "xiaomi-token-plan-cn"
144
147
  | "zenmux"
145
148
  | "lm-studio";
146
149
  export type Provider = KnownProvider | string;
@@ -1,6 +1,6 @@
1
1
  import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "@gajae-code/ai";
2
2
  import * as z from "zod/v4";
3
- import type { Api, Model, Provider } from "../../types";
3
+ import type { Api, FetchImpl, Model, Provider } from "../../types";
4
4
 
5
5
  const MODELS_PATH = "/models";
6
6
 
@@ -81,7 +81,9 @@ export interface FetchOpenAICompatibleModelsOptions<TApi extends Api> {
81
81
  /** Optional AbortSignal for request cancellation. */
82
82
  signal?: AbortSignal;
83
83
  /** Optional fetch implementation override for testing/custom runtimes. */
84
- fetch?: typeof globalThis.fetch;
84
+ fetch?: FetchImpl;
85
+ /** Optional HTTP status predicate for provider-specific hard failures. */
86
+ throwOnStatus?: (response: Response) => Error | undefined;
85
87
  /**
86
88
  * Optional post-normalization filter.
87
89
  * Return false to skip a model.
@@ -133,6 +135,10 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
133
135
  }
134
136
 
135
137
  if (!response.ok) {
138
+ const hardFailure = options.throwOnStatus?.(response);
139
+ if (hardFailure) {
140
+ throw hardFailure;
141
+ }
136
142
  return null;
137
143
  }
138
144
 
@@ -140,6 +140,21 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
140
140
  name: "Xiaomi MiMo",
141
141
  available: true,
142
142
  },
143
+ {
144
+ id: "xiaomi-token-plan-sgp",
145
+ name: "Xiaomi Token Plan (Singapore)",
146
+ available: true,
147
+ },
148
+ {
149
+ id: "xiaomi-token-plan-ams",
150
+ name: "Xiaomi Token Plan (Europe)",
151
+ available: true,
152
+ },
153
+ {
154
+ id: "xiaomi-token-plan-cn",
155
+ name: "Xiaomi Token Plan (China)",
156
+ available: true,
157
+ },
143
158
  {
144
159
  id: "opencode-zen",
145
160
  name: "OpenCode Zen",
@@ -50,6 +50,9 @@ export type OAuthProvider =
50
50
  | "vllm"
51
51
  | "xai"
52
52
  | "xiaomi"
53
+ | "xiaomi-token-plan-sgp"
54
+ | "xiaomi-token-plan-ams"
55
+ | "xiaomi-token-plan-cn"
53
56
  | "zenmux"
54
57
  | "zai";
55
58
 
@@ -78,6 +81,7 @@ export interface OAuthController {
78
81
  onManualCodeInput?(): Promise<string>;
79
82
  onPrompt?(prompt: OAuthPrompt): Promise<string>;
80
83
  signal?: AbortSignal;
84
+ fetch?: typeof globalThis.fetch;
81
85
  }
82
86
 
83
87
  export interface OAuthLoginCallbacks extends OAuthController {