@gajae-code/ai 0.5.2 → 0.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,24 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.5.4] - 2026-06-17
6
+
7
+ ### Fixed
8
+
9
+ - Made the "No API key for provider" error from `stream`/`complete` actionable for OpenCode Go/Zen subscription providers in headless runs (#755). The subscription is itself an API key (`OPENCODE_API_KEY`, created at https://opencode.ai/auth), not a separate OAuth/session token; the new `formatProviderCredentialHint` helper (composed into `formatMissingApiKeyError`) names the env var GJC reads, warns that a project `.env` is intentionally ignored for provider credentials, and points OpenCode users at the one-time interactive `gjc auth-broker login <provider>` credential capture to run before headless/print mode. No auth behavior changed.
10
+
11
+ ## [0.5.3] - 2026-06-16
12
+
13
+ ### Added
14
+
15
+ - Added opt-in `AuthStorageOptions.credentialRankingMode` (`balanced` (default) | `earliest-reset`) for multi-account OAuth credential selection. `earliest-reset` ranks non-blocked credentials earliest-expiry-first — draining the soonest-to-reset account before its perishable tumbling-window quota (e.g. Claude 5h/7d) is lost at reset — keeping the existing drain-rate/used-fraction metrics as tiebreakers. `balanced` is byte-identical to prior behavior, and ranking only runs at session start (or when the session's preferred credential is blocked), so this never thrashes accounts mid-session.
16
+
17
+ ### Fixed
18
+
19
+ - Allowed `openai-codex-responses` custom backends to use opaque `apiKey` bearer tokens by omitting `chatgpt-account-id` when the token does not expose a Codex account id.
20
+ - Fixed OpenAI code websocket continuations to treat codex-lb's `codex_previous_response_stale` response failures as expired `previous_response_id` anchors and retry with full context instead of surfacing the transient failure.
21
+ - Bounded the Cursor provider's conversation cache with an LRU(64) + 1h TTL and added `disposeCursorConversation`, so long-running sessions no longer retain Cursor conversation state without limit (#717).
22
+
5
23
  ## [0.5.2] - 2026-06-15
6
24
 
7
25
  ### Changed
@@ -229,9 +229,26 @@ export interface CredentialDisabledEvent {
229
229
  provider: string;
230
230
  disabledCause: string;
231
231
  }
232
+ /**
233
+ * How {@link AuthStorage} orders multiple healthy OAuth credentials of the same
234
+ * provider:type pool when selecting one for a (new) session.
235
+ *
236
+ * - `balanced` (default): prefer the least-used / lowest-drain-rate account.
237
+ * Spreads load across accounts and keeps burst headroom on every account.
238
+ * - `earliest-reset`: prefer the non-blocked account whose usage window resets
239
+ * soonest (earliest-expiry-first). Tumbling-window quota is perishable —
240
+ * unused quota is lost at reset — so draining the soonest-to-reset account
241
+ * first minimizes wasted quota. Drain/used metrics remain tiebreakers.
242
+ *
243
+ * Only affects ranking, which the `shouldRank` guard already limits to session
244
+ * start (or when the session's preferred credential is blocked), so this never
245
+ * thrashes accounts mid-session / cold-starts the server-side prompt cache.
246
+ */
247
+ export type CredentialRankingMode = "balanced" | "earliest-reset";
232
248
  export type AuthStorageOptions = {
233
249
  usageProviderResolver?: (provider: Provider) => UsageProvider | undefined;
234
250
  rankingStrategyResolver?: (provider: Provider) => CredentialRankingStrategy | undefined;
251
+ credentialRankingMode?: CredentialRankingMode;
235
252
  usageFetch?: typeof fetch;
236
253
  usageRequestTimeoutMs?: number;
237
254
  usageLogger?: UsageLogger;
@@ -1,5 +1,5 @@
1
1
  import type { ModelManagerOptions } from "../model-manager";
2
- import type { Api, Model } from "../types";
2
+ import type { Api, FetchImpl, Model, Provider } from "../types";
3
3
  export interface ModelsDevModel {
4
4
  id?: string;
5
5
  name?: string;
@@ -161,10 +161,17 @@ export interface CloudflareAiGatewayModelManagerConfig {
161
161
  baseUrl?: string;
162
162
  }
163
163
  export declare function cloudflareAiGatewayModelManagerOptions(config?: CloudflareAiGatewayModelManagerConfig): ModelManagerOptions<"anthropic-messages">;
164
+ /** Region codes for Xiaomi Token Plan clusters exposed as separate login providers. */
165
+ export type XiaomiTokenPlanRegion = "sgp" | "ams" | "cn";
166
+ /** Configures Xiaomi standard or regional Token Plan OpenAI-compatible model discovery. */
164
167
  export interface XiaomiModelManagerConfig {
165
168
  apiKey?: string;
166
169
  baseUrl?: string;
170
+ fetch?: FetchImpl;
171
+ providerId?: Provider;
172
+ tokenPlanRegion?: XiaomiTokenPlanRegion;
167
173
  }
174
+ /** Builds a Xiaomi model manager, preserving Token Plan region provider ids during discovery. */
168
175
  export declare function xiaomiModelManagerOptions(config?: XiaomiModelManagerConfig): ModelManagerOptions<"openai-completions">;
169
176
  export interface LiteLLMModelManagerConfig {
170
177
  apiKey?: string;
@@ -2,6 +2,8 @@ import { type JsonValue } from "@bufbuild/protobuf";
2
2
  import type { CursorExecHandlerResult, CursorExecHandlers, CursorToolResultHandler, Message, StreamFunction, StreamOptions, ToolResultMessage } from "../types";
3
3
  export declare const CURSOR_API_URL = "https://api2.cursor.sh";
4
4
  export declare const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
5
+ /** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
6
+ export declare function disposeCursorConversation(conversationId: string): void;
5
7
  export interface CursorOptions extends StreamOptions {
6
8
  customSystemPrompt?: string;
7
9
  conversationId?: string;
@@ -15,6 +15,25 @@ export declare function getEnvApiKey(provider: string): string | undefined;
15
15
  * that should be uploaded to the broker.
16
16
  */
17
17
  export declare function listProvidersWithEnvKey(): string[];
18
+ /**
19
+ * Provider-specific credential guidance appended to "no credential" errors.
20
+ *
21
+ * Headless GJC has no interactive `/login` TUI, so a bare "No API key" /
22
+ * "No credentials" error left users — OpenCode Go subscribers especially
23
+ * (#755) — unsure what signal GJC actually reads. OpenCode subscriptions are
24
+ * themselves API keys, so this names the env var GJC reads for the provider,
25
+ * warns that a project `.env` is intentionally ignored for provider
26
+ * credentials, and points OpenCode users at one-time interactive CLI credential capture.
27
+ *
28
+ * Returns an empty string when the provider has no env-var key and no special
29
+ * handling, so callers can append it unconditionally.
30
+ */
31
+ export declare function formatProviderCredentialHint(provider: string): string;
32
+ /**
33
+ * Build an actionable "missing API key" error for a provider, used by the
34
+ * low-level `stream`/`complete` entry points (#755).
35
+ */
36
+ export declare function formatMissingApiKeyError(provider: string): string;
18
37
  export declare function stream<TApi extends Api>(model: Model<TApi>, context: Context, options?: OptionsForApi<TApi>): AssistantMessageEventStream;
19
38
  export declare function complete<TApi extends Api>(model: Model<TApi>, context: Context, options?: OptionsForApi<TApi>): Promise<AssistantMessage>;
20
39
  export declare function streamSimple<TApi extends Api>(model: Model<TApi>, context: Context, options?: SimpleStreamOptions): AssistantMessageEventStream;
@@ -48,7 +48,7 @@ export interface ThinkingConfig {
48
48
  /** Provider-specific transport used to encode the selected effort. */
49
49
  mode: ThinkingControlMode;
50
50
  }
51
- export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "zenmux" | "lm-studio";
51
+ export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
52
52
  export type Provider = KnownProvider | string;
53
53
  import type { Effort } from "./model-thinking";
54
54
  /** Token budgets for each thinking level (token-based providers only) */
@@ -1,4 +1,4 @@
1
- import type { Api, Model, Provider } from "../../types";
1
+ import type { Api, FetchImpl, Model, Provider } from "../../types";
2
2
  /**
3
3
  * Minimal OpenAI-style model entry shape consumed by discovery.
4
4
  *
@@ -51,7 +51,9 @@ export interface FetchOpenAICompatibleModelsOptions<TApi extends Api> {
51
51
  /** Optional AbortSignal for request cancellation. */
52
52
  signal?: AbortSignal;
53
53
  /** Optional fetch implementation override for testing/custom runtimes. */
54
- fetch?: typeof globalThis.fetch;
54
+ fetch?: FetchImpl;
55
+ /** Optional HTTP status predicate for provider-specific hard failures. */
56
+ throwOnStatus?: (response: Response) => Error | undefined;
55
57
  /**
56
58
  * Optional post-normalization filter.
57
59
  * Return false to skip a model.
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "xiaomi" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
@@ -29,6 +29,7 @@ export interface OAuthController {
29
29
  onManualCodeInput?(): Promise<string>;
30
30
  onPrompt?(prompt: OAuthPrompt): Promise<string>;
31
31
  signal?: AbortSignal;
32
+ fetch?: typeof globalThis.fetch;
32
33
  }
33
34
  export interface OAuthLoginCallbacks extends OAuthController {
34
35
  onAuth: (info: OAuthAuthInfo) => void;
@@ -4,12 +4,12 @@
4
4
  * Xiaomi MiMo provides OpenAI-compatible models via
5
5
  * https://api.xiaomimimo.com/v1.
6
6
  *
7
- * This is not OAuth - it's a simple API key flow:
8
- * 1. Open browser to Xiaomi MiMo API key console
9
- * 2. User copies their API key
10
- * 3. User pastes the API key into the CLI
7
+ * Standard Xiaomi login opens the pay-as-you-go API key console. Token Plan
8
+ * login opens plan management so users copy the regional `tp-...` key.
11
9
  */
12
10
  import type { OAuthController } from "./types";
11
+ /** Region codes accepted by the Xiaomi Token Plan login flow. */
12
+ export type XiaomiTokenPlanRegion = "sgp" | "ams" | "cn";
13
13
  /**
14
14
  * Login to Xiaomi MiMo.
15
15
  *
@@ -17,3 +17,9 @@ import type { OAuthController } from "./types";
17
17
  * Returns the API key directly (not OAuthCredentials - this isn't OAuth).
18
18
  */
19
19
  export declare function loginXiaomi(options: OAuthController): Promise<string>;
20
+ /**
21
+ * Login to a regional Xiaomi Token Plan endpoint.
22
+ *
23
+ * Prompts for a token-plan API key and validates it against the selected region.
24
+ */
25
+ export declare function loginXiaomiTokenPlan(options: OAuthController, region: XiaomiTokenPlanRegion): Promise<string>;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.5.2",
4
+ "version": "0.5.4",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gaebal-gajae.dev",
7
7
  "author": "Yeachan-Heo",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@gajae-code/utils": "0.5.2",
46
+ "@gajae-code/utils": "0.5.4",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -288,9 +288,27 @@ export interface CredentialDisabledEvent {
288
288
  disabledCause: string;
289
289
  }
290
290
 
291
+ /**
292
+ * How {@link AuthStorage} orders multiple healthy OAuth credentials of the same
293
+ * provider:type pool when selecting one for a (new) session.
294
+ *
295
+ * - `balanced` (default): prefer the least-used / lowest-drain-rate account.
296
+ * Spreads load across accounts and keeps burst headroom on every account.
297
+ * - `earliest-reset`: prefer the non-blocked account whose usage window resets
298
+ * soonest (earliest-expiry-first). Tumbling-window quota is perishable —
299
+ * unused quota is lost at reset — so draining the soonest-to-reset account
300
+ * first minimizes wasted quota. Drain/used metrics remain tiebreakers.
301
+ *
302
+ * Only affects ranking, which the `shouldRank` guard already limits to session
303
+ * start (or when the session's preferred credential is blocked), so this never
304
+ * thrashes accounts mid-session / cold-starts the server-side prompt cache.
305
+ */
306
+ export type CredentialRankingMode = "balanced" | "earliest-reset";
307
+
291
308
  export type AuthStorageOptions = {
292
309
  usageProviderResolver?: (provider: Provider) => UsageProvider | undefined;
293
310
  rankingStrategyResolver?: (provider: Provider) => CredentialRankingStrategy | undefined;
311
+ credentialRankingMode?: CredentialRankingMode;
294
312
  usageFetch?: typeof fetch;
295
313
  usageRequestTimeoutMs?: number;
296
314
  usageLogger?: UsageLogger;
@@ -655,6 +673,7 @@ export class AuthStorage {
655
673
  #usageReportsInFlight: Map<string, Promise<UsageReport[] | null>> = new Map();
656
674
  #usageFetch: typeof fetch;
657
675
  #usageRequestTimeoutMs: number;
676
+ #credentialRankingMode: CredentialRankingMode = "balanced";
658
677
  #usageLogger?: UsageLogger;
659
678
  #fallbackResolver?: (provider: string) => string | undefined;
660
679
  #store: AuthCredentialStore;
@@ -686,6 +705,7 @@ export class AuthStorage {
686
705
  this.#usageCache = new AuthStorageUsageCache(this.#store);
687
706
  this.#usageFetch = options.usageFetch ?? fetch;
688
707
  this.#usageRequestTimeoutMs = options.usageRequestTimeoutMs ?? DEFAULT_USAGE_REQUEST_TIMEOUT_MS;
708
+ this.#credentialRankingMode = options.credentialRankingMode ?? "balanced";
689
709
  this.#refreshOAuthCredentialOverride = options.refreshOAuthCredential;
690
710
  this.#fetchUsageReportsOverride = options.fetchUsageReports;
691
711
  this.#sourceLabel = options.sourceLabel;
@@ -1621,6 +1641,24 @@ export class AuthStorage {
1621
1641
  await saveApiKeyCredential(apiKey);
1622
1642
  return;
1623
1643
  }
1644
+ case "xiaomi-token-plan-sgp": {
1645
+ const { loginXiaomiTokenPlan } = await import("./utils/oauth/xiaomi");
1646
+ const apiKey = await loginXiaomiTokenPlan(ctrl, "sgp");
1647
+ await saveApiKeyCredential(apiKey);
1648
+ return;
1649
+ }
1650
+ case "xiaomi-token-plan-ams": {
1651
+ const { loginXiaomiTokenPlan } = await import("./utils/oauth/xiaomi");
1652
+ const apiKey = await loginXiaomiTokenPlan(ctrl, "ams");
1653
+ await saveApiKeyCredential(apiKey);
1654
+ return;
1655
+ }
1656
+ case "xiaomi-token-plan-cn": {
1657
+ const { loginXiaomiTokenPlan } = await import("./utils/oauth/xiaomi");
1658
+ const apiKey = await loginXiaomiTokenPlan(ctrl, "cn");
1659
+ await saveApiKeyCredential(apiKey);
1660
+ return;
1661
+ }
1624
1662
  case "zenmux": {
1625
1663
  const { loginZenMux } = await import("./utils/oauth/zenmux");
1626
1664
  const apiKey = await loginZenMux(ctrl);
@@ -2453,6 +2491,7 @@ export class AuthStorage {
2453
2491
  secondaryDrainRate: number;
2454
2492
  primaryUsed: number;
2455
2493
  primaryDrainRate: number;
2494
+ resetAtMs: number;
2456
2495
  orderPos: number;
2457
2496
  }> = [];
2458
2497
  // Pre-fetch usage reports in parallel for non-blocked credentials.
@@ -2522,6 +2561,10 @@ export class AuthStorage {
2522
2561
  ),
2523
2562
  primaryUsed: this.#normalizeUsageFraction(primary),
2524
2563
  primaryDrainRate: this.#computeWindowDrainRate(primary, nowMs, strategy.windowDefaults.primaryMs),
2564
+ resetAtMs:
2565
+ this.#resolveWindowResetAt(primary?.window) ??
2566
+ this.#resolveWindowResetAt(secondary?.window) ??
2567
+ Number.POSITIVE_INFINITY,
2525
2568
  orderPos,
2526
2569
  });
2527
2570
  }
@@ -2539,6 +2582,11 @@ export class AuthStorage {
2539
2582
  if (leftPlanPriority !== rightPlanPriority) return leftPlanPriority - rightPlanPriority;
2540
2583
  }
2541
2584
  if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1;
2585
+ if (this.#credentialRankingMode === "earliest-reset" && left.resetAtMs !== right.resetAtMs) {
2586
+ // Earliest-expiry-first: drain the soonest-to-reset account before
2587
+ // its perishable tumbling-window quota is lost at reset.
2588
+ return left.resetAtMs - right.resetAtMs;
2589
+ }
2542
2590
  if (left.secondaryDrainRate !== right.secondaryDrainRate)
2543
2591
  return left.secondaryDrainRate - right.secondaryDrainRate;
2544
2592
  if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed;
@@ -1,5 +1,5 @@
1
1
  import { readModelCache, writeModelCache } from "./model-cache";
2
- import { enrichModelThinking } from "./model-thinking";
2
+ import { applyGeneratedModelPolicies, enrichModelThinking } from "./model-thinking";
3
3
  import { type GeneratedProvider, getBundledModels } from "./models";
4
4
  import type { Api, Model, Provider } from "./types";
5
5
  import { isRecord } from "./utils";
@@ -90,6 +90,7 @@ function passModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
90
90
  }
91
91
  out.push(enrichModelThinking(item as Model<TApi>));
92
92
  }
93
+ applyGeneratedModelPolicies(out as Model<Api>[]);
93
94
  return out;
94
95
  }
95
96
 
@@ -322,7 +323,7 @@ function mergeDynamicModel<TApi extends Api>(existingModel: Model<TApi>, dynamic
322
323
  // (issue #489). Keep the existing api, and only take the dynamic baseUrl
323
324
  // when the api matches (same transport, same URL shape).
324
325
  const baseUrl = existingModel.api === dynamicModel.api ? dynamicModel.baseUrl : existingModel.baseUrl;
325
- return enrichModelThinking({
326
+ const merged = enrichModelThinking({
326
327
  ...existingModel,
327
328
  ...dynamicModel,
328
329
  api: existingModel.api,
@@ -342,6 +343,9 @@ function mergeDynamicModel<TApi extends Api>(existingModel: Model<TApi>, dynamic
342
343
  compat: dynamicModel.compat ?? existingModel.compat,
343
344
  contextPromotionTarget: dynamicModel.contextPromotionTarget ?? existingModel.contextPromotionTarget,
344
345
  });
346
+ const policyModels = [merged as Model<Api>];
347
+ applyGeneratedModelPolicies(policyModels);
348
+ return policyModels[0] as Model<TApi>;
345
349
  }
346
350
 
347
351
  function preferDiscoveryCost(discoveryCost: number, fallbackCost: number): number {
@@ -699,7 +699,7 @@ function parseAnthropicModel(modelId: string): AnthropicModel | null {
699
699
  }
700
700
 
701
701
  function parseOpenAIModel(modelId: string): OpenAIModel | null {
702
- const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|mini|max|nano))?\b/.exec(modelId);
702
+ const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|mini|max|nano))?$/.exec(modelId);
703
703
  if (!match) {
704
704
  return null;
705
705
  }