@gajae-code/ai 0.5.2 → 0.5.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/dist/types/auth-storage.d.ts +17 -0
- package/dist/types/provider-models/openai-compat.d.ts +8 -1
- package/dist/types/providers/cursor.d.ts +2 -0
- package/dist/types/stream.d.ts +19 -0
- package/dist/types/types.d.ts +1 -1
- package/dist/types/utils/discovery/openai-compatible.d.ts +4 -2
- package/dist/types/utils/oauth/types.d.ts +2 -1
- package/dist/types/utils/oauth/xiaomi.d.ts +10 -4
- package/package.json +2 -2
- package/src/auth-storage.ts +48 -0
- package/src/model-manager.ts +6 -2
- package/src/model-thinking.ts +1 -1
- package/src/models.json +584 -20
- package/src/provider-models/descriptors.ts +18 -0
- package/src/provider-models/openai-compat.ts +107 -44
- package/src/providers/cursor.ts +34 -0
- package/src/providers/openai-codex-responses.ts +38 -19
- package/src/stream.ts +57 -3
- package/src/types.ts +3 -0
- package/src/utils/discovery/openai-compatible.ts +8 -2
- package/src/utils/oauth/index.ts +15 -0
- package/src/utils/oauth/types.ts +4 -0
- package/src/utils/oauth/xiaomi.ts +80 -18
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,24 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.5.4] - 2026-06-17
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Made the "No API key for provider" error from `stream`/`complete` actionable for OpenCode Go/Zen subscription providers in headless runs (#755). The subscription is itself an API key (`OPENCODE_API_KEY`, created at https://opencode.ai/auth), not a separate OAuth/session token; the new `formatProviderCredentialHint` helper (composed into `formatMissingApiKeyError`) names the env var GJC reads, warns that a project `.env` is intentionally ignored for provider credentials, and points OpenCode users at the one-time interactive `gjc auth-broker login <provider>` credential capture to run before headless/print mode. No auth behavior changed.
|
|
10
|
+
|
|
11
|
+
## [0.5.3] - 2026-06-16
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- Added opt-in `AuthStorageOptions.credentialRankingMode` (`balanced` (default) | `earliest-reset`) for multi-account OAuth credential selection. `earliest-reset` ranks non-blocked credentials earliest-expiry-first — draining the soonest-to-reset account before its perishable tumbling-window quota (e.g. Claude 5h/7d) is lost at reset — keeping the existing drain-rate/used-fraction metrics as tiebreakers. `balanced` is byte-identical to prior behavior, and ranking only runs at session start (or when the session's preferred credential is blocked), so this never thrashes accounts mid-session.
|
|
16
|
+
|
|
17
|
+
### Fixed
|
|
18
|
+
|
|
19
|
+
- Allowed `openai-codex-responses` custom backends to use opaque `apiKey` bearer tokens by omitting `chatgpt-account-id` when the token does not expose a Codex account id.
|
|
20
|
+
- Fixed OpenAI code websocket continuations to treat codex-lb's `codex_previous_response_stale` response failures as expired `previous_response_id` anchors and retry with full context instead of surfacing the transient failure.
|
|
21
|
+
- Bounded the Cursor provider's conversation cache with an LRU(64) + 1h TTL and added `disposeCursorConversation`, so long-running sessions no longer retain Cursor conversation state without limit (#717).
|
|
22
|
+
|
|
5
23
|
## [0.5.2] - 2026-06-15
|
|
6
24
|
|
|
7
25
|
### Changed
|
|
@@ -229,9 +229,26 @@ export interface CredentialDisabledEvent {
|
|
|
229
229
|
provider: string;
|
|
230
230
|
disabledCause: string;
|
|
231
231
|
}
|
|
232
|
+
/**
|
|
233
|
+
* How {@link AuthStorage} orders multiple healthy OAuth credentials of the same
|
|
234
|
+
* provider:type pool when selecting one for a (new) session.
|
|
235
|
+
*
|
|
236
|
+
* - `balanced` (default): prefer the least-used / lowest-drain-rate account.
|
|
237
|
+
* Spreads load across accounts and keeps burst headroom on every account.
|
|
238
|
+
* - `earliest-reset`: prefer the non-blocked account whose usage window resets
|
|
239
|
+
* soonest (earliest-expiry-first). Tumbling-window quota is perishable —
|
|
240
|
+
* unused quota is lost at reset — so draining the soonest-to-reset account
|
|
241
|
+
* first minimizes wasted quota. Drain/used metrics remain tiebreakers.
|
|
242
|
+
*
|
|
243
|
+
* Only affects ranking, which the `shouldRank` guard already limits to session
|
|
244
|
+
* start (or when the session's preferred credential is blocked), so this never
|
|
245
|
+
* thrashes accounts mid-session / cold-starts the server-side prompt cache.
|
|
246
|
+
*/
|
|
247
|
+
export type CredentialRankingMode = "balanced" | "earliest-reset";
|
|
232
248
|
export type AuthStorageOptions = {
|
|
233
249
|
usageProviderResolver?: (provider: Provider) => UsageProvider | undefined;
|
|
234
250
|
rankingStrategyResolver?: (provider: Provider) => CredentialRankingStrategy | undefined;
|
|
251
|
+
credentialRankingMode?: CredentialRankingMode;
|
|
235
252
|
usageFetch?: typeof fetch;
|
|
236
253
|
usageRequestTimeoutMs?: number;
|
|
237
254
|
usageLogger?: UsageLogger;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { ModelManagerOptions } from "../model-manager";
|
|
2
|
-
import type { Api, Model } from "../types";
|
|
2
|
+
import type { Api, FetchImpl, Model, Provider } from "../types";
|
|
3
3
|
export interface ModelsDevModel {
|
|
4
4
|
id?: string;
|
|
5
5
|
name?: string;
|
|
@@ -161,10 +161,17 @@ export interface CloudflareAiGatewayModelManagerConfig {
|
|
|
161
161
|
baseUrl?: string;
|
|
162
162
|
}
|
|
163
163
|
export declare function cloudflareAiGatewayModelManagerOptions(config?: CloudflareAiGatewayModelManagerConfig): ModelManagerOptions<"anthropic-messages">;
|
|
164
|
+
/** Region codes for Xiaomi Token Plan clusters exposed as separate login providers. */
|
|
165
|
+
export type XiaomiTokenPlanRegion = "sgp" | "ams" | "cn";
|
|
166
|
+
/** Configures Xiaomi standard or regional Token Plan OpenAI-compatible model discovery. */
|
|
164
167
|
export interface XiaomiModelManagerConfig {
|
|
165
168
|
apiKey?: string;
|
|
166
169
|
baseUrl?: string;
|
|
170
|
+
fetch?: FetchImpl;
|
|
171
|
+
providerId?: Provider;
|
|
172
|
+
tokenPlanRegion?: XiaomiTokenPlanRegion;
|
|
167
173
|
}
|
|
174
|
+
/** Builds a Xiaomi model manager, preserving Token Plan region provider ids during discovery. */
|
|
168
175
|
export declare function xiaomiModelManagerOptions(config?: XiaomiModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
169
176
|
export interface LiteLLMModelManagerConfig {
|
|
170
177
|
apiKey?: string;
|
|
@@ -2,6 +2,8 @@ import { type JsonValue } from "@bufbuild/protobuf";
|
|
|
2
2
|
import type { CursorExecHandlerResult, CursorExecHandlers, CursorToolResultHandler, Message, StreamFunction, StreamOptions, ToolResultMessage } from "../types";
|
|
3
3
|
export declare const CURSOR_API_URL = "https://api2.cursor.sh";
|
|
4
4
|
export declare const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
|
|
5
|
+
/** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
|
|
6
|
+
export declare function disposeCursorConversation(conversationId: string): void;
|
|
5
7
|
export interface CursorOptions extends StreamOptions {
|
|
6
8
|
customSystemPrompt?: string;
|
|
7
9
|
conversationId?: string;
|
package/dist/types/stream.d.ts
CHANGED
|
@@ -15,6 +15,25 @@ export declare function getEnvApiKey(provider: string): string | undefined;
|
|
|
15
15
|
* that should be uploaded to the broker.
|
|
16
16
|
*/
|
|
17
17
|
export declare function listProvidersWithEnvKey(): string[];
|
|
18
|
+
/**
|
|
19
|
+
* Provider-specific credential guidance appended to "no credential" errors.
|
|
20
|
+
*
|
|
21
|
+
* Headless GJC has no interactive `/login` TUI, so a bare "No API key" /
|
|
22
|
+
* "No credentials" error left users — OpenCode Go subscribers especially
|
|
23
|
+
* (#755) — unsure what signal GJC actually reads. OpenCode subscriptions are
|
|
24
|
+
* themselves API keys, so this names the env var GJC reads for the provider,
|
|
25
|
+
* warns that a project `.env` is intentionally ignored for provider
|
|
26
|
+
* credentials, and points OpenCode users at one-time interactive CLI credential capture.
|
|
27
|
+
*
|
|
28
|
+
* Returns an empty string when the provider has no env-var key and no special
|
|
29
|
+
* handling, so callers can append it unconditionally.
|
|
30
|
+
*/
|
|
31
|
+
export declare function formatProviderCredentialHint(provider: string): string;
|
|
32
|
+
/**
|
|
33
|
+
* Build an actionable "missing API key" error for a provider, used by the
|
|
34
|
+
* low-level `stream`/`complete` entry points (#755).
|
|
35
|
+
*/
|
|
36
|
+
export declare function formatMissingApiKeyError(provider: string): string;
|
|
18
37
|
export declare function stream<TApi extends Api>(model: Model<TApi>, context: Context, options?: OptionsForApi<TApi>): AssistantMessageEventStream;
|
|
19
38
|
export declare function complete<TApi extends Api>(model: Model<TApi>, context: Context, options?: OptionsForApi<TApi>): Promise<AssistantMessage>;
|
|
20
39
|
export declare function streamSimple<TApi extends Api>(model: Model<TApi>, context: Context, options?: SimpleStreamOptions): AssistantMessageEventStream;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -48,7 +48,7 @@ export interface ThinkingConfig {
|
|
|
48
48
|
/** Provider-specific transport used to encode the selected effort. */
|
|
49
49
|
mode: ThinkingControlMode;
|
|
50
50
|
}
|
|
51
|
-
export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "zenmux" | "lm-studio";
|
|
51
|
+
export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
52
52
|
export type Provider = KnownProvider | string;
|
|
53
53
|
import type { Effort } from "./model-thinking";
|
|
54
54
|
/** Token budgets for each thinking level (token-based providers only) */
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Api, Model, Provider } from "../../types";
|
|
1
|
+
import type { Api, FetchImpl, Model, Provider } from "../../types";
|
|
2
2
|
/**
|
|
3
3
|
* Minimal OpenAI-style model entry shape consumed by discovery.
|
|
4
4
|
*
|
|
@@ -51,7 +51,9 @@ export interface FetchOpenAICompatibleModelsOptions<TApi extends Api> {
|
|
|
51
51
|
/** Optional AbortSignal for request cancellation. */
|
|
52
52
|
signal?: AbortSignal;
|
|
53
53
|
/** Optional fetch implementation override for testing/custom runtimes. */
|
|
54
|
-
fetch?:
|
|
54
|
+
fetch?: FetchImpl;
|
|
55
|
+
/** Optional HTTP status predicate for provider-specific hard failures. */
|
|
56
|
+
throwOnStatus?: (response: Response) => Error | undefined;
|
|
55
57
|
/**
|
|
56
58
|
* Optional post-normalization filter.
|
|
57
59
|
* Return false to skip a model.
|
|
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
|
|
|
7
7
|
email?: string;
|
|
8
8
|
accountId?: string;
|
|
9
9
|
};
|
|
10
|
-
export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "xiaomi" | "zenmux" | "zai";
|
|
10
|
+
export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
11
11
|
export type OAuthProviderId = OAuthProvider | (string & {});
|
|
12
12
|
export type OAuthPrompt = {
|
|
13
13
|
message: string;
|
|
@@ -29,6 +29,7 @@ export interface OAuthController {
|
|
|
29
29
|
onManualCodeInput?(): Promise<string>;
|
|
30
30
|
onPrompt?(prompt: OAuthPrompt): Promise<string>;
|
|
31
31
|
signal?: AbortSignal;
|
|
32
|
+
fetch?: typeof globalThis.fetch;
|
|
32
33
|
}
|
|
33
34
|
export interface OAuthLoginCallbacks extends OAuthController {
|
|
34
35
|
onAuth: (info: OAuthAuthInfo) => void;
|
|
@@ -4,12 +4,12 @@
|
|
|
4
4
|
* Xiaomi MiMo provides OpenAI-compatible models via
|
|
5
5
|
* https://api.xiaomimimo.com/v1.
|
|
6
6
|
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
* 2. User copies their API key
|
|
10
|
-
* 3. User pastes the API key into the CLI
|
|
7
|
+
* Standard Xiaomi login opens the pay-as-you-go API key console. Token Plan
|
|
8
|
+
* login opens plan management so users copy the regional `tp-...` key.
|
|
11
9
|
*/
|
|
12
10
|
import type { OAuthController } from "./types";
|
|
11
|
+
/** Region codes accepted by the Xiaomi Token Plan login flow. */
|
|
12
|
+
export type XiaomiTokenPlanRegion = "sgp" | "ams" | "cn";
|
|
13
13
|
/**
|
|
14
14
|
* Login to Xiaomi MiMo.
|
|
15
15
|
*
|
|
@@ -17,3 +17,9 @@ import type { OAuthController } from "./types";
|
|
|
17
17
|
* Returns the API key directly (not OAuthCredentials - this isn't OAuth).
|
|
18
18
|
*/
|
|
19
19
|
export declare function loginXiaomi(options: OAuthController): Promise<string>;
|
|
20
|
+
/**
|
|
21
|
+
* Login to a regional Xiaomi Token Plan endpoint.
|
|
22
|
+
*
|
|
23
|
+
* Prompts for a token-plan API key and validates it against the selected region.
|
|
24
|
+
*/
|
|
25
|
+
export declare function loginXiaomiTokenPlan(options: OAuthController, region: XiaomiTokenPlanRegion): Promise<string>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.5.
|
|
4
|
+
"version": "0.5.4",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gaebal-gajae.dev",
|
|
7
7
|
"author": "Yeachan-Heo",
|
|
@@ -43,7 +43,7 @@
|
|
|
43
43
|
"dependencies": {
|
|
44
44
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
45
45
|
"@bufbuild/protobuf": "^2.12.0",
|
|
46
|
-
"@gajae-code/utils": "0.5.
|
|
46
|
+
"@gajae-code/utils": "0.5.4",
|
|
47
47
|
"openai": "^6.36.0",
|
|
48
48
|
"partial-json": "^0.1.7",
|
|
49
49
|
"zod": "4.4.3"
|
package/src/auth-storage.ts
CHANGED
|
@@ -288,9 +288,27 @@ export interface CredentialDisabledEvent {
|
|
|
288
288
|
disabledCause: string;
|
|
289
289
|
}
|
|
290
290
|
|
|
291
|
+
/**
|
|
292
|
+
* How {@link AuthStorage} orders multiple healthy OAuth credentials of the same
|
|
293
|
+
* provider:type pool when selecting one for a (new) session.
|
|
294
|
+
*
|
|
295
|
+
* - `balanced` (default): prefer the least-used / lowest-drain-rate account.
|
|
296
|
+
* Spreads load across accounts and keeps burst headroom on every account.
|
|
297
|
+
* - `earliest-reset`: prefer the non-blocked account whose usage window resets
|
|
298
|
+
* soonest (earliest-expiry-first). Tumbling-window quota is perishable —
|
|
299
|
+
* unused quota is lost at reset — so draining the soonest-to-reset account
|
|
300
|
+
* first minimizes wasted quota. Drain/used metrics remain tiebreakers.
|
|
301
|
+
*
|
|
302
|
+
* Only affects ranking, which the `shouldRank` guard already limits to session
|
|
303
|
+
* start (or when the session's preferred credential is blocked), so this never
|
|
304
|
+
* thrashes accounts mid-session / cold-starts the server-side prompt cache.
|
|
305
|
+
*/
|
|
306
|
+
export type CredentialRankingMode = "balanced" | "earliest-reset";
|
|
307
|
+
|
|
291
308
|
export type AuthStorageOptions = {
|
|
292
309
|
usageProviderResolver?: (provider: Provider) => UsageProvider | undefined;
|
|
293
310
|
rankingStrategyResolver?: (provider: Provider) => CredentialRankingStrategy | undefined;
|
|
311
|
+
credentialRankingMode?: CredentialRankingMode;
|
|
294
312
|
usageFetch?: typeof fetch;
|
|
295
313
|
usageRequestTimeoutMs?: number;
|
|
296
314
|
usageLogger?: UsageLogger;
|
|
@@ -655,6 +673,7 @@ export class AuthStorage {
|
|
|
655
673
|
#usageReportsInFlight: Map<string, Promise<UsageReport[] | null>> = new Map();
|
|
656
674
|
#usageFetch: typeof fetch;
|
|
657
675
|
#usageRequestTimeoutMs: number;
|
|
676
|
+
#credentialRankingMode: CredentialRankingMode = "balanced";
|
|
658
677
|
#usageLogger?: UsageLogger;
|
|
659
678
|
#fallbackResolver?: (provider: string) => string | undefined;
|
|
660
679
|
#store: AuthCredentialStore;
|
|
@@ -686,6 +705,7 @@ export class AuthStorage {
|
|
|
686
705
|
this.#usageCache = new AuthStorageUsageCache(this.#store);
|
|
687
706
|
this.#usageFetch = options.usageFetch ?? fetch;
|
|
688
707
|
this.#usageRequestTimeoutMs = options.usageRequestTimeoutMs ?? DEFAULT_USAGE_REQUEST_TIMEOUT_MS;
|
|
708
|
+
this.#credentialRankingMode = options.credentialRankingMode ?? "balanced";
|
|
689
709
|
this.#refreshOAuthCredentialOverride = options.refreshOAuthCredential;
|
|
690
710
|
this.#fetchUsageReportsOverride = options.fetchUsageReports;
|
|
691
711
|
this.#sourceLabel = options.sourceLabel;
|
|
@@ -1621,6 +1641,24 @@ export class AuthStorage {
|
|
|
1621
1641
|
await saveApiKeyCredential(apiKey);
|
|
1622
1642
|
return;
|
|
1623
1643
|
}
|
|
1644
|
+
case "xiaomi-token-plan-sgp": {
|
|
1645
|
+
const { loginXiaomiTokenPlan } = await import("./utils/oauth/xiaomi");
|
|
1646
|
+
const apiKey = await loginXiaomiTokenPlan(ctrl, "sgp");
|
|
1647
|
+
await saveApiKeyCredential(apiKey);
|
|
1648
|
+
return;
|
|
1649
|
+
}
|
|
1650
|
+
case "xiaomi-token-plan-ams": {
|
|
1651
|
+
const { loginXiaomiTokenPlan } = await import("./utils/oauth/xiaomi");
|
|
1652
|
+
const apiKey = await loginXiaomiTokenPlan(ctrl, "ams");
|
|
1653
|
+
await saveApiKeyCredential(apiKey);
|
|
1654
|
+
return;
|
|
1655
|
+
}
|
|
1656
|
+
case "xiaomi-token-plan-cn": {
|
|
1657
|
+
const { loginXiaomiTokenPlan } = await import("./utils/oauth/xiaomi");
|
|
1658
|
+
const apiKey = await loginXiaomiTokenPlan(ctrl, "cn");
|
|
1659
|
+
await saveApiKeyCredential(apiKey);
|
|
1660
|
+
return;
|
|
1661
|
+
}
|
|
1624
1662
|
case "zenmux": {
|
|
1625
1663
|
const { loginZenMux } = await import("./utils/oauth/zenmux");
|
|
1626
1664
|
const apiKey = await loginZenMux(ctrl);
|
|
@@ -2453,6 +2491,7 @@ export class AuthStorage {
|
|
|
2453
2491
|
secondaryDrainRate: number;
|
|
2454
2492
|
primaryUsed: number;
|
|
2455
2493
|
primaryDrainRate: number;
|
|
2494
|
+
resetAtMs: number;
|
|
2456
2495
|
orderPos: number;
|
|
2457
2496
|
}> = [];
|
|
2458
2497
|
// Pre-fetch usage reports in parallel for non-blocked credentials.
|
|
@@ -2522,6 +2561,10 @@ export class AuthStorage {
|
|
|
2522
2561
|
),
|
|
2523
2562
|
primaryUsed: this.#normalizeUsageFraction(primary),
|
|
2524
2563
|
primaryDrainRate: this.#computeWindowDrainRate(primary, nowMs, strategy.windowDefaults.primaryMs),
|
|
2564
|
+
resetAtMs:
|
|
2565
|
+
this.#resolveWindowResetAt(primary?.window) ??
|
|
2566
|
+
this.#resolveWindowResetAt(secondary?.window) ??
|
|
2567
|
+
Number.POSITIVE_INFINITY,
|
|
2525
2568
|
orderPos,
|
|
2526
2569
|
});
|
|
2527
2570
|
}
|
|
@@ -2539,6 +2582,11 @@ export class AuthStorage {
|
|
|
2539
2582
|
if (leftPlanPriority !== rightPlanPriority) return leftPlanPriority - rightPlanPriority;
|
|
2540
2583
|
}
|
|
2541
2584
|
if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1;
|
|
2585
|
+
if (this.#credentialRankingMode === "earliest-reset" && left.resetAtMs !== right.resetAtMs) {
|
|
2586
|
+
// Earliest-expiry-first: drain the soonest-to-reset account before
|
|
2587
|
+
// its perishable tumbling-window quota is lost at reset.
|
|
2588
|
+
return left.resetAtMs - right.resetAtMs;
|
|
2589
|
+
}
|
|
2542
2590
|
if (left.secondaryDrainRate !== right.secondaryDrainRate)
|
|
2543
2591
|
return left.secondaryDrainRate - right.secondaryDrainRate;
|
|
2544
2592
|
if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed;
|
package/src/model-manager.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { readModelCache, writeModelCache } from "./model-cache";
|
|
2
|
-
import { enrichModelThinking } from "./model-thinking";
|
|
2
|
+
import { applyGeneratedModelPolicies, enrichModelThinking } from "./model-thinking";
|
|
3
3
|
import { type GeneratedProvider, getBundledModels } from "./models";
|
|
4
4
|
import type { Api, Model, Provider } from "./types";
|
|
5
5
|
import { isRecord } from "./utils";
|
|
@@ -90,6 +90,7 @@ function passModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
|
|
|
90
90
|
}
|
|
91
91
|
out.push(enrichModelThinking(item as Model<TApi>));
|
|
92
92
|
}
|
|
93
|
+
applyGeneratedModelPolicies(out as Model<Api>[]);
|
|
93
94
|
return out;
|
|
94
95
|
}
|
|
95
96
|
|
|
@@ -322,7 +323,7 @@ function mergeDynamicModel<TApi extends Api>(existingModel: Model<TApi>, dynamic
|
|
|
322
323
|
// (issue #489). Keep the existing api, and only take the dynamic baseUrl
|
|
323
324
|
// when the api matches (same transport, same URL shape).
|
|
324
325
|
const baseUrl = existingModel.api === dynamicModel.api ? dynamicModel.baseUrl : existingModel.baseUrl;
|
|
325
|
-
|
|
326
|
+
const merged = enrichModelThinking({
|
|
326
327
|
...existingModel,
|
|
327
328
|
...dynamicModel,
|
|
328
329
|
api: existingModel.api,
|
|
@@ -342,6 +343,9 @@ function mergeDynamicModel<TApi extends Api>(existingModel: Model<TApi>, dynamic
|
|
|
342
343
|
compat: dynamicModel.compat ?? existingModel.compat,
|
|
343
344
|
contextPromotionTarget: dynamicModel.contextPromotionTarget ?? existingModel.contextPromotionTarget,
|
|
344
345
|
});
|
|
346
|
+
const policyModels = [merged as Model<Api>];
|
|
347
|
+
applyGeneratedModelPolicies(policyModels);
|
|
348
|
+
return policyModels[0] as Model<TApi>;
|
|
345
349
|
}
|
|
346
350
|
|
|
347
351
|
function preferDiscoveryCost(discoveryCost: number, fallbackCost: number): number {
|
package/src/model-thinking.ts
CHANGED
|
@@ -699,7 +699,7 @@ function parseAnthropicModel(modelId: string): AnthropicModel | null {
|
|
|
699
699
|
}
|
|
700
700
|
|
|
701
701
|
function parseOpenAIModel(modelId: string): OpenAIModel | null {
|
|
702
|
-
const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|mini|max|nano))
|
|
702
|
+
const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|mini|max|nano))?$/.exec(modelId);
|
|
703
703
|
if (!match) {
|
|
704
704
|
return null;
|
|
705
705
|
}
|