@gajae-code/ai 0.12.1 → 0.12.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/README.md +3 -0
- package/dist/types/provider-models/openai-compat.d.ts +9 -0
- package/dist/types/types.d.ts +1 -1
- package/dist/types/utils/oauth/mara.d.ts +1 -0
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +2 -2
- package/src/auth-storage.ts +98 -15
- package/src/cli.ts +1 -0
- package/src/models.json +89 -1
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +16 -0
- package/src/stream.ts +1 -0
- package/src/types.ts +1 -0
- package/src/utils/oauth/index.ts +6 -0
- package/src/utils/oauth/mara.ts +16 -0
- package/src/utils/oauth/types.ts +1 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,20 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.12.4] - 2026-07-30
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Mara Cloud login now validates pasted credentials against the authenticated chat-completions endpoint instead of the public `/v1/models` catalog. The catalog returns `200` even for random invalid bearer tokens, so the previous check could persist unusable keys.
|
|
10
|
+
|
|
11
|
+
## [0.12.3] - 2026-07-30
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- Added first-class support for **Mara Cloud**, an OpenAI-compatible enterprise AI inference platform. Registers the `mara` provider descriptor, `/login` entry (API-key paste validated against `https://api.cloud.mara.com/v1/models`), `MARA_API_KEY` environment resolution, and bundled `models.json` seed models. Models are discovered dynamically from `GET /v1/models` (base URL `https://api.cloud.mara.com/v1`).
|
|
16
|
+
|
|
17
|
+
## [0.12.2] - 2026-07-30
|
|
18
|
+
|
|
5
19
|
## [0.12.1] - 2026-07-29
|
|
6
20
|
|
|
7
21
|
### Fixed
|
|
@@ -19,6 +33,7 @@
|
|
|
19
33
|
- The Anthropic "thinking blocks in the latest assistant message cannot be modified" 400 now escalates its one-shot replay repair. The error names the latest assistant message but its cited `messages.N.content.M` path can point at an earlier replayed turn, so the latest-only repair was rejected identically and killed the turn; recovery now retries once more with thinking dropped from every replayed assistant message.
|
|
20
34
|
- Anthropic adaptive-thinking `display` support is now decided by the canonical model-version parser instead of a provider-local `claude-opus-(\d+)-(\d+)` regex. The regex only matched two-component ids, so a single-component alias such as `claude-opus-5` was classified as pre-4.7 while its dated snapshot `claude-opus-5-20260101` was not: the alias sent `thinking: { type: "adaptive" }` without `display: "summarized"`, additionally requested the `interleaved-thinking-2025-05-14` beta, and had its returned thinking blocks recorded as raw rather than summarized. Both Anthropic and Bedrock providers now share `supportsAnthropicAdaptiveThinkingDisplay`, so alias and dated ids of the same model send an identical request shape.
|
|
21
35
|
- Anthropic requests that force a tool choice no longer replay signed thinking blocks. Forcing `tool_choice` strips `thinking` from the request (the API rejects the combination), but the converted history still carried native `thinking`/`redacted_thinking` blocks from thinking-enabled turns, so eager tool-forcing turns (e.g. the todo bootstrap) sent a request whose history contradicted its own thinking setting and drew a 400. The replay now degrades in the same rebuild; the forced request trades its prompt-cache prefix for a shape the API accepts.
|
|
36
|
+
- A definitively failed OAuth refresh can no longer loop forever instead of disabling the credential. The refresh-failure path disables the row with a CAS conditioned on its serialized `data`, and treated a lost CAS as proof that a peer had rotated the token: it reloaded the store and re-resolved, without bound. That predicate also misses when nothing was rotated — an account switcher that replaces the provider's rows leaves the attempted id gone, and an unrelated identity-metadata write leaves the row byte-different — so a revoked credential was never disabled and every subsequent request re-issued the same `invalid_grant` refresh (observed in the wild as ~3k `OAuth token refresh failed` / `disable lost CAS` log pairs in 3.5 hours, one wasted refresh round-trip per request). When the row still holds the refresh token that just failed, it is now disabled by id (no peer rotation exists to clobber); otherwise the reload-and-retry recovery is capped, so resolution terminates instead of recursing until the runtime dies.
|
|
22
37
|
|
|
23
38
|
## [0.12.0] - 2026-07-28
|
|
24
39
|
|
package/README.md
CHANGED
|
@@ -71,6 +71,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
|
|
71
71
|
- **ZenMux** (requires `ZENMUX_API_KEY`)
|
|
72
72
|
- **OpenGateway by Sionic AI** (requires `OPENGATEWAY_API_KEY`)
|
|
73
73
|
- **BizRouter** (requires `BIZROUTER_API_KEY`)
|
|
74
|
+
- **Mara Cloud** (requires `MARA_API_KEY`)
|
|
74
75
|
- **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
|
|
75
76
|
- **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
|
|
76
77
|
- **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
|
|
@@ -958,6 +959,7 @@ In Node.js environments, you can set environment variables to avoid passing API
|
|
|
958
959
|
| ZenMux | `ZENMUX_API_KEY` |
|
|
959
960
|
| OpenGateway | `OPENGATEWAY_API_KEY` |
|
|
960
961
|
| BizRouter | `BIZROUTER_API_KEY` |
|
|
962
|
+
| Mara Cloud | `MARA_API_KEY` |
|
|
961
963
|
| vLLM | `VLLM_API_KEY` |
|
|
962
964
|
| Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
|
|
963
965
|
| GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
|
|
@@ -983,6 +985,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
|
|
|
983
985
|
- ZenMux (Anthropic models): `https://zenmux.ai/api/anthropic`
|
|
984
986
|
- OpenGateway by Sionic AI: `https://apis.opengateway.ai/v1`
|
|
985
987
|
- BizRouter: `https://api.bizrouter.ai/v1`
|
|
988
|
+
- Mara Cloud: `https://api.cloud.mara.com/v1`
|
|
986
989
|
- vLLM: `http://127.0.0.1:8000/v1`
|
|
987
990
|
- Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
|
|
988
991
|
- Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
|
|
@@ -128,6 +128,15 @@ export interface BizRouterModelManagerConfig {
|
|
|
128
128
|
baseUrl?: string;
|
|
129
129
|
}
|
|
130
130
|
export declare function bizrouterModelManagerOptions(config?: BizRouterModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
131
|
+
export interface MaraModelManagerConfig {
|
|
132
|
+
apiKey?: string;
|
|
133
|
+
baseUrl?: string;
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Mara Cloud — an OpenAI-compatible enterprise AI inference platform. Models
|
|
137
|
+
* are discovered from the OpenAI-compatible `/v1/models` endpoint.
|
|
138
|
+
*/
|
|
139
|
+
export declare function maraModelManagerOptions(config?: MaraModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
131
140
|
export interface KiloModelManagerConfig {
|
|
132
141
|
apiKey?: string;
|
|
133
142
|
baseUrl?: string;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -51,7 +51,7 @@ export interface ThinkingConfig {
|
|
|
51
51
|
/** Provider-specific transport used to encode the selected effort. */
|
|
52
52
|
mode: ThinkingControlMode;
|
|
53
53
|
}
|
|
54
|
-
export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
54
|
+
export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "mara" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
55
55
|
export type Provider = KnownProvider | string;
|
|
56
56
|
import type { Effort } from "./model-thinking";
|
|
57
57
|
/** Token budgets for each thinking level (token-based providers only) */
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export declare const loginMara: (options: import("./types").OAuthController) => Promise<string>;
|
|
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
|
|
|
7
7
|
email?: string;
|
|
8
8
|
accountId?: string;
|
|
9
9
|
};
|
|
10
|
-
export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
10
|
+
export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "mara" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
11
11
|
export type OAuthProviderId = OAuthProvider | (string & {});
|
|
12
12
|
export type OAuthPrompt = {
|
|
13
13
|
message: string;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.12.
|
|
4
|
+
"version": "0.12.4",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.12.
|
|
43
|
+
"@gajae-code/utils": "0.12.4",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/auth-storage.ts
CHANGED
|
@@ -572,6 +572,16 @@ const OAUTH_REFRESH_SKEW_MS = 60_000;
|
|
|
572
572
|
* pathological detach-without-reattach loops can't grow memory unboundedly.
|
|
573
573
|
*/
|
|
574
574
|
const MAX_PENDING_DISABLED_EVENTS = 32;
|
|
575
|
+
/**
|
|
576
|
+
* Cap on how many times an OAuth resolution may reload the credential store and
|
|
577
|
+
* re-resolve after a failed refresh. Each retry exists to recover from a peer
|
|
578
|
+
* process rotating (or replacing) the row under us, which is a bounded event:
|
|
579
|
+
* the peer either published a usable credential we pick up on the next pass, or
|
|
580
|
+
* it did not. Without a cap, a credential whose disable can never be applied
|
|
581
|
+
* (row replaced by an account switcher, CAS predicate that can never match)
|
|
582
|
+
* makes the recovery path re-issue the same failing token refresh forever.
|
|
583
|
+
*/
|
|
584
|
+
const MAX_OAUTH_RESOLUTION_RELOADS = 3;
|
|
575
585
|
|
|
576
586
|
type UsageCacheEntry<T> = {
|
|
577
587
|
value: T;
|
|
@@ -1426,6 +1436,34 @@ export class AuthStorage {
|
|
|
1426
1436
|
return true;
|
|
1427
1437
|
}
|
|
1428
1438
|
|
|
1439
|
+
/**
|
|
1440
|
+
* Whether the persisted row `credentialId` is still an OAuth credential holding
|
|
1441
|
+
* `refreshToken`. Used by the refresh-failure path to tell "a peer rotated this
|
|
1442
|
+
* row" (retry is worthwhile) apart from "the row is unchanged but the CAS
|
|
1443
|
+
* predicate cannot match it" (retry replays the same failing refresh).
|
|
1444
|
+
*/
|
|
1445
|
+
#credentialRowHoldsRefreshToken(provider: string, credentialId: number, refreshToken: string): boolean {
|
|
1446
|
+
const row = this.#store.listAuthCredentials(provider).find(entry => entry.id === credentialId);
|
|
1447
|
+
const credential = row?.credential;
|
|
1448
|
+
return credential?.type === "oauth" && credential.refresh === refreshToken;
|
|
1449
|
+
}
|
|
1450
|
+
|
|
1451
|
+
/**
|
|
1452
|
+
* Soft-deletes a row by id, bypassing the data-equality CAS. Only safe when the
|
|
1453
|
+
* caller has confirmed the row still holds the credential it attempted to
|
|
1454
|
+
* refresh, so no peer rotation can be clobbered.
|
|
1455
|
+
*/
|
|
1456
|
+
#disableCredentialById(provider: string, credentialId: number, disabledCause: string): void {
|
|
1457
|
+
this.#store.deleteAuthCredential(credentialId, disabledCause);
|
|
1458
|
+
const entries = this.#getStoredCredentials(provider);
|
|
1459
|
+
this.#setStoredCredentials(
|
|
1460
|
+
provider,
|
|
1461
|
+
entries.filter(entry => entry.id !== credentialId),
|
|
1462
|
+
);
|
|
1463
|
+
this.#resetProviderAssignments(provider);
|
|
1464
|
+
this.#emitCredentialDisabled({ provider, disabledCause });
|
|
1465
|
+
}
|
|
1466
|
+
|
|
1429
1467
|
#emitCredentialDisabled(event: CredentialDisabledEvent): void {
|
|
1430
1468
|
if (this.#credentialDisabledListeners.size === 0) {
|
|
1431
1469
|
// No subscribers — buffer for later replay. Cap the backlog so a process that runs
|
|
@@ -2004,6 +2042,12 @@ export class AuthStorage {
|
|
|
2004
2042
|
await saveApiKeyCredential(apiKey);
|
|
2005
2043
|
return;
|
|
2006
2044
|
}
|
|
2045
|
+
case "mara": {
|
|
2046
|
+
const { loginMara } = await import("./utils/oauth/mara");
|
|
2047
|
+
const apiKey = await loginMara(ctrl);
|
|
2048
|
+
await saveApiKeyCredential(apiKey);
|
|
2049
|
+
return;
|
|
2050
|
+
}
|
|
2007
2051
|
case "opengateway": {
|
|
2008
2052
|
const { loginOpenGateway } = await import("./utils/oauth/opengateway");
|
|
2009
2053
|
const apiKey = await loginOpenGateway(ctrl);
|
|
@@ -2989,7 +3033,15 @@ export class AuthStorage {
|
|
|
2989
3033
|
provider: string,
|
|
2990
3034
|
sessionId?: string,
|
|
2991
3035
|
options?: AuthApiKeyOptions,
|
|
3036
|
+
reloadsUsed = 0,
|
|
2992
3037
|
): Promise<OAuthResolutionResult | undefined> {
|
|
3038
|
+
if (reloadsUsed > MAX_OAUTH_RESOLUTION_RELOADS) {
|
|
3039
|
+
logger.warn("OAuth credential resolution exhausted its reload budget", {
|
|
3040
|
+
provider,
|
|
3041
|
+
reloadsUsed,
|
|
3042
|
+
});
|
|
3043
|
+
return undefined;
|
|
3044
|
+
}
|
|
2993
3045
|
const selectedCredential = this.#resolveSelectedStoredCredential(provider, options);
|
|
2994
3046
|
const selectedOAuthCredential =
|
|
2995
3047
|
selectedCredential?.credential.type === "oauth"
|
|
@@ -3087,18 +3139,27 @@ export class AuthStorage {
|
|
|
3087
3139
|
usagePrechecked: candidate.usageChecked,
|
|
3088
3140
|
enforceProRequirement,
|
|
3089
3141
|
},
|
|
3142
|
+
reloadsUsed,
|
|
3090
3143
|
);
|
|
3091
3144
|
if (resolved) return resolved;
|
|
3092
3145
|
}
|
|
3093
3146
|
|
|
3094
3147
|
if (fallback && this.#isCredentialBlocked(providerKey, fallback.selection.index)) {
|
|
3095
|
-
return this.#tryOAuthCredential(
|
|
3096
|
-
|
|
3097
|
-
|
|
3098
|
-
|
|
3099
|
-
|
|
3100
|
-
|
|
3101
|
-
|
|
3148
|
+
return this.#tryOAuthCredential(
|
|
3149
|
+
provider,
|
|
3150
|
+
fallback.selection,
|
|
3151
|
+
providerKey,
|
|
3152
|
+
sessionId,
|
|
3153
|
+
options,
|
|
3154
|
+
{
|
|
3155
|
+
checkUsage,
|
|
3156
|
+
allowBlocked: true,
|
|
3157
|
+
prefetchedUsage: fallback.usage,
|
|
3158
|
+
usagePrechecked: fallback.usageChecked,
|
|
3159
|
+
enforceProRequirement,
|
|
3160
|
+
},
|
|
3161
|
+
reloadsUsed,
|
|
3162
|
+
);
|
|
3102
3163
|
}
|
|
3103
3164
|
|
|
3104
3165
|
return undefined;
|
|
@@ -3219,6 +3280,7 @@ export class AuthStorage {
|
|
|
3219
3280
|
usagePrechecked?: boolean;
|
|
3220
3281
|
enforceProRequirement?: boolean;
|
|
3221
3282
|
},
|
|
3283
|
+
reloadsUsed = 0,
|
|
3222
3284
|
): Promise<OAuthResolutionResult | undefined> {
|
|
3223
3285
|
const {
|
|
3224
3286
|
checkUsage,
|
|
@@ -3357,7 +3419,7 @@ export class AuthStorage {
|
|
|
3357
3419
|
credentialId: attemptedCredentialId,
|
|
3358
3420
|
});
|
|
3359
3421
|
await this.reload();
|
|
3360
|
-
return this.#resolveOAuthSelection(provider, sessionId, options);
|
|
3422
|
+
return this.#resolveOAuthSelection(provider, sessionId, options, reloadsUsed + 1);
|
|
3361
3423
|
}
|
|
3362
3424
|
}
|
|
3363
3425
|
// Only remove credentials for definitive auth failures
|
|
@@ -3387,18 +3449,39 @@ export class AuthStorage {
|
|
|
3387
3449
|
`oauth refresh failed: ${errorMsg}`,
|
|
3388
3450
|
);
|
|
3389
3451
|
if (!disabled) {
|
|
3390
|
-
|
|
3391
|
-
|
|
3392
|
-
|
|
3393
|
-
|
|
3394
|
-
|
|
3395
|
-
|
|
3452
|
+
// The CAS predicate compares the row's serialized `data`, so it also
|
|
3453
|
+
// misses when nothing was rotated: the row may have been replaced by
|
|
3454
|
+
// a peer (account switcher rewriting the provider's credentials, so
|
|
3455
|
+
// our snapshot's id no longer exists) or updated with unrelated
|
|
3456
|
+
// identity metadata. Reload-and-retry only makes progress in the
|
|
3457
|
+
// rotation case; otherwise the same revoked token is re-refreshed on
|
|
3458
|
+
// every request forever. When the row is still present with the very
|
|
3459
|
+
// refresh token we just tried, disabling by id is safe — there is no
|
|
3460
|
+
// peer rotation to clobber — so apply it directly instead of looping.
|
|
3461
|
+
const stillHoldsAttemptedToken =
|
|
3462
|
+
attemptedCredentialId !== undefined &&
|
|
3463
|
+
this.#credentialRowHoldsRefreshToken(provider, attemptedCredentialId, selection.credential.refresh);
|
|
3464
|
+
if (stillHoldsAttemptedToken && attemptedCredentialId !== undefined) {
|
|
3465
|
+
logger.warn("OAuth refresh disable CAS mismatched an unrotated row; disabling by id", {
|
|
3466
|
+
provider,
|
|
3467
|
+
index: selection.index,
|
|
3468
|
+
credentialId: attemptedCredentialId,
|
|
3469
|
+
});
|
|
3470
|
+
this.#disableCredentialById(provider, attemptedCredentialId, `oauth refresh failed: ${errorMsg}`);
|
|
3471
|
+
} else {
|
|
3472
|
+
logger.debug("OAuth refresh disable lost CAS; reloading after peer rotation", {
|
|
3473
|
+
provider,
|
|
3474
|
+
index: selection.index,
|
|
3475
|
+
});
|
|
3476
|
+
await this.reload();
|
|
3477
|
+
return this.#resolveOAuthSelection(provider, sessionId, options, reloadsUsed + 1);
|
|
3478
|
+
}
|
|
3396
3479
|
}
|
|
3397
3480
|
if (
|
|
3398
3481
|
!this.#getCredentialSelector(provider, options) &&
|
|
3399
3482
|
this.#getCredentialsForProvider(provider).some(credential => credential.type === "oauth")
|
|
3400
3483
|
) {
|
|
3401
|
-
return this.#resolveOAuthSelection(provider, sessionId, options);
|
|
3484
|
+
return this.#resolveOAuthSelection(provider, sessionId, options, reloadsUsed);
|
|
3402
3485
|
}
|
|
3403
3486
|
} else {
|
|
3404
3487
|
// Block temporarily for transient failures (5 minutes)
|
package/src/cli.ts
CHANGED
package/src/models.json
CHANGED
|
@@ -39882,6 +39882,94 @@
|
|
|
39882
39882
|
"maxTokens": 8888
|
|
39883
39883
|
}
|
|
39884
39884
|
},
|
|
39885
|
+
"mara": {
|
|
39886
|
+
"DeepSeek-V3.1": {
|
|
39887
|
+
"id": "DeepSeek-V3.1",
|
|
39888
|
+
"name": "DeepSeek V3.1",
|
|
39889
|
+
"api": "openai-completions",
|
|
39890
|
+
"provider": "mara",
|
|
39891
|
+
"baseUrl": "https://api.cloud.mara.com/v1",
|
|
39892
|
+
"reasoning": true,
|
|
39893
|
+
"input": [
|
|
39894
|
+
"text"
|
|
39895
|
+
],
|
|
39896
|
+
"cost": {
|
|
39897
|
+
"input": 0.6,
|
|
39898
|
+
"output": 1.7,
|
|
39899
|
+
"cacheRead": 0,
|
|
39900
|
+
"cacheWrite": 0
|
|
39901
|
+
},
|
|
39902
|
+
"contextWindow": 131072,
|
|
39903
|
+
"maxTokens": 16384,
|
|
39904
|
+
"thinking": {
|
|
39905
|
+
"mode": "effort",
|
|
39906
|
+
"minLevel": "minimal",
|
|
39907
|
+
"maxLevel": "xhigh"
|
|
39908
|
+
}
|
|
39909
|
+
},
|
|
39910
|
+
"MiniMax-M2.5": {
|
|
39911
|
+
"id": "MiniMax-M2.5",
|
|
39912
|
+
"name": "MiniMax M2.5",
|
|
39913
|
+
"api": "openai-completions",
|
|
39914
|
+
"provider": "mara",
|
|
39915
|
+
"baseUrl": "https://api.cloud.mara.com/v1",
|
|
39916
|
+
"reasoning": false,
|
|
39917
|
+
"input": [
|
|
39918
|
+
"text"
|
|
39919
|
+
],
|
|
39920
|
+
"cost": {
|
|
39921
|
+
"input": 0.3,
|
|
39922
|
+
"output": 1.2,
|
|
39923
|
+
"cacheRead": 0,
|
|
39924
|
+
"cacheWrite": 0
|
|
39925
|
+
},
|
|
39926
|
+
"contextWindow": 196608,
|
|
39927
|
+
"maxTokens": 16384
|
|
39928
|
+
},
|
|
39929
|
+
"MiniMax-M2.7": {
|
|
39930
|
+
"id": "MiniMax-M2.7",
|
|
39931
|
+
"name": "MiniMax M2.7",
|
|
39932
|
+
"api": "openai-completions",
|
|
39933
|
+
"provider": "mara",
|
|
39934
|
+
"baseUrl": "https://api.cloud.mara.com/v1",
|
|
39935
|
+
"reasoning": false,
|
|
39936
|
+
"input": [
|
|
39937
|
+
"text"
|
|
39938
|
+
],
|
|
39939
|
+
"cost": {
|
|
39940
|
+
"input": 0.3,
|
|
39941
|
+
"output": 1.2,
|
|
39942
|
+
"cacheRead": 0,
|
|
39943
|
+
"cacheWrite": 0
|
|
39944
|
+
},
|
|
39945
|
+
"contextWindow": 196608,
|
|
39946
|
+
"maxTokens": 16384
|
|
39947
|
+
},
|
|
39948
|
+
"gpt-oss-120b": {
|
|
39949
|
+
"id": "gpt-oss-120b",
|
|
39950
|
+
"name": "GPT OSS 120B",
|
|
39951
|
+
"api": "openai-completions",
|
|
39952
|
+
"provider": "mara",
|
|
39953
|
+
"baseUrl": "https://api.cloud.mara.com/v1",
|
|
39954
|
+
"reasoning": true,
|
|
39955
|
+
"input": [
|
|
39956
|
+
"text"
|
|
39957
|
+
],
|
|
39958
|
+
"cost": {
|
|
39959
|
+
"input": 0.15,
|
|
39960
|
+
"output": 0.75,
|
|
39961
|
+
"cacheRead": 0,
|
|
39962
|
+
"cacheWrite": 0
|
|
39963
|
+
},
|
|
39964
|
+
"contextWindow": 131072,
|
|
39965
|
+
"maxTokens": 16384,
|
|
39966
|
+
"thinking": {
|
|
39967
|
+
"mode": "effort",
|
|
39968
|
+
"minLevel": "minimal",
|
|
39969
|
+
"maxLevel": "xhigh"
|
|
39970
|
+
}
|
|
39971
|
+
}
|
|
39972
|
+
},
|
|
39885
39973
|
"minimax": {
|
|
39886
39974
|
"MiniMax-M2": {
|
|
39887
39975
|
"id": "MiniMax-M2",
|
|
@@ -85423,4 +85511,4 @@
|
|
|
85423
85511
|
}
|
|
85424
85512
|
}
|
|
85425
85513
|
}
|
|
85426
|
-
}
|
|
85514
|
+
}
|
|
@@ -26,6 +26,7 @@ import {
|
|
|
26
26
|
kimiCodeModelManagerOptions,
|
|
27
27
|
litellmModelManagerOptions,
|
|
28
28
|
lmStudioModelManagerOptions,
|
|
29
|
+
maraModelManagerOptions,
|
|
29
30
|
mistralModelManagerOptions,
|
|
30
31
|
moonshotModelManagerOptions,
|
|
31
32
|
nanoGptModelManagerOptions,
|
|
@@ -326,6 +327,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
|
|
|
326
327
|
config => bizrouterModelManagerOptions(config),
|
|
327
328
|
catalog("BizRouter", ["BIZROUTER_API_KEY"]),
|
|
328
329
|
),
|
|
330
|
+
catalogDescriptor(
|
|
331
|
+
"mara",
|
|
332
|
+
"DeepSeek-V3.1",
|
|
333
|
+
config => maraModelManagerOptions(config),
|
|
334
|
+
catalog("Mara Cloud", ["MARA_API_KEY"]),
|
|
335
|
+
),
|
|
329
336
|
catalogDescriptor("zai", "glm-5.2", config => zaiModelManagerOptions(config), catalog("zAI", ["ZAI_API_KEY"])),
|
|
330
337
|
catalogDescriptor(
|
|
331
338
|
"glm-zcode",
|
|
@@ -1180,6 +1180,22 @@ export function bizrouterModelManagerOptions(
|
|
|
1180
1180
|
};
|
|
1181
1181
|
}
|
|
1182
1182
|
|
|
1183
|
+
// ---------------------------------------------------------------------------
|
|
1184
|
+
// 10.5.3 Mara Cloud
|
|
1185
|
+
// ---------------------------------------------------------------------------
|
|
1186
|
+
|
|
1187
|
+
export interface MaraModelManagerConfig {
|
|
1188
|
+
apiKey?: string;
|
|
1189
|
+
baseUrl?: string;
|
|
1190
|
+
}
|
|
1191
|
+
|
|
1192
|
+
/**
|
|
1193
|
+
* Mara Cloud — an OpenAI-compatible enterprise AI inference platform. Models
|
|
1194
|
+
* are discovered from the OpenAI-compatible `/v1/models` endpoint.
|
|
1195
|
+
*/
|
|
1196
|
+
export function maraModelManagerOptions(config?: MaraModelManagerConfig): ModelManagerOptions<"openai-completions"> {
|
|
1197
|
+
return createSimpleOpenAICompletionsOptions("mara", "https://api.cloud.mara.com/v1", config);
|
|
1198
|
+
}
|
|
1183
1199
|
// ---------------------------------------------------------------------------
|
|
1184
1200
|
// 10.6 Kilo Gateway
|
|
1185
1201
|
// ---------------------------------------------------------------------------
|
package/src/stream.ts
CHANGED
|
@@ -164,6 +164,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
164
164
|
zenmux: "ZENMUX_API_KEY",
|
|
165
165
|
opengateway: "OPENGATEWAY_API_KEY",
|
|
166
166
|
bizrouter: "BIZROUTER_API_KEY",
|
|
167
|
+
mara: "MARA_API_KEY",
|
|
167
168
|
venice: "VENICE_API_KEY",
|
|
168
169
|
vllm: "VLLM_API_KEY",
|
|
169
170
|
xiaomi: "XIAOMI_API_KEY",
|
package/src/types.ts
CHANGED
package/src/utils/oauth/index.ts
CHANGED
|
@@ -245,6 +245,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
|
|
|
245
245
|
name: "BizRouter",
|
|
246
246
|
available: true,
|
|
247
247
|
},
|
|
248
|
+
{
|
|
249
|
+
id: "mara",
|
|
250
|
+
name: "Mara Cloud",
|
|
251
|
+
available: true,
|
|
252
|
+
},
|
|
248
253
|
{
|
|
249
254
|
id: "opengateway",
|
|
250
255
|
name: "OpenGateway by Sionic AI",
|
|
@@ -393,6 +398,7 @@ export async function refreshOAuthToken(
|
|
|
393
398
|
case "moonshot":
|
|
394
399
|
case "kagi":
|
|
395
400
|
case "cloudflare-ai-gateway":
|
|
401
|
+
case "mara":
|
|
396
402
|
case "vercel-ai-gateway":
|
|
397
403
|
case "qwen-portal":
|
|
398
404
|
case "zenmux":
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/** Mara Cloud login flow (API key paste, validated via chat completions). */
|
|
2
|
+
import { createApiKeyLogin } from "./api-key-login";
|
|
3
|
+
|
|
4
|
+
export const loginMara = createApiKeyLogin({
|
|
5
|
+
providerLabel: "Mara Cloud",
|
|
6
|
+
authUrl: "https://cloud.mara.com/apis",
|
|
7
|
+
instructions: "Create or copy your Mara Cloud API key",
|
|
8
|
+
promptMessage: "Paste your Mara Cloud API key",
|
|
9
|
+
placeholder: "<your-mara-api-key>",
|
|
10
|
+
validation: {
|
|
11
|
+
kind: "chat-completions",
|
|
12
|
+
provider: "Mara Cloud",
|
|
13
|
+
baseUrl: "https://api.cloud.mara.com/v1",
|
|
14
|
+
model: "DeepSeek-V3.1",
|
|
15
|
+
},
|
|
16
|
+
});
|