@gajae-code/ai 0.12.1 → 0.12.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,20 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.12.4] - 2026-07-30
6
+
7
+ ### Fixed
8
+
9
+ - Mara Cloud login now validates pasted credentials against the authenticated chat-completions endpoint instead of the public `/v1/models` catalog. The catalog returns `200` even for random invalid bearer tokens, so the previous check could persist unusable keys.
10
+
11
+ ## [0.12.3] - 2026-07-30
12
+
13
+ ### Added
14
+
15
+ - Added first-class support for **Mara Cloud**, an OpenAI-compatible enterprise AI inference platform. Registers the `mara` provider descriptor, `/login` entry (API-key paste validated against `https://api.cloud.mara.com/v1/models`), `MARA_API_KEY` environment resolution, and bundled `models.json` seed models. Models are discovered dynamically from `GET /v1/models` (base URL `https://api.cloud.mara.com/v1`).
16
+
17
+ ## [0.12.2] - 2026-07-30
18
+
5
19
  ## [0.12.1] - 2026-07-29
6
20
 
7
21
  ### Fixed
@@ -19,6 +33,7 @@
19
33
  - The Anthropic "thinking blocks in the latest assistant message cannot be modified" 400 now escalates its one-shot replay repair. The error names the latest assistant message but its cited `messages.N.content.M` path can point at an earlier replayed turn, so the latest-only repair was rejected identically and killed the turn; recovery now retries once more with thinking dropped from every replayed assistant message.
20
34
  - Anthropic adaptive-thinking `display` support is now decided by the canonical model-version parser instead of a provider-local `claude-opus-(\d+)-(\d+)` regex. The regex only matched two-component ids, so a single-component alias such as `claude-opus-5` was classified as pre-4.7 while its dated snapshot `claude-opus-5-20260101` was not: the alias sent `thinking: { type: "adaptive" }` without `display: "summarized"`, additionally requested the `interleaved-thinking-2025-05-14` beta, and had its returned thinking blocks recorded as raw rather than summarized. Both Anthropic and Bedrock providers now share `supportsAnthropicAdaptiveThinkingDisplay`, so alias and dated ids of the same model send an identical request shape.
21
35
  - Anthropic requests that force a tool choice no longer replay signed thinking blocks. Forcing `tool_choice` strips `thinking` from the request (the API rejects the combination), but the converted history still carried native `thinking`/`redacted_thinking` blocks from thinking-enabled turns, so eager tool-forcing turns (e.g. the todo bootstrap) sent a request whose history contradicted its own thinking setting and drew a 400. The replay now degrades in the same rebuild; the forced request trades its prompt-cache prefix for a shape the API accepts.
36
+ - A definitively failed OAuth refresh can no longer loop forever instead of disabling the credential. The refresh-failure path disables the row with a CAS conditioned on its serialized `data`, and treated a lost CAS as proof that a peer had rotated the token: it reloaded the store and re-resolved, without bound. That predicate also misses when nothing was rotated — an account switcher that replaces the provider's rows leaves the attempted id gone, and an unrelated identity-metadata write leaves the row byte-different — so a revoked credential was never disabled and every subsequent request re-issued the same `invalid_grant` refresh (observed in the wild as ~3k `OAuth token refresh failed` / `disable lost CAS` log pairs in 3.5 hours, one wasted refresh round-trip per request). When the row still holds the refresh token that just failed, it is now disabled by id (no peer rotation exists to clobber); otherwise the reload-and-retry recovery is capped, so resolution terminates instead of recursing until the runtime dies.
22
37
 
23
38
  ## [0.12.0] - 2026-07-28
24
39
 
package/README.md CHANGED
@@ -71,6 +71,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
71
71
  - **ZenMux** (requires `ZENMUX_API_KEY`)
72
72
  - **OpenGateway by Sionic AI** (requires `OPENGATEWAY_API_KEY`)
73
73
  - **BizRouter** (requires `BIZROUTER_API_KEY`)
74
+ - **Mara Cloud** (requires `MARA_API_KEY`)
74
75
  - **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
75
76
  - **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
76
77
  - **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
@@ -958,6 +959,7 @@ In Node.js environments, you can set environment variables to avoid passing API
958
959
  | ZenMux | `ZENMUX_API_KEY` |
959
960
  | OpenGateway | `OPENGATEWAY_API_KEY` |
960
961
  | BizRouter | `BIZROUTER_API_KEY` |
962
+ | Mara Cloud | `MARA_API_KEY` |
961
963
  | vLLM | `VLLM_API_KEY` |
962
964
  | Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
963
965
  | GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
@@ -983,6 +985,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
983
985
  - ZenMux (Anthropic models): `https://zenmux.ai/api/anthropic`
984
986
  - OpenGateway by Sionic AI: `https://apis.opengateway.ai/v1`
985
987
  - BizRouter: `https://api.bizrouter.ai/v1`
988
+ - Mara Cloud: `https://api.cloud.mara.com/v1`
986
989
  - vLLM: `http://127.0.0.1:8000/v1`
987
990
  - Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
988
991
  - Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
@@ -128,6 +128,15 @@ export interface BizRouterModelManagerConfig {
128
128
  baseUrl?: string;
129
129
  }
130
130
  export declare function bizrouterModelManagerOptions(config?: BizRouterModelManagerConfig): ModelManagerOptions<"openai-completions">;
131
+ export interface MaraModelManagerConfig {
132
+ apiKey?: string;
133
+ baseUrl?: string;
134
+ }
135
+ /**
136
+ * Mara Cloud — an OpenAI-compatible enterprise AI inference platform. Models
137
+ * are discovered from the OpenAI-compatible `/v1/models` endpoint.
138
+ */
139
+ export declare function maraModelManagerOptions(config?: MaraModelManagerConfig): ModelManagerOptions<"openai-completions">;
131
140
  export interface KiloModelManagerConfig {
132
141
  apiKey?: string;
133
142
  baseUrl?: string;
@@ -51,7 +51,7 @@ export interface ThinkingConfig {
51
51
  /** Provider-specific transport used to encode the selected effort. */
52
52
  mode: ThinkingControlMode;
53
53
  }
54
- export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
54
+ export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "mara" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
55
55
  export type Provider = KnownProvider | string;
56
56
  import type { Effort } from "./model-thinking";
57
57
  /** Token budgets for each thinking level (token-based providers only) */
@@ -0,0 +1 @@
1
+ export declare const loginMara: (options: import("./types").OAuthController) => Promise<string>;
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "mara" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.12.1",
4
+ "version": "0.12.4",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.12.1",
43
+ "@gajae-code/utils": "0.12.4",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -572,6 +572,16 @@ const OAUTH_REFRESH_SKEW_MS = 60_000;
572
572
  * pathological detach-without-reattach loops can't grow memory unboundedly.
573
573
  */
574
574
  const MAX_PENDING_DISABLED_EVENTS = 32;
575
+ /**
576
+ * Cap on how many times an OAuth resolution may reload the credential store and
577
+ * re-resolve after a failed refresh. Each retry exists to recover from a peer
578
+ * process rotating (or replacing) the row under us, which is a bounded event:
579
+ * the peer either published a usable credential we pick up on the next pass, or
580
+ * it did not. Without a cap, a credential whose disable can never be applied
581
+ * (row replaced by an account switcher, CAS predicate that can never match)
582
+ * makes the recovery path re-issue the same failing token refresh forever.
583
+ */
584
+ const MAX_OAUTH_RESOLUTION_RELOADS = 3;
575
585
 
576
586
  type UsageCacheEntry<T> = {
577
587
  value: T;
@@ -1426,6 +1436,34 @@ export class AuthStorage {
1426
1436
  return true;
1427
1437
  }
1428
1438
 
1439
+ /**
1440
+ * Whether the persisted row `credentialId` is still an OAuth credential holding
1441
+ * `refreshToken`. Used by the refresh-failure path to tell "a peer rotated this
1442
+ * row" (retry is worthwhile) apart from "the row is unchanged but the CAS
1443
+ * predicate cannot match it" (retry replays the same failing refresh).
1444
+ */
1445
+ #credentialRowHoldsRefreshToken(provider: string, credentialId: number, refreshToken: string): boolean {
1446
+ const row = this.#store.listAuthCredentials(provider).find(entry => entry.id === credentialId);
1447
+ const credential = row?.credential;
1448
+ return credential?.type === "oauth" && credential.refresh === refreshToken;
1449
+ }
1450
+
1451
+ /**
1452
+ * Soft-deletes a row by id, bypassing the data-equality CAS. Only safe when the
1453
+ * caller has confirmed the row still holds the credential it attempted to
1454
+ * refresh, so no peer rotation can be clobbered.
1455
+ */
1456
+ #disableCredentialById(provider: string, credentialId: number, disabledCause: string): void {
1457
+ this.#store.deleteAuthCredential(credentialId, disabledCause);
1458
+ const entries = this.#getStoredCredentials(provider);
1459
+ this.#setStoredCredentials(
1460
+ provider,
1461
+ entries.filter(entry => entry.id !== credentialId),
1462
+ );
1463
+ this.#resetProviderAssignments(provider);
1464
+ this.#emitCredentialDisabled({ provider, disabledCause });
1465
+ }
1466
+
1429
1467
  #emitCredentialDisabled(event: CredentialDisabledEvent): void {
1430
1468
  if (this.#credentialDisabledListeners.size === 0) {
1431
1469
  // No subscribers — buffer for later replay. Cap the backlog so a process that runs
@@ -2004,6 +2042,12 @@ export class AuthStorage {
2004
2042
  await saveApiKeyCredential(apiKey);
2005
2043
  return;
2006
2044
  }
2045
+ case "mara": {
2046
+ const { loginMara } = await import("./utils/oauth/mara");
2047
+ const apiKey = await loginMara(ctrl);
2048
+ await saveApiKeyCredential(apiKey);
2049
+ return;
2050
+ }
2007
2051
  case "opengateway": {
2008
2052
  const { loginOpenGateway } = await import("./utils/oauth/opengateway");
2009
2053
  const apiKey = await loginOpenGateway(ctrl);
@@ -2989,7 +3033,15 @@ export class AuthStorage {
2989
3033
  provider: string,
2990
3034
  sessionId?: string,
2991
3035
  options?: AuthApiKeyOptions,
3036
+ reloadsUsed = 0,
2992
3037
  ): Promise<OAuthResolutionResult | undefined> {
3038
+ if (reloadsUsed > MAX_OAUTH_RESOLUTION_RELOADS) {
3039
+ logger.warn("OAuth credential resolution exhausted its reload budget", {
3040
+ provider,
3041
+ reloadsUsed,
3042
+ });
3043
+ return undefined;
3044
+ }
2993
3045
  const selectedCredential = this.#resolveSelectedStoredCredential(provider, options);
2994
3046
  const selectedOAuthCredential =
2995
3047
  selectedCredential?.credential.type === "oauth"
@@ -3087,18 +3139,27 @@ export class AuthStorage {
3087
3139
  usagePrechecked: candidate.usageChecked,
3088
3140
  enforceProRequirement,
3089
3141
  },
3142
+ reloadsUsed,
3090
3143
  );
3091
3144
  if (resolved) return resolved;
3092
3145
  }
3093
3146
 
3094
3147
  if (fallback && this.#isCredentialBlocked(providerKey, fallback.selection.index)) {
3095
- return this.#tryOAuthCredential(provider, fallback.selection, providerKey, sessionId, options, {
3096
- checkUsage,
3097
- allowBlocked: true,
3098
- prefetchedUsage: fallback.usage,
3099
- usagePrechecked: fallback.usageChecked,
3100
- enforceProRequirement,
3101
- });
3148
+ return this.#tryOAuthCredential(
3149
+ provider,
3150
+ fallback.selection,
3151
+ providerKey,
3152
+ sessionId,
3153
+ options,
3154
+ {
3155
+ checkUsage,
3156
+ allowBlocked: true,
3157
+ prefetchedUsage: fallback.usage,
3158
+ usagePrechecked: fallback.usageChecked,
3159
+ enforceProRequirement,
3160
+ },
3161
+ reloadsUsed,
3162
+ );
3102
3163
  }
3103
3164
 
3104
3165
  return undefined;
@@ -3219,6 +3280,7 @@ export class AuthStorage {
3219
3280
  usagePrechecked?: boolean;
3220
3281
  enforceProRequirement?: boolean;
3221
3282
  },
3283
+ reloadsUsed = 0,
3222
3284
  ): Promise<OAuthResolutionResult | undefined> {
3223
3285
  const {
3224
3286
  checkUsage,
@@ -3357,7 +3419,7 @@ export class AuthStorage {
3357
3419
  credentialId: attemptedCredentialId,
3358
3420
  });
3359
3421
  await this.reload();
3360
- return this.#resolveOAuthSelection(provider, sessionId, options);
3422
+ return this.#resolveOAuthSelection(provider, sessionId, options, reloadsUsed + 1);
3361
3423
  }
3362
3424
  }
3363
3425
  // Only remove credentials for definitive auth failures
@@ -3387,18 +3449,39 @@ export class AuthStorage {
3387
3449
  `oauth refresh failed: ${errorMsg}`,
3388
3450
  );
3389
3451
  if (!disabled) {
3390
- logger.debug("OAuth refresh disable lost CAS; reloading after peer rotation", {
3391
- provider,
3392
- index: selection.index,
3393
- });
3394
- await this.reload();
3395
- return this.#resolveOAuthSelection(provider, sessionId, options);
3452
+ // The CAS predicate compares the row's serialized `data`, so it also
3453
+ // misses when nothing was rotated: the row may have been replaced by
3454
+ // a peer (account switcher rewriting the provider's credentials, so
3455
+ // our snapshot's id no longer exists) or updated with unrelated
3456
+ // identity metadata. Reload-and-retry only makes progress in the
3457
+ // rotation case; otherwise the same revoked token is re-refreshed on
3458
+ // every request forever. When the row is still present with the very
3459
+ // refresh token we just tried, disabling by id is safe — there is no
3460
+ // peer rotation to clobber — so apply it directly instead of looping.
3461
+ const stillHoldsAttemptedToken =
3462
+ attemptedCredentialId !== undefined &&
3463
+ this.#credentialRowHoldsRefreshToken(provider, attemptedCredentialId, selection.credential.refresh);
3464
+ if (stillHoldsAttemptedToken && attemptedCredentialId !== undefined) {
3465
+ logger.warn("OAuth refresh disable CAS mismatched an unrotated row; disabling by id", {
3466
+ provider,
3467
+ index: selection.index,
3468
+ credentialId: attemptedCredentialId,
3469
+ });
3470
+ this.#disableCredentialById(provider, attemptedCredentialId, `oauth refresh failed: ${errorMsg}`);
3471
+ } else {
3472
+ logger.debug("OAuth refresh disable lost CAS; reloading after peer rotation", {
3473
+ provider,
3474
+ index: selection.index,
3475
+ });
3476
+ await this.reload();
3477
+ return this.#resolveOAuthSelection(provider, sessionId, options, reloadsUsed + 1);
3478
+ }
3396
3479
  }
3397
3480
  if (
3398
3481
  !this.#getCredentialSelector(provider, options) &&
3399
3482
  this.#getCredentialsForProvider(provider).some(credential => credential.type === "oauth")
3400
3483
  ) {
3401
- return this.#resolveOAuthSelection(provider, sessionId, options);
3484
+ return this.#resolveOAuthSelection(provider, sessionId, options, reloadsUsed);
3402
3485
  }
3403
3486
  } else {
3404
3487
  // Block temporarily for transient failures (5 minutes)
package/src/cli.ts CHANGED
@@ -120,6 +120,7 @@ Providers:
120
120
  zenmux ZenMux
121
121
  opengateway OpenGateway by Sionic AI
122
122
  bizrouter BizRouter
123
+ mara Mara Cloud
123
124
  ollama-cloud Ollama Cloud
124
125
 
125
126
  Examples:
package/src/models.json CHANGED
@@ -39882,6 +39882,94 @@
39882
39882
  "maxTokens": 8888
39883
39883
  }
39884
39884
  },
39885
+ "mara": {
39886
+ "DeepSeek-V3.1": {
39887
+ "id": "DeepSeek-V3.1",
39888
+ "name": "DeepSeek V3.1",
39889
+ "api": "openai-completions",
39890
+ "provider": "mara",
39891
+ "baseUrl": "https://api.cloud.mara.com/v1",
39892
+ "reasoning": true,
39893
+ "input": [
39894
+ "text"
39895
+ ],
39896
+ "cost": {
39897
+ "input": 0.6,
39898
+ "output": 1.7,
39899
+ "cacheRead": 0,
39900
+ "cacheWrite": 0
39901
+ },
39902
+ "contextWindow": 131072,
39903
+ "maxTokens": 16384,
39904
+ "thinking": {
39905
+ "mode": "effort",
39906
+ "minLevel": "minimal",
39907
+ "maxLevel": "xhigh"
39908
+ }
39909
+ },
39910
+ "MiniMax-M2.5": {
39911
+ "id": "MiniMax-M2.5",
39912
+ "name": "MiniMax M2.5",
39913
+ "api": "openai-completions",
39914
+ "provider": "mara",
39915
+ "baseUrl": "https://api.cloud.mara.com/v1",
39916
+ "reasoning": false,
39917
+ "input": [
39918
+ "text"
39919
+ ],
39920
+ "cost": {
39921
+ "input": 0.3,
39922
+ "output": 1.2,
39923
+ "cacheRead": 0,
39924
+ "cacheWrite": 0
39925
+ },
39926
+ "contextWindow": 196608,
39927
+ "maxTokens": 16384
39928
+ },
39929
+ "MiniMax-M2.7": {
39930
+ "id": "MiniMax-M2.7",
39931
+ "name": "MiniMax M2.7",
39932
+ "api": "openai-completions",
39933
+ "provider": "mara",
39934
+ "baseUrl": "https://api.cloud.mara.com/v1",
39935
+ "reasoning": false,
39936
+ "input": [
39937
+ "text"
39938
+ ],
39939
+ "cost": {
39940
+ "input": 0.3,
39941
+ "output": 1.2,
39942
+ "cacheRead": 0,
39943
+ "cacheWrite": 0
39944
+ },
39945
+ "contextWindow": 196608,
39946
+ "maxTokens": 16384
39947
+ },
39948
+ "gpt-oss-120b": {
39949
+ "id": "gpt-oss-120b",
39950
+ "name": "GPT OSS 120B",
39951
+ "api": "openai-completions",
39952
+ "provider": "mara",
39953
+ "baseUrl": "https://api.cloud.mara.com/v1",
39954
+ "reasoning": true,
39955
+ "input": [
39956
+ "text"
39957
+ ],
39958
+ "cost": {
39959
+ "input": 0.15,
39960
+ "output": 0.75,
39961
+ "cacheRead": 0,
39962
+ "cacheWrite": 0
39963
+ },
39964
+ "contextWindow": 131072,
39965
+ "maxTokens": 16384,
39966
+ "thinking": {
39967
+ "mode": "effort",
39968
+ "minLevel": "minimal",
39969
+ "maxLevel": "xhigh"
39970
+ }
39971
+ }
39972
+ },
39885
39973
  "minimax": {
39886
39974
  "MiniMax-M2": {
39887
39975
  "id": "MiniMax-M2",
@@ -85423,4 +85511,4 @@
85423
85511
  }
85424
85512
  }
85425
85513
  }
85426
- }
85514
+ }
@@ -26,6 +26,7 @@ import {
26
26
  kimiCodeModelManagerOptions,
27
27
  litellmModelManagerOptions,
28
28
  lmStudioModelManagerOptions,
29
+ maraModelManagerOptions,
29
30
  mistralModelManagerOptions,
30
31
  moonshotModelManagerOptions,
31
32
  nanoGptModelManagerOptions,
@@ -326,6 +327,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
326
327
  config => bizrouterModelManagerOptions(config),
327
328
  catalog("BizRouter", ["BIZROUTER_API_KEY"]),
328
329
  ),
330
+ catalogDescriptor(
331
+ "mara",
332
+ "DeepSeek-V3.1",
333
+ config => maraModelManagerOptions(config),
334
+ catalog("Mara Cloud", ["MARA_API_KEY"]),
335
+ ),
329
336
  catalogDescriptor("zai", "glm-5.2", config => zaiModelManagerOptions(config), catalog("zAI", ["ZAI_API_KEY"])),
330
337
  catalogDescriptor(
331
338
  "glm-zcode",
@@ -1180,6 +1180,22 @@ export function bizrouterModelManagerOptions(
1180
1180
  };
1181
1181
  }
1182
1182
 
1183
+ // ---------------------------------------------------------------------------
1184
+ // 10.5.3 Mara Cloud
1185
+ // ---------------------------------------------------------------------------
1186
+
1187
+ export interface MaraModelManagerConfig {
1188
+ apiKey?: string;
1189
+ baseUrl?: string;
1190
+ }
1191
+
1192
+ /**
1193
+ * Mara Cloud — an OpenAI-compatible enterprise AI inference platform. Models
1194
+ * are discovered from the OpenAI-compatible `/v1/models` endpoint.
1195
+ */
1196
+ export function maraModelManagerOptions(config?: MaraModelManagerConfig): ModelManagerOptions<"openai-completions"> {
1197
+ return createSimpleOpenAICompletionsOptions("mara", "https://api.cloud.mara.com/v1", config);
1198
+ }
1183
1199
  // ---------------------------------------------------------------------------
1184
1200
  // 10.6 Kilo Gateway
1185
1201
  // ---------------------------------------------------------------------------
package/src/stream.ts CHANGED
@@ -164,6 +164,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
164
164
  zenmux: "ZENMUX_API_KEY",
165
165
  opengateway: "OPENGATEWAY_API_KEY",
166
166
  bizrouter: "BIZROUTER_API_KEY",
167
+ mara: "MARA_API_KEY",
167
168
  venice: "VENICE_API_KEY",
168
169
  vllm: "VLLM_API_KEY",
169
170
  xiaomi: "XIAOMI_API_KEY",
package/src/types.ts CHANGED
@@ -149,6 +149,7 @@ export type KnownProvider =
149
149
  | "opencode-zen"
150
150
  | "opengateway"
151
151
  | "bizrouter"
152
+ | "mara"
152
153
  | "synthetic"
153
154
  | "cloudflare-ai-gateway"
154
155
  | "huggingface"
@@ -245,6 +245,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
245
245
  name: "BizRouter",
246
246
  available: true,
247
247
  },
248
+ {
249
+ id: "mara",
250
+ name: "Mara Cloud",
251
+ available: true,
252
+ },
248
253
  {
249
254
  id: "opengateway",
250
255
  name: "OpenGateway by Sionic AI",
@@ -393,6 +398,7 @@ export async function refreshOAuthToken(
393
398
  case "moonshot":
394
399
  case "kagi":
395
400
  case "cloudflare-ai-gateway":
401
+ case "mara":
396
402
  case "vercel-ai-gateway":
397
403
  case "qwen-portal":
398
404
  case "zenmux":
@@ -0,0 +1,16 @@
1
+ /** Mara Cloud login flow (API key paste, validated via chat completions). */
2
+ import { createApiKeyLogin } from "./api-key-login";
3
+
4
+ export const loginMara = createApiKeyLogin({
5
+ providerLabel: "Mara Cloud",
6
+ authUrl: "https://cloud.mara.com/apis",
7
+ instructions: "Create or copy your Mara Cloud API key",
8
+ promptMessage: "Paste your Mara Cloud API key",
9
+ placeholder: "<your-mara-api-key>",
10
+ validation: {
11
+ kind: "chat-completions",
12
+ provider: "Mara Cloud",
13
+ baseUrl: "https://api.cloud.mara.com/v1",
14
+ model: "DeepSeek-V3.1",
15
+ },
16
+ });
@@ -12,6 +12,7 @@ export type OAuthProvider =
12
12
  | "alibaba-token-plan"
13
13
  | "anthropic"
14
14
  | "bizrouter"
15
+ | "mara"
15
16
  | "cerebras"
16
17
  | "cloudflare-ai-gateway"
17
18
  | "cursor"