@gajae-code/ai 0.12.2 → 0.12.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,28 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.12.5] - 2026-07-30
6
+ ### Fixed
7
+
8
+ - Alibaba Token Plan requests now carry Qwen Code's canonical DashScope request fingerprint on both transports. The built-in `alibaba-token-plan` provider (openai-responses `qwen3.8-max-preview` and openai-completions `glm-5.2`/`deepseek-v4-pro`) now emits the four upstream identity/cache/auth headers (`User-Agent`, `X-DashScope-CacheControl: enable`, `X-DashScope-UserAgent`, `X-DashScope-AuthType: openai`) matching `QwenLM/qwen-code` v0.21.1 (commit `f4cd6e1`) exactly, via a shared helper. DashScope is compatibility-sensitive to this client fingerprint, so a non-identical set can cause request instability and affect first-event latency. Caller headers still win per key (upstream `{...default, ...customHeaders}` precedence); non-Alibaba providers are byte-unchanged (#3557).
9
+
10
+ ### Added
11
+
12
+ - Reproducible Alibaba Token Plan header-parity A/B latency benchmark (`packages/ai/scripts/alibaba-token-plan-latency-ab.ts`): a fixed-seed interleaved A/B comparison of legacy vs Qwen-identical headers against a deterministic local HTTP server, reporting n/success/error/timeout and TTFT/total latency median/p90/p95/mean/stddev. No live credentials are required; a public-safe blocked-live-data receipt is included (`packages/ai/test/fixtures/alibaba-token-plan-latency-blocked-receipt.md`) (#3557).
13
+
14
+
15
+ ## [0.12.4] - 2026-07-30
16
+
17
+ ### Fixed
18
+
19
+ - Mara Cloud login now validates pasted credentials against the authenticated chat-completions endpoint instead of the public `/v1/models` catalog. The catalog returns `200` even for random invalid bearer tokens, so the previous check could persist unusable keys.
20
+
21
+ ## [0.12.3] - 2026-07-30
22
+
23
+ ### Added
24
+
25
+ - Added first-class support for **Mara Cloud**, an OpenAI-compatible enterprise AI inference platform. Registers the `mara` provider descriptor, `/login` entry (API-key paste validated against `https://api.cloud.mara.com/v1/models`), `MARA_API_KEY` environment resolution, and bundled `models.json` seed models. Models are discovered dynamically from `GET /v1/models` (base URL `https://api.cloud.mara.com/v1`).
26
+
5
27
  ## [0.12.2] - 2026-07-30
6
28
 
7
29
  ## [0.12.1] - 2026-07-29
package/README.md CHANGED
@@ -71,6 +71,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
71
71
  - **ZenMux** (requires `ZENMUX_API_KEY`)
72
72
  - **OpenGateway by Sionic AI** (requires `OPENGATEWAY_API_KEY`)
73
73
  - **BizRouter** (requires `BIZROUTER_API_KEY`)
74
+ - **Mara Cloud** (requires `MARA_API_KEY`)
74
75
  - **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
75
76
  - **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
76
77
  - **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
@@ -958,6 +959,7 @@ In Node.js environments, you can set environment variables to avoid passing API
958
959
  | ZenMux | `ZENMUX_API_KEY` |
959
960
  | OpenGateway | `OPENGATEWAY_API_KEY` |
960
961
  | BizRouter | `BIZROUTER_API_KEY` |
962
+ | Mara Cloud | `MARA_API_KEY` |
961
963
  | vLLM | `VLLM_API_KEY` |
962
964
  | Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
963
965
  | GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
@@ -983,6 +985,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
983
985
  - ZenMux (Anthropic models): `https://zenmux.ai/api/anthropic`
984
986
  - OpenGateway by Sionic AI: `https://apis.opengateway.ai/v1`
985
987
  - BizRouter: `https://api.bizrouter.ai/v1`
988
+ - Mara Cloud: `https://api.cloud.mara.com/v1`
986
989
  - vLLM: `http://127.0.0.1:8000/v1`
987
990
  - Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
988
991
  - Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
@@ -128,6 +128,15 @@ export interface BizRouterModelManagerConfig {
128
128
  baseUrl?: string;
129
129
  }
130
130
  export declare function bizrouterModelManagerOptions(config?: BizRouterModelManagerConfig): ModelManagerOptions<"openai-completions">;
131
+ export interface MaraModelManagerConfig {
132
+ apiKey?: string;
133
+ baseUrl?: string;
134
+ }
135
+ /**
136
+ * Mara Cloud — an OpenAI-compatible enterprise AI inference platform. Models
137
+ * are discovered from the OpenAI-compatible `/v1/models` endpoint.
138
+ */
139
+ export declare function maraModelManagerOptions(config?: MaraModelManagerConfig): ModelManagerOptions<"openai-completions">;
131
140
  export interface KiloModelManagerConfig {
132
141
  apiKey?: string;
133
142
  baseUrl?: string;
@@ -0,0 +1,57 @@
1
+ /**
2
+ * DashScope Token Plan canonical request headers.
3
+ *
4
+ * Reproduces QwenLM/qwen-code's DashScopeOpenAICompatibleProvider.buildHeaders()
5
+ * defaultHeaders so the built-in `alibaba-token-plan` provider emits the same
6
+ * client identity / cache / auth-type fingerprint upstream sends. DashScope is
7
+ * compatibility-sensitive to this fingerprint; a non-identical set can cause
8
+ * request instability and affect first-event latency (gajae-code #3557).
9
+ *
10
+ * Upstream pin (reproduce EXACTLY here):
11
+ * Repository: QwenLM/qwen-code
12
+ * Commit: f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb
13
+ * Version: 0.21.1
14
+ * Source: packages/core/src/core/openaiContentGenerator/provider/dashscope.ts
15
+ * buildHeaders():
16
+ * const userAgent = `QwenCode/${version} (${process.platform}; ${process.arch})`;
17
+ * const defaultHeaders = {
18
+ * 'User-Agent': userAgent,
19
+ * 'X-DashScope-CacheControl': 'enable',
20
+ * 'X-DashScope-UserAgent': userAgent,
21
+ * 'X-DashScope-AuthType': authType,
22
+ * };
23
+ * return customHeaders ? { ...defaultHeaders, ...customHeaders } : defaultHeaders;
24
+ *
25
+ * The Token Plan preset authenticates with AuthType.USE_OPENAI ('openai'), so
26
+ * X-DashScope-AuthType is the constant 'openai'. Pin the version so an upstream
27
+ * bump is an explicit parity update rather than silent drift.
28
+ */
29
+ export declare const QWEN_CODE_UPSTREAM_REPO = "QwenLM/qwen-code";
30
+ export declare const QWEN_CODE_UPSTREAM_COMMIT = "f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb";
31
+ export declare const QWEN_CODE_UPSTREAM_VERSION = "0.21.1";
32
+ /**
33
+ * The Qwen Code CLI version string used in identity headers. Pinned to the
34
+ * upstream version at {@link QWEN_CODE_UPSTREAM_COMMIT}; change both together
35
+ * as an explicit parity update.
36
+ */
37
+ export declare function qwenCodeUserAgent(version?: string): string;
38
+ /**
39
+ * Canonical DashScope Token Plan headers (upstream defaultHeaders, no caller
40
+ * overrides applied). Exposed for tests/fixtures so the pinned wire set lives
41
+ * in exactly one place.
42
+ */
43
+ export declare function dashscopeTokenPlanDefaultHeaders(version?: string): Readonly<Record<string, string>>;
44
+ /**
45
+ * Merge canonical DashScope Token Plan identity headers onto a caller's header
46
+ * map, reproducing upstream buildHeaders() precedence EXACTLY:
47
+ * `{ ...defaultHeaders, ...customHeaders }` — caller wins per header.
48
+ *
49
+ * This mirrors GJC's existing kimi-code injection order
50
+ * (`headers = { ...getKimiCommonHeaders(), ...headers }`): canonical identity as
51
+ * the base, caller-supplied headers overriding individual keys. A caller that
52
+ * pins `User-Agent` takes that key; the other canonicals still apply.
53
+ *
54
+ * A null/undefined `callerHeaders` returns the canonical set alone (upstream
55
+ * `customHeaders ? {...} : defaultHeaders` shortcut).
56
+ */
57
+ export declare function mergeDashScopeTokenPlanHeaders(callerHeaders: Record<string, string> | undefined, version?: string): Record<string, string>;
@@ -51,7 +51,7 @@ export interface ThinkingConfig {
51
51
  /** Provider-specific transport used to encode the selected effort. */
52
52
  mode: ThinkingControlMode;
53
53
  }
54
- export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
54
+ export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "mara" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
55
55
  export type Provider = KnownProvider | string;
56
56
  import type { Effort } from "./model-thinking";
57
57
  /** Token budgets for each thinking level (token-based providers only) */
@@ -0,0 +1 @@
1
+ export declare const loginMara: (options: import("./types").OAuthController) => Promise<string>;
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "mara" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.12.2",
4
+ "version": "0.12.5",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.12.2",
43
+ "@gajae-code/utils": "0.12.5",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -2042,6 +2042,12 @@ export class AuthStorage {
2042
2042
  await saveApiKeyCredential(apiKey);
2043
2043
  return;
2044
2044
  }
2045
+ case "mara": {
2046
+ const { loginMara } = await import("./utils/oauth/mara");
2047
+ const apiKey = await loginMara(ctrl);
2048
+ await saveApiKeyCredential(apiKey);
2049
+ return;
2050
+ }
2045
2051
  case "opengateway": {
2046
2052
  const { loginOpenGateway } = await import("./utils/oauth/opengateway");
2047
2053
  const apiKey = await loginOpenGateway(ctrl);
package/src/cli.ts CHANGED
@@ -120,6 +120,7 @@ Providers:
120
120
  zenmux ZenMux
121
121
  opengateway OpenGateway by Sionic AI
122
122
  bizrouter BizRouter
123
+ mara Mara Cloud
123
124
  ollama-cloud Ollama Cloud
124
125
 
125
126
  Examples:
package/src/models.json CHANGED
@@ -39882,6 +39882,94 @@
39882
39882
  "maxTokens": 8888
39883
39883
  }
39884
39884
  },
39885
+ "mara": {
39886
+ "DeepSeek-V3.1": {
39887
+ "id": "DeepSeek-V3.1",
39888
+ "name": "DeepSeek V3.1",
39889
+ "api": "openai-completions",
39890
+ "provider": "mara",
39891
+ "baseUrl": "https://api.cloud.mara.com/v1",
39892
+ "reasoning": true,
39893
+ "input": [
39894
+ "text"
39895
+ ],
39896
+ "cost": {
39897
+ "input": 0.6,
39898
+ "output": 1.7,
39899
+ "cacheRead": 0,
39900
+ "cacheWrite": 0
39901
+ },
39902
+ "contextWindow": 131072,
39903
+ "maxTokens": 16384,
39904
+ "thinking": {
39905
+ "mode": "effort",
39906
+ "minLevel": "minimal",
39907
+ "maxLevel": "xhigh"
39908
+ }
39909
+ },
39910
+ "MiniMax-M2.5": {
39911
+ "id": "MiniMax-M2.5",
39912
+ "name": "MiniMax M2.5",
39913
+ "api": "openai-completions",
39914
+ "provider": "mara",
39915
+ "baseUrl": "https://api.cloud.mara.com/v1",
39916
+ "reasoning": false,
39917
+ "input": [
39918
+ "text"
39919
+ ],
39920
+ "cost": {
39921
+ "input": 0.3,
39922
+ "output": 1.2,
39923
+ "cacheRead": 0,
39924
+ "cacheWrite": 0
39925
+ },
39926
+ "contextWindow": 196608,
39927
+ "maxTokens": 16384
39928
+ },
39929
+ "MiniMax-M2.7": {
39930
+ "id": "MiniMax-M2.7",
39931
+ "name": "MiniMax M2.7",
39932
+ "api": "openai-completions",
39933
+ "provider": "mara",
39934
+ "baseUrl": "https://api.cloud.mara.com/v1",
39935
+ "reasoning": false,
39936
+ "input": [
39937
+ "text"
39938
+ ],
39939
+ "cost": {
39940
+ "input": 0.3,
39941
+ "output": 1.2,
39942
+ "cacheRead": 0,
39943
+ "cacheWrite": 0
39944
+ },
39945
+ "contextWindow": 196608,
39946
+ "maxTokens": 16384
39947
+ },
39948
+ "gpt-oss-120b": {
39949
+ "id": "gpt-oss-120b",
39950
+ "name": "GPT OSS 120B",
39951
+ "api": "openai-completions",
39952
+ "provider": "mara",
39953
+ "baseUrl": "https://api.cloud.mara.com/v1",
39954
+ "reasoning": true,
39955
+ "input": [
39956
+ "text"
39957
+ ],
39958
+ "cost": {
39959
+ "input": 0.15,
39960
+ "output": 0.75,
39961
+ "cacheRead": 0,
39962
+ "cacheWrite": 0
39963
+ },
39964
+ "contextWindow": 131072,
39965
+ "maxTokens": 16384,
39966
+ "thinking": {
39967
+ "mode": "effort",
39968
+ "minLevel": "minimal",
39969
+ "maxLevel": "xhigh"
39970
+ }
39971
+ }
39972
+ },
39885
39973
  "minimax": {
39886
39974
  "MiniMax-M2": {
39887
39975
  "id": "MiniMax-M2",
@@ -85423,4 +85511,4 @@
85423
85511
  }
85424
85512
  }
85425
85513
  }
85426
- }
85514
+ }
@@ -26,6 +26,7 @@ import {
26
26
  kimiCodeModelManagerOptions,
27
27
  litellmModelManagerOptions,
28
28
  lmStudioModelManagerOptions,
29
+ maraModelManagerOptions,
29
30
  mistralModelManagerOptions,
30
31
  moonshotModelManagerOptions,
31
32
  nanoGptModelManagerOptions,
@@ -326,6 +327,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
326
327
  config => bizrouterModelManagerOptions(config),
327
328
  catalog("BizRouter", ["BIZROUTER_API_KEY"]),
328
329
  ),
330
+ catalogDescriptor(
331
+ "mara",
332
+ "DeepSeek-V3.1",
333
+ config => maraModelManagerOptions(config),
334
+ catalog("Mara Cloud", ["MARA_API_KEY"]),
335
+ ),
329
336
  catalogDescriptor("zai", "glm-5.2", config => zaiModelManagerOptions(config), catalog("zAI", ["ZAI_API_KEY"])),
330
337
  catalogDescriptor(
331
338
  "glm-zcode",
@@ -1180,6 +1180,22 @@ export function bizrouterModelManagerOptions(
1180
1180
  };
1181
1181
  }
1182
1182
 
1183
+ // ---------------------------------------------------------------------------
1184
+ // 10.5.3 Mara Cloud
1185
+ // ---------------------------------------------------------------------------
1186
+
1187
+ export interface MaraModelManagerConfig {
1188
+ apiKey?: string;
1189
+ baseUrl?: string;
1190
+ }
1191
+
1192
+ /**
1193
+ * Mara Cloud — an OpenAI-compatible enterprise AI inference platform. Models
1194
+ * are discovered from the OpenAI-compatible `/v1/models` endpoint.
1195
+ */
1196
+ export function maraModelManagerOptions(config?: MaraModelManagerConfig): ModelManagerOptions<"openai-completions"> {
1197
+ return createSimpleOpenAICompletionsOptions("mara", "https://api.cloud.mara.com/v1", config);
1198
+ }
1183
1199
  // ---------------------------------------------------------------------------
1184
1200
  // 10.6 Kilo Gateway
1185
1201
  // ---------------------------------------------------------------------------
@@ -0,0 +1,84 @@
1
+ /**
2
+ * DashScope Token Plan canonical request headers.
3
+ *
4
+ * Reproduces QwenLM/qwen-code's DashScopeOpenAICompatibleProvider.buildHeaders()
5
+ * defaultHeaders so the built-in `alibaba-token-plan` provider emits the same
6
+ * client identity / cache / auth-type fingerprint upstream sends. DashScope is
7
+ * compatibility-sensitive to this fingerprint; a non-identical set can cause
8
+ * request instability and affect first-event latency (gajae-code #3557).
9
+ *
10
+ * Upstream pin (reproduce EXACTLY here):
11
+ * Repository: QwenLM/qwen-code
12
+ * Commit: f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb
13
+ * Version: 0.21.1
14
+ * Source: packages/core/src/core/openaiContentGenerator/provider/dashscope.ts
15
+ * buildHeaders():
16
+ * const userAgent = `QwenCode/${version} (${process.platform}; ${process.arch})`;
17
+ * const defaultHeaders = {
18
+ * 'User-Agent': userAgent,
19
+ * 'X-DashScope-CacheControl': 'enable',
20
+ * 'X-DashScope-UserAgent': userAgent,
21
+ * 'X-DashScope-AuthType': authType,
22
+ * };
23
+ * return customHeaders ? { ...defaultHeaders, ...customHeaders } : defaultHeaders;
24
+ *
25
+ * The Token Plan preset authenticates with AuthType.USE_OPENAI ('openai'), so
26
+ * X-DashScope-AuthType is the constant 'openai'. Pin the version so an upstream
27
+ * bump is an explicit parity update rather than silent drift.
28
+ */
29
+ export const QWEN_CODE_UPSTREAM_REPO = "QwenLM/qwen-code";
30
+ export const QWEN_CODE_UPSTREAM_COMMIT = "f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb";
31
+ export const QWEN_CODE_UPSTREAM_VERSION = "0.21.1";
32
+
33
+ // Upstream Token Plan preset uses AuthType.USE_OPENAI = 'openai'.
34
+ const QWEN_CODE_TOKEN_PLAN_AUTH_TYPE = "openai";
35
+
36
+ /**
37
+ * The Qwen Code CLI version string used in identity headers. Pinned to the
38
+ * upstream version at {@link QWEN_CODE_UPSTREAM_COMMIT}; change both together
39
+ * as an explicit parity update.
40
+ */
41
+ export function qwenCodeUserAgent(version: string = QWEN_CODE_UPSTREAM_VERSION): string {
42
+ // process.platform / process.arch are read verbatim, matching upstream
43
+ // (e.g. "linux", "darwin", "win32"; "x64", "arm64"). No normalization.
44
+ return `QwenCode/${version} (${process.platform}; ${process.arch})`;
45
+ }
46
+
47
+ /**
48
+ * Canonical DashScope Token Plan headers (upstream defaultHeaders, no caller
49
+ * overrides applied). Exposed for tests/fixtures so the pinned wire set lives
50
+ * in exactly one place.
51
+ */
52
+ export function dashscopeTokenPlanDefaultHeaders(
53
+ version: string = QWEN_CODE_UPSTREAM_VERSION,
54
+ ): Readonly<Record<string, string>> {
55
+ const userAgent = qwenCodeUserAgent(version);
56
+ return Object.freeze({
57
+ "User-Agent": userAgent,
58
+ "X-DashScope-CacheControl": "enable",
59
+ "X-DashScope-UserAgent": userAgent,
60
+ "X-DashScope-AuthType": QWEN_CODE_TOKEN_PLAN_AUTH_TYPE,
61
+ });
62
+ }
63
+
64
+ /**
65
+ * Merge canonical DashScope Token Plan identity headers onto a caller's header
66
+ * map, reproducing upstream buildHeaders() precedence EXACTLY:
67
+ * `{ ...defaultHeaders, ...customHeaders }` — caller wins per header.
68
+ *
69
+ * This mirrors GJC's existing kimi-code injection order
70
+ * (`headers = { ...getKimiCommonHeaders(), ...headers }`): canonical identity as
71
+ * the base, caller-supplied headers overriding individual keys. A caller that
72
+ * pins `User-Agent` takes that key; the other canonicals still apply.
73
+ *
74
+ * A null/undefined `callerHeaders` returns the canonical set alone (upstream
75
+ * `customHeaders ? {...} : defaultHeaders` shortcut).
76
+ */
77
+ export function mergeDashScopeTokenPlanHeaders(
78
+ callerHeaders: Record<string, string> | undefined,
79
+ version: string = QWEN_CODE_UPSTREAM_VERSION,
80
+ ): Record<string, string> {
81
+ const defaults = dashscopeTokenPlanDefaultHeaders(version);
82
+ if (!callerHeaders) return { ...defaults };
83
+ return { ...defaults, ...callerHeaders };
84
+ }
@@ -43,6 +43,7 @@ import {
43
43
  createOpenAIResponsesHistoryPayload,
44
44
  getOpenAIResponsesHistoryItems,
45
45
  getOpenAIResponsesHistoryPayload,
46
+ neutralizeReservedControlTokens,
46
47
  neutralizeResponsesInputControlTokens,
47
48
  normalizeSystemPrompts,
48
49
  sanitizeOpenAIResponsesHistoryItemsForReplay,
@@ -746,7 +747,12 @@ async function buildTransformedCodexRequestBody(
746
747
  }
747
748
  }
748
749
 
749
- const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
750
+ // Neutralize leaked Harmony control tokens in the system prompt too:
751
+ // `params.instructions` and the developer messages prepended inside
752
+ // `transformRequestBody` bypass the `input` sanitizer above, so a poisoned
753
+ // system prompt rejects every turn with
754
+ // `Request blocked (code=invalid_prompt)`.
755
+ const systemPrompts = normalizeSystemPrompts(context.systemPrompt).map(neutralizeReservedControlTokens);
750
756
  if (systemPrompts.length > 0) {
751
757
  params.instructions = systemPrompts[0];
752
758
  }
@@ -68,6 +68,7 @@ import {
68
68
  resolveToolChoice,
69
69
  } from "../utils/tool-choice-capability";
70
70
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
71
+ import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
71
72
  import {
72
73
  buildCopilotDynamicHeaders,
73
74
  hasCopilotVisionInput,
@@ -1071,6 +1072,14 @@ async function createClient(
1071
1072
  if (model.provider === "kimi-code") {
1072
1073
  headers = { ...getKimiCommonHeaders(), ...headers };
1073
1074
  }
1075
+ if (model.provider === "alibaba-token-plan") {
1076
+ // Emit Qwen Code's canonical DashScope request fingerprint (User-Agent /
1077
+ // X-DashScope-CacheControl / X-DashScope-UserAgent / X-DashScope-AuthType)
1078
+ // so DashScope treats the caller identically to upstream QwenLM/qwen-code.
1079
+ // Canonical identity is the base; caller headers win per key (upstream
1080
+ // `{...default, ...customHeaders}`). #3557.
1081
+ headers = mergeDashScopeTokenPlanHeaders(headers);
1082
+ }
1074
1083
  headers = applyOpenAIRequestTransformHeaders(headers, model.requestTransform, `Gajae-Code/${packageJson.version}`);
1075
1084
  let copilotPremiumRequests: number | undefined;
1076
1085
 
@@ -27,6 +27,7 @@ import {
27
27
  getOpenAIResponsesHistoryItems,
28
28
  getOpenAIResponsesHistoryPayload,
29
29
  isInvalidPromptError,
30
+ neutralizeReservedControlTokens,
30
31
  neutralizeResponsesInputControlTokens,
31
32
  normalizeSystemPrompts,
32
33
  resolveCacheRetention,
@@ -60,6 +61,7 @@ import {
60
61
  resolveToolChoice,
61
62
  } from "../utils/tool-choice-capability";
62
63
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
64
+ import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
63
65
  import {
64
66
  buildCopilotDynamicHeaders,
65
67
  hasCopilotVisionInput,
@@ -446,8 +448,17 @@ function createClient(
446
448
  }
447
449
  const rawApiKey = apiKey;
448
450
 
451
+ const baseHeaders =
452
+ model.provider === "alibaba-token-plan"
453
+ ? // Emit Qwen Code's canonical DashScope request fingerprint (User-Agent /
454
+ // X-DashScope-CacheControl / X-DashScope-UserAgent / X-DashScope-AuthType)
455
+ // so DashScope treats the caller identically to upstream QwenLM/qwen-code.
456
+ // Canonical identity is the base; caller headers win per key (upstream
457
+ // `{...default, ...customHeaders}`). #3557.
458
+ mergeDashScopeTokenPlanHeaders({ ...(model.headers ?? {}), ...(extraHeaders ?? {}) })
459
+ : { ...(model.headers ?? {}), ...(extraHeaders ?? {}) };
449
460
  const headers = applyOpenAIRequestTransformHeaders(
450
- { ...(model.headers ?? {}), ...(extraHeaders ?? {}) },
461
+ baseHeaders,
451
462
  model.requestTransform,
452
463
  `Gajae-Code/${packageJson.version}`,
453
464
  );
@@ -523,7 +534,13 @@ function buildParams(
523
534
  );
524
535
  const messages: ResponseInput = neutralizeResponsesInputControlTokens(conversationMessages);
525
536
 
526
- const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
537
+ // Neutralize leaked Harmony control tokens in the system prompt too: the
538
+ // `instructions` field and developer-role messages bypass the `input`
539
+ // request-boundary sanitizer above, and a poisoned system prompt (e.g.
540
+ // injected project context quoting `<|channel|>` markers) rejects EVERY
541
+ // turn with `Request blocked (code=invalid_prompt)` — unrepairable by the
542
+ // history circuit breaker.
543
+ const systemPrompts = normalizeSystemPrompts(context.systemPrompt).map(neutralizeReservedControlTokens);
527
544
  if (isComposerHarnessModel(model.id)) {
528
545
  systemPrompts.unshift(COMPOSER_EDIT_DISCIPLINE_PROMPT);
529
546
  }
package/src/stream.ts CHANGED
@@ -164,6 +164,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
164
164
  zenmux: "ZENMUX_API_KEY",
165
165
  opengateway: "OPENGATEWAY_API_KEY",
166
166
  bizrouter: "BIZROUTER_API_KEY",
167
+ mara: "MARA_API_KEY",
167
168
  venice: "VENICE_API_KEY",
168
169
  vllm: "VLLM_API_KEY",
169
170
  xiaomi: "XIAOMI_API_KEY",
package/src/types.ts CHANGED
@@ -149,6 +149,7 @@ export type KnownProvider =
149
149
  | "opencode-zen"
150
150
  | "opengateway"
151
151
  | "bizrouter"
152
+ | "mara"
152
153
  | "synthetic"
153
154
  | "cloudflare-ai-gateway"
154
155
  | "huggingface"
@@ -245,6 +245,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
245
245
  name: "BizRouter",
246
246
  available: true,
247
247
  },
248
+ {
249
+ id: "mara",
250
+ name: "Mara Cloud",
251
+ available: true,
252
+ },
248
253
  {
249
254
  id: "opengateway",
250
255
  name: "OpenGateway by Sionic AI",
@@ -393,6 +398,7 @@ export async function refreshOAuthToken(
393
398
  case "moonshot":
394
399
  case "kagi":
395
400
  case "cloudflare-ai-gateway":
401
+ case "mara":
396
402
  case "vercel-ai-gateway":
397
403
  case "qwen-portal":
398
404
  case "zenmux":
@@ -0,0 +1,16 @@
1
+ /** Mara Cloud login flow (API key paste, validated via chat completions). */
2
+ import { createApiKeyLogin } from "./api-key-login";
3
+
4
+ export const loginMara = createApiKeyLogin({
5
+ providerLabel: "Mara Cloud",
6
+ authUrl: "https://cloud.mara.com/apis",
7
+ instructions: "Create or copy your Mara Cloud API key",
8
+ promptMessage: "Paste your Mara Cloud API key",
9
+ placeholder: "<your-mara-api-key>",
10
+ validation: {
11
+ kind: "chat-completions",
12
+ provider: "Mara Cloud",
13
+ baseUrl: "https://api.cloud.mara.com/v1",
14
+ model: "DeepSeek-V3.1",
15
+ },
16
+ });
@@ -12,6 +12,7 @@ export type OAuthProvider =
12
12
  | "alibaba-token-plan"
13
13
  | "anthropic"
14
14
  | "bizrouter"
15
+ | "mara"
15
16
  | "cerebras"
16
17
  | "cloudflare-ai-gateway"
17
18
  | "cursor"