@gajae-code/ai 0.12.2 → 0.12.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/README.md +3 -0
- package/dist/types/provider-models/openai-compat.d.ts +9 -0
- package/dist/types/providers/dashscope-token-plan-headers.d.ts +57 -0
- package/dist/types/types.d.ts +1 -1
- package/dist/types/utils/oauth/mara.d.ts +1 -0
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +2 -2
- package/src/auth-storage.ts +6 -0
- package/src/cli.ts +1 -0
- package/src/models.json +89 -1
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +16 -0
- package/src/providers/dashscope-token-plan-headers.ts +84 -0
- package/src/providers/openai-codex-responses.ts +7 -1
- package/src/providers/openai-completions.ts +9 -0
- package/src/providers/openai-responses.ts +19 -2
- package/src/stream.ts +1 -0
- package/src/types.ts +1 -0
- package/src/utils/oauth/index.ts +6 -0
- package/src/utils/oauth/mara.ts +16 -0
- package/src/utils/oauth/types.ts +1 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,28 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.12.5] - 2026-07-30
|
|
6
|
+
### Fixed
|
|
7
|
+
|
|
8
|
+
- Alibaba Token Plan requests now carry Qwen Code's canonical DashScope request fingerprint on both transports. The built-in `alibaba-token-plan` provider (openai-responses `qwen3.8-max-preview` and openai-completions `glm-5.2`/`deepseek-v4-pro`) now emits the four upstream identity/cache/auth headers (`User-Agent`, `X-DashScope-CacheControl: enable`, `X-DashScope-UserAgent`, `X-DashScope-AuthType: openai`) matching `QwenLM/qwen-code` v0.21.1 (commit `f4cd6e1`) exactly, via a shared helper. DashScope is compatibility-sensitive to this client fingerprint, so a non-identical set can cause request instability and affect first-event latency. Caller headers still win per key (upstream `{...default, ...customHeaders}` precedence); non-Alibaba providers are byte-unchanged (#3557).
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- Reproducible Alibaba Token Plan header-parity A/B latency benchmark (`packages/ai/scripts/alibaba-token-plan-latency-ab.ts`): a fixed-seed interleaved A/B comparison of legacy vs Qwen-identical headers against a deterministic local HTTP server, reporting n/success/error/timeout and TTFT/total latency median/p90/p95/mean/stddev. No live credentials are required; a public-safe blocked-live-data receipt is included (`packages/ai/test/fixtures/alibaba-token-plan-latency-blocked-receipt.md`) (#3557).
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
## [0.12.4] - 2026-07-30
|
|
16
|
+
|
|
17
|
+
### Fixed
|
|
18
|
+
|
|
19
|
+
- Mara Cloud login now validates pasted credentials against the authenticated chat-completions endpoint instead of the public `/v1/models` catalog. The catalog returns `200` even for random invalid bearer tokens, so the previous check could persist unusable keys.
|
|
20
|
+
|
|
21
|
+
## [0.12.3] - 2026-07-30
|
|
22
|
+
|
|
23
|
+
### Added
|
|
24
|
+
|
|
25
|
+
- Added first-class support for **Mara Cloud**, an OpenAI-compatible enterprise AI inference platform. Registers the `mara` provider descriptor, `/login` entry (API-key paste validated against `https://api.cloud.mara.com/v1/models`), `MARA_API_KEY` environment resolution, and bundled `models.json` seed models. Models are discovered dynamically from `GET /v1/models` (base URL `https://api.cloud.mara.com/v1`).
|
|
26
|
+
|
|
5
27
|
## [0.12.2] - 2026-07-30
|
|
6
28
|
|
|
7
29
|
## [0.12.1] - 2026-07-29
|
package/README.md
CHANGED
|
@@ -71,6 +71,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
|
|
71
71
|
- **ZenMux** (requires `ZENMUX_API_KEY`)
|
|
72
72
|
- **OpenGateway by Sionic AI** (requires `OPENGATEWAY_API_KEY`)
|
|
73
73
|
- **BizRouter** (requires `BIZROUTER_API_KEY`)
|
|
74
|
+
- **Mara Cloud** (requires `MARA_API_KEY`)
|
|
74
75
|
- **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
|
|
75
76
|
- **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
|
|
76
77
|
- **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
|
|
@@ -958,6 +959,7 @@ In Node.js environments, you can set environment variables to avoid passing API
|
|
|
958
959
|
| ZenMux | `ZENMUX_API_KEY` |
|
|
959
960
|
| OpenGateway | `OPENGATEWAY_API_KEY` |
|
|
960
961
|
| BizRouter | `BIZROUTER_API_KEY` |
|
|
962
|
+
| Mara Cloud | `MARA_API_KEY` |
|
|
961
963
|
| vLLM | `VLLM_API_KEY` |
|
|
962
964
|
| Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
|
|
963
965
|
| GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
|
|
@@ -983,6 +985,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
|
|
|
983
985
|
- ZenMux (Anthropic models): `https://zenmux.ai/api/anthropic`
|
|
984
986
|
- OpenGateway by Sionic AI: `https://apis.opengateway.ai/v1`
|
|
985
987
|
- BizRouter: `https://api.bizrouter.ai/v1`
|
|
988
|
+
- Mara Cloud: `https://api.cloud.mara.com/v1`
|
|
986
989
|
- vLLM: `http://127.0.0.1:8000/v1`
|
|
987
990
|
- Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
|
|
988
991
|
- Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
|
|
@@ -128,6 +128,15 @@ export interface BizRouterModelManagerConfig {
|
|
|
128
128
|
baseUrl?: string;
|
|
129
129
|
}
|
|
130
130
|
export declare function bizrouterModelManagerOptions(config?: BizRouterModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
131
|
+
export interface MaraModelManagerConfig {
|
|
132
|
+
apiKey?: string;
|
|
133
|
+
baseUrl?: string;
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Mara Cloud — an OpenAI-compatible enterprise AI inference platform. Models
|
|
137
|
+
* are discovered from the OpenAI-compatible `/v1/models` endpoint.
|
|
138
|
+
*/
|
|
139
|
+
export declare function maraModelManagerOptions(config?: MaraModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
131
140
|
export interface KiloModelManagerConfig {
|
|
132
141
|
apiKey?: string;
|
|
133
142
|
baseUrl?: string;
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DashScope Token Plan canonical request headers.
|
|
3
|
+
*
|
|
4
|
+
* Reproduces QwenLM/qwen-code's DashScopeOpenAICompatibleProvider.buildHeaders()
|
|
5
|
+
* defaultHeaders so the built-in `alibaba-token-plan` provider emits the same
|
|
6
|
+
* client identity / cache / auth-type fingerprint upstream sends. DashScope is
|
|
7
|
+
* compatibility-sensitive to this fingerprint; a non-identical set can cause
|
|
8
|
+
* request instability and affect first-event latency (gajae-code #3557).
|
|
9
|
+
*
|
|
10
|
+
* Upstream pin (reproduce EXACTLY here):
|
|
11
|
+
* Repository: QwenLM/qwen-code
|
|
12
|
+
* Commit: f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb
|
|
13
|
+
* Version: 0.21.1
|
|
14
|
+
* Source: packages/core/src/core/openaiContentGenerator/provider/dashscope.ts
|
|
15
|
+
* buildHeaders():
|
|
16
|
+
* const userAgent = `QwenCode/${version} (${process.platform}; ${process.arch})`;
|
|
17
|
+
* const defaultHeaders = {
|
|
18
|
+
* 'User-Agent': userAgent,
|
|
19
|
+
* 'X-DashScope-CacheControl': 'enable',
|
|
20
|
+
* 'X-DashScope-UserAgent': userAgent,
|
|
21
|
+
* 'X-DashScope-AuthType': authType,
|
|
22
|
+
* };
|
|
23
|
+
* return customHeaders ? { ...defaultHeaders, ...customHeaders } : defaultHeaders;
|
|
24
|
+
*
|
|
25
|
+
* The Token Plan preset authenticates with AuthType.USE_OPENAI ('openai'), so
|
|
26
|
+
* X-DashScope-AuthType is the constant 'openai'. Pin the version so an upstream
|
|
27
|
+
* bump is an explicit parity update rather than silent drift.
|
|
28
|
+
*/
|
|
29
|
+
export declare const QWEN_CODE_UPSTREAM_REPO = "QwenLM/qwen-code";
|
|
30
|
+
export declare const QWEN_CODE_UPSTREAM_COMMIT = "f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb";
|
|
31
|
+
export declare const QWEN_CODE_UPSTREAM_VERSION = "0.21.1";
|
|
32
|
+
/**
|
|
33
|
+
* The Qwen Code CLI version string used in identity headers. Pinned to the
|
|
34
|
+
* upstream version at {@link QWEN_CODE_UPSTREAM_COMMIT}; change both together
|
|
35
|
+
* as an explicit parity update.
|
|
36
|
+
*/
|
|
37
|
+
export declare function qwenCodeUserAgent(version?: string): string;
|
|
38
|
+
/**
|
|
39
|
+
* Canonical DashScope Token Plan headers (upstream defaultHeaders, no caller
|
|
40
|
+
* overrides applied). Exposed for tests/fixtures so the pinned wire set lives
|
|
41
|
+
* in exactly one place.
|
|
42
|
+
*/
|
|
43
|
+
export declare function dashscopeTokenPlanDefaultHeaders(version?: string): Readonly<Record<string, string>>;
|
|
44
|
+
/**
|
|
45
|
+
* Merge canonical DashScope Token Plan identity headers onto a caller's header
|
|
46
|
+
* map, reproducing upstream buildHeaders() precedence EXACTLY:
|
|
47
|
+
* `{ ...defaultHeaders, ...customHeaders }` — caller wins per header.
|
|
48
|
+
*
|
|
49
|
+
* This mirrors GJC's existing kimi-code injection order
|
|
50
|
+
* (`headers = { ...getKimiCommonHeaders(), ...headers }`): canonical identity as
|
|
51
|
+
* the base, caller-supplied headers overriding individual keys. A caller that
|
|
52
|
+
* pins `User-Agent` takes that key; the other canonicals still apply.
|
|
53
|
+
*
|
|
54
|
+
* A null/undefined `callerHeaders` returns the canonical set alone (upstream
|
|
55
|
+
* `customHeaders ? {...} : defaultHeaders` shortcut).
|
|
56
|
+
*/
|
|
57
|
+
export declare function mergeDashScopeTokenPlanHeaders(callerHeaders: Record<string, string> | undefined, version?: string): Record<string, string>;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -51,7 +51,7 @@ export interface ThinkingConfig {
|
|
|
51
51
|
/** Provider-specific transport used to encode the selected effort. */
|
|
52
52
|
mode: ThinkingControlMode;
|
|
53
53
|
}
|
|
54
|
-
export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
54
|
+
export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "mara" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
55
55
|
export type Provider = KnownProvider | string;
|
|
56
56
|
import type { Effort } from "./model-thinking";
|
|
57
57
|
/** Token budgets for each thinking level (token-based providers only) */
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export declare const loginMara: (options: import("./types").OAuthController) => Promise<string>;
|
|
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
|
|
|
7
7
|
email?: string;
|
|
8
8
|
accountId?: string;
|
|
9
9
|
};
|
|
10
|
-
export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
10
|
+
export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "mara" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
11
11
|
export type OAuthProviderId = OAuthProvider | (string & {});
|
|
12
12
|
export type OAuthPrompt = {
|
|
13
13
|
message: string;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.12.
|
|
4
|
+
"version": "0.12.5",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.12.
|
|
43
|
+
"@gajae-code/utils": "0.12.5",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/auth-storage.ts
CHANGED
|
@@ -2042,6 +2042,12 @@ export class AuthStorage {
|
|
|
2042
2042
|
await saveApiKeyCredential(apiKey);
|
|
2043
2043
|
return;
|
|
2044
2044
|
}
|
|
2045
|
+
case "mara": {
|
|
2046
|
+
const { loginMara } = await import("./utils/oauth/mara");
|
|
2047
|
+
const apiKey = await loginMara(ctrl);
|
|
2048
|
+
await saveApiKeyCredential(apiKey);
|
|
2049
|
+
return;
|
|
2050
|
+
}
|
|
2045
2051
|
case "opengateway": {
|
|
2046
2052
|
const { loginOpenGateway } = await import("./utils/oauth/opengateway");
|
|
2047
2053
|
const apiKey = await loginOpenGateway(ctrl);
|
package/src/cli.ts
CHANGED
package/src/models.json
CHANGED
|
@@ -39882,6 +39882,94 @@
|
|
|
39882
39882
|
"maxTokens": 8888
|
|
39883
39883
|
}
|
|
39884
39884
|
},
|
|
39885
|
+
"mara": {
|
|
39886
|
+
"DeepSeek-V3.1": {
|
|
39887
|
+
"id": "DeepSeek-V3.1",
|
|
39888
|
+
"name": "DeepSeek V3.1",
|
|
39889
|
+
"api": "openai-completions",
|
|
39890
|
+
"provider": "mara",
|
|
39891
|
+
"baseUrl": "https://api.cloud.mara.com/v1",
|
|
39892
|
+
"reasoning": true,
|
|
39893
|
+
"input": [
|
|
39894
|
+
"text"
|
|
39895
|
+
],
|
|
39896
|
+
"cost": {
|
|
39897
|
+
"input": 0.6,
|
|
39898
|
+
"output": 1.7,
|
|
39899
|
+
"cacheRead": 0,
|
|
39900
|
+
"cacheWrite": 0
|
|
39901
|
+
},
|
|
39902
|
+
"contextWindow": 131072,
|
|
39903
|
+
"maxTokens": 16384,
|
|
39904
|
+
"thinking": {
|
|
39905
|
+
"mode": "effort",
|
|
39906
|
+
"minLevel": "minimal",
|
|
39907
|
+
"maxLevel": "xhigh"
|
|
39908
|
+
}
|
|
39909
|
+
},
|
|
39910
|
+
"MiniMax-M2.5": {
|
|
39911
|
+
"id": "MiniMax-M2.5",
|
|
39912
|
+
"name": "MiniMax M2.5",
|
|
39913
|
+
"api": "openai-completions",
|
|
39914
|
+
"provider": "mara",
|
|
39915
|
+
"baseUrl": "https://api.cloud.mara.com/v1",
|
|
39916
|
+
"reasoning": false,
|
|
39917
|
+
"input": [
|
|
39918
|
+
"text"
|
|
39919
|
+
],
|
|
39920
|
+
"cost": {
|
|
39921
|
+
"input": 0.3,
|
|
39922
|
+
"output": 1.2,
|
|
39923
|
+
"cacheRead": 0,
|
|
39924
|
+
"cacheWrite": 0
|
|
39925
|
+
},
|
|
39926
|
+
"contextWindow": 196608,
|
|
39927
|
+
"maxTokens": 16384
|
|
39928
|
+
},
|
|
39929
|
+
"MiniMax-M2.7": {
|
|
39930
|
+
"id": "MiniMax-M2.7",
|
|
39931
|
+
"name": "MiniMax M2.7",
|
|
39932
|
+
"api": "openai-completions",
|
|
39933
|
+
"provider": "mara",
|
|
39934
|
+
"baseUrl": "https://api.cloud.mara.com/v1",
|
|
39935
|
+
"reasoning": false,
|
|
39936
|
+
"input": [
|
|
39937
|
+
"text"
|
|
39938
|
+
],
|
|
39939
|
+
"cost": {
|
|
39940
|
+
"input": 0.3,
|
|
39941
|
+
"output": 1.2,
|
|
39942
|
+
"cacheRead": 0,
|
|
39943
|
+
"cacheWrite": 0
|
|
39944
|
+
},
|
|
39945
|
+
"contextWindow": 196608,
|
|
39946
|
+
"maxTokens": 16384
|
|
39947
|
+
},
|
|
39948
|
+
"gpt-oss-120b": {
|
|
39949
|
+
"id": "gpt-oss-120b",
|
|
39950
|
+
"name": "GPT OSS 120B",
|
|
39951
|
+
"api": "openai-completions",
|
|
39952
|
+
"provider": "mara",
|
|
39953
|
+
"baseUrl": "https://api.cloud.mara.com/v1",
|
|
39954
|
+
"reasoning": true,
|
|
39955
|
+
"input": [
|
|
39956
|
+
"text"
|
|
39957
|
+
],
|
|
39958
|
+
"cost": {
|
|
39959
|
+
"input": 0.15,
|
|
39960
|
+
"output": 0.75,
|
|
39961
|
+
"cacheRead": 0,
|
|
39962
|
+
"cacheWrite": 0
|
|
39963
|
+
},
|
|
39964
|
+
"contextWindow": 131072,
|
|
39965
|
+
"maxTokens": 16384,
|
|
39966
|
+
"thinking": {
|
|
39967
|
+
"mode": "effort",
|
|
39968
|
+
"minLevel": "minimal",
|
|
39969
|
+
"maxLevel": "xhigh"
|
|
39970
|
+
}
|
|
39971
|
+
}
|
|
39972
|
+
},
|
|
39885
39973
|
"minimax": {
|
|
39886
39974
|
"MiniMax-M2": {
|
|
39887
39975
|
"id": "MiniMax-M2",
|
|
@@ -85423,4 +85511,4 @@
|
|
|
85423
85511
|
}
|
|
85424
85512
|
}
|
|
85425
85513
|
}
|
|
85426
|
-
}
|
|
85514
|
+
}
|
|
@@ -26,6 +26,7 @@ import {
|
|
|
26
26
|
kimiCodeModelManagerOptions,
|
|
27
27
|
litellmModelManagerOptions,
|
|
28
28
|
lmStudioModelManagerOptions,
|
|
29
|
+
maraModelManagerOptions,
|
|
29
30
|
mistralModelManagerOptions,
|
|
30
31
|
moonshotModelManagerOptions,
|
|
31
32
|
nanoGptModelManagerOptions,
|
|
@@ -326,6 +327,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
|
|
|
326
327
|
config => bizrouterModelManagerOptions(config),
|
|
327
328
|
catalog("BizRouter", ["BIZROUTER_API_KEY"]),
|
|
328
329
|
),
|
|
330
|
+
catalogDescriptor(
|
|
331
|
+
"mara",
|
|
332
|
+
"DeepSeek-V3.1",
|
|
333
|
+
config => maraModelManagerOptions(config),
|
|
334
|
+
catalog("Mara Cloud", ["MARA_API_KEY"]),
|
|
335
|
+
),
|
|
329
336
|
catalogDescriptor("zai", "glm-5.2", config => zaiModelManagerOptions(config), catalog("zAI", ["ZAI_API_KEY"])),
|
|
330
337
|
catalogDescriptor(
|
|
331
338
|
"glm-zcode",
|
|
@@ -1180,6 +1180,22 @@ export function bizrouterModelManagerOptions(
|
|
|
1180
1180
|
};
|
|
1181
1181
|
}
|
|
1182
1182
|
|
|
1183
|
+
// ---------------------------------------------------------------------------
|
|
1184
|
+
// 10.5.3 Mara Cloud
|
|
1185
|
+
// ---------------------------------------------------------------------------
|
|
1186
|
+
|
|
1187
|
+
export interface MaraModelManagerConfig {
|
|
1188
|
+
apiKey?: string;
|
|
1189
|
+
baseUrl?: string;
|
|
1190
|
+
}
|
|
1191
|
+
|
|
1192
|
+
/**
|
|
1193
|
+
* Mara Cloud — an OpenAI-compatible enterprise AI inference platform. Models
|
|
1194
|
+
* are discovered from the OpenAI-compatible `/v1/models` endpoint.
|
|
1195
|
+
*/
|
|
1196
|
+
export function maraModelManagerOptions(config?: MaraModelManagerConfig): ModelManagerOptions<"openai-completions"> {
|
|
1197
|
+
return createSimpleOpenAICompletionsOptions("mara", "https://api.cloud.mara.com/v1", config);
|
|
1198
|
+
}
|
|
1183
1199
|
// ---------------------------------------------------------------------------
|
|
1184
1200
|
// 10.6 Kilo Gateway
|
|
1185
1201
|
// ---------------------------------------------------------------------------
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DashScope Token Plan canonical request headers.
|
|
3
|
+
*
|
|
4
|
+
* Reproduces QwenLM/qwen-code's DashScopeOpenAICompatibleProvider.buildHeaders()
|
|
5
|
+
* defaultHeaders so the built-in `alibaba-token-plan` provider emits the same
|
|
6
|
+
* client identity / cache / auth-type fingerprint upstream sends. DashScope is
|
|
7
|
+
* compatibility-sensitive to this fingerprint; a non-identical set can cause
|
|
8
|
+
* request instability and affect first-event latency (gajae-code #3557).
|
|
9
|
+
*
|
|
10
|
+
* Upstream pin (reproduce EXACTLY here):
|
|
11
|
+
* Repository: QwenLM/qwen-code
|
|
12
|
+
* Commit: f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb
|
|
13
|
+
* Version: 0.21.1
|
|
14
|
+
* Source: packages/core/src/core/openaiContentGenerator/provider/dashscope.ts
|
|
15
|
+
* buildHeaders():
|
|
16
|
+
* const userAgent = `QwenCode/${version} (${process.platform}; ${process.arch})`;
|
|
17
|
+
* const defaultHeaders = {
|
|
18
|
+
* 'User-Agent': userAgent,
|
|
19
|
+
* 'X-DashScope-CacheControl': 'enable',
|
|
20
|
+
* 'X-DashScope-UserAgent': userAgent,
|
|
21
|
+
* 'X-DashScope-AuthType': authType,
|
|
22
|
+
* };
|
|
23
|
+
* return customHeaders ? { ...defaultHeaders, ...customHeaders } : defaultHeaders;
|
|
24
|
+
*
|
|
25
|
+
* The Token Plan preset authenticates with AuthType.USE_OPENAI ('openai'), so
|
|
26
|
+
* X-DashScope-AuthType is the constant 'openai'. Pin the version so an upstream
|
|
27
|
+
* bump is an explicit parity update rather than silent drift.
|
|
28
|
+
*/
|
|
29
|
+
export const QWEN_CODE_UPSTREAM_REPO = "QwenLM/qwen-code";
|
|
30
|
+
export const QWEN_CODE_UPSTREAM_COMMIT = "f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb";
|
|
31
|
+
export const QWEN_CODE_UPSTREAM_VERSION = "0.21.1";
|
|
32
|
+
|
|
33
|
+
// Upstream Token Plan preset uses AuthType.USE_OPENAI = 'openai'.
|
|
34
|
+
const QWEN_CODE_TOKEN_PLAN_AUTH_TYPE = "openai";
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* The Qwen Code CLI version string used in identity headers. Pinned to the
|
|
38
|
+
* upstream version at {@link QWEN_CODE_UPSTREAM_COMMIT}; change both together
|
|
39
|
+
* as an explicit parity update.
|
|
40
|
+
*/
|
|
41
|
+
export function qwenCodeUserAgent(version: string = QWEN_CODE_UPSTREAM_VERSION): string {
|
|
42
|
+
// process.platform / process.arch are read verbatim, matching upstream
|
|
43
|
+
// (e.g. "linux", "darwin", "win32"; "x64", "arm64"). No normalization.
|
|
44
|
+
return `QwenCode/${version} (${process.platform}; ${process.arch})`;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Canonical DashScope Token Plan headers (upstream defaultHeaders, no caller
|
|
49
|
+
* overrides applied). Exposed for tests/fixtures so the pinned wire set lives
|
|
50
|
+
* in exactly one place.
|
|
51
|
+
*/
|
|
52
|
+
export function dashscopeTokenPlanDefaultHeaders(
|
|
53
|
+
version: string = QWEN_CODE_UPSTREAM_VERSION,
|
|
54
|
+
): Readonly<Record<string, string>> {
|
|
55
|
+
const userAgent = qwenCodeUserAgent(version);
|
|
56
|
+
return Object.freeze({
|
|
57
|
+
"User-Agent": userAgent,
|
|
58
|
+
"X-DashScope-CacheControl": "enable",
|
|
59
|
+
"X-DashScope-UserAgent": userAgent,
|
|
60
|
+
"X-DashScope-AuthType": QWEN_CODE_TOKEN_PLAN_AUTH_TYPE,
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Merge canonical DashScope Token Plan identity headers onto a caller's header
|
|
66
|
+
* map, reproducing upstream buildHeaders() precedence EXACTLY:
|
|
67
|
+
* `{ ...defaultHeaders, ...customHeaders }` — caller wins per header.
|
|
68
|
+
*
|
|
69
|
+
* This mirrors GJC's existing kimi-code injection order
|
|
70
|
+
* (`headers = { ...getKimiCommonHeaders(), ...headers }`): canonical identity as
|
|
71
|
+
* the base, caller-supplied headers overriding individual keys. A caller that
|
|
72
|
+
* pins `User-Agent` takes that key; the other canonicals still apply.
|
|
73
|
+
*
|
|
74
|
+
* A null/undefined `callerHeaders` returns the canonical set alone (upstream
|
|
75
|
+
* `customHeaders ? {...} : defaultHeaders` shortcut).
|
|
76
|
+
*/
|
|
77
|
+
export function mergeDashScopeTokenPlanHeaders(
|
|
78
|
+
callerHeaders: Record<string, string> | undefined,
|
|
79
|
+
version: string = QWEN_CODE_UPSTREAM_VERSION,
|
|
80
|
+
): Record<string, string> {
|
|
81
|
+
const defaults = dashscopeTokenPlanDefaultHeaders(version);
|
|
82
|
+
if (!callerHeaders) return { ...defaults };
|
|
83
|
+
return { ...defaults, ...callerHeaders };
|
|
84
|
+
}
|
|
@@ -43,6 +43,7 @@ import {
|
|
|
43
43
|
createOpenAIResponsesHistoryPayload,
|
|
44
44
|
getOpenAIResponsesHistoryItems,
|
|
45
45
|
getOpenAIResponsesHistoryPayload,
|
|
46
|
+
neutralizeReservedControlTokens,
|
|
46
47
|
neutralizeResponsesInputControlTokens,
|
|
47
48
|
normalizeSystemPrompts,
|
|
48
49
|
sanitizeOpenAIResponsesHistoryItemsForReplay,
|
|
@@ -746,7 +747,12 @@ async function buildTransformedCodexRequestBody(
|
|
|
746
747
|
}
|
|
747
748
|
}
|
|
748
749
|
|
|
749
|
-
|
|
750
|
+
// Neutralize leaked Harmony control tokens in the system prompt too:
|
|
751
|
+
// `params.instructions` and the developer messages prepended inside
|
|
752
|
+
// `transformRequestBody` bypass the `input` sanitizer above, so a poisoned
|
|
753
|
+
// system prompt rejects every turn with
|
|
754
|
+
// `Request blocked (code=invalid_prompt)`.
|
|
755
|
+
const systemPrompts = normalizeSystemPrompts(context.systemPrompt).map(neutralizeReservedControlTokens);
|
|
750
756
|
if (systemPrompts.length > 0) {
|
|
751
757
|
params.instructions = systemPrompts[0];
|
|
752
758
|
}
|
|
@@ -68,6 +68,7 @@ import {
|
|
|
68
68
|
resolveToolChoice,
|
|
69
69
|
} from "../utils/tool-choice-capability";
|
|
70
70
|
import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
71
|
+
import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
|
|
71
72
|
import {
|
|
72
73
|
buildCopilotDynamicHeaders,
|
|
73
74
|
hasCopilotVisionInput,
|
|
@@ -1071,6 +1072,14 @@ async function createClient(
|
|
|
1071
1072
|
if (model.provider === "kimi-code") {
|
|
1072
1073
|
headers = { ...getKimiCommonHeaders(), ...headers };
|
|
1073
1074
|
}
|
|
1075
|
+
if (model.provider === "alibaba-token-plan") {
|
|
1076
|
+
// Emit Qwen Code's canonical DashScope request fingerprint (User-Agent /
|
|
1077
|
+
// X-DashScope-CacheControl / X-DashScope-UserAgent / X-DashScope-AuthType)
|
|
1078
|
+
// so DashScope treats the caller identically to upstream QwenLM/qwen-code.
|
|
1079
|
+
// Canonical identity is the base; caller headers win per key (upstream
|
|
1080
|
+
// `{...default, ...customHeaders}`). #3557.
|
|
1081
|
+
headers = mergeDashScopeTokenPlanHeaders(headers);
|
|
1082
|
+
}
|
|
1074
1083
|
headers = applyOpenAIRequestTransformHeaders(headers, model.requestTransform, `Gajae-Code/${packageJson.version}`);
|
|
1075
1084
|
let copilotPremiumRequests: number | undefined;
|
|
1076
1085
|
|
|
@@ -27,6 +27,7 @@ import {
|
|
|
27
27
|
getOpenAIResponsesHistoryItems,
|
|
28
28
|
getOpenAIResponsesHistoryPayload,
|
|
29
29
|
isInvalidPromptError,
|
|
30
|
+
neutralizeReservedControlTokens,
|
|
30
31
|
neutralizeResponsesInputControlTokens,
|
|
31
32
|
normalizeSystemPrompts,
|
|
32
33
|
resolveCacheRetention,
|
|
@@ -60,6 +61,7 @@ import {
|
|
|
60
61
|
resolveToolChoice,
|
|
61
62
|
} from "../utils/tool-choice-capability";
|
|
62
63
|
import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
64
|
+
import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
|
|
63
65
|
import {
|
|
64
66
|
buildCopilotDynamicHeaders,
|
|
65
67
|
hasCopilotVisionInput,
|
|
@@ -446,8 +448,17 @@ function createClient(
|
|
|
446
448
|
}
|
|
447
449
|
const rawApiKey = apiKey;
|
|
448
450
|
|
|
451
|
+
const baseHeaders =
|
|
452
|
+
model.provider === "alibaba-token-plan"
|
|
453
|
+
? // Emit Qwen Code's canonical DashScope request fingerprint (User-Agent /
|
|
454
|
+
// X-DashScope-CacheControl / X-DashScope-UserAgent / X-DashScope-AuthType)
|
|
455
|
+
// so DashScope treats the caller identically to upstream QwenLM/qwen-code.
|
|
456
|
+
// Canonical identity is the base; caller headers win per key (upstream
|
|
457
|
+
// `{...default, ...customHeaders}`). #3557.
|
|
458
|
+
mergeDashScopeTokenPlanHeaders({ ...(model.headers ?? {}), ...(extraHeaders ?? {}) })
|
|
459
|
+
: { ...(model.headers ?? {}), ...(extraHeaders ?? {}) };
|
|
449
460
|
const headers = applyOpenAIRequestTransformHeaders(
|
|
450
|
-
|
|
461
|
+
baseHeaders,
|
|
451
462
|
model.requestTransform,
|
|
452
463
|
`Gajae-Code/${packageJson.version}`,
|
|
453
464
|
);
|
|
@@ -523,7 +534,13 @@ function buildParams(
|
|
|
523
534
|
);
|
|
524
535
|
const messages: ResponseInput = neutralizeResponsesInputControlTokens(conversationMessages);
|
|
525
536
|
|
|
526
|
-
|
|
537
|
+
// Neutralize leaked Harmony control tokens in the system prompt too: the
|
|
538
|
+
// `instructions` field and developer-role messages bypass the `input`
|
|
539
|
+
// request-boundary sanitizer above, and a poisoned system prompt (e.g.
|
|
540
|
+
// injected project context quoting `<|channel|>` markers) rejects EVERY
|
|
541
|
+
// turn with `Request blocked (code=invalid_prompt)` — unrepairable by the
|
|
542
|
+
// history circuit breaker.
|
|
543
|
+
const systemPrompts = normalizeSystemPrompts(context.systemPrompt).map(neutralizeReservedControlTokens);
|
|
527
544
|
if (isComposerHarnessModel(model.id)) {
|
|
528
545
|
systemPrompts.unshift(COMPOSER_EDIT_DISCIPLINE_PROMPT);
|
|
529
546
|
}
|
package/src/stream.ts
CHANGED
|
@@ -164,6 +164,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
164
164
|
zenmux: "ZENMUX_API_KEY",
|
|
165
165
|
opengateway: "OPENGATEWAY_API_KEY",
|
|
166
166
|
bizrouter: "BIZROUTER_API_KEY",
|
|
167
|
+
mara: "MARA_API_KEY",
|
|
167
168
|
venice: "VENICE_API_KEY",
|
|
168
169
|
vllm: "VLLM_API_KEY",
|
|
169
170
|
xiaomi: "XIAOMI_API_KEY",
|
package/src/types.ts
CHANGED
package/src/utils/oauth/index.ts
CHANGED
|
@@ -245,6 +245,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
|
|
|
245
245
|
name: "BizRouter",
|
|
246
246
|
available: true,
|
|
247
247
|
},
|
|
248
|
+
{
|
|
249
|
+
id: "mara",
|
|
250
|
+
name: "Mara Cloud",
|
|
251
|
+
available: true,
|
|
252
|
+
},
|
|
248
253
|
{
|
|
249
254
|
id: "opengateway",
|
|
250
255
|
name: "OpenGateway by Sionic AI",
|
|
@@ -393,6 +398,7 @@ export async function refreshOAuthToken(
|
|
|
393
398
|
case "moonshot":
|
|
394
399
|
case "kagi":
|
|
395
400
|
case "cloudflare-ai-gateway":
|
|
401
|
+
case "mara":
|
|
396
402
|
case "vercel-ai-gateway":
|
|
397
403
|
case "qwen-portal":
|
|
398
404
|
case "zenmux":
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/** Mara Cloud login flow (API key paste, validated via chat completions). */
|
|
2
|
+
import { createApiKeyLogin } from "./api-key-login";
|
|
3
|
+
|
|
4
|
+
export const loginMara = createApiKeyLogin({
|
|
5
|
+
providerLabel: "Mara Cloud",
|
|
6
|
+
authUrl: "https://cloud.mara.com/apis",
|
|
7
|
+
instructions: "Create or copy your Mara Cloud API key",
|
|
8
|
+
promptMessage: "Paste your Mara Cloud API key",
|
|
9
|
+
placeholder: "<your-mara-api-key>",
|
|
10
|
+
validation: {
|
|
11
|
+
kind: "chat-completions",
|
|
12
|
+
provider: "Mara Cloud",
|
|
13
|
+
baseUrl: "https://api.cloud.mara.com/v1",
|
|
14
|
+
model: "DeepSeek-V3.1",
|
|
15
|
+
},
|
|
16
|
+
});
|