@bitkyc08/opencodex 2.55.0-preview.20260914 → 2.56.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/gui/dist/assets/{index-DH2PUHqr.js → index-D4zuyIxQ.js} +1 -1
- package/gui/dist/index.html +1 -1
- package/package.json +2 -1
- package/src/adapters/base.ts +21 -0
- package/src/adapters/cursor/transport-retry.ts +46 -1
- package/src/adapters/cursor.ts +4 -0
- package/src/adapters/kiro/adapter.ts +42 -1
- package/src/adapters/kiro-retry.ts +23 -4
- package/src/adapters/openai-chat/errors.ts +116 -0
- package/src/adapters/openai-chat/messages.ts +346 -0
- package/src/adapters/openai-chat/passthrough.ts +146 -0
- package/src/adapters/openai-chat/response-events.ts +117 -0
- package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
- package/src/adapters/openai-chat/tool-schema.ts +477 -0
- package/src/adapters/openai-chat/wire.ts +50 -0
- package/src/adapters/openai-chat.ts +33 -1445
- package/src/adapters/openai-responses/canonical-forward.ts +202 -0
- package/src/adapters/openai-responses/image-gen.ts +406 -0
- package/src/adapters/openai-responses/internal.ts +3 -0
- package/src/adapters/openai-responses/passthrough.ts +611 -0
- package/src/adapters/openai-responses/prompt-cache.ts +83 -0
- package/src/adapters/openai-responses/reasoning.ts +220 -0
- package/src/adapters/openai-responses/request-strips.ts +185 -0
- package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
- package/src/adapters/openai-responses/tool-schema.ts +293 -0
- package/src/adapters/openai-responses/web-search.ts +156 -0
- package/src/adapters/openai-responses.ts +4 -2625
- package/src/bridge/errors.ts +34 -0
- package/src/bridge/internal.ts +174 -0
- package/src/bridge/response-json.ts +624 -0
- package/src/bridge/sse.ts +1444 -0
- package/src/bridge.ts +5 -2204
- package/src/chat/inbound.ts +12 -1
- package/src/codex/account-lifecycle.ts +3 -0
- package/src/codex/account-store.ts +71 -9
- package/src/codex/auth-api/account-list.ts +507 -0
- package/src/codex/auth-api/http.ts +32 -0
- package/src/codex/auth-api/login-flow.ts +554 -0
- package/src/codex/auth-api/login-state.ts +64 -0
- package/src/codex/auth-api/main-account-probe.ts +331 -0
- package/src/codex/auth-api/pool-mode-gate.ts +274 -0
- package/src/codex/auth-api/pool-quota-probe.ts +512 -0
- package/src/codex/auth-api/reset-credit-service.ts +422 -0
- package/src/codex/auth-api/routes.ts +425 -0
- package/src/codex/auth-api/runtime-config.ts +48 -0
- package/src/codex/auth-api.ts +27 -3118
- package/src/codex/auth-context.ts +95 -28
- package/src/codex/catalog/auto-review.ts +507 -0
- package/src/codex/catalog/build-entries.ts +981 -0
- package/src/codex/catalog/combo-member.ts +375 -0
- package/src/codex/catalog/derive-entry.ts +229 -0
- package/src/codex/catalog/effort.ts +0 -1
- package/src/codex/catalog/gated-native-warn.ts +63 -0
- package/src/codex/catalog/gather-capture.ts +533 -0
- package/src/codex/catalog/model-hints.ts +691 -0
- package/src/codex/catalog/model-visibility.ts +304 -0
- package/src/codex/catalog/provider-fetch.ts +52 -2942
- package/src/codex/catalog/provider-models.ts +685 -0
- package/src/codex/catalog/restore.ts +132 -0
- package/src/codex/catalog/retained-sync.ts +706 -0
- package/src/codex/catalog/routed-gather.ts +858 -0
- package/src/codex/catalog/subagent-roster.ts +176 -0
- package/src/codex/catalog/sync.ts +52 -2698
- package/src/codex/inject/config-toml.ts +563 -0
- package/src/codex/inject/remove.ts +192 -0
- package/src/codex/inject/restore.ts +540 -0
- package/src/codex/inject/routing-classify.ts +109 -0
- package/src/codex/inject/routing-target.ts +125 -0
- package/src/codex/inject.ts +81 -1436
- package/src/codex/lineage.ts +458 -0
- package/src/codex/pool-refresh-backoff.ts +152 -0
- package/src/codex/routing/active-account.ts +194 -0
- package/src/codex/routing/cooldown-math.ts +275 -0
- package/src/codex/routing/health-store.ts +402 -0
- package/src/codex/routing/probe-lease.ts +358 -0
- package/src/codex/routing/selection.ts +703 -0
- package/src/codex/routing/thread-affinity.ts +538 -0
- package/src/codex/routing.ts +353 -2234
- package/src/codex/shim-fingerprint.ts +223 -0
- package/src/codex/shim-inspect.ts +175 -0
- package/src/codex/shim-probe.ts +367 -0
- package/src/codex/shim-restore-lock.ts +169 -0
- package/src/codex/shim-state-file.ts +151 -0
- package/src/codex/shim-templates.ts +265 -0
- package/src/codex/shim.ts +48 -1268
- package/src/config/diagnostics.ts +705 -0
- package/src/config/feature-flags.ts +55 -0
- package/src/config/live-reconcile.ts +403 -0
- package/src/config/load-degrade.ts +880 -0
- package/src/config/mutation-lock.ts +244 -0
- package/src/config/openai-tier-backup.ts +268 -0
- package/src/config/persist-unlocked.ts +92 -0
- package/src/config/proxy-env.ts +188 -0
- package/src/config/salvage.ts +244 -0
- package/src/config/schema/config-schema.ts +640 -0
- package/src/config/schema/leaf-validators.ts +855 -0
- package/src/config/warn-memo.ts +28 -0
- package/src/config.ts +234 -4481
- package/src/generated/compatibility-version.json +539 -39
- package/src/lib/request-execution-budget.ts +69 -20
- package/src/lib/spend-reservation-ledger.ts +940 -0
- package/src/lib/upstream-retry.ts +55 -11
- package/src/lib/workflow-budget.ts +553 -30
- package/src/providers/quota/account-cache.ts +441 -0
- package/src/providers/quota/antigravity.ts +295 -0
- package/src/providers/quota/report-cache.ts +320 -0
- package/src/providers/quota/vendor-probes-key.ts +1243 -0
- package/src/providers/quota/vendor-probes-oauth.ts +590 -0
- package/src/providers/quota.ts +324 -3079
- package/src/providers/registry/entries-core.ts +1221 -0
- package/src/providers/registry/entries-extended.ts +1204 -0
- package/src/providers/registry/model-seeds.ts +908 -0
- package/src/providers/registry/types.ts +352 -0
- package/src/providers/registry.ts +24 -3536
- package/src/responses/continuation-ownership.ts +29 -0
- package/src/responses/state/replay-fingerprint.ts +80 -0
- package/src/responses/state/snapshot-codec.ts +104 -0
- package/src/responses/state/spill-failure.ts +118 -0
- package/src/responses/state/spill-queue.ts +665 -0
- package/src/responses/state/temp-recovery.ts +257 -0
- package/src/responses/state.ts +82 -1143
- package/src/routing/identity-domains.ts +449 -0
- package/src/routing/probe-lease.ts +511 -0
- package/src/server/index/bounded-request.ts +88 -0
- package/src/server/index/live-sideband.ts +565 -0
- package/src/server/index/serve-options.ts +1766 -0
- package/src/server/index/startup-warnings.ts +213 -0
- package/src/server/index/websocket-handler.ts +335 -0
- package/src/server/index.ts +40 -2547
- package/src/server/management/route-registry.ts +26 -23
- package/src/server/management/shared.ts +8 -5
- package/src/server/management/workflow-budget-routes.ts +133 -0
- package/src/server/management-api.ts +12 -0
- package/src/server/request-log-conversation.ts +9 -7
- package/src/server/request-log.ts +245 -1
- package/src/server/responses/account-change-state.ts +233 -0
- package/src/server/responses/adapter-continuation.ts +514 -0
- package/src/server/responses/adapter-delivery.ts +214 -0
- package/src/server/responses/adapter-dispatch.ts +971 -0
- package/src/server/responses/compact.ts +59 -4
- package/src/server/responses/completion-policy.ts +33 -0
- package/src/server/responses/core-auth.ts +527 -0
- package/src/server/responses/core-codex-account.ts +859 -0
- package/src/server/responses/core-combo-failure.ts +210 -0
- package/src/server/responses/core-combo.ts +707 -0
- package/src/server/responses/core-errors.ts +152 -0
- package/src/server/responses/core-lifetime.ts +95 -0
- package/src/server/responses/core-normalize.ts +350 -0
- package/src/server/responses/core-opaque-recovery.ts +380 -0
- package/src/server/responses/core-options.ts +159 -0
- package/src/server/responses/core-replay.ts +225 -0
- package/src/server/responses/core.ts +192 -8893
- package/src/server/responses/passthrough-delivery.ts +856 -0
- package/src/server/responses/passthrough-dispatch.ts +1476 -0
- package/src/server/responses/passthrough-execution.ts +54 -0
- package/src/server/responses/request-prepare.ts +970 -0
- package/src/server/responses/request-send-budget.ts +164 -0
- package/src/server/responses/request-sidecar-auth.ts +149 -0
- package/src/server/responses/request-transport.ts +744 -0
- package/src/server/responses/response-effects.ts +157 -0
- package/src/server/responses/run-turn-execution.ts +448 -0
- package/src/server/responses/sidecar-execution.ts +469 -0
- package/src/server/responses-image-gen-repair.ts +1 -1
- package/src/server/workflow-refusal.ts +84 -0
- package/src/types/config.ts +30 -0
- package/src/usage/log.ts +146 -0
- package/src/usage/summary.ts +171 -21
|
@@ -0,0 +1,1204 @@
|
|
|
1
|
+
import {
|
|
2
|
+
QWEN_CLOUD_BASE_URL_CHOICES,
|
|
3
|
+
QWEN_CLOUD_TOKEN_PLAN_BASE_URL,
|
|
4
|
+
ALIBABA_INTL_BASE_URL_CHOICES,
|
|
5
|
+
ALIBABA_INTL_TOKEN_PLAN_BASE_URL,
|
|
6
|
+
ALIBABA_CODING_BASE_URL_CHOICES,
|
|
7
|
+
ALIBABA_CODING_INTL_BASE_URL,
|
|
8
|
+
MOONSHOT_BASE_URL_CHOICES,
|
|
9
|
+
MOONSHOT_INTL_BASE_URL,
|
|
10
|
+
} from "../base-url-choices";
|
|
11
|
+
import { COMMAND_CODE_MODEL_REASONING_EFFORTS } from "../command-code-efforts";
|
|
12
|
+
import {
|
|
13
|
+
CODEBUDDY_CN_MODELS,
|
|
14
|
+
CODEBUDDY_CN_MODEL_CONTEXT_WINDOWS,
|
|
15
|
+
CODEBUDDY_CN_MODEL_DEFAULT_REASONING_EFFORTS,
|
|
16
|
+
CODEBUDDY_CN_MODEL_MAX_OUTPUT_TOKENS,
|
|
17
|
+
CODEBUDDY_CN_MODEL_REASONING_EFFORTS,
|
|
18
|
+
CODEBUDDY_CN_NO_VISION_MODELS,
|
|
19
|
+
CODEBUDDY_GLOBAL_MODELS,
|
|
20
|
+
CODEBUDDY_GLOBAL_MODEL_CONTEXT_WINDOWS,
|
|
21
|
+
CODEBUDDY_GLOBAL_MODEL_DEFAULT_REASONING_EFFORTS,
|
|
22
|
+
CODEBUDDY_GLOBAL_MODEL_MAX_OUTPUT_TOKENS,
|
|
23
|
+
CODEBUDDY_GLOBAL_MODEL_REASONING_EFFORTS,
|
|
24
|
+
CODEBUDDY_REASONING_EFFORTS,
|
|
25
|
+
} from "../codebuddy-models";
|
|
26
|
+
import { QODER_CN_MODELS, QODER_GLOBAL_MODELS, QODER_REASONING_EFFORTS } from "../qoder-models";
|
|
27
|
+
import type { ProviderRegistryEntry } from "./types";
|
|
28
|
+
import {
|
|
29
|
+
ZAI_GLM_53_MODELS,
|
|
30
|
+
ZAI_GLM_5X_MODELS,
|
|
31
|
+
ZAI_GLM_5X_SIDECAR_VISION_MODELS,
|
|
32
|
+
ZAI_GLM_5X_INPUT_MODALITIES,
|
|
33
|
+
ZAI_GLM_52_REASONING_EFFORTS,
|
|
34
|
+
ZAI_GLM_53_REASONING_EFFORTS,
|
|
35
|
+
ZAI_GLM_5X_REASONING_EFFORTS,
|
|
36
|
+
MINIMAX_MODELS,
|
|
37
|
+
MINIMAX_MODEL_CONTEXT_WINDOWS,
|
|
38
|
+
MINIMAX_M3_REASONING_EFFORTS,
|
|
39
|
+
MINIMAX_M3_REASONING_EFFORT_MAP,
|
|
40
|
+
THINKING_TOGGLE_EFFORTS,
|
|
41
|
+
THINKING_TOGGLE_MAP,
|
|
42
|
+
ZHIPU_BIGMODEL_MODELS,
|
|
43
|
+
ZHIPU_BIGMODEL_INPUT_MODALITIES,
|
|
44
|
+
ZHIPU_BIGMODEL_THINKING_TOGGLE_MODELS,
|
|
45
|
+
THINKING_BUDGET_EFFORTS,
|
|
46
|
+
QWEN38_REASONING_EFFORTS,
|
|
47
|
+
DEEPSEEK_V4_LEGACY_MODELS,
|
|
48
|
+
DEEPSEEK_GATEWAY_THINKING_MODELS,
|
|
49
|
+
DEEPSEEK_VISION_PREVIEW_MODEL,
|
|
50
|
+
COMMAND_CODE_MODEL_INPUT_MODALITIES,
|
|
51
|
+
OPENCODE_FREE_DEEPSEEK_MODELS,
|
|
52
|
+
OPENCODE_ZEN_TEXT_ONLY_MODELS,
|
|
53
|
+
OPENCODE_ZEN_IMAGE_MODELS,
|
|
54
|
+
deepseekThinkingEffortsFor,
|
|
55
|
+
deepseekReasoningMapFor,
|
|
56
|
+
ALIBABA_TOKEN_PLAN_MODELS,
|
|
57
|
+
ALIBABA_TOKEN_PLAN_QWEN_MODELS,
|
|
58
|
+
ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
59
|
+
ALIBABA_INTL_TOKEN_PLAN_MODELS,
|
|
60
|
+
ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS,
|
|
61
|
+
TENCENT_CODING_PLAN_MODELS,
|
|
62
|
+
VOLCENGINE_ARK_MODELS,
|
|
63
|
+
VOLCENGINE_DOUBAO_THINKING_MODELS,
|
|
64
|
+
VOLCENGINE_CODING_PLAN_MODELS,
|
|
65
|
+
VOLCENGINE_AGENT_PLAN_MODELS,
|
|
66
|
+
VOLCENGINE_PLAN_INPUT_MODALITIES,
|
|
67
|
+
VOLCENGINE_PLAN_TEXT_ONLY_MODELS,
|
|
68
|
+
ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES,
|
|
69
|
+
KIMI_API_MODELS,
|
|
70
|
+
KIMI_CODING_MODELS,
|
|
71
|
+
KIMI_THINKING_MODELS,
|
|
72
|
+
KIMI_CODING_NO_REASONING_MODELS,
|
|
73
|
+
KIMI_API_NO_REASONING_MODELS,
|
|
74
|
+
KIMI_CODING_REASONING_EFFORTS,
|
|
75
|
+
KIMI_CODING_DEFAULT_REASONING_EFFORTS,
|
|
76
|
+
KIMI_CODING_REASONING_EFFORT_MAPS,
|
|
77
|
+
KIMI_API_REASONING_EFFORTS,
|
|
78
|
+
KIMI_LOCKED_PARAMETER_MODELS,
|
|
79
|
+
KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS,
|
|
80
|
+
KIMI_API_MODEL_CONTEXT_WINDOWS,
|
|
81
|
+
KIMI_API_MODEL_INPUT_MODALITIES,
|
|
82
|
+
NVIDIA_NIM_KIMI_THINKING_MODELS,
|
|
83
|
+
NVIDIA_NIM_KIMI_MODELS,
|
|
84
|
+
NVIDIA_NIM_VISION_MODELS,
|
|
85
|
+
NVIDIA_NIM_VISION_INPUT_MODALITIES,
|
|
86
|
+
NVIDIA_NIM_NO_VISION_MODELS,
|
|
87
|
+
KIMI_CODING_MODEL_CONTEXT_WINDOWS,
|
|
88
|
+
KIMI_CODING_MODEL_INPUT_MODALITIES,
|
|
89
|
+
BASETEN_MODEL_REASONING_EFFORTS,
|
|
90
|
+
BASETEN_MODEL_REASONING_EFFORT_MAP,
|
|
91
|
+
BASETEN_MODEL_DEFAULT_REASONING_EFFORTS,
|
|
92
|
+
BASETEN_MODEL_INPUT_MODALITIES,
|
|
93
|
+
DIGITALOCEAN_CHAT_COMPLETION_MODELS,
|
|
94
|
+
SCALEWAY_SERVERLESS_CHAT_MODELS,
|
|
95
|
+
SCALEWAY_MODEL_INPUT_MODALITIES,
|
|
96
|
+
} from "./model-seeds";
|
|
97
|
+
|
|
98
|
+
export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
99
|
+
{
|
|
100
|
+
id: "baseten",
|
|
101
|
+
label: "Baseten Model APIs",
|
|
102
|
+
baseUrl: "https://inference.baseten.co/v1",
|
|
103
|
+
adapter: "openai-chat",
|
|
104
|
+
authKind: "key",
|
|
105
|
+
dashboardUrl: "https://app.baseten.co/settings/api_keys",
|
|
106
|
+
liveModels: true,
|
|
107
|
+
preserveCustomDestination: true,
|
|
108
|
+
// Baseten's Chat Completions contract documents parallel_tool_calls as default-on.
|
|
109
|
+
parallelToolCalls: true,
|
|
110
|
+
// Baseten says models outside its reasoning table do not support reasoning. Keep
|
|
111
|
+
// unknown/new live slugs conservative until an official-docs registry refresh proves it.
|
|
112
|
+
reasoningEfforts: [],
|
|
113
|
+
modelReasoningEfforts: BASETEN_MODEL_REASONING_EFFORTS,
|
|
114
|
+
modelReasoningEffortMap: BASETEN_MODEL_REASONING_EFFORT_MAP,
|
|
115
|
+
modelDefaultReasoningEfforts: BASETEN_MODEL_DEFAULT_REASONING_EFFORTS,
|
|
116
|
+
modelInputModalities: BASETEN_MODEL_INPUT_MODALITIES,
|
|
117
|
+
modelDiscovery: {
|
|
118
|
+
path: "models",
|
|
119
|
+
maxResponseBytes: 1_048_576,
|
|
120
|
+
maxModels: 256,
|
|
121
|
+
},
|
|
122
|
+
note: "Shared Model APIs only (personal API key, or team key with Call Model APIs access); dedicated Truss predict endpoints are outside this preset.",
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
id: "commandcode",
|
|
126
|
+
label: "Command Code - API",
|
|
127
|
+
adapter: "openai-chat",
|
|
128
|
+
baseUrl: "https://api.commandcode.ai/provider/v1",
|
|
129
|
+
authKind: "key",
|
|
130
|
+
dashboardUrl: "https://commandcode.ai/studio/",
|
|
131
|
+
liveModels: true,
|
|
132
|
+
preserveCustomDestination: true,
|
|
133
|
+
defaultModel: "deepseek/deepseek-v4-flash",
|
|
134
|
+
promptCacheKey: true,
|
|
135
|
+
// The default is also the cold-start seed: live discovery failure must not empty the catalog
|
|
136
|
+
// for a freshly configured provider with no stale cache (issue #308 pattern).
|
|
137
|
+
models: ["deepseek/deepseek-v4-flash"],
|
|
138
|
+
// The public model catalog is unauthenticated, so a Bearer probe cannot prove key validity.
|
|
139
|
+
apiKeyValidation: "unknown",
|
|
140
|
+
// The public catalog reports ids/context windows only; no trustworthy reasoning contract.
|
|
141
|
+
reasoningEfforts: [],
|
|
142
|
+
// Official Command Code model-profile reasoning facts (shared with the OAuth
|
|
143
|
+
// `command-code` entry). Without them the API-key preset never advertises a
|
|
144
|
+
// reasoning picker, and the router's known-ids decode source misses the native
|
|
145
|
+
// slash ids — so a Codex-facing slug like `commandcode/deepseek-deepseek-v4-flash`
|
|
146
|
+
// is sent upstream verbatim and rejected with `unsupported_model`.
|
|
147
|
+
modelReasoningEfforts: COMMAND_CODE_MODEL_REASONING_EFFORTS,
|
|
148
|
+
// The DeepSeek vision preview id is preemptive for when the catalog serves it
|
|
149
|
+
// (merges into v4-flash later).
|
|
150
|
+
modelContextWindows: {
|
|
151
|
+
[`deepseek/${DEEPSEEK_VISION_PREVIEW_MODEL}`]: 1_048_576,
|
|
152
|
+
},
|
|
153
|
+
modelInputModalities: COMMAND_CODE_MODEL_INPUT_MODALITIES,
|
|
154
|
+
modelDiscovery: {
|
|
155
|
+
path: "models",
|
|
156
|
+
maxResponseBytes: 256 * 1024,
|
|
157
|
+
maxModels: 256,
|
|
158
|
+
},
|
|
159
|
+
// Verified 2026-08-03: public /provider/v1/models returns 51 rows; /chat/completions returns
|
|
160
|
+
// 401 UNAUTHORIZED without a Bearer key. Primary source: https://commandcode.ai/docs/provider.
|
|
161
|
+
note: "Command Code Provider API (OpenAI-compatible); API access requires the Provider plan. Use `ocx login command-code` for OAuth account login (imports an existing local Command Code CLI credential when present). Docs: https://commandcode.ai/docs/provider.",
|
|
162
|
+
},
|
|
163
|
+
{
|
|
164
|
+
id: "sambanova",
|
|
165
|
+
label: "SambaNova Cloud",
|
|
166
|
+
baseUrl: "https://api.sambanova.ai/v1",
|
|
167
|
+
adapter: "openai-chat",
|
|
168
|
+
authKind: "key",
|
|
169
|
+
dashboardUrl: "https://cloud.sambanova.ai/apis",
|
|
170
|
+
liveModels: true,
|
|
171
|
+
preserveCustomDestination: true,
|
|
172
|
+
apiKeyValidation: "unknown",
|
|
173
|
+
// SambaNova documents this request field but does not yet support parallel function calls.
|
|
174
|
+
parallelToolCalls: false,
|
|
175
|
+
// The public catalog does not report a trustworthy per-model reasoning contract.
|
|
176
|
+
reasoningEfforts: [],
|
|
177
|
+
modelDiscovery: {
|
|
178
|
+
path: "models",
|
|
179
|
+
maxResponseBytes: 128 * 1024,
|
|
180
|
+
maxModels: 128,
|
|
181
|
+
},
|
|
182
|
+
note: "SambaNova Cloud text-generation models only; private SambaStudio deployment endpoints are outside this preset.",
|
|
183
|
+
},
|
|
184
|
+
{
|
|
185
|
+
id: "nebius",
|
|
186
|
+
label: "Nebius Token Factory",
|
|
187
|
+
baseUrl: "https://api.tokenfactory.nebius.com/v1",
|
|
188
|
+
adapter: "openai-chat",
|
|
189
|
+
authKind: "key",
|
|
190
|
+
dashboardUrl: "https://tokenfactory.nebius.com",
|
|
191
|
+
liveModels: true,
|
|
192
|
+
preserveCustomDestination: true,
|
|
193
|
+
// The public tools guide documents single function selection, not parallel tool calls.
|
|
194
|
+
parallelToolCalls: false,
|
|
195
|
+
// Missing reasoning metadata must not promote a model to Codex's full fallback ladder.
|
|
196
|
+
reasoningEfforts: [],
|
|
197
|
+
modelDiscovery: {
|
|
198
|
+
path: "models",
|
|
199
|
+
query: { verbose: "true" },
|
|
200
|
+
maxResponseBytes: 512 * 1024,
|
|
201
|
+
maxModels: 512,
|
|
202
|
+
filter: {
|
|
203
|
+
// Keep rows whose reported architecture output includes text (for example,
|
|
204
|
+
// text->text or text+image->text); embedding and image-generation rows are excluded.
|
|
205
|
+
allOf: [{ path: ["architecture", "modality"], containsAny: ["->text"] }],
|
|
206
|
+
},
|
|
207
|
+
},
|
|
208
|
+
note: "Shared Token Factory text-output inference only; live discovery excludes embedding and image-generation rows.",
|
|
209
|
+
},
|
|
210
|
+
{
|
|
211
|
+
id: "digitalocean",
|
|
212
|
+
label: "DigitalOcean Serverless Inference",
|
|
213
|
+
baseUrl: "https://inference.do-ai.run/v1",
|
|
214
|
+
adapter: "openai-chat",
|
|
215
|
+
authKind: "key",
|
|
216
|
+
dashboardUrl: "https://cloud.digitalocean.com/model-studio/manage-keys",
|
|
217
|
+
liveModels: true,
|
|
218
|
+
preserveCustomDestination: true,
|
|
219
|
+
// The Chat Completions contract documents function calls but not universal parallel support.
|
|
220
|
+
parallelToolCalls: false,
|
|
221
|
+
// Unknown catalog rows must not inherit Codex's full fallback reasoning ladder.
|
|
222
|
+
reasoningEfforts: [],
|
|
223
|
+
modelDiscovery: {
|
|
224
|
+
path: "models",
|
|
225
|
+
maxResponseBytes: 256 * 1024,
|
|
226
|
+
maxModels: 256,
|
|
227
|
+
filter: {
|
|
228
|
+
allOf: [{ path: ["id"], equalsAny: DIGITALOCEAN_CHAT_COMPLETION_MODELS }],
|
|
229
|
+
},
|
|
230
|
+
},
|
|
231
|
+
note: "Shared Serverless Inference Chat Completions only; agent-specific, dedicated, Responses-only, embedding, and media-generation models are outside this preset.",
|
|
232
|
+
},
|
|
233
|
+
{
|
|
234
|
+
id: "scaleway",
|
|
235
|
+
label: "Scaleway Generative APIs",
|
|
236
|
+
baseUrl: "https://api.scaleway.ai/v1",
|
|
237
|
+
adapter: "openai-chat",
|
|
238
|
+
authKind: "key",
|
|
239
|
+
dashboardUrl: "https://console.scaleway.com/generative-api",
|
|
240
|
+
liveModels: true,
|
|
241
|
+
freeTier: true,
|
|
242
|
+
preserveCustomDestination: true,
|
|
243
|
+
// Parallel support varies by model; avoid advertising it as a provider-wide capability.
|
|
244
|
+
parallelToolCalls: false,
|
|
245
|
+
// The generic `/models` rows carry no trustworthy reasoning metadata.
|
|
246
|
+
reasoningEfforts: [],
|
|
247
|
+
modelInputModalities: SCALEWAY_MODEL_INPUT_MODALITIES,
|
|
248
|
+
modelDiscovery: {
|
|
249
|
+
path: "models",
|
|
250
|
+
maxResponseBytes: 128 * 1024,
|
|
251
|
+
maxModels: 128,
|
|
252
|
+
filter: {
|
|
253
|
+
allOf: [{ path: ["id"], equalsAny: SCALEWAY_SERVERLESS_CHAT_MODELS }],
|
|
254
|
+
},
|
|
255
|
+
},
|
|
256
|
+
note: "Shared Generative APIs Serverless Chat Completions only; project-qualified and dedicated deployment hosts require a custom provider.",
|
|
257
|
+
},
|
|
258
|
+
{
|
|
259
|
+
// Primary sources checked 2026-08-08:
|
|
260
|
+
// - https://featherless.ai/docs/api-overview-and-common-options documents the fixed
|
|
261
|
+
// OpenAI-compatible base URL, Bearer keys, and Chat Completions.
|
|
262
|
+
// - https://featherless.ai/docs/api-reference-models documents authenticated plan filtering,
|
|
263
|
+
// chat capability filtering, popularity sorting, pagination, and per-row tool metadata.
|
|
264
|
+
// - https://featherless.ai/legal/terms-of-service identifies Featherless as a Delaware LLC,
|
|
265
|
+
// covers developers building on its APIs, and reserves arbitrary applications for Scale
|
|
266
|
+
// plans. Maintainer: @olddonkey; no affiliation with Featherless.
|
|
267
|
+
id: "featherless",
|
|
268
|
+
label: "Featherless AI",
|
|
269
|
+
baseUrl: "https://api.featherless.ai/v1",
|
|
270
|
+
adapter: "openai-chat",
|
|
271
|
+
authKind: "key",
|
|
272
|
+
dashboardUrl: "https://featherless.ai/account/api-keys",
|
|
273
|
+
liveModels: true,
|
|
274
|
+
preserveCustomDestination: true,
|
|
275
|
+
// /v1/models is documented as callable authenticated or unauthenticated, so a 2xx catalog
|
|
276
|
+
// response cannot prove that the supplied Bearer key is valid.
|
|
277
|
+
apiKeyValidation: "unknown",
|
|
278
|
+
// Featherless documents tool calling, but not a provider-wide parallel tool-call contract.
|
|
279
|
+
parallelToolCalls: false,
|
|
280
|
+
// Reasoning controls use model-specific chat_template_kwargs, not OpenAI reasoning_effort.
|
|
281
|
+
reasoningEfforts: [],
|
|
282
|
+
modelDiscovery: {
|
|
283
|
+
path: "models",
|
|
284
|
+
query: {
|
|
285
|
+
available_on_current_plan: "true",
|
|
286
|
+
capabilities: "chat",
|
|
287
|
+
page: "1",
|
|
288
|
+
per_page: "100",
|
|
289
|
+
sort: "-popularity",
|
|
290
|
+
},
|
|
291
|
+
maxResponseBytes: 128 * 1024,
|
|
292
|
+
maxModels: 100,
|
|
293
|
+
filter: {
|
|
294
|
+
// Treat server-side filters as a size optimization, not an authority boundary. A row must
|
|
295
|
+
// independently prove plan availability, no separate Hugging Face gate, and tool support.
|
|
296
|
+
allOf: [
|
|
297
|
+
{ path: ["available_on_current_plan"], equalsAny: [true] },
|
|
298
|
+
{ path: ["is_gated"], equalsAny: [false] },
|
|
299
|
+
{ path: ["features", "tool_use"], equalsAny: [true] },
|
|
300
|
+
],
|
|
301
|
+
},
|
|
302
|
+
},
|
|
303
|
+
note: "Authenticated first page of popular chat models only; live discovery admits at most 100 plan-available, ungated rows whose metadata explicitly reports tool use.",
|
|
304
|
+
},
|
|
305
|
+
{
|
|
306
|
+
// Primary sources checked 2026-08-08:
|
|
307
|
+
// - https://novita.ai/docs/api-reference/model-apis-llm-create-chat-completion and
|
|
308
|
+
// https://novita.ai/docs/api-reference/model-apis-llm-list-models document the fixed
|
|
309
|
+
// OpenAI-compatible Chat Completions and model-list endpoints.
|
|
310
|
+
// - https://novita.ai/docs/api-reference/basic-authentication documents Bearer API keys.
|
|
311
|
+
// - https://novita.ai/legal/terms-of-service (updated 2026-08-05) expressly covers AI
|
|
312
|
+
// inference APIs, third-party Model Providers, and customer Input/Output processing.
|
|
313
|
+
// - https://huggingface.co/docs/inference-providers/main/providers/novita lists Novita as an
|
|
314
|
+
// Inference Providers partner for chat/VLM traffic, independently supporting routing use.
|
|
315
|
+
// - https://tsdr.uspto.gov/statusview/sn99255805 is the official use-in-commerce record
|
|
316
|
+
// connecting the NOVITA AI mark to Hivemind Labs, Inc., a Delaware corporation. The mark
|
|
317
|
+
// application is now abandoned; it is cited only as the public operator-identity record.
|
|
318
|
+
// Maintainer: @olddonkey; no affiliation with Novita AI or Hivemind Labs, Inc.
|
|
319
|
+
id: "novita",
|
|
320
|
+
label: "Novita AI",
|
|
321
|
+
baseUrl: "https://api.novita.ai/openai/v1",
|
|
322
|
+
adapter: "openai-chat",
|
|
323
|
+
authKind: "key",
|
|
324
|
+
dashboardUrl: "https://novita.ai/settings/key-management",
|
|
325
|
+
liveModels: true,
|
|
326
|
+
preserveCustomDestination: true,
|
|
327
|
+
// The live catalog is public even though the reference shows an Authorization header, so a
|
|
328
|
+
// successful model fetch cannot prove that a supplied key is valid.
|
|
329
|
+
apiKeyValidation: "unknown",
|
|
330
|
+
// The request reference documents tools but not a provider-wide parallel-tool contract.
|
|
331
|
+
parallelToolCalls: false,
|
|
332
|
+
// Novita exposes model-specific thinking flags, not an OpenAI reasoning_effort contract.
|
|
333
|
+
reasoningEfforts: [],
|
|
334
|
+
modelDiscovery: {
|
|
335
|
+
path: "models",
|
|
336
|
+
maxResponseBytes: 512 * 1024,
|
|
337
|
+
maxModels: 256,
|
|
338
|
+
filter: {
|
|
339
|
+
// Require both Novita's chat classification and the exact configured wire endpoint.
|
|
340
|
+
allOf: [
|
|
341
|
+
{ path: ["model_type"], equalsAny: ["chat"] },
|
|
342
|
+
{ path: ["endpoints"], containsAny: ["chat/completions"] },
|
|
343
|
+
],
|
|
344
|
+
},
|
|
345
|
+
},
|
|
346
|
+
note: "Public live catalog filtered to rows that explicitly report chat type and Chat Completions support; key validity remains unknown until an authenticated inference request.",
|
|
347
|
+
},
|
|
348
|
+
// FREEZE 2026-07-10: exact serverless ids remain auth-gated/unverified. Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
|
|
349
|
+
{ id: "together", label: "Together", baseUrl: "https://api.together.xyz/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://api.together.xyz/settings/api-keys" },
|
|
350
|
+
{ id: "fireworks", label: "Fireworks", baseUrl: "https://api.fireworks.ai/inference/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://fireworks.ai/account/api-keys" },
|
|
351
|
+
{
|
|
352
|
+
id: "firepass", label: "Fire Pass (Fireworks Kimi)", baseUrl: "https://api.fireworks.ai/inference/v1", adapter: "openai-chat", authKind: "key",
|
|
353
|
+
dashboardUrl: "https://fireworks.ai/account/api-keys",
|
|
354
|
+
note: "Model data frozen pending Tier-2 entitlement proof",
|
|
355
|
+
},
|
|
356
|
+
{
|
|
357
|
+
id: "moonshot", label: "Moonshot (Kimi API)", baseUrl: MOONSHOT_INTL_BASE_URL, adapter: "openai-chat", authKind: "key",
|
|
358
|
+
allowBaseUrlOverride: true,
|
|
359
|
+
baseUrlChoices: MOONSHOT_BASE_URL_CHOICES,
|
|
360
|
+
dashboardUrl: "https://platform.moonshot.ai/console/api-keys", defaultModel: "kimi-k2.7-code", jawcodeBundle: "moonshot",
|
|
361
|
+
models: KIMI_API_MODELS,
|
|
362
|
+
modelContextWindows: KIMI_API_MODEL_CONTEXT_WINDOWS,
|
|
363
|
+
modelInputModalities: KIMI_API_MODEL_INPUT_MODALITIES,
|
|
364
|
+
noReasoningModels: KIMI_API_NO_REASONING_MODELS,
|
|
365
|
+
modelReasoningEfforts: KIMI_API_REASONING_EFFORTS,
|
|
366
|
+
noTemperatureModels: KIMI_API_MODELS,
|
|
367
|
+
noTopPModels: KIMI_API_MODELS,
|
|
368
|
+
noPenaltyModels: KIMI_API_MODELS,
|
|
369
|
+
autoToolChoiceOnlyModels: ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
|
|
370
|
+
preserveReasoningContentModels: KIMI_API_MODELS,
|
|
371
|
+
note: "International default (api.moonshot.ai). China accounts: choose China (.cn) or Custom for api.moonshot.cn.",
|
|
372
|
+
},
|
|
373
|
+
{ id: "huggingface", label: "Hugging Face", baseUrl: "https://router.huggingface.co/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://huggingface.co/settings/tokens" },
|
|
374
|
+
// 260715 NIM hardening (issue #126, devlog/_plan/260715_issue126_nim_kimi):
|
|
375
|
+
// - NIM kimi rejects `parallel_tool_calls: true` with 400 "This model only supports single
|
|
376
|
+
// tool-calls at once!" (openclaw#37048). NVIDIA's own function-calling docs default the
|
|
377
|
+
// Boolean to false, so provider-wide `false` is the documented-safe wire value.
|
|
378
|
+
// - `reasoning_effort` is not portable on NIM (models use chat_template_kwargs); the kimi
|
|
379
|
+
// family is live-discovered with no capability metadata, so Codex would otherwise send
|
|
380
|
+
// reasoning_effort=medium. Exact-id lists per modelInList semantics; gpt-oss on NIM keeps
|
|
381
|
+
// its working reasoning_effort. Future kimi ids must be appended individually.
|
|
382
|
+
{
|
|
383
|
+
id: "nvidia", label: "NVIDIA NIM", baseUrl: "https://integrate.api.nvidia.com/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://build.nvidia.com",
|
|
384
|
+
// Free pricing, but an API key is still required (free key from build.nvidia.com).
|
|
385
|
+
freeTier: true,
|
|
386
|
+
parallelToolCalls: false,
|
|
387
|
+
// 260804 issue #956: NIM exposes no input modalities, so vision capability is
|
|
388
|
+
// classified here. Both lists are verified per-model; unlisted ids stay unclassified
|
|
389
|
+
// by design (see the comment on NVIDIA_NIM_VISION_MODELS).
|
|
390
|
+
noVisionModels: NVIDIA_NIM_NO_VISION_MODELS,
|
|
391
|
+
modelInputModalities: NVIDIA_NIM_VISION_INPUT_MODALITIES,
|
|
392
|
+
noReasoningModels: NVIDIA_NIM_KIMI_MODELS,
|
|
393
|
+
modelReasoningEfforts: Object.fromEntries(NVIDIA_NIM_KIMI_MODELS.map(id => [id, []])),
|
|
394
|
+
preserveReasoningContentModels: NVIDIA_NIM_KIMI_THINKING_MODELS,
|
|
395
|
+
note: "Free tier on NVIDIA NIM — API key still required (get a free key at build.nvidia.com).",
|
|
396
|
+
},
|
|
397
|
+
{ id: "venice", label: "Venice", baseUrl: "https://api.venice.ai/api/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://venice.ai/settings/api" },
|
|
398
|
+
// 260710 GLM-5.2 context and path-specific ids: Tier-2 evidence in
|
|
399
|
+
// devlog/_plan/260710_provider_hardening/002_research_cn.md.
|
|
400
|
+
// 260814: glm-5.3 / glm-5.3[1m] added per docs.z.ai/devpack/latest-model, which lists them as
|
|
401
|
+
// Coding Plan ids on this same endpoint.
|
|
402
|
+
// 260815: docs.z.ai/guides/llm/glm-5.3 now publishes the capability table (thinking, streaming,
|
|
403
|
+
// function calling, caching, structured output) and a 128K output budget, recorded here as the
|
|
404
|
+
// exact 131_072 every other source in this repo uses for that model. Coding Plan pricing stays
|
|
405
|
+
// unpublished, so no cost entry is asserted.
|
|
406
|
+
{
|
|
407
|
+
id: "zai", label: "Z.AI — GLM Coding Plan", baseUrl: "https://api.z.ai", adapter: "openai-responses", authKind: "key",
|
|
408
|
+
// One subscription and one key, three protocols. docs.z.ai/guides/llm/glm-5.3 lists them:
|
|
409
|
+
// Chat Completions at /api/coding/paas/v4, Responses at /api/v1, Anthropic Messages at
|
|
410
|
+
// /api/anthropic. docs.z.ai/devpack/latest-model points Codex-family clients at /api/v1,
|
|
411
|
+
// and the Chat path is the one that misbehaves in practice.
|
|
412
|
+
//
|
|
413
|
+
// Responses is the default and Chat stays reachable per model through `modelAdapters`.
|
|
414
|
+
// The two wires sit under different prefixes, and a wire override swaps the adapter
|
|
415
|
+
// without touching baseUrl, so each wire carries its own relative send path.
|
|
416
|
+
//
|
|
417
|
+
// Measured 2026-09-12 against a live key: every roster id answers 200 on
|
|
418
|
+
// /api/v1/responses, and every one also answers 200 on the Chat prefix, so no model
|
|
419
|
+
// needs a `modelWireDefaults` pin. /api/v1/chat/completions returns 403
|
|
420
|
+
// model_access_denied, which is why the Chat path cannot simply hang off the new base.
|
|
421
|
+
responsesPath: "/api/v1/responses",
|
|
422
|
+
chatCompletionsPath: "/api/coding/paas/v4/chat/completions",
|
|
423
|
+
// The address this row occupied before the move. A saved custom provider still pointing
|
|
424
|
+
// at the Chat endpoint keeps receiving this row's metadata (#1100).
|
|
425
|
+
destinationAliases: [{ baseUrl: "https://api.z.ai/api/coding/paas/v4", adapter: "openai-chat" }],
|
|
426
|
+
dashboardUrl: "https://z.ai/manage-apikey/apikey-list", defaultModel: "glm-5.3",
|
|
427
|
+
note: "GLM-5.3 coding subscription",
|
|
428
|
+
models: ["glm-5.3", "glm-5.3[1m]", "glm-5.3-flash", "glm-5.2", "glm-5.2[1m]", "glm-5.1", "glm-5", "glm-4.6"],
|
|
429
|
+
// The upstream catalog reports 1_048_576 for the 5.3 family, which is what the domestic
|
|
430
|
+
// Responses row already carries. Both are documented as "1M"; this is that number.
|
|
431
|
+
modelContextWindows: { "glm-5.3": 1_048_576, "glm-5.3[1m]": 1_048_576, "glm-5.3-flash": 1_048_576, "glm-5.2": 1_000_000, "glm-5.2[1m]": 1_000_000 },
|
|
432
|
+
// Z.AI returns 400 for bracketed model ids on both wires; the aliases are local.
|
|
433
|
+
modelSuffixBracketStrip: true,
|
|
434
|
+
noVisionModels: ZAI_GLM_5X_SIDECAR_VISION_MODELS,
|
|
435
|
+
modelInputModalities: ZAI_GLM_5X_INPUT_MODALITIES,
|
|
436
|
+
modelReasoningEfforts: ZAI_GLM_5X_REASONING_EFFORTS,
|
|
437
|
+
modelDefaultReasoningEfforts: Object.fromEntries(ZAI_GLM_53_MODELS.map(id => [id, "max"])),
|
|
438
|
+
modelMaxOutputTokens: Object.fromEntries(ZAI_GLM_53_MODELS.map(id => [id, 131_072])),
|
|
439
|
+
modelSupportsReasoningSummaries: Object.fromEntries(ZAI_GLM_5X_MODELS.map(id => [id, true])),
|
|
440
|
+
preserveReasoningContentModels: ZAI_GLM_5X_MODELS,
|
|
441
|
+
// Responses replay uses this provider-level flag; the model list above still covers a
|
|
442
|
+
// caller who opts back into Chat.
|
|
443
|
+
preserveResponsesReasoningContent: true,
|
|
444
|
+
},
|
|
445
|
+
// Zhipu's domestic BigModel platform: OpenAI-compatible pay-as-you-go on open.bigmodel.cn — a
|
|
446
|
+
// different host and billing product from the `zai` coding-plan subscription above.
|
|
447
|
+
// The id is deliberately NOT `glm` or `glm-cn`: both are already bound in FREE_PROVIDER_DIRECTORY
|
|
448
|
+
// (to api.z.ai and to the BigModel *coding* path), and routedProviderConfig() canonicalizes a
|
|
449
|
+
// saved provider onto the registry baseUrl — reusing either id would silently retarget an
|
|
450
|
+
// existing config's endpoint and send its API key to another host.
|
|
451
|
+
// Evidence: docs.bigmodel.cn/api-reference (OpenAI-compatible chat completions),
|
|
452
|
+
// docs.bigmodel.cn/cn/guide/models/text/glm-4.6 (thinking: {type: enabled|disabled}).
|
|
453
|
+
// Originally proposed in #536 by @Lucinegogo.
|
|
454
|
+
{
|
|
455
|
+
id: "zhipu-bigmodel",
|
|
456
|
+
label: "Zhipu AI — BigModel",
|
|
457
|
+
baseUrl: "https://open.bigmodel.cn/api/paas/v4",
|
|
458
|
+
adapter: "openai-chat",
|
|
459
|
+
authKind: "key",
|
|
460
|
+
dashboardUrl: "https://bigmodel.cn/console/usercenter/apikeys",
|
|
461
|
+
defaultModel: "glm-4.6",
|
|
462
|
+
models: ZHIPU_BIGMODEL_MODELS,
|
|
463
|
+
// The GLM families here are the same ones the `zai` metadata bundle already describes, so the
|
|
464
|
+
// bundle owns context windows and modalities for the whole list instead of a hand-copied table.
|
|
465
|
+
jawcodeBundle: "zai",
|
|
466
|
+
// Declared explicitly for the default model so its window survives a bundle-lookup miss:
|
|
467
|
+
// without it, catalog normalization falls back to a generic 128k and compacts ~76,800 early.
|
|
468
|
+
modelContextWindows: { "glm-4.6": 204_800 },
|
|
469
|
+
modelInputModalities: ZHIPU_BIGMODEL_INPUT_MODALITIES,
|
|
470
|
+
// GLM exposes a binary thinking knob, not an effort ladder: the adapter emits
|
|
471
|
+
// `thinking: {type}` for these ids and would otherwise send a rejected reasoning_effort.
|
|
472
|
+
thinkingToggleModels: ZHIPU_BIGMODEL_THINKING_TOGGLE_MODELS,
|
|
473
|
+
modelReasoningEfforts: Object.fromEntries(
|
|
474
|
+
ZHIPU_BIGMODEL_THINKING_TOGGLE_MODELS.map(id => [id, THINKING_TOGGLE_EFFORTS]),
|
|
475
|
+
),
|
|
476
|
+
modelReasoningEffortMap: Object.fromEntries(
|
|
477
|
+
ZHIPU_BIGMODEL_THINKING_TOGGLE_MODELS.map(id => [id, THINKING_TOGGLE_MAP]),
|
|
478
|
+
),
|
|
479
|
+
modelSupportsReasoningSummaries: Object.fromEntries(
|
|
480
|
+
ZHIPU_BIGMODEL_THINKING_TOGGLE_MODELS.map(id => [id, true]),
|
|
481
|
+
),
|
|
482
|
+
preserveReasoningContentModels: ZHIPU_BIGMODEL_THINKING_TOGGLE_MODELS,
|
|
483
|
+
// GLM thinking is a binary toggle (low maps to disabled), so a legitimate
|
|
484
|
+
// tool round can carry no reasoning at all; never fabricate a placeholder
|
|
485
|
+
// for it, only replay real recorded text (P2 on #1205).
|
|
486
|
+
requiresReasoningPlaceholderModels: [],
|
|
487
|
+
// No liveModels: GET /api/paas/v4/models has not been observed to answer on this host, and a
|
|
488
|
+
// false live claim yields an empty picker at runtime. Flip it on once someone verifies it.
|
|
489
|
+
note: "Domestic BigModel pay-as-you-go endpoint (open.bigmodel.cn)",
|
|
490
|
+
},
|
|
491
|
+
// BigModel's Coding Plan is a SEPARATE endpoint from the pay-as-you-go row above, and that is
|
|
492
|
+
// the whole reason this one exists. #1100 was reported against
|
|
493
|
+
// `https://open.bigmodel.cn/api/coding/paas/v4`; the row above covers only `/api/paas/v4`, so
|
|
494
|
+
// destination enrichment matched nothing, `modelSupportsReasoningSummaries` stayed unset, and
|
|
495
|
+
// Codex kept dropping the inbound reasoning object — effort displayed as `-`.
|
|
496
|
+
//
|
|
497
|
+
// A prefix or fuzzy endpoint match would have been the shortcut. It is also how a config
|
|
498
|
+
// pointed at one vendor route silently inherits another route's metadata, so endpoints stay
|
|
499
|
+
// exact and each one gets its own row.
|
|
500
|
+
//
|
|
501
|
+
// The id is NOT `glm-cn`, which the free-provider directory already binds to this same coding
|
|
502
|
+
// path: registering it here would let routedProviderConfig() canonicalize a saved `glm-cn`
|
|
503
|
+
// config onto this baseUrl. Same reasoning as `zhipu-bigmodel` above.
|
|
504
|
+
//
|
|
505
|
+
// Models follow Z.AI's coding-plan list rather than the pay-as-you-go one. This endpoint is
|
|
506
|
+
// the subscription product, and the reporter's `glm-5.2` is only on that side.
|
|
507
|
+
{
|
|
508
|
+
id: "zhipu-bigmodel-coding",
|
|
509
|
+
label: "Zhipu AI — BigModel Coding Plan",
|
|
510
|
+
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
|
|
511
|
+
adapter: "openai-chat",
|
|
512
|
+
authKind: "key",
|
|
513
|
+
dashboardUrl: "https://bigmodel.cn/console/usercenter/apikeys",
|
|
514
|
+
defaultModel: "glm-5.3",
|
|
515
|
+
models: ["glm-5.3", "glm-5.3[1m]", "glm-5.3-flash", "glm-5.2", "glm-5.2[1m]", "glm-5.1", "glm-5", "glm-4.6"],
|
|
516
|
+
jawcodeBundle: "zai",
|
|
517
|
+
modelContextWindows: { "glm-5.3": 1_000_000, "glm-5.3[1m]": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000, "glm-5.2[1m]": 1_000_000 },
|
|
518
|
+
modelSuffixBracketStrip: true,
|
|
519
|
+
noVisionModels: ZAI_GLM_5X_SIDECAR_VISION_MODELS,
|
|
520
|
+
modelInputModalities: ZAI_GLM_5X_INPUT_MODALITIES,
|
|
521
|
+
modelReasoningEfforts: ZAI_GLM_5X_REASONING_EFFORTS,
|
|
522
|
+
modelSupportsReasoningSummaries: Object.fromEntries(ZAI_GLM_5X_MODELS.map(id => [id, true])),
|
|
523
|
+
preserveReasoningContentModels: ZAI_GLM_5X_MODELS,
|
|
524
|
+
// No liveModels: the same reasoning as the pay-as-you-go row — an unverified live claim
|
|
525
|
+
// yields an empty picker at runtime.
|
|
526
|
+
note: "Domestic BigModel Coding Plan endpoint (open.bigmodel.cn)",
|
|
527
|
+
},
|
|
528
|
+
// Narrowed carry of #3641: the official Codex example declares a local static catalog,
|
|
529
|
+
// not an HTTP /models contract. Keep Responses separate from the Chat endpoint above.
|
|
530
|
+
// Source: https://docs.bigmodel.cn/cn/coding-plan/tool/codex.md (checked 2026-09-07).
|
|
531
|
+
//
|
|
532
|
+
// #4201 completes the roster. The `models.json` example on that Codex page is a *starter
|
|
533
|
+
// catalog*, not the set of models the endpoint serves, and reading it as the latter is what
|
|
534
|
+
// left Flash off a subscription that sells it. Three upstream pages say so directly, all
|
|
535
|
+
// checked 2026-09-11:
|
|
536
|
+
// - coding-plan/latest-model.md pins Codex to THIS baseUrl
|
|
537
|
+
// (`Codex:https://open.bigmodel.cn/api/v1`) and opens with GLM Coding Plan supporting
|
|
538
|
+
// GLM-5.3 and GLM-5.3-Flash for every tier (Max & Pro & Lite), then treats
|
|
539
|
+
// `glm-5.3-flash` as an already-callable id in that same tool.
|
|
540
|
+
// - coding-plan/overview.md: every plan supports GLM-5.3 and GLM-5.3-Flash, and calls to
|
|
541
|
+
// GLM-5-Turbo are auto-switched to GLM-5.3-Flash. Turbo below is therefore an alias of
|
|
542
|
+
// the very model this row omitted, which is the clearest statement that the endpoint
|
|
543
|
+
// serves Flash: it was already serving it under another name.
|
|
544
|
+
// - guide/models/vlm/glm-5.3-flash.md: native multimodal input, 1M context, and text
|
|
545
|
+
// parameters explicitly "consistent with GLM-5.3".
|
|
546
|
+
// No authenticated /models probe is implied by any of this, so `liveModels` and
|
|
547
|
+
// `apiKeyValidation` below are deliberately unchanged.
|
|
548
|
+
{
|
|
549
|
+
id: "zhipu-bigmodel-responses",
|
|
550
|
+
label: "Zhipu AI — BigModel Coding Plan (Responses)",
|
|
551
|
+
baseUrl: "https://open.bigmodel.cn/api/v1",
|
|
552
|
+
adapter: "openai-responses",
|
|
553
|
+
authKind: "key",
|
|
554
|
+
dashboardUrl: "https://bigmodel.cn/console/usercenter/apikeys",
|
|
555
|
+
defaultModel: "glm-5.3",
|
|
556
|
+
models: ["glm-5.3", "glm-5.3-flash", "glm-5-turbo"],
|
|
557
|
+
liveModels: false,
|
|
558
|
+
// The local Codex catalog does not establish an authenticated HTTP /models contract.
|
|
559
|
+
apiKeyValidation: "unknown",
|
|
560
|
+
jawcodeBundle: "zai",
|
|
561
|
+
// A pre-existing same-named custom provider must retain its destination and key boundary.
|
|
562
|
+
preserveCustomDestination: true,
|
|
563
|
+
// Flash tracks its 5.3 sibling on this row rather than the Chat row's 1_000_000. Both
|
|
564
|
+
// models are documented as "1M", and this preset expresses that family's 1M the way
|
|
565
|
+
// BigModel's own Codex declaration does. Splitting the two would leave one preset
|
|
566
|
+
// claiming two different sizes for one documented window.
|
|
567
|
+
modelContextWindows: { "glm-5.3": 1_048_576, "glm-5.3-flash": 1_048_576, "glm-5-turbo": 204_800 },
|
|
568
|
+
// Flash is the only row here that can actually see an image. Its siblings are declared
|
|
569
|
+
// text-only and get `image` back from the vision sidecar at catalog-build time; declaring
|
|
570
|
+
// Flash text-only would route a native VLM's pictures through a describe-it-first detour
|
|
571
|
+
// and hand the model prose about an image it could have read (same defect
|
|
572
|
+
// ZAI_GLM_5X_SIDECAR_VISION_MODELS exists to prevent on the Chat rows).
|
|
573
|
+
modelInputModalities: { "glm-5.3": ["text"], "glm-5.3-flash": ["text", "image"], "glm-5-turbo": ["text"] },
|
|
574
|
+
modelReasoningEfforts: {
|
|
575
|
+
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
576
|
+
// Same three effective tiers: upstream documents Flash's text parameters as identical
|
|
577
|
+
// to GLM-5.3, and the Codex effort table folds every inbound value into low/high/max.
|
|
578
|
+
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
579
|
+
// Explicitly empty: Turbo must not inherit the generic selectable effort ladder.
|
|
580
|
+
"glm-5-turbo": [],
|
|
581
|
+
},
|
|
582
|
+
modelDefaultReasoningEfforts: { "glm-5.3": "max", "glm-5.3-flash": "max", "glm-5-turbo": "max" },
|
|
583
|
+
modelSupportsReasoningSummaries: { "glm-5.3": true, "glm-5.3-flash": true, "glm-5-turbo": true },
|
|
584
|
+
// Responses replay uses this provider-level flag, not the Chat-path model list.
|
|
585
|
+
preserveResponsesReasoningContent: true,
|
|
586
|
+
note: "Domestic BigModel Coding Plan Responses endpoint; static model roster",
|
|
587
|
+
},
|
|
588
|
+
{ id: "nanogpt", label: "NanoGPT", baseUrl: "https://nano-gpt.com/api/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://nano-gpt.com/api" },
|
|
589
|
+
{ id: "synthetic", label: "Synthetic", baseUrl: "https://api.synthetic.new/openai/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://synthetic.new" },
|
|
590
|
+
// SiliconFlow publishes an OpenAI-compatible chat endpoint and a dynamic model catalog. Do not
|
|
591
|
+
// freeze reasoning controls here: enable_thinking/thinking_budget support and limits vary by
|
|
592
|
+
// model, so live metadata or an explicit user override must own those capabilities.
|
|
593
|
+
// Evidence: https://docs.siliconflow.cn/en/api-reference/chat-completions/chat-completions
|
|
594
|
+
{
|
|
595
|
+
id: "siliconflow",
|
|
596
|
+
label: "SiliconFlow",
|
|
597
|
+
baseUrl: "https://api.siliconflow.cn/v1",
|
|
598
|
+
adapter: "openai-chat",
|
|
599
|
+
authKind: "key",
|
|
600
|
+
dashboardUrl: "https://cloud.siliconflow.cn/account/ak",
|
|
601
|
+
liveModels: true,
|
|
602
|
+
note: "OpenAI-compatible live model catalog; reasoning controls vary by model.",
|
|
603
|
+
},
|
|
604
|
+
// Qwen Cloud: token plan is the preset default; GUI offers pay-as-you-go + custom via baseUrlChoices.
|
|
605
|
+
// Formerly `qwen-portal` / portal.qwen.ai — that host is outdated.
|
|
606
|
+
{
|
|
607
|
+
id: "qwen-cloud",
|
|
608
|
+
label: "Qwen Cloud",
|
|
609
|
+
baseUrl: QWEN_CLOUD_TOKEN_PLAN_BASE_URL,
|
|
610
|
+
adapter: "openai-chat",
|
|
611
|
+
authKind: "key",
|
|
612
|
+
allowBaseUrlOverride: true,
|
|
613
|
+
baseUrlChoices: QWEN_CLOUD_BASE_URL_CHOICES,
|
|
614
|
+
dashboardUrl: "https://docs.qwencloud.com",
|
|
615
|
+
note: "Pick token plan, pay as you go, or a custom compatible-mode base URL",
|
|
616
|
+
},
|
|
617
|
+
{
|
|
618
|
+
id: "tencent-coding-plan",
|
|
619
|
+
label: "Tencent Cloud Coding Plan",
|
|
620
|
+
baseUrl: "https://api.lkeap.cloud.tencent.com/coding/v3",
|
|
621
|
+
adapter: "openai-chat",
|
|
622
|
+
authKind: "key",
|
|
623
|
+
dashboardUrl: "https://console.cloud.tencent.com/tokenhub/codingplan",
|
|
624
|
+
defaultModel: "tc-code-latest",
|
|
625
|
+
models: TENCENT_CODING_PLAN_MODELS,
|
|
626
|
+
liveModels: true,
|
|
627
|
+
modelInputModalities: Object.fromEntries(TENCENT_CODING_PLAN_MODELS.map(id => [id, ["text"]])),
|
|
628
|
+
noVisionModels: TENCENT_CODING_PLAN_MODELS,
|
|
629
|
+
note: "Coding tools only. Tencent forbids general API automation, custom backends, and non-interactive batch use.",
|
|
630
|
+
},
|
|
631
|
+
{
|
|
632
|
+
id: "volcengine",
|
|
633
|
+
label: "Volcengine Ark",
|
|
634
|
+
baseUrl: "https://ark.cn-beijing.volces.com/api/v3",
|
|
635
|
+
adapter: "openai-chat",
|
|
636
|
+
authKind: "key",
|
|
637
|
+
preserveCustomDestination: true,
|
|
638
|
+
dashboardUrl: "https://console.volcengine.com/ark/region:ark+cn-beijing/apikey",
|
|
639
|
+
defaultModel: "doubao-seed-2-1-pro-260628",
|
|
640
|
+
models: VOLCENGINE_ARK_MODELS,
|
|
641
|
+
liveModels: false,
|
|
642
|
+
modelReasoningEfforts: Object.fromEntries(
|
|
643
|
+
VOLCENGINE_DOUBAO_THINKING_MODELS.map(id => [id, THINKING_TOGGLE_EFFORTS]),
|
|
644
|
+
),
|
|
645
|
+
modelReasoningEffortMap: Object.fromEntries(
|
|
646
|
+
VOLCENGINE_DOUBAO_THINKING_MODELS.map(id => [id, THINKING_TOGGLE_MAP]),
|
|
647
|
+
),
|
|
648
|
+
thinkingToggleModels: VOLCENGINE_DOUBAO_THINKING_MODELS,
|
|
649
|
+
preserveReasoningContentModels: [
|
|
650
|
+
"deepseek-v4-flash-260425",
|
|
651
|
+
"glm-5-2-260617",
|
|
652
|
+
"glm-4-7-251222",
|
|
653
|
+
],
|
|
654
|
+
noVisionModels: [
|
|
655
|
+
"deepseek-v4-flash-260425",
|
|
656
|
+
"deepseek-v3-2-251201",
|
|
657
|
+
"glm-5-2-260617",
|
|
658
|
+
"glm-4-7-251222",
|
|
659
|
+
],
|
|
660
|
+
note: "Pay-as-you-go Ark API with a curated text/agent catalog. Calls on this endpoint do not consume Coding Plan or Agent Plan quota.",
|
|
661
|
+
},
|
|
662
|
+
{
|
|
663
|
+
id: "volcengine-coding-plan",
|
|
664
|
+
label: "Volcengine Ark Coding Plan",
|
|
665
|
+
baseUrl: "https://ark.cn-beijing.volces.com/api/coding/v3",
|
|
666
|
+
adapter: "openai-chat",
|
|
667
|
+
authKind: "key",
|
|
668
|
+
preserveCustomDestination: true,
|
|
669
|
+
dashboardUrl: "https://console.volcengine.com/ark/region:ark+cn-beijing/overview",
|
|
670
|
+
defaultModel: "ark-code-latest",
|
|
671
|
+
models: VOLCENGINE_CODING_PLAN_MODELS,
|
|
672
|
+
liveModels: false,
|
|
673
|
+
modelInputModalities: VOLCENGINE_PLAN_INPUT_MODALITIES,
|
|
674
|
+
noVisionModels: VOLCENGINE_PLAN_TEXT_ONLY_MODELS,
|
|
675
|
+
modelReasoningEfforts: Object.fromEntries(
|
|
676
|
+
DEEPSEEK_V4_LEGACY_MODELS.map(id => [id, deepseekThinkingEffortsFor(id)]),
|
|
677
|
+
),
|
|
678
|
+
modelReasoningEffortMap: Object.fromEntries(
|
|
679
|
+
DEEPSEEK_V4_LEGACY_MODELS.map(id => [id, deepseekReasoningMapFor(id)]),
|
|
680
|
+
),
|
|
681
|
+
preserveReasoningContentModels: DEEPSEEK_V4_LEGACY_MODELS,
|
|
682
|
+
note: "Coding tools only. Volcengine restricts Coding Plan quota to supported AI coding tools and warns that using this key for general API calls may suspend the subscription or ban the account. Use the plan key issued by the Ark console.",
|
|
683
|
+
},
|
|
684
|
+
{
|
|
685
|
+
id: "volcengine-agent-plan",
|
|
686
|
+
label: "Volcengine Ark Agent Plan",
|
|
687
|
+
baseUrl: "https://ark.cn-beijing.volces.com/api/plan/v3",
|
|
688
|
+
responsesPath: "/responses",
|
|
689
|
+
adapter: "openai-responses",
|
|
690
|
+
authKind: "key",
|
|
691
|
+
// Ark's plan route does not document `service_tier`; fail closed like DeepSeek.
|
|
692
|
+
supportsServiceTier: false,
|
|
693
|
+
preserveCustomDestination: true,
|
|
694
|
+
dashboardUrl: "https://console.volcengine.com/ark/region:ark+cn-beijing/overview",
|
|
695
|
+
// Was `deepseek-v4-pro` until DeepSeek retired it; the plan roster's other DeepSeek
|
|
696
|
+
// entry takes over so a fresh install still lands on a working default.
|
|
697
|
+
defaultModel: "deepseek-v4-flash",
|
|
698
|
+
models: VOLCENGINE_AGENT_PLAN_MODELS,
|
|
699
|
+
liveModels: false,
|
|
700
|
+
modelInputModalities: VOLCENGINE_PLAN_INPUT_MODALITIES,
|
|
701
|
+
noVisionModels: VOLCENGINE_PLAN_TEXT_ONLY_MODELS,
|
|
702
|
+
note: "Coding tools only. Agent Plan is a subscription endpoint over the native Responses API with a static fallback catalog; Ark plan quota is intended for supported AI coding and agent tools, so avoid using this key as a general-purpose API key.",
|
|
703
|
+
},
|
|
704
|
+
// 2026-07-10: docs unverified; model data frozen. Evidence: devlog/_plan/260710_provider_hardening/002_research_cn.md.
|
|
705
|
+
{ id: "qianfan", label: "Qianfan (Baidu)", baseUrl: "https://qianfan.baidubce.com/v2", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://console.bce.baidu.com/iam/#/iam/apikey/list" },
|
|
706
|
+
// 2026-07-10: docs unverified; model data frozen. Evidence: devlog/_plan/260710_provider_hardening/002_research_cn.md.
|
|
707
|
+
{ id: "alibaba", label: "Alibaba Coding Plan", baseUrl: ALIBABA_CODING_INTL_BASE_URL, adapter: "openai-chat", authKind: "key", allowBaseUrlOverride: true, baseUrlChoices: ALIBABA_CODING_BASE_URL_CHOICES, dashboardUrl: "https://dashscope.console.aliyun.com/apiKey" },
|
|
708
|
+
{
|
|
709
|
+
id: "alibaba-token-plan",
|
|
710
|
+
label: "Alibaba Token Plan (Beijing)",
|
|
711
|
+
baseUrl: "https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1",
|
|
712
|
+
adapter: "openai-chat",
|
|
713
|
+
authKind: "key",
|
|
714
|
+
dashboardUrl: "https://bailian.console.aliyun.com/cn-beijing?tab=plan",
|
|
715
|
+
defaultModel: "qwen3.8-max",
|
|
716
|
+
models: ALIBABA_TOKEN_PLAN_MODELS,
|
|
717
|
+
liveModels: false,
|
|
718
|
+
note: "Token Plan Personal Edition · China (Beijing)",
|
|
719
|
+
modelInputModalities: ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
720
|
+
modelContextWindows: {
|
|
721
|
+
"qwen3.8-max": 983_616, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
|
|
722
|
+
"qwen3.6-flash": 1_000_000, "glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000,
|
|
723
|
+
},
|
|
724
|
+
modelReasoningEfforts: {
|
|
725
|
+
...Object.fromEntries(ALIBABA_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
|
|
726
|
+
"qwen3.8-max": QWEN38_REASONING_EFFORTS,
|
|
727
|
+
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
728
|
+
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
729
|
+
"glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
|
|
730
|
+
},
|
|
731
|
+
modelDefaultReasoningEfforts: { "qwen3.8-max": "xhigh" },
|
|
732
|
+
directReasoningEffortModels: ["qwen3.8-max"],
|
|
733
|
+
thinkingBudgetModels: ALIBABA_TOKEN_PLAN_QWEN_MODELS.filter(id => id !== "qwen3.8-max"),
|
|
734
|
+
preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash"],
|
|
735
|
+
noVisionModels: ["glm-5.3", "glm-5.2"],
|
|
736
|
+
},
|
|
737
|
+
{
|
|
738
|
+
id: "alibaba-token-plan-intl",
|
|
739
|
+
label: "Alibaba Token Plan (International)",
|
|
740
|
+
baseUrl: ALIBABA_INTL_TOKEN_PLAN_BASE_URL,
|
|
741
|
+
adapter: "openai-chat",
|
|
742
|
+
authKind: "key",
|
|
743
|
+
allowBaseUrlOverride: true,
|
|
744
|
+
baseUrlChoices: ALIBABA_INTL_BASE_URL_CHOICES,
|
|
745
|
+
dashboardUrl: "https://modelstudio.console.alibabacloud.com/?tab=api#/api",
|
|
746
|
+
defaultModel: "qwen3.7-max",
|
|
747
|
+
models: ALIBABA_INTL_TOKEN_PLAN_MODELS,
|
|
748
|
+
liveModels: false,
|
|
749
|
+
note: "Token Plan Team Edition · Singapore (ap-southeast-1)",
|
|
750
|
+
metadataModelIdNormalize: "case-insensitive",
|
|
751
|
+
modelInputModalities: ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES,
|
|
752
|
+
modelContextWindows: {
|
|
753
|
+
"qwen3.8-max": 983_616,
|
|
754
|
+
"qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000, "qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
|
|
755
|
+
"deepseek-v4-flash": 1_000_000, "deepseek-v3.2": 131_072,
|
|
756
|
+
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
|
|
757
|
+
"glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000, "glm-5.1": 1_000_000, "glm-5": 1_000_000,
|
|
758
|
+
"MiniMax-M2.5": 204_800,
|
|
759
|
+
},
|
|
760
|
+
modelReasoningEfforts: {
|
|
761
|
+
...Object.fromEntries(ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
|
|
762
|
+
"qwen3.8-max": QWEN38_REASONING_EFFORTS,
|
|
763
|
+
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
764
|
+
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
765
|
+
"glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
|
|
766
|
+
"deepseek-v4-flash": deepseekThinkingEffortsFor("deepseek-v4-flash"),
|
|
767
|
+
},
|
|
768
|
+
modelReasoningEffortMap: {
|
|
769
|
+
"deepseek-v4-flash": deepseekReasoningMapFor("deepseek-v4-flash"),
|
|
770
|
+
},
|
|
771
|
+
directReasoningEffortModels: ["qwen3.8-max"],
|
|
772
|
+
thinkingBudgetModels: ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.filter(id => id !== "qwen3.8-max"),
|
|
773
|
+
preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "deepseek-v4-flash", "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash"],
|
|
774
|
+
noVisionModels: ["deepseek-v4-flash", "deepseek-v3.2", "glm-5.3", "glm-5.2", "glm-5.1", "glm-5", "MiniMax-M2.5"],
|
|
775
|
+
noReasoningModels: ["kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "deepseek-v3.2", "glm-5.1", "glm-5", "MiniMax-M2.5"],
|
|
776
|
+
modelDefaultReasoningEfforts: { "qwen3.8-max": "xhigh" },
|
|
777
|
+
},
|
|
778
|
+
// NEEDS_HUMAN 2026-07-10: kept for config compatibility, but this is a dashboard URL,
|
|
779
|
+
// no /models endpoint is documented, and tools are silently ignored upstream per docs.parallel.ai.
|
|
780
|
+
// Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
|
|
781
|
+
{ id: "parallel", label: "Parallel", baseUrl: "https://platform.parallel.ai", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://platform.parallel.ai" },
|
|
782
|
+
// ZenMux native ids are vendor-namespaced (`<vendor>/<model>`), verified live against
|
|
783
|
+
// https://zenmux.ai/api/v1/models on 2026-07-18. The static seed doubles as the
|
|
784
|
+
// cold-cache decode source for the Codex slug codec (src/providers/slug-codec.ts);
|
|
785
|
+
// live discovery still owns the full catalog.
|
|
786
|
+
{
|
|
787
|
+
id: "zenmux", label: "ZenMux", baseUrl: "https://zenmux.ai/api/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://zenmux.ai",
|
|
788
|
+
models: ["moonshotai/kimi-k3-free", "moonshotai/kimi-k3"],
|
|
789
|
+
},
|
|
790
|
+
{
|
|
791
|
+
id: "litellm", label: "LiteLLM (self-hosted)", baseUrl: "http://localhost:4000/v1", adapter: "openai-chat", authKind: "key",
|
|
792
|
+
dashboardUrl: "https://docs.litellm.ai/docs/proxy/quick_start",
|
|
793
|
+
allowPrivateNetworkByDefault: true,
|
|
794
|
+
allowBaseUrlOverride: true,
|
|
795
|
+
// A self-hosted proxy may legitimately run without a master key.
|
|
796
|
+
keyOptional: true,
|
|
797
|
+
},
|
|
798
|
+
{
|
|
799
|
+
id: "ollama-cloud",
|
|
800
|
+
label: "Ollama Cloud",
|
|
801
|
+
// The upstream /v1 spelling is deliberately unchanged: ollamaNativeChatUrl() normalizes it
|
|
802
|
+
// to /api/chat, and live model discovery declares its own /v1/models path against the origin,
|
|
803
|
+
// so the native transport needs no base-URL edit here or in the free-provider directory.
|
|
804
|
+
baseUrl: "https://ollama.com/v1",
|
|
805
|
+
// The native transport must be declared HERE, not in configuration. routedProviderConfig()
|
|
806
|
+
// overwrites provider.adapter with the registry adapter for every row whose transport
|
|
807
|
+
// matches, so a config-level adapter is silently discarded.
|
|
808
|
+
adapter: "ollama-native",
|
|
809
|
+
authKind: "key",
|
|
810
|
+
dashboardUrl: "https://ollama.com/settings/keys",
|
|
811
|
+
// Live IDs verified 2026-07-10; qwen3-coder:480b retires 2026-07-15.
|
|
812
|
+
models: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "qwen3-coder:480b", "gpt-oss:120b", "kimi-k2.6", "minimax-m3", "qwen3.5:397b", "gemma4:31b"],
|
|
813
|
+
defaultModel: "glm-5.3",
|
|
814
|
+
// Owner-audited exact outage fallback: these current Ollama Cloud GLM-5.3 rows have
|
|
815
|
+
// 1,048,576-token context windows. Live discovery and successful /api/show enrichment keep
|
|
816
|
+
// their existing precedence; these values prevent a failed show from becoming generic.
|
|
817
|
+
modelContextWindows: { "glm-5.3": 1_048_576, "glm-5.3-flash": 1_048_576 },
|
|
818
|
+
noVisionModels: [
|
|
819
|
+
// glm-5.3-flash is absent on purpose: native VLM
|
|
820
|
+
// (docs.z.ai/guides/vlm/glm-5.3-flash), so its images skip the sidecar.
|
|
821
|
+
"glm-5.3", "glm-5.2", "glm-5.1", "glm-5", "glm-4.7",
|
|
822
|
+
"minimax-m2.7", "minimax-m2.5", "minimax-m2.1",
|
|
823
|
+
"nemotron-3-ultra", "nemotron-3-super",
|
|
824
|
+
"deepseek-v4-flash",
|
|
825
|
+
"gpt-oss", "qwen3-coder:480b",
|
|
826
|
+
],
|
|
827
|
+
// Ollama's native chat API has no `text.verbosity` equivalent and the ollama-native adapter
|
|
828
|
+
// never emits one, so a routed row must not inherit the Codex template's verbosity picker.
|
|
829
|
+
// Provider-wide rather than per-model: this catalog is discovery-authoritative, so ids that
|
|
830
|
+
// arrive later from live discovery must opt out too (the live-discovery gap closed by #2578).
|
|
831
|
+
supportsVerbosity: false,
|
|
832
|
+
// Live model discovery: Ollama serves the standard OpenAI-style data[] envelope at /v1/models,
|
|
833
|
+
// so the generic discovery pipeline needs no special-casing. The path is spelled against the
|
|
834
|
+
// ORIGIN (model-discovery resolves a leading-slash path against base.origin). A discovery
|
|
835
|
+
// spec is REQUIRED here: without one the pipeline probes https://ollama.com/models, which
|
|
836
|
+
// 307-redirects to /search and discovery falls back to the configured list.
|
|
837
|
+
modelDiscovery: {
|
|
838
|
+
path: "/v1/models",
|
|
839
|
+
},
|
|
840
|
+
},
|
|
841
|
+
// FREEZE 2026-07-10: codestral-latest is unconfirmed behind auth. Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
|
|
842
|
+
{ id: "mistral", label: "Mistral", baseUrl: "https://api.mistral.ai/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://console.mistral.ai/api-keys", defaultModel: "codestral-latest" },
|
|
843
|
+
{
|
|
844
|
+
id: "minimax", label: "MiniMax — Coding Plan", baseUrl: "https://api.minimax.io/v1", adapter: "openai-chat", authKind: "key",
|
|
845
|
+
dashboardUrl: "https://platform.minimax.io", defaultModel: "MiniMax-M3", models: MINIMAX_MODELS,
|
|
846
|
+
modelContextWindows: MINIMAX_MODEL_CONTEXT_WINDOWS,
|
|
847
|
+
modelReasoningEfforts: { "MiniMax-M3": MINIMAX_M3_REASONING_EFFORTS },
|
|
848
|
+
modelDefaultReasoningEfforts: { "MiniMax-M3": "medium" },
|
|
849
|
+
modelReasoningEffortMap: { "MiniMax-M3": MINIMAX_M3_REASONING_EFFORT_MAP },
|
|
850
|
+
preserveReasoningContentModels: MINIMAX_MODELS,
|
|
851
|
+
// MiniMax-M3 low effort maps to thinking disabled, so a legitimate tool
|
|
852
|
+
// round can carry no reasoning at all; only replay real recorded text,
|
|
853
|
+
// never a fabricated placeholder (chatgpt-codex-connector P2 on #1205).
|
|
854
|
+
requiresReasoningPlaceholderModels: [],
|
|
855
|
+
reasoningSplitModels: MINIMAX_MODELS,
|
|
856
|
+
// With reasoning_split the upstream returns thinking as a structured
|
|
857
|
+
// reasoning_details array (cumulative text snapshots per stream chunk) and
|
|
858
|
+
// requires that array back verbatim on the next turn — a reasoning_content
|
|
859
|
+
// string replay is the native-format pass-back the docs say is unsupported.
|
|
860
|
+
// Evidence: platform.minimax.io/docs/guides/text-m3-function-call and
|
|
861
|
+
// /docs/api-reference/text-openai-api (verified 2026-09-01).
|
|
862
|
+
reasoningDetailsModels: MINIMAX_MODELS,
|
|
863
|
+
thinkingToggleModels: ["MiniMax-M3"],
|
|
864
|
+
jawcodeBundle: "minimax", metadataModelIdNormalize: "case-insensitive", note: "Subscription Key or API Key",
|
|
865
|
+
},
|
|
866
|
+
{
|
|
867
|
+
id: "minimax-cn", label: "MiniMax — Coding Plan (CN)", baseUrl: "https://api.minimaxi.com/v1", adapter: "openai-chat", authKind: "key",
|
|
868
|
+
dashboardUrl: "https://platform.minimaxi.com", defaultModel: "MiniMax-M3", models: MINIMAX_MODELS,
|
|
869
|
+
modelContextWindows: MINIMAX_MODEL_CONTEXT_WINDOWS,
|
|
870
|
+
modelReasoningEfforts: { "MiniMax-M3": MINIMAX_M3_REASONING_EFFORTS },
|
|
871
|
+
modelDefaultReasoningEfforts: { "MiniMax-M3": "medium" },
|
|
872
|
+
modelReasoningEffortMap: { "MiniMax-M3": MINIMAX_M3_REASONING_EFFORT_MAP },
|
|
873
|
+
preserveReasoningContentModels: MINIMAX_MODELS,
|
|
874
|
+
requiresReasoningPlaceholderModels: [],
|
|
875
|
+
reasoningSplitModels: MINIMAX_MODELS,
|
|
876
|
+
reasoningDetailsModels: MINIMAX_MODELS,
|
|
877
|
+
thinkingToggleModels: ["MiniMax-M3"],
|
|
878
|
+
jawcodeBundle: "minimax", metadataModelIdNormalize: "case-insensitive", note: "中国区 Subscription Key",
|
|
879
|
+
},
|
|
880
|
+
{
|
|
881
|
+
id: "kimi-code", label: "Kimi (coding)", baseUrl: "https://api.kimi.com/coding/v1", adapter: "openai-chat", authKind: "key",
|
|
882
|
+
dashboardUrl: "https://platform.moonshot.cn/console/api-keys", defaultModel: "kimi-k2.7-code",
|
|
883
|
+
modelSuffixBracketStrip: true,
|
|
884
|
+
// API-key form of the same Kimi Code Plan transport; keep cache affinity identical to OAuth.
|
|
885
|
+
promptCacheKey: true,
|
|
886
|
+
models: KIMI_CODING_MODELS,
|
|
887
|
+
modelContextWindows: KIMI_CODING_MODEL_CONTEXT_WINDOWS,
|
|
888
|
+
modelInputModalities: KIMI_CODING_MODEL_INPUT_MODALITIES,
|
|
889
|
+
noReasoningModels: KIMI_CODING_NO_REASONING_MODELS,
|
|
890
|
+
modelReasoningEfforts: KIMI_CODING_REASONING_EFFORTS,
|
|
891
|
+
modelDefaultReasoningEfforts: KIMI_CODING_DEFAULT_REASONING_EFFORTS,
|
|
892
|
+
modelReasoningEffortMap: KIMI_CODING_REASONING_EFFORT_MAPS,
|
|
893
|
+
noTemperatureModels: KIMI_LOCKED_PARAMETER_MODELS,
|
|
894
|
+
noTopPModels: KIMI_LOCKED_PARAMETER_MODELS,
|
|
895
|
+
noPenaltyModels: KIMI_LOCKED_PARAMETER_MODELS,
|
|
896
|
+
autoToolChoiceOnlyModels: KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS,
|
|
897
|
+
preserveReasoningContentModels: KIMI_THINKING_MODELS,
|
|
898
|
+
},
|
|
899
|
+
{
|
|
900
|
+
id: "opencode-zen", label: "opencode zen", baseUrl: "https://opencode.ai/zen/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://opencode.ai/auth",
|
|
901
|
+
// Same opencode.ai/zen/v1 gateway as `opencode-free` (keyed tier): DeepSeek thinking mode
|
|
902
|
+
// requires the assistant's original reasoning_content to be replayed on tool-call
|
|
903
|
+
// continuations, or the gateway answers HTTP 400 (issues #950/#994). Mirror the DeepSeek
|
|
904
|
+
// reasoning + thinking metadata so `opencode-zen/deepseek-v4-flash-free` — and the other
|
|
905
|
+
// Zen DeepSeek thinking models — never serialize a bare tool-call turn.
|
|
906
|
+
note: "Keyed OpenCode Zen gateway. Free models on this tier are often short-window rate-limited at roughly 15-20 requests/minute (community-measured; OpenCode does not publish RPM). Zen may return generic 429s without Retry-After / X-RateLimit headers; when Retry-After is omitted, opencodex adds a synthetic backoff hint (upstream Retry-After still wins). Distinct from the keyless opencode-free desktop quota (~200 Big Pickle/free-model requests per 5 hours). Docs: https://opencode.ai/docs/zen/. Free-model prompts may be retained for training — do not send confidential material.",
|
|
907
|
+
modelReasoningEfforts: Object.fromEntries(
|
|
908
|
+
[...DEEPSEEK_GATEWAY_THINKING_MODELS, ...OPENCODE_FREE_DEEPSEEK_MODELS].map(id => [id, deepseekThinkingEffortsFor(id)]),
|
|
909
|
+
),
|
|
910
|
+
modelReasoningEffortMap: Object.fromEntries(
|
|
911
|
+
[...DEEPSEEK_GATEWAY_THINKING_MODELS, ...OPENCODE_FREE_DEEPSEEK_MODELS].map(id => [id, deepseekReasoningMapFor(id)]),
|
|
912
|
+
),
|
|
913
|
+
preserveReasoningContentModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS, ...OPENCODE_FREE_DEEPSEEK_MODELS],
|
|
914
|
+
// Same Zen gateway as opencode-free: the DeepSeek vision preview id
|
|
915
|
+
// (merges into deepseek-v4-flash later).
|
|
916
|
+
modelContextWindows: {
|
|
917
|
+
[DEEPSEEK_VISION_PREVIEW_MODEL]: 1_048_576,
|
|
918
|
+
},
|
|
919
|
+
modelInputModalities: {
|
|
920
|
+
[DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
|
|
921
|
+
...Object.fromEntries(OPENCODE_ZEN_IMAGE_MODELS.map(id => [id, ["text", "image"] as string[]])),
|
|
922
|
+
},
|
|
923
|
+
noVisionModels: [...OPENCODE_ZEN_TEXT_ONLY_MODELS, ...DEEPSEEK_GATEWAY_THINKING_MODELS],
|
|
924
|
+
// Same DeepSeek routes as the Go preset above, behind the same vendor, so they carry
|
|
925
|
+
// the same json_schema rejection (#1338 / #1415).
|
|
926
|
+
noJsonSchemaModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS, ...OPENCODE_FREE_DEEPSEEK_MODELS],
|
|
927
|
+
},
|
|
928
|
+
{ id: "vercel-ai-gateway", label: "Vercel AI Gateway", baseUrl: "https://ai-gateway.vercel.sh/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://vercel.com/dashboard" },
|
|
929
|
+
{
|
|
930
|
+
id: "opencode-free",
|
|
931
|
+
label: "OpenCode Free",
|
|
932
|
+
adapter: "openai-chat",
|
|
933
|
+
baseUrl: "https://opencode.ai/zen/v1",
|
|
934
|
+
authKind: "key",
|
|
935
|
+
keyOptional: true,
|
|
936
|
+
featured: true,
|
|
937
|
+
liveModels: true,
|
|
938
|
+
note: "No key needed, but OpenCode now gates this tier to its own client: Zen refuses any request that arrives without an x-opencode-session header (error type MissingSessionID, \"OpenCode's free tier can only be used in OpenCode\"). opencodex does not mint that header or claim an OpenCode client identity, because no upstream contract authorizes a third-party agent to present itself as OpenCode. Until OpenCode publishes a third-party integration path for the keyless tier, use the keyed opencode-zen provider instead (https://opencode.ai/auth). Quota figures for when the tier admitted a request: OpenCode advertises about 200 Big Pickle/free-model requests per 5 hours, and the same Zen gateway can short-window rate-limit free models at roughly 15-20 requests/minute, and may return generic 429s without Retry-After (opencodex synthesizes backoff only when that header is omitted). Free models are discovered live from Zen. Data use: per OpenCode's Zen docs (https://opencode.ai/docs/zen/), prompts sent to free models may be retained and used for training/improvement — do not send confidential material through this provider.",
|
|
939
|
+
dashboardUrl: "https://opencode.ai",
|
|
940
|
+
staticHeaders: {
|
|
941
|
+
// Zen answers a bare runtime User-Agent (Bun/x.y.z) more aggressively than a client
|
|
942
|
+
// that identifies itself, which is what the 429 in #2067 traced to. The value is
|
|
943
|
+
// deliberately unversioned: a pinned "opencode-cli/<version>" is a claim about an
|
|
944
|
+
// install we do not have and goes stale on the vendor's schedule, not ours.
|
|
945
|
+
// Corroboration, not authority: OmniRoute — an independent open-source broker against
|
|
946
|
+
// the same Zen upstream — defaults to exactly this pair (userAgent "opencode", client
|
|
947
|
+
// "desktop") in open-sse/executors/opencode.ts, and got there by RETREATING from its
|
|
948
|
+
// own earlier "opencode-cli/1.0.0" pin. An operator can still override either value
|
|
949
|
+
// through the provider headers API; user headers win case-insensitively at route time.
|
|
950
|
+
"User-Agent": "opencode",
|
|
951
|
+
"x-opencode-client": "desktop",
|
|
952
|
+
},
|
|
953
|
+
modelReasoningEfforts: Object.fromEntries(OPENCODE_FREE_DEEPSEEK_MODELS.map(id => [id, deepseekThinkingEffortsFor(id)])),
|
|
954
|
+
modelReasoningEffortMap: Object.fromEntries(OPENCODE_FREE_DEEPSEEK_MODELS.map(id => [id, deepseekReasoningMapFor(id)])),
|
|
955
|
+
preserveReasoningContentModels: OPENCODE_FREE_DEEPSEEK_MODELS,
|
|
956
|
+
// The DeepSeek vision preview id is preemptive metadata for when Zen starts
|
|
957
|
+
// serving it (merges into v4-flash later).
|
|
958
|
+
modelContextWindows: {
|
|
959
|
+
[DEEPSEEK_VISION_PREVIEW_MODEL]: 1_048_576,
|
|
960
|
+
},
|
|
961
|
+
modelInputModalities: {
|
|
962
|
+
[DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
|
|
963
|
+
...Object.fromEntries(OPENCODE_ZEN_IMAGE_MODELS.map(id => [id, ["text", "image"] as string[]])),
|
|
964
|
+
},
|
|
965
|
+
// Same Zen roster behind the same base URL, so it carries the same measured
|
|
966
|
+
// text-only list rather than only its DeepSeek member (#1043).
|
|
967
|
+
noVisionModels: OPENCODE_ZEN_TEXT_ONLY_MODELS,
|
|
968
|
+
// Same reasoning: the free tier is the same Zen roster, so its DeepSeek members get
|
|
969
|
+
// the keyed tier's json_schema treatment and its reasoning contract rather than a
|
|
970
|
+
// narrower table that silently falls behind whenever the keyed one is updated.
|
|
971
|
+
noJsonSchemaModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS, ...OPENCODE_FREE_DEEPSEEK_MODELS],
|
|
972
|
+
},
|
|
973
|
+
{ id: "xiaomi", label: "Xiaomi MiMo", baseUrl: "https://api.xiaomimimo.com/anthropic", adapter: "anthropic", authKind: "key", dashboardUrl: "https://xiaomimimo.com", defaultModel: "mimo-v2.5-pro" },
|
|
974
|
+
// Xiaomi's public OpenAI-compatible endpoint is a distinct transport from both the Anthropic
|
|
975
|
+
// preset above and the paid token-plan host below. Keep a separate fixed-destination contract
|
|
976
|
+
// so existing custom providers are never retargeted while the official route receives the
|
|
977
|
+
// strict reasoning ladder its validator enforces (#1483).
|
|
978
|
+
{
|
|
979
|
+
id: "xiaomi-mimo",
|
|
980
|
+
label: "Xiaomi MiMo (OpenAI Chat)",
|
|
981
|
+
baseUrl: "https://api.xiaomimimo.com/v1",
|
|
982
|
+
adapter: "openai-chat",
|
|
983
|
+
authKind: "key",
|
|
984
|
+
dashboardUrl: "https://platform.xiaomimimo.com/console/balance",
|
|
985
|
+
defaultModel: "mimo-v2.5",
|
|
986
|
+
models: ["mimo-v2.5"],
|
|
987
|
+
reasoningEfforts: ["low", "medium", "high"],
|
|
988
|
+
reasoningEffortMap: { xhigh: "high", max: "high", ultra: "high" },
|
|
989
|
+
preserveCustomDestination: true,
|
|
990
|
+
note: "Official Xiaomi MiMo OpenAI-compatible Chat endpoint. The upstream validator accepts reasoning_effort none/low/medium/high; higher Codex tiers are clamped to high.",
|
|
991
|
+
},
|
|
992
|
+
{ id: "kilo", label: "Kilo", baseUrl: "https://api.kilo.ai/api/gateway", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://kilo.ai" },
|
|
993
|
+
{
|
|
994
|
+
id: "mimo-free",
|
|
995
|
+
label: "MiMo Free",
|
|
996
|
+
adapter: "mimo-free",
|
|
997
|
+
baseUrl: "https://api.xiaomimimo.com/api/free-ai/openai/chat",
|
|
998
|
+
authKind: "key",
|
|
999
|
+
keyOptional: true,
|
|
1000
|
+
featured: true,
|
|
1001
|
+
liveModels: true,
|
|
1002
|
+
dashboardUrl: "https://xiaomimimo.com",
|
|
1003
|
+
defaultModel: "mimo-auto",
|
|
1004
|
+
models: ["mimo-auto"],
|
|
1005
|
+
reasoningEfforts: ["low", "medium", "high"],
|
|
1006
|
+
reasoningEffortMap: { xhigh: "high", max: "high", ultra: "high" },
|
|
1007
|
+
note: "No key needed — uses Xiaomi MiMo's free public tier (limited-time offer). A JWT is bootstrapped automatically with an anonymous random client id stored locally. The endpoint contract mirrors the official MiMoCode client and is not publicly documented — Xiaomi may change or restrict it at any time. Prompts may be processed/retained by Xiaomi; do not send confidential material.",
|
|
1008
|
+
},
|
|
1009
|
+
// Xiaomi MiMo paid token plan. Separate host and wire from both `xiaomi` (Anthropic) and
|
|
1010
|
+
// `mimo-free` (free tier, bespoke adapter), so it needs its own entry rather than a variant.
|
|
1011
|
+
//
|
|
1012
|
+
// Pinned to openai-chat deliberately (#1158). The endpoint answers the Responses wire for
|
|
1013
|
+
// plain turns, which is why users configuring it by hand pick `openai-responses` — MiMo
|
|
1014
|
+
// documents Responses support. But its gateway rejects `type: "custom"` tools with
|
|
1015
|
+
// `400 responses_feature_not_supported`, and `apply_patch` is a custom tool, so every agentic
|
|
1016
|
+
// turn fails while chat turns succeed. The Chat path lowers custom tools to `{input: string}`
|
|
1017
|
+
// functions and restores them as `custom_tool_call`, so the capability survives intact.
|
|
1018
|
+
// Stripping the tools instead would stop the 400 and disable the agent loop.
|
|
1019
|
+
{
|
|
1020
|
+
id: "mimo",
|
|
1021
|
+
label: "Xiaomi MiMo (token plan)",
|
|
1022
|
+
baseUrl: "https://token-plan-cn.xiaomimimo.com/v1",
|
|
1023
|
+
adapter: "openai-chat",
|
|
1024
|
+
authKind: "key",
|
|
1025
|
+
dashboardUrl: "https://xiaomimimo.com",
|
|
1026
|
+
defaultModel: "mimo-v2.5-pro",
|
|
1027
|
+
models: ["mimo-v2.5-pro", "mimo-v2.5"],
|
|
1028
|
+
// The gateway validates the ladder strictly and rejects anything above `high`.
|
|
1029
|
+
reasoningEfforts: ["low", "medium", "high"],
|
|
1030
|
+
reasoningEffortMap: { xhigh: "high", max: "high", ultra: "high" },
|
|
1031
|
+
// Live token-plan verification (#1927): the Pro route rejects image input while
|
|
1032
|
+
// mimo-v2.5 accepts it natively. Keep this provider-scoped so a hand-rolled
|
|
1033
|
+
// provider with the same id but another destination does not inherit the claim.
|
|
1034
|
+
noVisionModels: ["mimo-v2.5-pro"],
|
|
1035
|
+
// A user may already have hand-rolled a provider under this id against a different host;
|
|
1036
|
+
// without this, routedProviderConfig() would canonicalize their base URL onto ours and send
|
|
1037
|
+
// their key somewhere they did not choose.
|
|
1038
|
+
preserveCustomDestination: true,
|
|
1039
|
+
note: "Xiaomi MiMo paid token plan. Pinned to the Chat wire: the Responses endpoint rejects freeform (custom) tools such as apply_patch with 400 responses_feature_not_supported, so agentic turns fail there while plain turns succeed. Reasoning tiers above high are clamped.",
|
|
1040
|
+
},
|
|
1041
|
+
{ id: "cloudflare-ai-gateway", label: "Cloudflare AI Gateway", baseUrl: "https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic", adapter: "anthropic", authKind: "key", dashboardUrl: "https://dash.cloudflare.com/?to=/:account/ai/ai-gateway" },
|
|
1042
|
+
{
|
|
1043
|
+
// Cloudflare Workers AI: OpenAI-compatible endpoint. The base URL contains {account_id}
|
|
1044
|
+
// which must be resolved by the user at setup time. Model IDs use the @cf/ prefix.
|
|
1045
|
+
// Live-verified 2026-07-21 against https://developers.cloudflare.com/workers-ai/models/
|
|
1046
|
+
// Official search is sibling to /ai/v1 (GET .../ai/models/search?format=openrouter).
|
|
1047
|
+
id: "cloudflare-workers-ai", label: "Cloudflare Workers AI",
|
|
1048
|
+
baseUrl: "https://api.cloudflare.com/client/v4/accounts/{account_id}/ai/v1",
|
|
1049
|
+
adapter: "openai-chat", authKind: "key", freeTier: true,
|
|
1050
|
+
dashboardUrl: "https://dash.cloudflare.com/?to=/:account/ai/workers-ai",
|
|
1051
|
+
defaultModel: "@cf/meta/llama-3.3-70b-instruct-fp8-fast",
|
|
1052
|
+
models: [
|
|
1053
|
+
"@cf/meta/llama-3.3-70b-instruct-fp8-fast",
|
|
1054
|
+
"@cf/qwen/qwq-32b",
|
|
1055
|
+
"@cf/deepseek-ai/deepseek-r1-distill-qwen-32b",
|
|
1056
|
+
"@cf/moonshotai/kimi-k2.7-code",
|
|
1057
|
+
"@cf/zai-org/glm-5.3",
|
|
1058
|
+
"@cf/zai-org/glm-5.3-flash",
|
|
1059
|
+
"@cf/zai-org/glm-5.2",
|
|
1060
|
+
"@cf/mistralai/mistral-small-3.1-24b-instruct",
|
|
1061
|
+
],
|
|
1062
|
+
liveModels: true,
|
|
1063
|
+
modelDiscovery: {
|
|
1064
|
+
path: "../models/search",
|
|
1065
|
+
query: { format: "openrouter", per_page: "1000" },
|
|
1066
|
+
stripIdPrefix: "workers-ai/",
|
|
1067
|
+
maxModels: 256,
|
|
1068
|
+
},
|
|
1069
|
+
note: "Workers AI · Free tier included · Account ID required in base URL",
|
|
1070
|
+
},
|
|
1071
|
+
// FREEZE 2026-07-10: /models was auth-gated under key login. OAuth device-flow + copilot_internal
|
|
1072
|
+
// exchange (issue #151) unlocks live discovery; static seed is a cold-start fallback only.
|
|
1073
|
+
{
|
|
1074
|
+
id: "github-copilot",
|
|
1075
|
+
label: "GitHub Copilot",
|
|
1076
|
+
baseUrl: "https://api.githubcopilot.com",
|
|
1077
|
+
adapter: "openai-chat",
|
|
1078
|
+
authKind: "oauth",
|
|
1079
|
+
allowKeyAuthOverride: true,
|
|
1080
|
+
featured: false,
|
|
1081
|
+
dashboardUrl: "https://github.com/settings/copilot",
|
|
1082
|
+
liveModels: true,
|
|
1083
|
+
models: ["gpt-4o", "gpt-4.1", "gpt-4.1-mini", "claude-sonnet-4", "gemini-2.5-pro", "gpt-5-mini", "gpt-5.3-codex", "gpt-5.4", "gpt-5.4-mini", "gpt-5.5", "gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.6-terra"],
|
|
1084
|
+
defaultModel: "gpt-4o",
|
|
1085
|
+
// Copilot fronts a mixed-wire catalog: these models reject /chat/completions for
|
|
1086
|
+
// real Codex-agent traffic (function tools + reasoning), so every inbound wire
|
|
1087
|
+
// rides Responses. Evidence: issue #748 field runs, pi.dev/models/github-copilot/*
|
|
1088
|
+
// wire declarations, BerriAI/litellm#23332 (gpt-5.4), JetBrains LLM-29711
|
|
1089
|
+
// (gpt-5.6-sol). gpt-5.4-nano is deliberately absent — it has no field report; a
|
|
1090
|
+
// user can opt it in with an explicit modelAdapters entry, which always wins.
|
|
1091
|
+
modelWireDefaults: {
|
|
1092
|
+
"gpt-5.3-codex": "openai-responses",
|
|
1093
|
+
"gpt-5.4": "openai-responses",
|
|
1094
|
+
"gpt-5.4-mini": "openai-responses",
|
|
1095
|
+
"gpt-5.5": "openai-responses",
|
|
1096
|
+
"gpt-5.6-luna": "openai-responses",
|
|
1097
|
+
"gpt-5.6-sol": "openai-responses",
|
|
1098
|
+
"gpt-5.6-terra": "openai-responses",
|
|
1099
|
+
"gpt-6-astra": "openai-responses",
|
|
1100
|
+
"grok-4.5": "openai-responses",
|
|
1101
|
+
"grok-4.6": "openai-responses",
|
|
1102
|
+
"mai-code-1.1-flash": "openai-responses",
|
|
1103
|
+
"mai-code-1-flash-picker": "openai-responses",
|
|
1104
|
+
},
|
|
1105
|
+
note: "Experimental unofficial Copilot bridge. Logs in via GitHub device flow using the public VS Code OAuth client id, then exchanges for a short-lived Copilot API token (copilot_internal). Requires an active Copilot subscription. GitHub may tighten or revoke this path; do not send confidential material you would not paste into Copilot Chat.",
|
|
1106
|
+
},
|
|
1107
|
+
// FREEZE 2026-07-10: no public OpenAI-compatible endpoint is documented. Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
|
|
1108
|
+
{ id: "gitlab-duo", label: "GitLab Duo", baseUrl: "https://cloud.gitlab.com/ai/v1/proxy/openai/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://gitlab.com/-/user_settings/personal_access_tokens" },
|
|
1109
|
+
{
|
|
1110
|
+
// Official Qoder Global CLI automation surface. The canonical URL is an identity boundary;
|
|
1111
|
+
// inference and model discovery are performed only by the installed vendor CLI. Authentication
|
|
1112
|
+
// uses the documented PAT environment variable and never imports desktop/session credentials.
|
|
1113
|
+
id: "qoder",
|
|
1114
|
+
label: "Qoder (Global)",
|
|
1115
|
+
adapter: "qoder",
|
|
1116
|
+
baseUrl: "https://qoder.com",
|
|
1117
|
+
authKind: "key",
|
|
1118
|
+
apiKeyValidation: "unknown",
|
|
1119
|
+
preserveCustomDestination: true,
|
|
1120
|
+
dashboardUrl: "https://qoder.com/account/integrations",
|
|
1121
|
+
defaultModel: "Qwen3.8-Max",
|
|
1122
|
+
models: [...QODER_GLOBAL_MODELS],
|
|
1123
|
+
liveModels: true,
|
|
1124
|
+
reasoningEfforts: [...QODER_REASONING_EFFORTS],
|
|
1125
|
+
noVisionModels: [...QODER_GLOBAL_MODELS],
|
|
1126
|
+
note: "Official Qoder Global CLI using QODER_PERSONAL_ACCESS_TOKEN. Models are discovered per account with `qoder --list-models`; the documented roster is a degraded fallback. The CLI runs single-turn with tools, MCP, settings hooks, and session persistence disabled. Requires `npm install -g @qoder-ai/qodercli`.",
|
|
1127
|
+
},
|
|
1128
|
+
{
|
|
1129
|
+
// Qoder CN is a separate credential, executable, destination, entitlement cache, and health
|
|
1130
|
+
// domain. It deliberately does not reuse the OAuth/private-protocol design from #3010.
|
|
1131
|
+
id: "qoder-cn",
|
|
1132
|
+
label: "Qoder CN",
|
|
1133
|
+
adapter: "qoder",
|
|
1134
|
+
baseUrl: "https://qoder.cn",
|
|
1135
|
+
authKind: "key",
|
|
1136
|
+
apiKeyValidation: "unknown",
|
|
1137
|
+
preserveCustomDestination: true,
|
|
1138
|
+
dashboardUrl: "https://qoder.cn/account/integrations",
|
|
1139
|
+
defaultModel: "Qwen3.8-Max",
|
|
1140
|
+
models: [...QODER_CN_MODELS],
|
|
1141
|
+
liveModels: true,
|
|
1142
|
+
reasoningEfforts: [...QODER_REASONING_EFFORTS],
|
|
1143
|
+
noVisionModels: [...QODER_CN_MODELS],
|
|
1144
|
+
note: "Official Qoder CN CLI using QODERCN_PERSONAL_ACCESS_TOKEN. Models are discovered per account with `qodercn --list-models`; the verified roster is a degraded fallback. The CLI runs single-turn with tools, MCP, settings hooks, and session persistence disabled. Requires `npm install -g @qodercn-ai/qoderclicn`.",
|
|
1145
|
+
},
|
|
1146
|
+
{
|
|
1147
|
+
// Official CodeBuddy Code CLI provider (Tencent Cloud), GLOBAL / `public` environment.
|
|
1148
|
+
// Transport is the vendor-documented headless CLI automation surface
|
|
1149
|
+
// (`codebuddy -p --output-format stream-json --tools ""`) authenticated with the official
|
|
1150
|
+
// `CODEBUDDY_API_KEY` (https://www.codebuddy.ai/profile/keys). It does NOT read desktop
|
|
1151
|
+
// session files, import desktop bearer tokens, impersonate the desktop client, or call the
|
|
1152
|
+
// private console endpoint — the approach closed in #687 and left in draft in #2244.
|
|
1153
|
+
// baseUrl is the canonical region identity: the adapter fails closed if it is overridden, so a
|
|
1154
|
+
// global key is never sent to the CN environment (that is the separate `codebuddy-cn` entry).
|
|
1155
|
+
// v1 runs tools-disabled so Codex keeps tool ownership; this provider is text/reasoning only
|
|
1156
|
+
// until the control-protocol tool bridge lands (see docs). Free/trial/promotional/subscription
|
|
1157
|
+
// credits draw from the same official API-key pool. Requires the CLI: `npm i -g @tencent-ai/codebuddy-code`.
|
|
1158
|
+
// GOVERNANCE: whether routing this vendor automation surface behind a proxy for a third-party
|
|
1159
|
+
// agent satisfies CodeBuddy's AUP is an open question flagged for maintainer security review.
|
|
1160
|
+
id: "codebuddy",
|
|
1161
|
+
label: "CodeBuddy (Global)",
|
|
1162
|
+
adapter: "codebuddy",
|
|
1163
|
+
baseUrl: "https://www.codebuddy.ai",
|
|
1164
|
+
authKind: "key",
|
|
1165
|
+
apiKeyValidation: "unknown",
|
|
1166
|
+
preserveCustomDestination: true,
|
|
1167
|
+
dashboardUrl: "https://www.codebuddy.ai/profile/keys",
|
|
1168
|
+
defaultModel: "default-model",
|
|
1169
|
+
models: CODEBUDDY_GLOBAL_MODELS,
|
|
1170
|
+
liveModels: false,
|
|
1171
|
+
modelContextWindows: CODEBUDDY_GLOBAL_MODEL_CONTEXT_WINDOWS,
|
|
1172
|
+
modelMaxOutputTokens: CODEBUDDY_GLOBAL_MODEL_MAX_OUTPUT_TOKENS,
|
|
1173
|
+
defaultMaxOutputTokens: 32_000,
|
|
1174
|
+
reasoningEfforts: CODEBUDDY_REASONING_EFFORTS,
|
|
1175
|
+
modelReasoningEfforts: CODEBUDDY_GLOBAL_MODEL_REASONING_EFFORTS,
|
|
1176
|
+
modelDefaultReasoningEfforts: CODEBUDDY_GLOBAL_MODEL_DEFAULT_REASONING_EFFORTS,
|
|
1177
|
+
note: "Official CodeBuddy Code CLI (Tencent Cloud), global/public environment. Uses the documented CODEBUDDY_API_KEY + headless CLI surface; never reads desktop sessions or private console endpoints. Region-isolated from codebuddy-cn. v1 disables CLI tools (--tools \"\") so Codex retains tool ownership: text/reasoning only for now. Requires `npm i -g @tencent-ai/codebuddy-code`. AUP/routing authorization flagged for maintainer security review.",
|
|
1178
|
+
},
|
|
1179
|
+
{
|
|
1180
|
+
// Official CodeBuddy Code CLI provider, CHINA / `internal` environment. Identical adapter and
|
|
1181
|
+
// binary as `codebuddy`; the region is fixed by the profile's CODEBUDDY_INTERNET_ENVIRONMENT
|
|
1182
|
+
// and this canonical baseUrl. CN key: https://copilot.tencent.com/profile/keys. The CN model
|
|
1183
|
+
// roster differs from Global (see codebuddy-models.ts) and is seeded separately (§八).
|
|
1184
|
+
id: "codebuddy-cn",
|
|
1185
|
+
label: "CodeBuddy (CN)",
|
|
1186
|
+
adapter: "codebuddy",
|
|
1187
|
+
baseUrl: "https://www.codebuddy.cn",
|
|
1188
|
+
authKind: "key",
|
|
1189
|
+
apiKeyValidation: "unknown",
|
|
1190
|
+
preserveCustomDestination: true,
|
|
1191
|
+
dashboardUrl: "https://copilot.tencent.com/profile/keys",
|
|
1192
|
+
defaultModel: "default",
|
|
1193
|
+
models: CODEBUDDY_CN_MODELS,
|
|
1194
|
+
liveModels: false,
|
|
1195
|
+
modelContextWindows: CODEBUDDY_CN_MODEL_CONTEXT_WINDOWS,
|
|
1196
|
+
modelMaxOutputTokens: CODEBUDDY_CN_MODEL_MAX_OUTPUT_TOKENS,
|
|
1197
|
+
defaultMaxOutputTokens: 32_000,
|
|
1198
|
+
reasoningEfforts: CODEBUDDY_REASONING_EFFORTS,
|
|
1199
|
+
modelReasoningEfforts: CODEBUDDY_CN_MODEL_REASONING_EFFORTS,
|
|
1200
|
+
modelDefaultReasoningEfforts: CODEBUDDY_CN_MODEL_DEFAULT_REASONING_EFFORTS,
|
|
1201
|
+
noVisionModels: CODEBUDDY_CN_NO_VISION_MODELS,
|
|
1202
|
+
note: "Official CodeBuddy Code CLI (Tencent Cloud), China/internal environment. Uses the documented CODEBUDDY_API_KEY + headless CLI surface; never reads desktop sessions or private console endpoints. Region-isolated from codebuddy (Global); credentials are never exchanged across regions. v1 disables CLI tools (--tools \"\"): text/reasoning only for now. Requires `npm i -g @tencent-ai/codebuddy-code`. AUP/routing authorization flagged for maintainer security review.",
|
|
1203
|
+
},
|
|
1204
|
+
];
|