@bitkyc08/opencodex 2.55.0 → 2.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/ocx.mjs +10 -0
- package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
- package/gui/dist/assets/{index-VuoiWj9J.js → index-Cz7CLdif.js} +21 -21
- package/gui/dist/index.html +2 -2
- package/package.json +4 -3
- package/src/adapters/base.ts +21 -0
- package/src/adapters/codebuddy/adapter.ts +2 -1
- package/src/adapters/codebuddy/scaffold-guard.ts +248 -0
- package/src/adapters/command-code.ts +1 -1
- package/src/adapters/cursor/envelope-echo.ts +8 -2
- package/src/adapters/cursor/transport-retry.ts +46 -1
- package/src/adapters/cursor.ts +4 -0
- package/src/adapters/google.ts +7 -7
- package/src/adapters/kiro/adapter.ts +42 -1
- package/src/adapters/kiro/payload.ts +17 -3
- package/src/adapters/kiro/reasoning.ts +70 -7
- package/src/adapters/kiro/stream.ts +8 -2
- package/src/adapters/kiro/wire.ts +2 -1
- package/src/adapters/kiro-events.ts +21 -13
- package/src/adapters/kiro-retry.ts +23 -4
- package/src/adapters/openai-chat/errors.ts +116 -0
- package/src/adapters/openai-chat/messages.ts +346 -0
- package/src/adapters/openai-chat/passthrough.ts +146 -0
- package/src/adapters/openai-chat/response-events.ts +117 -0
- package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
- package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
- package/src/adapters/openai-chat/tool-schema.ts +495 -0
- package/src/adapters/openai-chat/wire.ts +50 -0
- package/src/adapters/openai-chat.ts +40 -1452
- package/src/adapters/openai-responses/canonical-forward.ts +202 -0
- package/src/adapters/openai-responses/image-gen.ts +406 -0
- package/src/adapters/openai-responses/internal.ts +3 -0
- package/src/adapters/openai-responses/passthrough.ts +642 -0
- package/src/adapters/openai-responses/prompt-cache.ts +83 -0
- package/src/adapters/openai-responses/reasoning.ts +220 -0
- package/src/adapters/openai-responses/request-strips.ts +185 -0
- package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
- package/src/adapters/openai-responses/tool-schema.ts +293 -0
- package/src/adapters/openai-responses/web-search.ts +156 -0
- package/src/adapters/openai-responses.ts +4 -2625
- package/src/bridge/errors.ts +58 -0
- package/src/bridge/internal.ts +174 -0
- package/src/bridge/response-json.ts +630 -0
- package/src/bridge/sse.ts +1462 -0
- package/src/bridge.ts +5 -2204
- package/src/chat/inbound.ts +12 -1
- package/src/claude/desktop-profile.ts +66 -9
- package/src/claude/outbound.ts +18 -0
- package/src/cli/account-main.ts +1 -1
- package/src/cli/capabilities.ts +2 -2
- package/src/cli/combo.ts +10 -1
- package/src/cli/index.ts +48 -5
- package/src/cli/registry.ts +2 -1
- package/src/cli/system-command.ts +4 -4
- package/src/clients/config-export.ts +7 -3
- package/src/codex/account-label.ts +14 -3
- package/src/codex/account-lifecycle.ts +3 -0
- package/src/codex/account-store.ts +184 -35
- package/src/codex/account-usability.ts +21 -0
- package/src/codex/auth-api/account-list.ts +507 -0
- package/src/codex/auth-api/http.ts +32 -0
- package/src/codex/auth-api/login-flow.ts +566 -0
- package/src/codex/auth-api/login-state.ts +64 -0
- package/src/codex/auth-api/main-account-probe.ts +331 -0
- package/src/codex/auth-api/pool-mode-gate.ts +274 -0
- package/src/codex/auth-api/pool-quota-probe.ts +512 -0
- package/src/codex/auth-api/reset-credit-service.ts +431 -0
- package/src/codex/auth-api/routes.ts +425 -0
- package/src/codex/auth-api/runtime-config.ts +48 -0
- package/src/codex/auth-api.ts +27 -3118
- package/src/codex/auth-context.ts +252 -35
- package/src/codex/catalog/aggregation.ts +80 -1
- package/src/codex/catalog/auto-review.ts +507 -0
- package/src/codex/catalog/build-entries.ts +981 -0
- package/src/codex/catalog/combo-member.ts +375 -0
- package/src/codex/catalog/derive-entry.ts +229 -0
- package/src/codex/catalog/effort.ts +0 -1
- package/src/codex/catalog/gated-native-warn.ts +63 -0
- package/src/codex/catalog/gather-capture.ts +533 -0
- package/src/codex/catalog/model-hints.ts +691 -0
- package/src/codex/catalog/model-visibility.ts +305 -0
- package/src/codex/catalog/provider-fetch.ts +52 -2942
- package/src/codex/catalog/provider-models.ts +685 -0
- package/src/codex/catalog/remote.ts +30 -0
- package/src/codex/catalog/restore.ts +132 -0
- package/src/codex/catalog/retained-sync.ts +714 -0
- package/src/codex/catalog/routed-gather.ts +895 -0
- package/src/codex/catalog/subagent-roster.ts +176 -0
- package/src/codex/catalog/sync.ts +52 -2698
- package/src/codex/cli-install-provenance.ts +7 -1
- package/src/codex/convergence.ts +7 -2
- package/src/codex/desktop-app/types.ts +11 -2
- package/src/codex/desktop-app/windows.ts +5 -5
- package/src/codex/inject/config-toml.ts +563 -0
- package/src/codex/inject/remove.ts +192 -0
- package/src/codex/inject/restore.ts +567 -0
- package/src/codex/inject/routing-classify.ts +109 -0
- package/src/codex/inject/routing-target.ts +125 -0
- package/src/codex/inject.ts +89 -1444
- package/src/codex/lineage.ts +458 -0
- package/src/codex/model-entitlements.ts +152 -15
- package/src/codex/pool-refresh-backoff.ts +161 -0
- package/src/codex/quota-rejection.ts +104 -15
- package/src/codex/routing/active-account.ts +194 -0
- package/src/codex/routing/cache-affinity.ts +70 -0
- package/src/codex/routing/cooldown-math.ts +285 -0
- package/src/codex/routing/health-store.ts +402 -0
- package/src/codex/routing/probe-lease.ts +358 -0
- package/src/codex/routing/selection.ts +780 -0
- package/src/codex/routing/thread-affinity.ts +586 -0
- package/src/codex/routing/transient-hold-dispatch.ts +141 -0
- package/src/codex/routing.ts +370 -2271
- package/src/codex/shim-fingerprint.ts +223 -0
- package/src/codex/shim-inspect.ts +175 -0
- package/src/codex/shim-probe.ts +367 -0
- package/src/codex/shim-restore-lock.ts +169 -0
- package/src/codex/shim-state-file.ts +151 -0
- package/src/codex/shim-templates.ts +265 -0
- package/src/codex/shim.ts +48 -1268
- package/src/codex/warmup.ts +1 -1
- package/src/combos/failover.ts +85 -0
- package/src/combos/request.ts +17 -10
- package/src/combos/types.ts +23 -2
- package/src/config/diagnostics.ts +705 -0
- package/src/config/feature-flags.ts +55 -0
- package/src/config/live-reconcile.ts +403 -0
- package/src/config/load-degrade.ts +880 -0
- package/src/config/mutation-lock.ts +244 -0
- package/src/config/openai-tier-backup.ts +268 -0
- package/src/config/pending-teardown.ts +31 -0
- package/src/config/persist-unlocked.ts +92 -0
- package/src/config/proxy-env.ts +188 -0
- package/src/config/salvage.ts +244 -0
- package/src/config/schema/config-schema.ts +640 -0
- package/src/config/schema/leaf-validators.ts +855 -0
- package/src/config/warn-memo.ts +28 -0
- package/src/config.ts +234 -4481
- package/src/generated/compatibility-version.json +649 -121
- package/src/images/loop.ts +1 -1
- package/src/lib/errors.ts +17 -0
- package/src/lib/request-execution-budget.ts +198 -23
- package/src/lib/spend-reservation-ledger.ts +958 -0
- package/src/lib/state-store-registrations.ts +6 -2
- package/src/lib/test-home-guard.ts +85 -1
- package/src/lib/upstream-retry.ts +132 -21
- package/src/lib/windows-elevation.ts +76 -14
- package/src/lib/workflow-budget.ts +553 -30
- package/src/oauth/index.ts +2 -2
- package/src/oauth/key-providers.ts +2 -2
- package/src/providers/kiro-models.ts +4 -3
- package/src/providers/label.ts +19 -1
- package/src/providers/model-discovery.ts +16 -0
- package/src/providers/quota/account-cache.ts +441 -0
- package/src/providers/quota/antigravity.ts +295 -0
- package/src/providers/quota/report-cache.ts +320 -0
- package/src/providers/quota/vendor-probes-key.ts +1243 -0
- package/src/providers/quota/vendor-probes-oauth.ts +590 -0
- package/src/providers/quota.ts +324 -3079
- package/src/providers/registry/entries-core.ts +1228 -0
- package/src/providers/registry/entries-extended.ts +1213 -0
- package/src/providers/registry/model-seeds.ts +912 -0
- package/src/providers/registry/types.ts +352 -0
- package/src/providers/registry.ts +24 -3536
- package/src/responses/continuation-ownership.ts +29 -0
- package/src/responses/reasoning-envelope.ts +6 -3
- package/src/responses/state/replay-fingerprint.ts +80 -0
- package/src/responses/state/snapshot-codec.ts +104 -0
- package/src/responses/state/spill-failure.ts +118 -0
- package/src/responses/state/spill-queue.ts +665 -0
- package/src/responses/state/temp-recovery.ts +257 -0
- package/src/responses/state.ts +82 -1143
- package/src/routing/identity-domains.ts +456 -0
- package/src/routing/probe-lease.ts +613 -0
- package/src/server/chat-completions.ts +3 -1
- package/src/server/chat-native.ts +37 -9
- package/src/server/index/bounded-request.ts +88 -0
- package/src/server/index/live-sideband.ts +601 -0
- package/src/server/index/serve-options.ts +1766 -0
- package/src/server/index/startup-warnings.ts +213 -0
- package/src/server/index/websocket-handler.ts +339 -0
- package/src/server/index.ts +45 -2552
- package/src/server/inspection-tee.ts +107 -0
- package/src/server/live.ts +46 -1
- package/src/server/management/combo-routes.ts +10 -1
- package/src/server/management/route-registry.ts +26 -23
- package/src/server/management/shared.ts +8 -5
- package/src/server/management/workflow-budget-routes.ts +133 -0
- package/src/server/management-api.ts +12 -0
- package/src/server/relay-eager.ts +2 -0
- package/src/server/relay.ts +14 -19
- package/src/server/request-log-conversation.ts +9 -7
- package/src/server/request-log.ts +372 -4
- package/src/server/response-log-body.ts +153 -0
- package/src/server/responses/account-change-state.ts +307 -0
- package/src/server/responses/adapter-continuation.ts +540 -0
- package/src/server/responses/adapter-delivery.ts +208 -0
- package/src/server/responses/adapter-dispatch.ts +1042 -0
- package/src/server/responses/codex-ws-wire.ts +5 -0
- package/src/server/responses/collaboration.ts +74 -4
- package/src/server/responses/combo-session-recall.ts +68 -8
- package/src/server/responses/compact.ts +113 -17
- package/src/server/responses/completion-policy.ts +33 -0
- package/src/server/responses/core-auth.ts +529 -0
- package/src/server/responses/core-codex-account.ts +907 -0
- package/src/server/responses/core-combo-failure.ts +210 -0
- package/src/server/responses/core-combo.ts +787 -0
- package/src/server/responses/core-errors.ts +170 -0
- package/src/server/responses/core-lifetime.ts +95 -0
- package/src/server/responses/core-normalize.ts +350 -0
- package/src/server/responses/core-opaque-recovery.ts +380 -0
- package/src/server/responses/core-options.ts +159 -0
- package/src/server/responses/core-replay.ts +298 -0
- package/src/server/responses/core.ts +192 -8893
- package/src/server/responses/encrypted-payload.ts +0 -1
- package/src/server/responses/input-admission.ts +126 -6
- package/src/server/responses/passthrough-delivery.ts +869 -0
- package/src/server/responses/passthrough-dispatch.ts +1494 -0
- package/src/server/responses/passthrough-error.ts +38 -2
- package/src/server/responses/passthrough-execution.ts +54 -0
- package/src/server/responses/request-prepare.ts +1080 -0
- package/src/server/responses/request-send-budget.ts +259 -0
- package/src/server/responses/request-sidecar-auth.ts +149 -0
- package/src/server/responses/request-spend.ts +147 -0
- package/src/server/responses/request-transport.ts +803 -0
- package/src/server/responses/response-effects.ts +157 -0
- package/src/server/responses/run-turn-execution.ts +476 -0
- package/src/server/responses/sidecar-execution.ts +463 -0
- package/src/server/responses/terminal-guard.ts +65 -4
- package/src/server/responses-image-gen-repair.ts +1 -1
- package/src/server/responses-undeclared-tool-guard.ts +9 -5
- package/src/server/workflow-refusal.ts +84 -0
- package/src/service/windows-ops.ts +210 -16
- package/src/service/windows-scheduler.ts +28 -21
- package/src/service.ts +1 -1
- package/src/types/config.ts +34 -1
- package/src/types/request.ts +8 -5
- package/src/types/tools.ts +24 -0
- package/src/types.ts +2 -0
- package/src/update/index.ts +10 -0
- package/src/update/stop-contract.d.mts +1 -0
- package/src/update/stop-contract.mjs +19 -0
- package/src/update/stop-decision.d.mts +1 -1
- package/src/update/stop-decision.mjs +12 -3
- package/src/usage/log.ts +147 -1
- package/src/usage/summary.ts +171 -21
- package/src/vision/anthropic-describe.ts +1 -1
- package/src/vision/describe.ts +5 -5
- package/src/web-search/anthropic-executor.ts +1 -1
- package/src/web-search/exa-executor.ts +1 -1
- package/src/web-search/executor.ts +1 -1
- package/src/web-search/gemini-executor.ts +1 -1
- package/src/web-search/loop.ts +1 -1
- package/src/web-search/ollama-executor.ts +1 -1
- package/src/web-search/parse.ts +67 -14
- package/src/web-search/passthrough-bridge.ts +64 -31
- package/src/web-search/xai-executor.ts +1 -1
|
@@ -0,0 +1,912 @@
|
|
|
1
|
+
import type { ProviderModelDiscoverySpec } from "./types";
|
|
2
|
+
|
|
3
|
+
// Shared between the OAuth (Claude account) and API-key Anthropic entries so both expose the
|
|
4
|
+
// same static model seed.
|
|
5
|
+
// 260710 context refresh: Tier-2 evidence in
|
|
6
|
+
// devlog/_plan/260710_provider_hardening/001_research_frontier.md.
|
|
7
|
+
// 260902 Claude Fable 5.1 (`claude-fable-5-1`): 1M context / 128K output / adaptive thinking
|
|
8
|
+
// always on, per the official models overview and pricing page (platform.claude.com).
|
|
9
|
+
export const ANTHROPIC_MODELS = ["claude-fable-5-1", "claude-fable-5", "claude-sonnet-5", "claude-opus-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-sonnet-4-6", "claude-haiku-4-5"];
|
|
10
|
+
export const ANTHROPIC_MODEL_CONTEXT_WINDOWS: Record<string, number> = { "claude-fable-5-1": 1_000_000, "claude-sonnet-5": 1_000_000, "claude-fable-5": 1_000_000, "claude-opus-5": 1_000_000, "claude-opus-4-8": 1_000_000, "claude-opus-4-7": 1_000_000, "claude-opus-4-6": 1_000_000, "claude-sonnet-4-6": 1_000_000, "claude-haiku-4-5": 200_000 };
|
|
11
|
+
// All seeded Claude models support vision: https://platform.claude.com/docs/en/models/overview
|
|
12
|
+
export const ANTHROPIC_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
|
|
13
|
+
ANTHROPIC_MODELS.map(id => [id, ["text", "image"]]),
|
|
14
|
+
);
|
|
15
|
+
// Every current Claude family accepts at least 64k output tokens (Haiku 4.5 / Sonnet 4.x
|
|
16
|
+
// through Opus 5 and Fable 5). Anthropic caps max_tokens per model server-side, so a
|
|
17
|
+
// larger request never over-allocates; it only stops the 8192 truncation.
|
|
18
|
+
export const ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS = 64_000;
|
|
19
|
+
/**
|
|
20
|
+
* The effort rungs opencodex exposes for native Anthropic models. Without this the
|
|
21
|
+
* providers advertised no ladder at all, so every client that keys its effort control off
|
|
22
|
+
* `reasoningEfforts` — Aside and the rest of the Pi-shaped exports — wrote these models
|
|
23
|
+
* with no control, while the SAME Claude models routed through `cursor` or
|
|
24
|
+
* `google-antigravity` had one.
|
|
25
|
+
*
|
|
26
|
+
* This is an opencodex ladder, not a claim that each model takes `output_config.effort`.
|
|
27
|
+
* The adapter serves two wire shapes (src/adapters/anthropic.ts): adaptive families
|
|
28
|
+
* (fable, sonnet >= 5, opus >= 4.7) send the effort directly, while opus 4.6, sonnet 4.6
|
|
29
|
+
* and haiku 4.5 take the legacy path where `reasoningBudget` TRANSLATES each rung into
|
|
30
|
+
* `thinking.budget_tokens`. Anthropic documents `low|medium|high|max` for the 4.6 models
|
|
31
|
+
* and no effort parameter at all for haiku 4.5; the budget translation is what makes five
|
|
32
|
+
* rungs meaningful there, and it clamps below `max_tokens` so none of them 400.
|
|
33
|
+
*
|
|
34
|
+
* Deliberately excluded, each because advertising it would offer a control that does not
|
|
35
|
+
* do what it says:
|
|
36
|
+
* - `minimal`: `adaptiveEffort` rewrites it to `low` (the adaptive wire 400s on it), so
|
|
37
|
+
* it is not a distinct setting.
|
|
38
|
+
* - `none`: only sonnet >= 5 accepts an explicit thinking disable
|
|
39
|
+
* (`EXPLICIT_THINKING_DISABLE_FAMILY_MINIMUMS`); Fable rejects one outright.
|
|
40
|
+
* - `ultra`: not an Anthropic concept, and it is degraded to `max` at the request
|
|
41
|
+
* boundary anyway (src/responses/parser.ts).
|
|
42
|
+
*/
|
|
43
|
+
export const ANTHROPIC_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
44
|
+
export const ANTHROPIC_MODEL_REASONING_EFFORTS: Record<string, string[]> = Object.fromEntries(
|
|
45
|
+
ANTHROPIC_MODELS.map(id => [id, [...ANTHROPIC_REASONING_EFFORTS]]),
|
|
46
|
+
);
|
|
47
|
+
|
|
48
|
+
// 260814 GLM-5.3 is registered pre-emptively alongside 5.2 everywhere 5.2 appears. Z.AI's
|
|
49
|
+
// devpack "How to Switch Models" page (docs.z.ai/devpack/latest-model) lists glm-5.3 and
|
|
50
|
+
// glm-5.3[1m] as Coding Plan ids on the unchanged endpoints; the capability and pricing
|
|
51
|
+
// tables were not published yet, so every 5.3 row mirrors its 5.2 sibling until they settle.
|
|
52
|
+
// The non-Z.AI providers below are speculative on purpose: they carry 5.2 today and are
|
|
53
|
+
// expected to pick 5.3 up on their usual lag. Providers whose live /v1/models discovery is
|
|
54
|
+
// enabled self-correct on the next successful fetch; static ones need a follow-up refresh.
|
|
55
|
+
// Every 5.3 family member, so the effort ladder, the default effort and the output
|
|
56
|
+
// cap are derived in ONE place. `glm-5.3-flash` was seeded into the model list and
|
|
57
|
+
// the context map by hand and left out of this constant, which meant it advertised
|
|
58
|
+
// a 1M context with a null effort ladder, no default effort and no output cap while
|
|
59
|
+
// its siblings carried three tiers, a `max` default and 131072 tokens. A member
|
|
60
|
+
// added to the list but not to the family is a model whose metadata silently
|
|
61
|
+
// disappears.
|
|
62
|
+
export const ZAI_GLM_53_MODELS = ["glm-5.3", "glm-5.3[1m]", "glm-5.3-flash"];
|
|
63
|
+
export const ZAI_GLM_52_MODELS = ["glm-5.2", "glm-5.2[1m]"];
|
|
64
|
+
export const ZAI_GLM_5X_MODELS = [...ZAI_GLM_53_MODELS, ...ZAI_GLM_52_MODELS];
|
|
65
|
+
/**
|
|
66
|
+
* The 5.x rows whose images the PROXY has to describe, which is NOT the same set as
|
|
67
|
+
* the 5.x rows themselves.
|
|
68
|
+
*
|
|
69
|
+
* `glm-5.3-flash` is a native VLM (docs.z.ai/guides/vlm/glm-5.3-flash), so listing it
|
|
70
|
+
* in `noVisionModels` sent an image through the vision sidecar and handed the model a
|
|
71
|
+
* text description of a picture it could have read itself - no error, worse answer,
|
|
72
|
+
* extra call. The correction commit fixed the Alibaba entries and left the eight
|
|
73
|
+
* providers that reach this constant behind.
|
|
74
|
+
*
|
|
75
|
+
* Kept separate from ZAI_GLM_5X_MODELS rather than filtered at each use site: that
|
|
76
|
+
* constant also drives `modelSupportsReasoningSummaries` and
|
|
77
|
+
* `preserveReasoningContentModels`, where flash DOES belong.
|
|
78
|
+
*/
|
|
79
|
+
export const ZAI_GLM_5X_SIDECAR_VISION_MODELS = ZAI_GLM_5X_MODELS.filter(id => id !== "glm-5.3-flash");
|
|
80
|
+
/**
|
|
81
|
+
* Positive input-modality declaration for the Chat-path GLM rows.
|
|
82
|
+
*
|
|
83
|
+
* `noVisionModels` already keeps Flash out of the vision sidecar, but that is a NEGATIVE
|
|
84
|
+
* statement: it stops a detour without telling the catalog what the model can read. With
|
|
85
|
+
* no `modelInputModalities` entry, `configuredInputModalities` returns undefined and the
|
|
86
|
+
* catalog falls through to the `["text"]` floor, so every client export (ZCode, Pi, OMP)
|
|
87
|
+
* listed a native VLM as text-only and its picker refused to attach an image.
|
|
88
|
+
*
|
|
89
|
+
* The Responses sibling row below already declares this positively, so the same model was
|
|
90
|
+
* described two different ways in one registry.
|
|
91
|
+
*
|
|
92
|
+
* Authoritative source: `GET https://api.z.ai/api/v1/models` returns `input_modalities:
|
|
93
|
+
* ["text"]` for glm-5.3 and `["text", "image"]` for glm-5.3-flash (captured in
|
|
94
|
+
* devlog/_plan/260912_zcode_protocol_and_catalog/evidence/zai-responses-models.json).
|
|
95
|
+
* docs.z.ai/devpack/latest-model says the same in prose: "GLM-5.3 is a text-only model...
|
|
96
|
+
* GLM-5.3-FLASH is a multimodal model". Upstream also lists video and file for Flash;
|
|
97
|
+
* neither the internal vocabulary nor the export vocabulary can express them, so `image`
|
|
98
|
+
* is where this stops.
|
|
99
|
+
*/
|
|
100
|
+
export const ZAI_GLM_5X_INPUT_MODALITIES: Record<string, string[]> = {
|
|
101
|
+
...Object.fromEntries(ZAI_GLM_5X_SIDECAR_VISION_MODELS.map(id => [id, ["text"]])),
|
|
102
|
+
"glm-5.3-flash": ["text", "image"],
|
|
103
|
+
};
|
|
104
|
+
export const ZAI_GLM_52_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
105
|
+
/**
|
|
106
|
+
* GLM-5.3 does NOT share 5.2's five-tier ladder. docs.z.ai/devpack/latest-model folds every
|
|
107
|
+
* incoming effort into three effective tiers — low/minimal/light -> low, medium/high -> high,
|
|
108
|
+
* xhigh/max/ultra -> max — with max as both the default and the unknown-value fallback.
|
|
109
|
+
* Advertising five levels would publish two picker rows that are indistinguishable on the wire,
|
|
110
|
+
* so only the effective tiers are exposed (same treatment Cursor and Baseten already give GLM).
|
|
111
|
+
*/
|
|
112
|
+
export const ZAI_GLM_53_REASONING_EFFORTS = ["low", "high", "max"];
|
|
113
|
+
/** Per-model ladders for the Coding Plan rows: 5.3 gets its three effective tiers, 5.2 keeps five. */
|
|
114
|
+
export const ZAI_GLM_5X_REASONING_EFFORTS: Record<string, string[]> = {
|
|
115
|
+
...Object.fromEntries(ZAI_GLM_53_MODELS.map(id => [id, ZAI_GLM_53_REASONING_EFFORTS])),
|
|
116
|
+
...Object.fromEntries(ZAI_GLM_52_MODELS.map(id => [id, ZAI_GLM_52_REASONING_EFFORTS])),
|
|
117
|
+
};
|
|
118
|
+
// 260710 MiniMax models and context windows: Tier-2 evidence in
|
|
119
|
+
// devlog/_plan/260710_provider_hardening/002_research_cn.md.
|
|
120
|
+
export const MINIMAX_MODELS = [
|
|
121
|
+
"MiniMax-M3",
|
|
122
|
+
"MiniMax-M2.7", "MiniMax-M2.7-highspeed",
|
|
123
|
+
"MiniMax-M2.5", "MiniMax-M2.5-highspeed",
|
|
124
|
+
"MiniMax-M2.1", "MiniMax-M2.1-highspeed",
|
|
125
|
+
"MiniMax-M2",
|
|
126
|
+
];
|
|
127
|
+
export const MINIMAX_MODEL_CONTEXT_WINDOWS: Record<string, number> = Object.fromEntries(
|
|
128
|
+
MINIMAX_MODELS.map(id => [id, id === "MiniMax-M3" ? 1_000_000 : 204_800]),
|
|
129
|
+
);
|
|
130
|
+
export const MINIMAX_M3_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
131
|
+
export const MINIMAX_M3_REASONING_EFFORT_MAP: Record<string, string> = {
|
|
132
|
+
none: "disabled",
|
|
133
|
+
minimal: "disabled",
|
|
134
|
+
low: "disabled",
|
|
135
|
+
medium: "adaptive",
|
|
136
|
+
high: "adaptive",
|
|
137
|
+
xhigh: "adaptive",
|
|
138
|
+
max: "adaptive",
|
|
139
|
+
};
|
|
140
|
+
export const OPENAI_GPT56_MODELS = ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"];
|
|
141
|
+
export const OPENAI_GPT56_PRO_MODELS = ["gpt-5.6-sol-pro", "gpt-5.6-terra-pro", "gpt-5.6-luna-pro"];
|
|
142
|
+
export const OPENAI_API_GPT56_CONTEXT_WINDOW = 1_050_000;
|
|
143
|
+
export const OPENAI_API_GPT56_CONTEXT_WINDOWS: Record<string, number> = {
|
|
144
|
+
...Object.fromEntries([...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS].map(id => [id, OPENAI_API_GPT56_CONTEXT_WINDOW])),
|
|
145
|
+
"gpt-5.5": OPENAI_API_GPT56_CONTEXT_WINDOW,
|
|
146
|
+
};
|
|
147
|
+
export const OPENAI_API_GPT56_MAX_INPUT_TOKENS: Record<string, number> = {
|
|
148
|
+
...Object.fromEntries([...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS].map(id => [id, 922_000])),
|
|
149
|
+
"gpt-5.5": 922_000,
|
|
150
|
+
};
|
|
151
|
+
export const OPENAI_API_GPT56_VIRTUAL_MODELS: Record<string, { wireModelId: string; reasoningMode: "pro" }> = {
|
|
152
|
+
"gpt-5.6-sol-pro": { wireModelId: "gpt-5.6-sol", reasoningMode: "pro" },
|
|
153
|
+
"gpt-5.6-terra-pro": { wireModelId: "gpt-5.6-terra", reasoningMode: "pro" },
|
|
154
|
+
"gpt-5.6-luna-pro": { wireModelId: "gpt-5.6-luna", reasoningMode: "pro" },
|
|
155
|
+
};
|
|
156
|
+
export const OPENAI_API_GPT56_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
157
|
+
/*
|
|
158
|
+
* Meta Model API (https://api.meta.ai/v1) — published ladder, deliberately NOT the
|
|
159
|
+
* house set. dev.meta.ai/docs/reasoning lists "none", "minimal", "low", "medium",
|
|
160
|
+
* "high", "xhigh" and then excludes "none" for this family: "not supported by Muse
|
|
161
|
+
* Spark and returns HTTP 400". "max" and "ultra" are absent from the vendor's list
|
|
162
|
+
* entirely, so appending one by family resemblance would invent a wire value.
|
|
163
|
+
*
|
|
164
|
+
* Corroborated on a second surface: an unauthenticated OpenCode Zen probe of
|
|
165
|
+
* muse-spark-1.3-contributor-free (2026-09-03) accepted minimal..xhigh, rejected
|
|
166
|
+
* max/ultra with `unknown variant`, and rejected none with "does not support none
|
|
167
|
+
* with this model".
|
|
168
|
+
*/
|
|
169
|
+
export const META_MUSE_REASONING_EFFORTS = ["minimal", "low", "medium", "high", "xhigh"];
|
|
170
|
+
/*
|
|
171
|
+
* Identity wire map. `requestToCodexEffort` (src/reasoning-effort.ts) rewrites
|
|
172
|
+
* `minimal` to `low` unless a model-scoped wire map says otherwise, so without this
|
|
173
|
+
* the picker would advertise an effort the wire never sends — and a registry-array
|
|
174
|
+
* assertion would pass while the request body was wrong. Identity because Meta's
|
|
175
|
+
* values ARE the Codex names.
|
|
176
|
+
*/
|
|
177
|
+
export const META_MUSE_REASONING_EFFORT_MAP: Record<string, string> = Object.fromEntries(
|
|
178
|
+
META_MUSE_REASONING_EFFORTS.map(effort => [effort, effort]),
|
|
179
|
+
);
|
|
180
|
+
/** Both Muse Spark 1.3 tiers publish a 1,048,576-token window (dev.meta.ai/docs/models). */
|
|
181
|
+
export const META_MUSE_CONTEXT_WINDOW = 1_048_576;
|
|
182
|
+
export const META_MUSE_MODELS = ["muse-spark-1.3", "muse-spark-1.3-contributor"];
|
|
183
|
+
/**
|
|
184
|
+
* Daybreak program aliases. These `-latest` ids are the stable contract: OpenAI repoints
|
|
185
|
+
* them at newer snapshots over time (red -> gpt-5.6-cyber, blue -> gpt-5.6-sol as of
|
|
186
|
+
* 2026-08-11), so registering the ALIAS inherits future model swaps while a pinned
|
|
187
|
+
* snapshot id would silently go stale. Snapshot ids are deliberately absent here.
|
|
188
|
+
* Responses-only per both published endpoint tables (`v1/chat/completions` is marked
|
|
189
|
+
* Not supported) — never add these to a chat-completions provider. Access needs separate
|
|
190
|
+
* Daybreak approval and provisioning, so neither is ever a default.
|
|
191
|
+
* Verified 2026-08-11: developers.openai.com/api/docs/models/daybreak-red-latest.md
|
|
192
|
+
* and .../daybreak-blue-latest.md
|
|
193
|
+
*/
|
|
194
|
+
export const OPENAI_DAYBREAK_MODELS = ["daybreak-red-latest", "daybreak-blue-latest"];
|
|
195
|
+
export const OPENAI_DAYBREAK_CONTEXT_WINDOWS: Record<string, number> = {
|
|
196
|
+
"daybreak-red-latest": 400_000,
|
|
197
|
+
"daybreak-blue-latest": 1_050_000,
|
|
198
|
+
};
|
|
199
|
+
export const OPENAI_DAYBREAK_MAX_INPUT_TOKENS: Record<string, number> = {
|
|
200
|
+
"daybreak-red-latest": 272_000,
|
|
201
|
+
"daybreak-blue-latest": 922_000,
|
|
202
|
+
};
|
|
203
|
+
/**
|
|
204
|
+
* Neither Daybreak page publishes a reasoning-effort ladder. An explicit empty array means
|
|
205
|
+
* "expose no effort control"; OMITTING the key would instead fall back to the full routed
|
|
206
|
+
* ladder (`configuredReasoningEfforts` returns undefined -> `applyReasoningLevels` uses
|
|
207
|
+
* ROUTED_REASONING_LEVELS), which would advertise efforts the models never documented.
|
|
208
|
+
* `noReasoningModels` is wrong here: both pages document reasoning-token support, so these
|
|
209
|
+
* are reasoning models with no *selectable* ladder.
|
|
210
|
+
*/
|
|
211
|
+
export const OPENAI_DAYBREAK_REASONING_EFFORTS: Record<string, string[]> = Object.fromEntries(
|
|
212
|
+
OPENAI_DAYBREAK_MODELS.map(id => [id, [] as string[]]),
|
|
213
|
+
);
|
|
214
|
+
export const OPENROUTER_GPT56_MODELS = OPENAI_GPT56_MODELS.map(id => `openai/${id}`);
|
|
215
|
+
export const XAI_MODELS = [
|
|
216
|
+
"grok-4.6",
|
|
217
|
+
"grok-4.5",
|
|
218
|
+
"grok-4.3",
|
|
219
|
+
"grok-4.20-multi-agent-0309",
|
|
220
|
+
"grok-4.20-0309-reasoning",
|
|
221
|
+
"grok-4.20-0309-non-reasoning",
|
|
222
|
+
"grok-build-0.1",
|
|
223
|
+
"grok-composer-2.5-fast",
|
|
224
|
+
];
|
|
225
|
+
// OpenRouter's live /endpoints routes report 1,050,000; keep this separate from the
|
|
226
|
+
// unverified OpenAI API-key seed. Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
|
|
227
|
+
export const OPENROUTER_GPT56_CONTEXT_WINDOW = 1_050_000;
|
|
228
|
+
export const OPENROUTER_GPT56_CONTEXT_WINDOWS = {
|
|
229
|
+
"openai/gpt-5.6-sol": OPENROUTER_GPT56_CONTEXT_WINDOW,
|
|
230
|
+
"openai/gpt-5.6-terra": OPENROUTER_GPT56_CONTEXT_WINDOW,
|
|
231
|
+
"openai/gpt-5.6-luna": OPENROUTER_GPT56_CONTEXT_WINDOW,
|
|
232
|
+
};
|
|
233
|
+
|
|
234
|
+
/**
|
|
235
|
+
* Vendor thinking-toggle models (MiMo v2.x, GLM 5/5.1 on Zen Go): the wire knob is
|
|
236
|
+
* `thinking: {type: enabled|disabled}` — a binary. Advertise the full Codex picker ladder
|
|
237
|
+
* and map efforts onto the toggle. Zen Go
|
|
238
|
+
* pass-through probed live 2026-07-07 (glm-5.2 toggle verified; mimo/minimax accept shape).
|
|
239
|
+
*/
|
|
240
|
+
export const THINKING_TOGGLE_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
241
|
+
export const THINKING_TOGGLE_MAP: Record<string, string> = {
|
|
242
|
+
none: "disabled",
|
|
243
|
+
minimal: "disabled",
|
|
244
|
+
low: "disabled",
|
|
245
|
+
medium: "enabled",
|
|
246
|
+
high: "enabled",
|
|
247
|
+
xhigh: "enabled",
|
|
248
|
+
max: "enabled",
|
|
249
|
+
};
|
|
250
|
+
export const OPENCODE_GO_THINKING_TOGGLE_MODELS = [
|
|
251
|
+
"mimo-v2.5", "mimo-v2.5-pro", "glm-5", "glm-5.1",
|
|
252
|
+
];
|
|
253
|
+
/**
|
|
254
|
+
* Zhipu's domestic BigModel platform. Text families first, then the vision member: modalities are
|
|
255
|
+
* declared per model because `noVisionModels` means the opposite of "text only" here — it routes
|
|
256
|
+
* images through the proxy's vision sidecar (src/codex/catalog/provider-fetch.ts), a claim nobody
|
|
257
|
+
* has verified for BigModel-hosted GLM.
|
|
258
|
+
*/
|
|
259
|
+
// `glm-5.3-flash` is deliberately absent: it is a native VLM
|
|
260
|
+
// (docs.z.ai/guides/vlm/glm-5.3-flash), unlike glm-5.3 itself.
|
|
261
|
+
export const ZHIPU_BIGMODEL_TEXT_MODELS = ["glm-4.6", "glm-4.7", "glm-4.7-flash", "glm-5", "glm-5.1", "glm-5.2", "glm-5.3"];
|
|
262
|
+
export const ZHIPU_BIGMODEL_MODELS = [...ZHIPU_BIGMODEL_TEXT_MODELS, "glm-4.6v"];
|
|
263
|
+
export const ZHIPU_BIGMODEL_INPUT_MODALITIES: Record<string, string[]> = {
|
|
264
|
+
...Object.fromEntries(ZHIPU_BIGMODEL_TEXT_MODELS.map(id => [id, ["text"]])),
|
|
265
|
+
"glm-4.6v": ["text", "image"],
|
|
266
|
+
};
|
|
267
|
+
export const ZHIPU_BIGMODEL_THINKING_TOGGLE_MODELS = ["glm-4.6", "glm-4.7", "glm-5", "glm-5.1", "glm-5.2", "glm-5.3", "glm-5.3-flash"];
|
|
268
|
+
export const THINKING_BUDGET_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
269
|
+
// Qwen3.8-Max is the first Qwen3.x model with official direct `reasoning_effort` support.
|
|
270
|
+
// Evidence: https://qwen.ai/blog?id=qwen3.8
|
|
271
|
+
export const QWEN38_REASONING_EFFORTS = ["low", "medium", "xhigh"];
|
|
272
|
+
export const THINKING_BUDGET_MODELS = [
|
|
273
|
+
"qwen3.5-397b", "qwen3.6-35b",
|
|
274
|
+
"qwen3.5-plus", "qwen3.6-plus", "qwen3.7-max", "qwen3.7-plus",
|
|
275
|
+
];
|
|
276
|
+
export const OPENCODE_GO_THINKING_BUDGET_MODELS = ["qwen3.5-plus", "qwen3.6-plus", "qwen3.7-max", "qwen3.7-plus"];
|
|
277
|
+
/*
|
|
278
|
+
* DeepSeek moved the whole V4 name set on 2026-09-10. V4.1-Flash ships as deepseek-flash
|
|
279
|
+
* on the first-party API; deepseek-v4-flash and the vision preview retire as models but
|
|
280
|
+
* keep routing there as compatibility aliases, and deepseek-v4-pro follows from
|
|
281
|
+
* 2026-09-14 04:00 UTC. Evidence: https://api-docs.deepseek.com/news/news260910/.
|
|
282
|
+
*
|
|
283
|
+
* The spelling differs by who serves it, so one shared list cannot express it: the
|
|
284
|
+
* first-party API answers to deepseek-flash, while the Zen gateway exposes the route as
|
|
285
|
+
* deepseek-v4.1-flash (issue #4253, PR #4258). Vendor-hosted rosters (Volcengine plan
|
|
286
|
+
* snapshots, Alibaba) publish on their own schedule and keep the legacy set until they say
|
|
287
|
+
* otherwise - a first-party retirement notice does not end their deployment.
|
|
288
|
+
*/
|
|
289
|
+
export const DEEPSEEK_V4_LEGACY_MODELS = ["deepseek-v4-flash"];
|
|
290
|
+
/*
|
|
291
|
+
* `deepseek-v4-pro` is deliberately absent from both live sets. DeepSeek retires it from
|
|
292
|
+
* 2026-09-14 04:00 UTC and routes its requests to V4.1-Flash until a V4.1 Pro exists, so a
|
|
293
|
+
* row here would advertise a Pro context window and Pro pricing for a route that serves
|
|
294
|
+
* Flash. The retirement is followed through every roster in this file, including the
|
|
295
|
+
* vendor-hosted ones; providers that discover their models live are handled by
|
|
296
|
+
* `ROUTED_MODEL_COMPATIBILITY_EXCLUSIONS` because deleting a row there removes the
|
|
297
|
+
* model's capabilities rather than the model.
|
|
298
|
+
*/
|
|
299
|
+
export const DEEPSEEK_NATIVE_THINKING_MODELS = ["deepseek-flash", "deepseek-v4-flash"];
|
|
300
|
+
export const DEEPSEEK_GATEWAY_THINKING_MODELS = ["deepseek-v4.1-flash", "deepseek-v4-flash"];
|
|
301
|
+
/*
|
|
302
|
+
* DeepSeek's legacy vision preview id (released 2026-08-21). First-party probes
|
|
303
|
+
* in #4436 resolve it to image-capable `deepseek-flash`; retain the existing
|
|
304
|
+
* declarations because gateway support is specific to each served identifier.
|
|
305
|
+
*/
|
|
306
|
+
export const DEEPSEEK_VISION_PREVIEW_MODEL = "deepseek-v4-flash-vision-exp";
|
|
307
|
+
/**
|
|
308
|
+
* CommandCode routes verified to accept image input end-to-end (#2406).
|
|
309
|
+
*
|
|
310
|
+
* Verified-negative and therefore deliberately ABSENT: deepseek/deepseek-v4-flash,
|
|
311
|
+
* zai-org/GLM-5.2, zai-org/GLM-5.3, xai/grok-4.6. Those
|
|
312
|
+
* routes accept the request and drop the image, which is worse than declining it — the
|
|
313
|
+
* model answers about an image it never saw. Do not add an id here on family resemblance;
|
|
314
|
+
* capability intersection trusts this map.
|
|
315
|
+
*/
|
|
316
|
+
export const COMMAND_CODE_IMAGE_MODELS = [
|
|
317
|
+
`deepseek/${DEEPSEEK_VISION_PREVIEW_MODEL}`,
|
|
318
|
+
"gpt-5.6-luna",
|
|
319
|
+
"gpt-5.6-sol",
|
|
320
|
+
"MiniMaxAI/MiniMax-M3",
|
|
321
|
+
"moonshotai/Kimi-K3",
|
|
322
|
+
"meta/muse-spark-1.3",
|
|
323
|
+
"meta/muse-spark-1.3-contributor",
|
|
324
|
+
"meta/muse-spark-1.2",
|
|
325
|
+
"meta/muse-spark-1.2-contributor",
|
|
326
|
+
// Native Z.AI VLM (docs.z.ai/guides/vlm/glm-5.3-flash). This exact id is already
|
|
327
|
+
// classified as natively vision-capable in NVIDIA_NIM_VISION_MODELS in this file;
|
|
328
|
+
// it is not one of the verified-negative ids the header names (those are
|
|
329
|
+
// deepseek/deepseek-v4-flash, zai-org/GLM-5.2, zai-org/GLM-5.3, xai/grok-4.6 —
|
|
330
|
+
// different ids). Adding it on the shared GLM-5.3 prefix would be the family-
|
|
331
|
+
// resemblance mistake the header forbids; the VLM docs are the evidence (#4505).
|
|
332
|
+
"z-ai/glm-5.3-flash",
|
|
333
|
+
] as const;
|
|
334
|
+
/**
|
|
335
|
+
* Native image stays sourced from COMMAND_CODE_IMAGE_MODELS. Text-only routes
|
|
336
|
+
* sit beside that list so the catalog can still advertise sidecar coverage
|
|
337
|
+
* without claiming the gateway itself accepts a picture.
|
|
338
|
+
*
|
|
339
|
+
* The gateway-prefixed DeepSeek V4.1 Flash route has no verified native image
|
|
340
|
+
* support, so declaring it image-capable would hand it a picture it drops. A
|
|
341
|
+
* positive text-only declaration makes it a vision-sidecar consumer
|
|
342
|
+
* (src/vision/eligibility.ts), so the catalog advertises image input on its
|
|
343
|
+
* behalf and the four-target combo in #4505 intersects to ["text","image"]
|
|
344
|
+
* instead of ["text"] — without claiming native vision. modelInputModalities
|
|
345
|
+
* is per-key filled, so this reaches an existing install even when
|
|
346
|
+
* noVisionModels was persisted before the id joined that list.
|
|
347
|
+
*/
|
|
348
|
+
export const COMMAND_CODE_TEXT_ONLY_MODELS = [
|
|
349
|
+
"deepseek/deepseek-v4.1-flash",
|
|
350
|
+
] as const;
|
|
351
|
+
export const COMMAND_CODE_MODEL_INPUT_MODALITIES: Record<string, ["text"] | ["text", "image"]> = {
|
|
352
|
+
...Object.fromEntries(COMMAND_CODE_IMAGE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
|
|
353
|
+
...Object.fromEntries(COMMAND_CODE_TEXT_ONLY_MODELS.map(id => [id, ["text"] as ["text"]])),
|
|
354
|
+
};
|
|
355
|
+
export const OPENCODE_FREE_DEEPSEEK_MODELS = ["deepseek-v4-flash-free"];
|
|
356
|
+
/*
|
|
357
|
+
* Zen free models that reject `image_url` upstream (#1043, and the reproducible
|
|
358
|
+
* half of #1024).
|
|
359
|
+
*
|
|
360
|
+
* Zen publishes NO modality metadata — its `/v1/models` returns only id, object,
|
|
361
|
+
* created, owned_by — so this list is measured, not derived. Each id was probed
|
|
362
|
+
* once against https://opencode.ai/zen/v1 on 2026-08-05 with a text control first
|
|
363
|
+
* and then a 1x1 PNG; the six below failed the image request, four of them with
|
|
364
|
+
* `[404] No endpoints found that support image input` and `big-pickle` with the
|
|
365
|
+
* exact deserialize error quoted in #1043.
|
|
366
|
+
*
|
|
367
|
+
* `mimo-v2.5-free` and `longcat-2.0-free` ACCEPT images. They remain absent
|
|
368
|
+
* from the blind list and are recorded separately as positive input-modality evidence,
|
|
369
|
+
* so capability-positive dispatch can forward images without relying on blacklist absence.
|
|
370
|
+
*
|
|
371
|
+
* Zen's roster is discovered live while this list is static, so it is a dated
|
|
372
|
+
* exception list, not a capability model. Re-probe before extending it.
|
|
373
|
+
* Evidence: devlog/_fin/260805_bug_fix_stack/002_zen_modality_probe.md
|
|
374
|
+
*/
|
|
375
|
+
export const OPENCODE_ZEN_TEXT_ONLY_MODELS = [
|
|
376
|
+
"big-pickle",
|
|
377
|
+
"nemotron-3-ultra-free",
|
|
378
|
+
"ling-3.0-flash-free",
|
|
379
|
+
"north-mini-code-free",
|
|
380
|
+
"laguna-s-2.1-free",
|
|
381
|
+
"deepseek-v4-flash-free",
|
|
382
|
+
];
|
|
383
|
+
export const OPENCODE_ZEN_IMAGE_MODELS = ["mimo-v2.5-free", "longcat-2.0-free"] as const;
|
|
384
|
+
/*
|
|
385
|
+
* DeepSeek's Codex ladder is low/high/max. With the V4 Pro GA release
|
|
386
|
+
* (DeepSeek-V4-Pro-0813) the official thinking-mode table is IDENTICAL for both
|
|
387
|
+
* V4 models (api-docs.deepseek.com/guides/thinking_mode, verified 2026-08-13):
|
|
388
|
+
*
|
|
389
|
+
* requested | v4-flash | v4-pro
|
|
390
|
+
* low | low | low
|
|
391
|
+
* medium | high | high
|
|
392
|
+
* high | high | high
|
|
393
|
+
* xhigh | high | high
|
|
394
|
+
* max | max | max
|
|
395
|
+
*
|
|
396
|
+
* Before GA, Pro silently upgraded low->high and mapped xhigh->max (#1057-era
|
|
397
|
+
* table); the page's footnote about an early-August Pro mapping update landed
|
|
398
|
+
* with this GA, so Pro now advertises the same three real tiers as Flash.
|
|
399
|
+
*
|
|
400
|
+
* Two standing notes (#1057):
|
|
401
|
+
*
|
|
402
|
+
* - `xhigh` is a COMPATIBILITY ALIAS, not a native tier. It stays in the wire maps
|
|
403
|
+
* so existing requests and saved configs keep working, but it is not advertised.
|
|
404
|
+
* - `medium` has no row in the vendor table — mapping it to `high` is OUR
|
|
405
|
+
* compatibility choice for clients that only speak the OpenAI ladder.
|
|
406
|
+
*/
|
|
407
|
+
export const DEEPSEEK_FLASH_THINKING_EFFORTS = ["low", "high", "max"];
|
|
408
|
+
export const DEEPSEEK_PRO_THINKING_EFFORTS = ["low", "high", "max"];
|
|
409
|
+
export const DEEPSEEK_PRO_REASONING_MAP: Record<string, string> = {
|
|
410
|
+
low: "low",
|
|
411
|
+
medium: "high",
|
|
412
|
+
high: "high",
|
|
413
|
+
xhigh: "high",
|
|
414
|
+
max: "max",
|
|
415
|
+
};
|
|
416
|
+
export const DEEPSEEK_FLASH_REASONING_MAP: Record<string, string> = {
|
|
417
|
+
low: "low",
|
|
418
|
+
medium: "high",
|
|
419
|
+
high: "high",
|
|
420
|
+
xhigh: "high",
|
|
421
|
+
max: "max",
|
|
422
|
+
};
|
|
423
|
+
/**
|
|
424
|
+
* Flash-versus-Pro classification for DeepSeek V4 model ids, including prefixed
|
|
425
|
+
* (`deepseek/deepseek-v4.1-flash`) and suffixed (`deepseek-v4-flash-free`) forms.
|
|
426
|
+
* `tests/providers/provider-registry-parity.test.ts` enumerates every id the registry
|
|
427
|
+
* actually passes here, so a future id this substring test would misread cannot
|
|
428
|
+
* land silently.
|
|
429
|
+
*/
|
|
430
|
+
export const isDeepseekFlashModel = (modelId: string): boolean =>
|
|
431
|
+
modelId.toLowerCase().includes("flash");
|
|
432
|
+
export const deepseekThinkingEffortsFor = (modelId: string): string[] =>
|
|
433
|
+
isDeepseekFlashModel(modelId) ? DEEPSEEK_FLASH_THINKING_EFFORTS : DEEPSEEK_PRO_THINKING_EFFORTS;
|
|
434
|
+
export const deepseekReasoningMapFor = (modelId: string): Record<string, string> =>
|
|
435
|
+
isDeepseekFlashModel(modelId) ? DEEPSEEK_FLASH_REASONING_MAP : DEEPSEEK_PRO_REASONING_MAP;
|
|
436
|
+
// 260719 Alibaba Token Plan Personal Edition (China/Beijing). Keep it distinct from
|
|
437
|
+
// Coding Plan: the products use different exact allowlists and different base URLs.
|
|
438
|
+
// Evidence: https://help.aliyun.com/en/model-studio/token-plan-personal-overview
|
|
439
|
+
// https://help.aliyun.com/en/model-studio/token-plan-quickstart
|
|
440
|
+
export const ALIBABA_TOKEN_PLAN_MODELS = [
|
|
441
|
+
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
442
|
+
"glm-5.3", "glm-5.3-flash", "glm-5.2",
|
|
443
|
+
];
|
|
444
|
+
export const ALIBABA_TOKEN_PLAN_QWEN_MODELS = [
|
|
445
|
+
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
446
|
+
];
|
|
447
|
+
export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
448
|
+
"qwen3.8-max": ["text", "image"],
|
|
449
|
+
"qwen3.7-max": ["text", "image"],
|
|
450
|
+
"qwen3.7-plus": ["text", "image"],
|
|
451
|
+
"qwen3.6-flash": ["text", "image"],
|
|
452
|
+
"glm-5.3": ["text"],
|
|
453
|
+
"glm-5.3-flash": ["text", "image"],
|
|
454
|
+
"glm-5.2": ["text"],
|
|
455
|
+
};
|
|
456
|
+
|
|
457
|
+
// 260721 Alibaba Token Plan International (ap-southeast-1 / Singapore, hardened 260721).
|
|
458
|
+
// Multi-vendor lineup distinct from Beijing — includes DeepSeek V4 flash, Kimi K2.7, MiniMax.
|
|
459
|
+
// Evidence: https://www.alibabacloud.com/help/en/model-studio/token-plan-overview
|
|
460
|
+
// https://qwencloud.com/pricing/token-plan (qwen3.8 metadata)
|
|
461
|
+
export const ALIBABA_INTL_TOKEN_PLAN_MODELS = [
|
|
462
|
+
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
463
|
+
"deepseek-v4-flash", "deepseek-v3.2",
|
|
464
|
+
"kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5",
|
|
465
|
+
"glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5",
|
|
466
|
+
"MiniMax-M2.5",
|
|
467
|
+
];
|
|
468
|
+
export const ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS = [
|
|
469
|
+
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
470
|
+
];
|
|
471
|
+
|
|
472
|
+
// 260722 Tencent Cloud Coding Plan. The plan's model set is explicitly dynamic; these are the
|
|
473
|
+
// current documented ids and live discovery remains enabled so successful /models responses win.
|
|
474
|
+
// Tencent marks every Coding Plan model as text-only input and restricts plan keys to interactive
|
|
475
|
+
// coding tools (not custom application backends or non-interactive batch automation).
|
|
476
|
+
// Evidence: https://cloud.tencent.cn/document/product/1823/130092
|
|
477
|
+
export const TENCENT_CODING_PLAN_MODELS = ["tc-code-latest", "glm-5", "kimi-k2.5", "minimax-m2.5"];
|
|
478
|
+
// Volcengine's authenticated /api/v3/models catalog mixes chat models with embedding,
|
|
479
|
+
// image, video, and 3D generation resources. Keep the Codex-facing presets scoped to
|
|
480
|
+
// models documented for text/agent or Coding Plan use.
|
|
481
|
+
//
|
|
482
|
+
// Maintenance owner: @lidge-jun. Verified 2026-08-01 against the vendor's own docs —
|
|
483
|
+
// endpoints https://docs.volcengine.com/docs/82379/1528783 (Coding Plan) and
|
|
484
|
+
// https://docs.volcengine.com/docs/82379/2165245 (Agent Plan); Codex CLI integration
|
|
485
|
+
// https://www.volcengine.com/docs/82379/2556056; supported clients
|
|
486
|
+
// https://www.volcengine.com/docs/82379/2188957; terms https://www.volcengine.com/docs/6256/64903
|
|
487
|
+
// (北京火山引擎科技有限公司). Plan quota is restricted to supported AI coding tools and misuse
|
|
488
|
+
// is documented as grounds for suspension — see the `note` on both Plan entries.
|
|
489
|
+
// Report a break by opening an issue tagging the owner; the three things that rot first are the
|
|
490
|
+
// static catalogs (liveModels:false cannot self-heal), the base URLs, and those Plan terms.
|
|
491
|
+
// Full evidence ledger: devlog/_fin/260801_pr611_volcengine_evidence/000_evidence_ledger.md
|
|
492
|
+
export const VOLCENGINE_ARK_MODELS = [
|
|
493
|
+
"doubao-seed-2-1-pro-260628",
|
|
494
|
+
"doubao-seed-2-1-turbo-260628",
|
|
495
|
+
"doubao-seed-evolving",
|
|
496
|
+
"deepseek-v4-flash-260425",
|
|
497
|
+
"deepseek-v3-2-251201",
|
|
498
|
+
// No glm-5-3 row: Ark pins date-stamped snapshot ids (glm-5-2-260617) that cannot be
|
|
499
|
+
// guessed ahead of the vendor publishing them. Add it once /api/v3/models lists one.
|
|
500
|
+
"glm-5-2-260617",
|
|
501
|
+
"glm-4-7-251222",
|
|
502
|
+
];
|
|
503
|
+
export const VOLCENGINE_DOUBAO_THINKING_MODELS = [
|
|
504
|
+
"doubao-seed-2-1-pro-260628",
|
|
505
|
+
"doubao-seed-2-1-turbo-260628",
|
|
506
|
+
"doubao-seed-evolving",
|
|
507
|
+
];
|
|
508
|
+
export const VOLCENGINE_CODING_PLAN_MODELS = [
|
|
509
|
+
"ark-code-latest",
|
|
510
|
+
"doubao-seed-2.0-code",
|
|
511
|
+
"deepseek-v4-flash",
|
|
512
|
+
"glm-5.3",
|
|
513
|
+
"glm-5.3-flash",
|
|
514
|
+
"glm-5.2",
|
|
515
|
+
"kimi-k2.6",
|
|
516
|
+
"minimax-m3",
|
|
517
|
+
];
|
|
518
|
+
export const VOLCENGINE_AGENT_PLAN_MODELS = [
|
|
519
|
+
"deepseek-v4-flash",
|
|
520
|
+
"glm-5.3",
|
|
521
|
+
"glm-5.3-flash",
|
|
522
|
+
"glm-5.2",
|
|
523
|
+
"kimi-k2.6",
|
|
524
|
+
"minimax-m3",
|
|
525
|
+
"doubao-seed-2.0-pro",
|
|
526
|
+
];
|
|
527
|
+
export const VOLCENGINE_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
528
|
+
"kimi-k2.6": ["text", "image"],
|
|
529
|
+
"minimax-m3": ["text", "image"],
|
|
530
|
+
// Native VLM (docs.z.ai/guides/vlm/glm-5.3-flash), so it is declared here and left
|
|
531
|
+
// out of the text-only list below.
|
|
532
|
+
"glm-5.3-flash": ["text", "image"],
|
|
533
|
+
};
|
|
534
|
+
// Every other Plan model is text-only. Declaring this explicitly keeps the vision
|
|
535
|
+
// sidecar from advertising image input for models that cannot accept it — the same
|
|
536
|
+
// treatment tencent-coding-plan gives its (entirely text-only) plan catalog.
|
|
537
|
+
export const VOLCENGINE_PLAN_TEXT_ONLY_MODELS = [
|
|
538
|
+
"ark-code-latest",
|
|
539
|
+
"doubao-seed-2.0-code",
|
|
540
|
+
"deepseek-v4-flash",
|
|
541
|
+
"glm-5.3",
|
|
542
|
+
"glm-5.2",
|
|
543
|
+
"doubao-seed-2.0-pro",
|
|
544
|
+
];
|
|
545
|
+
export const ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
546
|
+
"qwen3.8-max": ["text", "image"],
|
|
547
|
+
"qwen3.7-max": ["text", "image"],
|
|
548
|
+
"qwen3.7-plus": ["text", "image"],
|
|
549
|
+
"qwen3.6-plus": ["text", "image"],
|
|
550
|
+
"qwen3.6-flash": ["text", "image"],
|
|
551
|
+
"deepseek-v4-flash": ["text"],
|
|
552
|
+
"deepseek-v3.2": ["text"],
|
|
553
|
+
"kimi-k2.7-code": ["text", "image"],
|
|
554
|
+
"kimi-k2.6": ["text", "image"],
|
|
555
|
+
"kimi-k2.5": ["text", "image"],
|
|
556
|
+
"glm-5.3": ["text"],
|
|
557
|
+
"glm-5.3-flash": ["text", "image"],
|
|
558
|
+
"glm-5.2": ["text"],
|
|
559
|
+
"glm-5.1": ["text"],
|
|
560
|
+
"glm-5": ["text"],
|
|
561
|
+
"MiniMax-M2.5": ["text"],
|
|
562
|
+
};
|
|
563
|
+
|
|
564
|
+
// 260717 Kimi K3: the subscription endpoint uses one upstream id (`k3`) for both
|
|
565
|
+
// entitlement tiers. Bare `k3` advertises the Moderato 256K ceiling; the local `[1m]`
|
|
566
|
+
// alias advertises Allegretto's 1M ceiling and is stripped before the upstream request.
|
|
567
|
+
// The separately billed Moonshot API uses `kimi-k3`.
|
|
568
|
+
// Evidence: https://www.kimi.com/code/docs/en/kimi-code/models.html
|
|
569
|
+
// https://www.kimi.com/code/docs/en/kimi-code/error-reference.html
|
|
570
|
+
export const KIMI_K3_STANDARD_CONTEXT_WINDOW = 262_144;
|
|
571
|
+
export const KIMI_K3_1M_CONTEXT_WINDOW = 1_048_576;
|
|
572
|
+
export const KIMI_CODING_K3_MODELS = ["k3", "k3[1m]"];
|
|
573
|
+
export const KIMI_LEGACY_API_MODELS = ["kimi-k2.7-code", "kimi-k2.7-code-highspeed", "kimi-k2.6", "kimi-k2.5"];
|
|
574
|
+
export const KIMI_API_MODELS = ["kimi-k3", ...KIMI_LEGACY_API_MODELS];
|
|
575
|
+
export const KIMI_CODING_MODELS = [...KIMI_CODING_K3_MODELS, ...KIMI_LEGACY_API_MODELS, "kimi-for-coding"];
|
|
576
|
+
export const KIMI_THINKING_MODELS = KIMI_CODING_MODELS;
|
|
577
|
+
export const KIMI_CODING_NO_REASONING_MODELS = KIMI_CODING_MODELS.filter(id => !KIMI_CODING_K3_MODELS.includes(id));
|
|
578
|
+
export const KIMI_API_NO_REASONING_MODELS = KIMI_API_MODELS.filter(id => id !== "kimi-k3");
|
|
579
|
+
export const KIMI_CODING_K3_REASONING_EFFORTS = ["low", "high", "max"];
|
|
580
|
+
export const KIMI_CODING_K3_REASONING_EFFORT_MAP: Record<string, string> = {
|
|
581
|
+
none: "none",
|
|
582
|
+
low: "low",
|
|
583
|
+
medium: "high",
|
|
584
|
+
high: "high",
|
|
585
|
+
xhigh: "max",
|
|
586
|
+
max: "max",
|
|
587
|
+
};
|
|
588
|
+
export const KIMI_CODING_REASONING_EFFORTS = Object.fromEntries(
|
|
589
|
+
KIMI_CODING_MODELS.map(id => [id, KIMI_CODING_K3_MODELS.includes(id) ? KIMI_CODING_K3_REASONING_EFFORTS : []]),
|
|
590
|
+
);
|
|
591
|
+
export const KIMI_CODING_DEFAULT_REASONING_EFFORTS = Object.fromEntries(
|
|
592
|
+
KIMI_CODING_K3_MODELS.map(id => [id, "max"]),
|
|
593
|
+
);
|
|
594
|
+
export const KIMI_CODING_REASONING_EFFORT_MAPS = Object.fromEntries(
|
|
595
|
+
KIMI_CODING_K3_MODELS.map(id => [id, KIMI_CODING_K3_REASONING_EFFORT_MAP]),
|
|
596
|
+
);
|
|
597
|
+
export const KIMI_API_REASONING_EFFORTS = Object.fromEntries(
|
|
598
|
+
KIMI_API_MODELS.map(id => [id, id === "kimi-k3" ? ["max"] : []]),
|
|
599
|
+
);
|
|
600
|
+
export const KIMI_LOCKED_PARAMETER_MODELS = KIMI_CODING_MODELS;
|
|
601
|
+
export const KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS = ["kimi-k2.7-code", "kimi-k2.7-code-highspeed", "kimi-for-coding"];
|
|
602
|
+
export const KIMI_API_MODEL_CONTEXT_WINDOWS: Record<string, number> = Object.fromEntries(
|
|
603
|
+
KIMI_API_MODELS.map(id => [id, id === "kimi-k3" ? KIMI_K3_1M_CONTEXT_WINDOW : 262_144]),
|
|
604
|
+
);
|
|
605
|
+
export const KIMI_API_MODEL_INPUT_MODALITIES = { "kimi-k3": ["text", "image"] };
|
|
606
|
+
|
|
607
|
+
// 260715 NVIDIA NIM kimi family (issue #126): documented served ids on integrate
|
|
608
|
+
// chat/completions per docs.api.nvidia.com/nim/reference/llm-apis; live /v1/models
|
|
609
|
+
// currently lists only kimi-k2.6 but the list is dynamic, so carry the documented family.
|
|
610
|
+
export const NVIDIA_NIM_KIMI_THINKING_MODELS = [
|
|
611
|
+
"moonshotai/kimi-k2.6", "moonshotai/kimi-k2.5", "moonshotai/kimi-k2-thinking",
|
|
612
|
+
];
|
|
613
|
+
export const NVIDIA_NIM_KIMI_MODELS = [
|
|
614
|
+
...NVIDIA_NIM_KIMI_THINKING_MODELS,
|
|
615
|
+
"moonshotai/kimi-k2-instruct", "moonshotai/kimi-k2-instruct-0905",
|
|
616
|
+
];
|
|
617
|
+
/**
|
|
618
|
+
* 260804 issue #956: NIM publishes no input-modality metadata on `/v1/models`, so the
|
|
619
|
+
* registry is the only source of truth for which models can see images.
|
|
620
|
+
*
|
|
621
|
+
* Two lists, both verified per-model against NVIDIA documentation on 2026-08-04
|
|
622
|
+
* (build.nvidia.com model pages and docs.api.nvidia.com/nim/reference/*). Evidence and
|
|
623
|
+
* the per-id audit: devlog/_fin/260804_stack7_service_vision/011_nim_id_audit.md.
|
|
624
|
+
*
|
|
625
|
+
* Read `noVisionModels` carefully — it lists models that CANNOT see images, which is
|
|
626
|
+
* what routes them through the proxy's vision sidecar (src/vision/index.ts) and makes the
|
|
627
|
+
* catalog advertise image input for them. Membership is wrong in BOTH directions:
|
|
628
|
+
* - a text-only model missing from it keeps issue #956 (images blocked or rejected);
|
|
629
|
+
* - a vision model wrongly IN it gets its image silently replaced by another model's
|
|
630
|
+
* text description — no error, worse answers, extra cost.
|
|
631
|
+
*
|
|
632
|
+
* A new NIM id must be classified DELIBERATELY against its NVIDIA page, never assumed
|
|
633
|
+
* from its name: `google/gemma-4-31b-it` carries no vision marker yet accepts images,
|
|
634
|
+
* `-vl` also appears on embedding/reranking models, and `google/codegemma-7b` is
|
|
635
|
+
* text-only while `google/codegemma-1.1-7b` has no current page at all. An unclassified
|
|
636
|
+
* id is intentionally left alone rather than defaulted, because NIM serves non-chat
|
|
637
|
+
* endpoints (embeddings, rerankers, guards, OCR) that reach the same code path.
|
|
638
|
+
*/
|
|
639
|
+
export const NVIDIA_NIM_VISION_MODELS = [
|
|
640
|
+
"meta/llama-3.2-11b-vision-instruct", "meta/llama-3.2-90b-vision-instruct",
|
|
641
|
+
"nvidia/llama-3.1-nemotron-nano-vl-8b-v1", "nvidia/nemotron-nano-12b-v2-vl",
|
|
642
|
+
"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "nvidia/cosmos3-nano-reasoner",
|
|
643
|
+
"nvidia/ising-calibration-1.5-31b", "nvidia/ising-calibration-1-35b-a3b",
|
|
644
|
+
"google/gemma-4-31b-it", "google/diffusiongemma-26b-a4b-it",
|
|
645
|
+
"minimaxai/minimax-m3", "moonshotai/kimi-k2.6", "moonshotai/kimi-k2.5",
|
|
646
|
+
"stepfun-ai/step-3.7-flash", "thinkingmachines/inkling",
|
|
647
|
+
"mistralai/mistral-medium-3.5-128b",
|
|
648
|
+
"z-ai/glm-5.3-flash",
|
|
649
|
+
];
|
|
650
|
+
/**
|
|
651
|
+
* The catalog advertises image input only for `noVisionModels` members, so a natively
|
|
652
|
+
* vision-capable model would otherwise be published as text-only and the Codex app would
|
|
653
|
+
* block attachments before the native path ever runs.
|
|
654
|
+
*/
|
|
655
|
+
export const NVIDIA_NIM_VISION_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
|
|
656
|
+
NVIDIA_NIM_VISION_MODELS.map(id => [id, ["text", "image"]]),
|
|
657
|
+
);
|
|
658
|
+
/**
|
|
659
|
+
* Text-only NIM chat models — 26 ids, each carrying an explicit `Input Modalities: Text`
|
|
660
|
+
* (or equivalent) on its NVIDIA page. PR #964 proposed ~64; six of those are natively
|
|
661
|
+
* image-capable and live in NVIDIA_NIM_VISION_MODELS above, and 32 more had no current
|
|
662
|
+
* NVIDIA page and were dropped rather than assumed.
|
|
663
|
+
*
|
|
664
|
+
* kimi-k2-thinking and kimi-k2-instruct are text-only while k2.5/k2.6 are not — vision
|
|
665
|
+
* and reasoning are independent axes, so all four stay in NVIDIA_NIM_KIMI_MODELS for
|
|
666
|
+
* reasoning suppression regardless of which list they appear in here.
|
|
667
|
+
*/
|
|
668
|
+
export const NVIDIA_NIM_NO_VISION_MODELS = [
|
|
669
|
+
"deepseek-ai/deepseek-v4-flash",
|
|
670
|
+
"google/codegemma-7b",
|
|
671
|
+
"meta/llama-3.1-70b-instruct", "meta/llama-3.1-8b-instruct",
|
|
672
|
+
"meta/llama-3.2-1b-instruct", "meta/llama-3.2-3b-instruct",
|
|
673
|
+
"meta/llama-3.3-70b-instruct", "meta/llama2-70b",
|
|
674
|
+
"mistralai/mistral-7b-instruct-v0.3", "mistralai/mistral-nemotron",
|
|
675
|
+
"moonshotai/kimi-k2-thinking", "moonshotai/kimi-k2-instruct",
|
|
676
|
+
"nvidia/llama-3.1-nemotron-nano-8b-v1", "nvidia/llama-3.1-nemotron-ultra-253b-v1",
|
|
677
|
+
"nvidia/llama-3.3-nemotron-super-49b-v1", "nvidia/llama-3.3-nemotron-super-49b-v1.5",
|
|
678
|
+
"nvidia/nemotron-3-nano-30b-a3b", "nvidia/nemotron-3-super-120b-a12b",
|
|
679
|
+
"nvidia/nemotron-3-ultra-550b-a55b", "nvidia/nemotron-mini-4b-instruct",
|
|
680
|
+
"nvidia/nvidia-nemotron-nano-9b-v2",
|
|
681
|
+
"openai/gpt-oss-120b", "openai/gpt-oss-20b",
|
|
682
|
+
// z-ai/glm-5.3-flash belongs in NVIDIA_NIM_VISION_MODELS, not here: Z.AI documents
|
|
683
|
+
// it under docs.z.ai/guides/vlm/. The header above says an id must be classified
|
|
684
|
+
// deliberately rather than assumed from its name, and inheriting glm-5.3's
|
|
685
|
+
// text-only verdict because of the shared prefix is exactly that mistake.
|
|
686
|
+
"poolside/laguna-xs-2.1", "z-ai/glm-5.3", "z-ai/glm-5.2",
|
|
687
|
+
];
|
|
688
|
+
export const KIMI_CODING_MODEL_CONTEXT_WINDOWS: Record<string, number> = Object.fromEntries(
|
|
689
|
+
KIMI_CODING_MODELS.map(id => [id, id === "k3[1m]" ? KIMI_K3_1M_CONTEXT_WINDOW : KIMI_K3_STANDARD_CONTEXT_WINDOW]),
|
|
690
|
+
);
|
|
691
|
+
export const KIMI_CODING_MODEL_INPUT_MODALITIES = Object.fromEntries(
|
|
692
|
+
KIMI_CODING_K3_MODELS.map(id => [id, ["text", "image"]]),
|
|
693
|
+
);
|
|
694
|
+
export const NEURALWATT_REASONING_HISTORY_MODELS = [
|
|
695
|
+
"glm-5.3", "glm-5.3-short", "glm-5.3-flash",
|
|
696
|
+
"glm-5.2", "glm-5.2-short",
|
|
697
|
+
"kimi-k2.6", "kimi-k2.7-code",
|
|
698
|
+
"qwen3.5-397b", "qwen3.6-35b",
|
|
699
|
+
];
|
|
700
|
+
|
|
701
|
+
// 260728 Baseten Model APIs: `/v1/models` owns the live lineup, while these hints
|
|
702
|
+
// describe only capabilities that Baseten documents per slug. Unlisted live models
|
|
703
|
+
// intentionally inherit the empty provider ladder instead of being advertised with
|
|
704
|
+
// opencodex's generic reasoning defaults. Audio is omitted because the current proxy
|
|
705
|
+
// request model does not carry OpenAI `audio_url` parts.
|
|
706
|
+
// Evidence: https://docs.baseten.co/inference/model-apis/reasoning
|
|
707
|
+
// https://docs.baseten.co/inference/model-apis/vision
|
|
708
|
+
export const BASETEN_FULL_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
709
|
+
export const BASETEN_MODEL_REASONING_EFFORTS: Record<string, string[]> = {
|
|
710
|
+
"thinkingmachines/inkling": BASETEN_FULL_REASONING_EFFORTS,
|
|
711
|
+
"openai/gpt-oss-120b": BASETEN_FULL_REASONING_EFFORTS,
|
|
712
|
+
"moonshotai/Kimi-K3": ["low", "high", "max"],
|
|
713
|
+
// 260814: GLM-5.3 honours low/high/max upstream, unlike 5.2's high/max on Baseten.
|
|
714
|
+
"zai-org/GLM-5.3": ["low", "high", "max"],
|
|
715
|
+
"zai-org/GLM-5.3-Fast": ["low", "high", "max"],
|
|
716
|
+
"zai-org/GLM-5.2": ["high", "max"],
|
|
717
|
+
"zai-org/GLM-5.2-Fast": ["high", "max"],
|
|
718
|
+
};
|
|
719
|
+
export const BASETEN_MODEL_REASONING_EFFORT_MAP: Record<string, Record<string, string>> = {
|
|
720
|
+
"thinkingmachines/inkling": { none: "none", minimal: "minimal" },
|
|
721
|
+
"openai/gpt-oss-120b": { none: "none", minimal: "minimal" },
|
|
722
|
+
"moonshotai/Kimi-K3": { none: "none" },
|
|
723
|
+
"zai-org/GLM-5.3": { none: "none" },
|
|
724
|
+
"zai-org/GLM-5.3-Fast": { none: "none" },
|
|
725
|
+
"zai-org/GLM-5.2": { none: "none" },
|
|
726
|
+
"zai-org/GLM-5.2-Fast": { none: "none" },
|
|
727
|
+
};
|
|
728
|
+
export const BASETEN_MODEL_DEFAULT_REASONING_EFFORTS: Record<string, string> = {
|
|
729
|
+
"thinkingmachines/inkling": "high",
|
|
730
|
+
"openai/gpt-oss-120b": "medium",
|
|
731
|
+
"moonshotai/Kimi-K3": "max",
|
|
732
|
+
};
|
|
733
|
+
export const BASETEN_MODEL_INPUT_MODALITIES: Record<string, string[]> = {
|
|
734
|
+
"thinkingmachines/inkling": ["text", "image"],
|
|
735
|
+
"moonshotai/Kimi-K2.6": ["text", "image"],
|
|
736
|
+
"moonshotai/Kimi-K2.7-Code": ["text", "image"],
|
|
737
|
+
"moonshotai/Kimi-K3": ["text", "image"],
|
|
738
|
+
};
|
|
739
|
+
|
|
740
|
+
// 260801 DigitalOcean and Scaleway expose OpenAI-shaped `/v1/models` rows with only
|
|
741
|
+
// id/object/created/owned_by, while their shared serverless catalogs also contain
|
|
742
|
+
// non-chat and endpoint-specific models. Fail closed by intersecting live discovery
|
|
743
|
+
// with ids that the providers' current first-party model tables establish for Chat
|
|
744
|
+
// Completions. A newly listed id therefore needs a docs-backed registry refresh before
|
|
745
|
+
// it can enter the Codex catalog.
|
|
746
|
+
// Evidence: https://docs.digitalocean.com/products/inference/details/models/
|
|
747
|
+
// https://docs.digitalocean.com/reference/api/reference/serverless-inference/
|
|
748
|
+
// https://www.scaleway.com/en/docs/generative-apis/reference-content/supported-models/
|
|
749
|
+
export const DIGITALOCEAN_CHAT_COMPLETION_MODELS = [
|
|
750
|
+
"arcee-trinity-large-thinking",
|
|
751
|
+
"openai-gpt-5.6-sol",
|
|
752
|
+
"openai-gpt-5.6-terra",
|
|
753
|
+
"openai-gpt-5.6-luna",
|
|
754
|
+
"qwen3-coder-flash",
|
|
755
|
+
"qwen3.5-397b-a17b",
|
|
756
|
+
"deepseek-4-flash",
|
|
757
|
+
"deepseek-3.2",
|
|
758
|
+
"gemma-4-31B-it",
|
|
759
|
+
"minimax-m2.5",
|
|
760
|
+
"kimi-k3",
|
|
761
|
+
"kimi-k2.6",
|
|
762
|
+
"kimi-k2.5",
|
|
763
|
+
"llama3.3-70b-instruct",
|
|
764
|
+
"llama-4-maverick",
|
|
765
|
+
"mistral-3-14B",
|
|
766
|
+
"nemotron-3-ultra-550b",
|
|
767
|
+
"nvidia-nemotron-3-super-120b",
|
|
768
|
+
"nemotron-3-nano-omni",
|
|
769
|
+
"nemotron-nano-12b-v2-vl",
|
|
770
|
+
"mimo-v2.5-pro",
|
|
771
|
+
"glm-5.3",
|
|
772
|
+
"glm-5.3-flash",
|
|
773
|
+
"glm-5.2",
|
|
774
|
+
"glm-5.1",
|
|
775
|
+
"glm-5",
|
|
776
|
+
// The API reference uses this native slash id in its Chat Completions example.
|
|
777
|
+
"meta-llama/Meta-Llama-3.1-8B-Instruct",
|
|
778
|
+
] as const;
|
|
779
|
+
export const SCALEWAY_SERVERLESS_CHAT_MODELS = [
|
|
780
|
+
"glm-5.3",
|
|
781
|
+
"glm-5.3-flash",
|
|
782
|
+
"glm-5.2",
|
|
783
|
+
// gpt-oss-120b is intentionally omitted: Scaleway requires Responses API for tool calling,
|
|
784
|
+
// while this preset routes Codex agent tools through Chat Completions.
|
|
785
|
+
"qwen3.6-35b-a3b",
|
|
786
|
+
"qwen3.5-397b-a17b",
|
|
787
|
+
"qwen3-235b-a22b-instruct-2507",
|
|
788
|
+
"qwen3-coder-30b-a3b-instruct",
|
|
789
|
+
"gemma-4-26b-a4b-it",
|
|
790
|
+
"llama-3.3-70b-instruct",
|
|
791
|
+
"mistral-medium-3.5-128b",
|
|
792
|
+
"mistral-small-3.2-24b-instruct-2506",
|
|
793
|
+
"pixtral-12b-2409",
|
|
794
|
+
] as const;
|
|
795
|
+
export const SCALEWAY_MODEL_INPUT_MODALITIES: Record<string, string[]> = {
|
|
796
|
+
"pixtral-12b-2409": ["text", "image"],
|
|
797
|
+
};
|
|
798
|
+
export const UMANS_MODELS = [
|
|
799
|
+
"umans-coder",
|
|
800
|
+
"umans-kimi-k2.7",
|
|
801
|
+
"umans-flash",
|
|
802
|
+
"umans-glm-5.3",
|
|
803
|
+
"umans-glm-5.3-flash",
|
|
804
|
+
"umans-glm-5.2",
|
|
805
|
+
"umans-glm-5.1",
|
|
806
|
+
"umans-qwen3.6-35b-a3b",
|
|
807
|
+
];
|
|
808
|
+
export const UMANS_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
809
|
+
export const UMANS_GLM_REASONING_EFFORTS = ["high", "xhigh", "max"];
|
|
810
|
+
// 260814: Z.AI folds GLM-5.3 efforts into low/high/max, so `low` is a real tier here and
|
|
811
|
+
// `xhigh` is not distinct from `max` (docs.z.ai/devpack/latest-model).
|
|
812
|
+
export const UMANS_GLM_53_REASONING_EFFORTS = ["low", "high", "max"];
|
|
813
|
+
// `umans-glm-5.3-flash` is NOT here: Z.AI documents glm-5.3-flash under
|
|
814
|
+
// docs.z.ai/guides/vlm/, so it takes images natively and does not need the proxy's
|
|
815
|
+
// vision sidecar. The seeding pass classified it from the family name and a later
|
|
816
|
+
// pass corrected only some of the providers; this is one it missed.
|
|
817
|
+
export const UMANS_TEXT_ONLY_MODELS = ["umans-glm-5.3", "umans-glm-5.2", "umans-glm-5.1"];
|
|
818
|
+
export const UMANS_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
|
|
819
|
+
"umans-coder": 262_144,
|
|
820
|
+
"umans-kimi-k2.7": 262_144,
|
|
821
|
+
"umans-flash": 262_144,
|
|
822
|
+
"umans-glm-5.3": 405_504,
|
|
823
|
+
// Mirrors the sibling this provider already carries. Umans has not published a
|
|
824
|
+
// separate window for the flash tier; asserting a different number would be a guess.
|
|
825
|
+
"umans-glm-5.3-flash": 405_504,
|
|
826
|
+
"umans-glm-5.2": 405_504,
|
|
827
|
+
"umans-glm-5.1": 202_752,
|
|
828
|
+
"umans-qwen3.6-35b-a3b": 262_144,
|
|
829
|
+
};
|
|
830
|
+
export const UMANS_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
|
|
831
|
+
UMANS_MODELS.map(id => [id, UMANS_TEXT_ONLY_MODELS.includes(id) ? ["text"] : ["text", "image"]]),
|
|
832
|
+
);
|
|
833
|
+
export const CLINE_PASS_MODELS = [
|
|
834
|
+
"cline-pass/glm-5.3",
|
|
835
|
+
"cline-pass/glm-5.3-flash",
|
|
836
|
+
"cline-pass/glm-5.2",
|
|
837
|
+
"cline-pass/kimi-k3",
|
|
838
|
+
"cline-pass/kimi-k2.7-code",
|
|
839
|
+
"cline-pass/kimi-k2.6",
|
|
840
|
+
"cline-pass/deepseek-v4-flash",
|
|
841
|
+
"cline-pass/mimo-v2.5",
|
|
842
|
+
"cline-pass/mimo-v2.5-pro",
|
|
843
|
+
"cline-pass/minimax-m3",
|
|
844
|
+
"cline-pass/qwen3.8-max",
|
|
845
|
+
"cline-pass/qwen3.7-max",
|
|
846
|
+
"cline-pass/qwen3.7-plus",
|
|
847
|
+
];
|
|
848
|
+
|
|
849
|
+
export const ORCAROUTER_MODEL_DISCOVERY: ProviderModelDiscoverySpec = {
|
|
850
|
+
path: "models",
|
|
851
|
+
query: { capability: "chat" },
|
|
852
|
+
maxResponseBytes: 512 * 1024,
|
|
853
|
+
maxModels: 512,
|
|
854
|
+
filter: {
|
|
855
|
+
anyOf: [{
|
|
856
|
+
path: ["supported_endpoint_types"],
|
|
857
|
+
containsAny: ["openai", "openai-response", "anthropic", "gemini"],
|
|
858
|
+
caseInsensitive: true,
|
|
859
|
+
}],
|
|
860
|
+
noneOf: [{
|
|
861
|
+
path: ["supported_endpoint_types"],
|
|
862
|
+
containsAny: ["image-generation", "openai-video", "jina-rerank"],
|
|
863
|
+
caseInsensitive: true,
|
|
864
|
+
}],
|
|
865
|
+
},
|
|
866
|
+
};
|
|
867
|
+
// Preserve the previously verified cold-start catalog. Live discovery remains authoritative
|
|
868
|
+
// when it succeeds, but a temporary catalog outage must not erase the provider's known-good
|
|
869
|
+
// selectors from the picker. `orcarouter/auto` is intentionally retained here even though the
|
|
870
|
+
// public catalog did not enumerate it at the latest verification (2026-09-07).
|
|
871
|
+
export const ORCAROUTER_MODELS = [
|
|
872
|
+
"openai/gpt-5.5",
|
|
873
|
+
"anthropic/claude-opus-4.8",
|
|
874
|
+
"google/gemini-3.5-flash",
|
|
875
|
+
"orcarouter/auto",
|
|
876
|
+
];
|
|
877
|
+
export const ORCAROUTER_MODEL_REASONING_EFFORTS = {
|
|
878
|
+
// Live /models currently exposes ids and modalities, not the accepted reasoning ladder.
|
|
879
|
+
"openai/gpt-5.5": ["low", "medium", "high", "xhigh"],
|
|
880
|
+
};
|
|
881
|
+
export const CLINE_PASS_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
|
|
882
|
+
"cline-pass/glm-5.3": 1_048_576,
|
|
883
|
+
"cline-pass/glm-5.3-flash": 1_048_576,
|
|
884
|
+
"cline-pass/glm-5.2": 1_048_576,
|
|
885
|
+
"cline-pass/kimi-k3": 1_048_576,
|
|
886
|
+
"cline-pass/kimi-k2.7-code": 262_144,
|
|
887
|
+
"cline-pass/kimi-k2.6": 262_144,
|
|
888
|
+
"cline-pass/deepseek-v4-flash": 1_048_576,
|
|
889
|
+
"cline-pass/mimo-v2.5": 1_050_000,
|
|
890
|
+
"cline-pass/mimo-v2.5-pro": 1_050_000,
|
|
891
|
+
"cline-pass/minimax-m3": 1_048_576,
|
|
892
|
+
"cline-pass/qwen3.7-max": 1_000_000,
|
|
893
|
+
"cline-pass/qwen3.7-plus": 1_000_000,
|
|
894
|
+
};
|
|
895
|
+
export const CLINE_PASS_IMAGE_MODELS = new Set([
|
|
896
|
+
"cline-pass/kimi-k3",
|
|
897
|
+
"cline-pass/kimi-k2.7-code",
|
|
898
|
+
"cline-pass/kimi-k2.6",
|
|
899
|
+
"cline-pass/mimo-v2.5",
|
|
900
|
+
"cline-pass/minimax-m3",
|
|
901
|
+
"cline-pass/qwen3.7-plus",
|
|
902
|
+
// Native VLM (docs.z.ai/guides/vlm/), so its images do not go through the proxy's
|
|
903
|
+
// sidecar. Adding it here moves it out of CLINE_PASS_TEXT_ONLY_MODELS and flips its
|
|
904
|
+
// declared modalities to ["text", "image"] in one edit, because both are derived
|
|
905
|
+
// from this set.
|
|
906
|
+
"cline-pass/glm-5.3-flash",
|
|
907
|
+
]);
|
|
908
|
+
export const CLINE_PASS_MODALITY_KNOWN_MODELS = CLINE_PASS_MODELS.filter(id => id !== "cline-pass/qwen3.8-max");
|
|
909
|
+
export const CLINE_PASS_TEXT_ONLY_MODELS = CLINE_PASS_MODALITY_KNOWN_MODELS.filter(id => !CLINE_PASS_IMAGE_MODELS.has(id));
|
|
910
|
+
export const CLINE_PASS_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
|
|
911
|
+
CLINE_PASS_MODALITY_KNOWN_MODELS.map(id => [id, CLINE_PASS_IMAGE_MODELS.has(id) ? ["text", "image"] : ["text"]]),
|
|
912
|
+
);
|