@bitkyc08/opencodex 2.55.0 → 2.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/ocx.mjs +10 -0
- package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
- package/gui/dist/assets/{index-VuoiWj9J.js → index-Cz7CLdif.js} +21 -21
- package/gui/dist/index.html +2 -2
- package/package.json +4 -3
- package/src/adapters/base.ts +21 -0
- package/src/adapters/codebuddy/adapter.ts +2 -1
- package/src/adapters/codebuddy/scaffold-guard.ts +248 -0
- package/src/adapters/command-code.ts +1 -1
- package/src/adapters/cursor/envelope-echo.ts +8 -2
- package/src/adapters/cursor/transport-retry.ts +46 -1
- package/src/adapters/cursor.ts +4 -0
- package/src/adapters/google.ts +7 -7
- package/src/adapters/kiro/adapter.ts +42 -1
- package/src/adapters/kiro/payload.ts +17 -3
- package/src/adapters/kiro/reasoning.ts +70 -7
- package/src/adapters/kiro/stream.ts +8 -2
- package/src/adapters/kiro/wire.ts +2 -1
- package/src/adapters/kiro-events.ts +21 -13
- package/src/adapters/kiro-retry.ts +23 -4
- package/src/adapters/openai-chat/errors.ts +116 -0
- package/src/adapters/openai-chat/messages.ts +346 -0
- package/src/adapters/openai-chat/passthrough.ts +146 -0
- package/src/adapters/openai-chat/response-events.ts +117 -0
- package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
- package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
- package/src/adapters/openai-chat/tool-schema.ts +495 -0
- package/src/adapters/openai-chat/wire.ts +50 -0
- package/src/adapters/openai-chat.ts +40 -1452
- package/src/adapters/openai-responses/canonical-forward.ts +202 -0
- package/src/adapters/openai-responses/image-gen.ts +406 -0
- package/src/adapters/openai-responses/internal.ts +3 -0
- package/src/adapters/openai-responses/passthrough.ts +642 -0
- package/src/adapters/openai-responses/prompt-cache.ts +83 -0
- package/src/adapters/openai-responses/reasoning.ts +220 -0
- package/src/adapters/openai-responses/request-strips.ts +185 -0
- package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
- package/src/adapters/openai-responses/tool-schema.ts +293 -0
- package/src/adapters/openai-responses/web-search.ts +156 -0
- package/src/adapters/openai-responses.ts +4 -2625
- package/src/bridge/errors.ts +58 -0
- package/src/bridge/internal.ts +174 -0
- package/src/bridge/response-json.ts +630 -0
- package/src/bridge/sse.ts +1462 -0
- package/src/bridge.ts +5 -2204
- package/src/chat/inbound.ts +12 -1
- package/src/claude/desktop-profile.ts +66 -9
- package/src/claude/outbound.ts +18 -0
- package/src/cli/account-main.ts +1 -1
- package/src/cli/capabilities.ts +2 -2
- package/src/cli/combo.ts +10 -1
- package/src/cli/index.ts +48 -5
- package/src/cli/registry.ts +2 -1
- package/src/cli/system-command.ts +4 -4
- package/src/clients/config-export.ts +7 -3
- package/src/codex/account-label.ts +14 -3
- package/src/codex/account-lifecycle.ts +3 -0
- package/src/codex/account-store.ts +184 -35
- package/src/codex/account-usability.ts +21 -0
- package/src/codex/auth-api/account-list.ts +507 -0
- package/src/codex/auth-api/http.ts +32 -0
- package/src/codex/auth-api/login-flow.ts +566 -0
- package/src/codex/auth-api/login-state.ts +64 -0
- package/src/codex/auth-api/main-account-probe.ts +331 -0
- package/src/codex/auth-api/pool-mode-gate.ts +274 -0
- package/src/codex/auth-api/pool-quota-probe.ts +512 -0
- package/src/codex/auth-api/reset-credit-service.ts +431 -0
- package/src/codex/auth-api/routes.ts +425 -0
- package/src/codex/auth-api/runtime-config.ts +48 -0
- package/src/codex/auth-api.ts +27 -3118
- package/src/codex/auth-context.ts +252 -35
- package/src/codex/catalog/aggregation.ts +80 -1
- package/src/codex/catalog/auto-review.ts +507 -0
- package/src/codex/catalog/build-entries.ts +981 -0
- package/src/codex/catalog/combo-member.ts +375 -0
- package/src/codex/catalog/derive-entry.ts +229 -0
- package/src/codex/catalog/effort.ts +0 -1
- package/src/codex/catalog/gated-native-warn.ts +63 -0
- package/src/codex/catalog/gather-capture.ts +533 -0
- package/src/codex/catalog/model-hints.ts +691 -0
- package/src/codex/catalog/model-visibility.ts +305 -0
- package/src/codex/catalog/provider-fetch.ts +52 -2942
- package/src/codex/catalog/provider-models.ts +685 -0
- package/src/codex/catalog/remote.ts +30 -0
- package/src/codex/catalog/restore.ts +132 -0
- package/src/codex/catalog/retained-sync.ts +714 -0
- package/src/codex/catalog/routed-gather.ts +895 -0
- package/src/codex/catalog/subagent-roster.ts +176 -0
- package/src/codex/catalog/sync.ts +52 -2698
- package/src/codex/cli-install-provenance.ts +7 -1
- package/src/codex/convergence.ts +7 -2
- package/src/codex/desktop-app/types.ts +11 -2
- package/src/codex/desktop-app/windows.ts +5 -5
- package/src/codex/inject/config-toml.ts +563 -0
- package/src/codex/inject/remove.ts +192 -0
- package/src/codex/inject/restore.ts +567 -0
- package/src/codex/inject/routing-classify.ts +109 -0
- package/src/codex/inject/routing-target.ts +125 -0
- package/src/codex/inject.ts +89 -1444
- package/src/codex/lineage.ts +458 -0
- package/src/codex/model-entitlements.ts +152 -15
- package/src/codex/pool-refresh-backoff.ts +161 -0
- package/src/codex/quota-rejection.ts +104 -15
- package/src/codex/routing/active-account.ts +194 -0
- package/src/codex/routing/cache-affinity.ts +70 -0
- package/src/codex/routing/cooldown-math.ts +285 -0
- package/src/codex/routing/health-store.ts +402 -0
- package/src/codex/routing/probe-lease.ts +358 -0
- package/src/codex/routing/selection.ts +780 -0
- package/src/codex/routing/thread-affinity.ts +586 -0
- package/src/codex/routing/transient-hold-dispatch.ts +141 -0
- package/src/codex/routing.ts +370 -2271
- package/src/codex/shim-fingerprint.ts +223 -0
- package/src/codex/shim-inspect.ts +175 -0
- package/src/codex/shim-probe.ts +367 -0
- package/src/codex/shim-restore-lock.ts +169 -0
- package/src/codex/shim-state-file.ts +151 -0
- package/src/codex/shim-templates.ts +265 -0
- package/src/codex/shim.ts +48 -1268
- package/src/codex/warmup.ts +1 -1
- package/src/combos/failover.ts +85 -0
- package/src/combos/request.ts +17 -10
- package/src/combos/types.ts +23 -2
- package/src/config/diagnostics.ts +705 -0
- package/src/config/feature-flags.ts +55 -0
- package/src/config/live-reconcile.ts +403 -0
- package/src/config/load-degrade.ts +880 -0
- package/src/config/mutation-lock.ts +244 -0
- package/src/config/openai-tier-backup.ts +268 -0
- package/src/config/pending-teardown.ts +31 -0
- package/src/config/persist-unlocked.ts +92 -0
- package/src/config/proxy-env.ts +188 -0
- package/src/config/salvage.ts +244 -0
- package/src/config/schema/config-schema.ts +640 -0
- package/src/config/schema/leaf-validators.ts +855 -0
- package/src/config/warn-memo.ts +28 -0
- package/src/config.ts +234 -4481
- package/src/generated/compatibility-version.json +649 -121
- package/src/images/loop.ts +1 -1
- package/src/lib/errors.ts +17 -0
- package/src/lib/request-execution-budget.ts +198 -23
- package/src/lib/spend-reservation-ledger.ts +958 -0
- package/src/lib/state-store-registrations.ts +6 -2
- package/src/lib/test-home-guard.ts +85 -1
- package/src/lib/upstream-retry.ts +132 -21
- package/src/lib/windows-elevation.ts +76 -14
- package/src/lib/workflow-budget.ts +553 -30
- package/src/oauth/index.ts +2 -2
- package/src/oauth/key-providers.ts +2 -2
- package/src/providers/kiro-models.ts +4 -3
- package/src/providers/label.ts +19 -1
- package/src/providers/model-discovery.ts +16 -0
- package/src/providers/quota/account-cache.ts +441 -0
- package/src/providers/quota/antigravity.ts +295 -0
- package/src/providers/quota/report-cache.ts +320 -0
- package/src/providers/quota/vendor-probes-key.ts +1243 -0
- package/src/providers/quota/vendor-probes-oauth.ts +590 -0
- package/src/providers/quota.ts +324 -3079
- package/src/providers/registry/entries-core.ts +1228 -0
- package/src/providers/registry/entries-extended.ts +1213 -0
- package/src/providers/registry/model-seeds.ts +912 -0
- package/src/providers/registry/types.ts +352 -0
- package/src/providers/registry.ts +24 -3536
- package/src/responses/continuation-ownership.ts +29 -0
- package/src/responses/reasoning-envelope.ts +6 -3
- package/src/responses/state/replay-fingerprint.ts +80 -0
- package/src/responses/state/snapshot-codec.ts +104 -0
- package/src/responses/state/spill-failure.ts +118 -0
- package/src/responses/state/spill-queue.ts +665 -0
- package/src/responses/state/temp-recovery.ts +257 -0
- package/src/responses/state.ts +82 -1143
- package/src/routing/identity-domains.ts +456 -0
- package/src/routing/probe-lease.ts +613 -0
- package/src/server/chat-completions.ts +3 -1
- package/src/server/chat-native.ts +37 -9
- package/src/server/index/bounded-request.ts +88 -0
- package/src/server/index/live-sideband.ts +601 -0
- package/src/server/index/serve-options.ts +1766 -0
- package/src/server/index/startup-warnings.ts +213 -0
- package/src/server/index/websocket-handler.ts +339 -0
- package/src/server/index.ts +45 -2552
- package/src/server/inspection-tee.ts +107 -0
- package/src/server/live.ts +46 -1
- package/src/server/management/combo-routes.ts +10 -1
- package/src/server/management/route-registry.ts +26 -23
- package/src/server/management/shared.ts +8 -5
- package/src/server/management/workflow-budget-routes.ts +133 -0
- package/src/server/management-api.ts +12 -0
- package/src/server/relay-eager.ts +2 -0
- package/src/server/relay.ts +14 -19
- package/src/server/request-log-conversation.ts +9 -7
- package/src/server/request-log.ts +372 -4
- package/src/server/response-log-body.ts +153 -0
- package/src/server/responses/account-change-state.ts +307 -0
- package/src/server/responses/adapter-continuation.ts +540 -0
- package/src/server/responses/adapter-delivery.ts +208 -0
- package/src/server/responses/adapter-dispatch.ts +1042 -0
- package/src/server/responses/codex-ws-wire.ts +5 -0
- package/src/server/responses/collaboration.ts +74 -4
- package/src/server/responses/combo-session-recall.ts +68 -8
- package/src/server/responses/compact.ts +113 -17
- package/src/server/responses/completion-policy.ts +33 -0
- package/src/server/responses/core-auth.ts +529 -0
- package/src/server/responses/core-codex-account.ts +907 -0
- package/src/server/responses/core-combo-failure.ts +210 -0
- package/src/server/responses/core-combo.ts +787 -0
- package/src/server/responses/core-errors.ts +170 -0
- package/src/server/responses/core-lifetime.ts +95 -0
- package/src/server/responses/core-normalize.ts +350 -0
- package/src/server/responses/core-opaque-recovery.ts +380 -0
- package/src/server/responses/core-options.ts +159 -0
- package/src/server/responses/core-replay.ts +298 -0
- package/src/server/responses/core.ts +192 -8893
- package/src/server/responses/encrypted-payload.ts +0 -1
- package/src/server/responses/input-admission.ts +126 -6
- package/src/server/responses/passthrough-delivery.ts +869 -0
- package/src/server/responses/passthrough-dispatch.ts +1494 -0
- package/src/server/responses/passthrough-error.ts +38 -2
- package/src/server/responses/passthrough-execution.ts +54 -0
- package/src/server/responses/request-prepare.ts +1080 -0
- package/src/server/responses/request-send-budget.ts +259 -0
- package/src/server/responses/request-sidecar-auth.ts +149 -0
- package/src/server/responses/request-spend.ts +147 -0
- package/src/server/responses/request-transport.ts +803 -0
- package/src/server/responses/response-effects.ts +157 -0
- package/src/server/responses/run-turn-execution.ts +476 -0
- package/src/server/responses/sidecar-execution.ts +463 -0
- package/src/server/responses/terminal-guard.ts +65 -4
- package/src/server/responses-image-gen-repair.ts +1 -1
- package/src/server/responses-undeclared-tool-guard.ts +9 -5
- package/src/server/workflow-refusal.ts +84 -0
- package/src/service/windows-ops.ts +210 -16
- package/src/service/windows-scheduler.ts +28 -21
- package/src/service.ts +1 -1
- package/src/types/config.ts +34 -1
- package/src/types/request.ts +8 -5
- package/src/types/tools.ts +24 -0
- package/src/types.ts +2 -0
- package/src/update/index.ts +10 -0
- package/src/update/stop-contract.d.mts +1 -0
- package/src/update/stop-contract.mjs +19 -0
- package/src/update/stop-decision.d.mts +1 -1
- package/src/update/stop-decision.mjs +12 -3
- package/src/usage/log.ts +147 -1
- package/src/usage/summary.ts +171 -21
- package/src/vision/anthropic-describe.ts +1 -1
- package/src/vision/describe.ts +5 -5
- package/src/web-search/anthropic-executor.ts +1 -1
- package/src/web-search/exa-executor.ts +1 -1
- package/src/web-search/executor.ts +1 -1
- package/src/web-search/gemini-executor.ts +1 -1
- package/src/web-search/loop.ts +1 -1
- package/src/web-search/ollama-executor.ts +1 -1
- package/src/web-search/parse.ts +67 -14
- package/src/web-search/passthrough-bridge.ts +64 -31
- package/src/web-search/xai-executor.ts +1 -1
|
@@ -0,0 +1,1228 @@
|
|
|
1
|
+
import { KIRO_MODELS, KIRO_MODEL_CONTEXT_WINDOWS, KIRO_MODEL_REASONING_EFFORTS } from "../kiro-models";
|
|
2
|
+
import { DEVIN_MODEL_CONTEXT_WINDOWS, DEVIN_MODEL_EFFORTS, DEVIN_DEFAULT_EFFORTS } from "../../adapters/devin/live-models";
|
|
3
|
+
import { ANTIGRAVITY_MODELS, ANTIGRAVITY_MODEL_CONTEXT_WINDOWS, ANTIGRAVITY_MODEL_EFFORTS, ANTIGRAVITY_MODEL_INPUT_MODALITIES } from "../antigravity-models";
|
|
4
|
+
import {
|
|
5
|
+
CURSOR_NO_VISION_MODELS,
|
|
6
|
+
CURSOR_STATIC_MODELS,
|
|
7
|
+
cursorModelContextWindows,
|
|
8
|
+
cursorModelDisplayNames,
|
|
9
|
+
cursorModelIds,
|
|
10
|
+
cursorModelInputModalities,
|
|
11
|
+
cursorModelReasoningEfforts,
|
|
12
|
+
} from "../../adapters/cursor/discovery";
|
|
13
|
+
import { cursorFastCapableBases } from "../../adapters/cursor/catalog";
|
|
14
|
+
import { COMMAND_CODE_MODEL_REASONING_EFFORTS } from "../command-code-efforts";
|
|
15
|
+
import { isCanonicalOpenRouterTarget } from "../openrouter-routing";
|
|
16
|
+
import type { ProviderRegistryEntry } from "./types";
|
|
17
|
+
import {
|
|
18
|
+
ANTHROPIC_MODELS,
|
|
19
|
+
ANTHROPIC_MODEL_CONTEXT_WINDOWS,
|
|
20
|
+
ANTHROPIC_MODEL_INPUT_MODALITIES,
|
|
21
|
+
ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
|
|
22
|
+
ANTHROPIC_MODEL_REASONING_EFFORTS,
|
|
23
|
+
ZAI_GLM_52_REASONING_EFFORTS,
|
|
24
|
+
ZAI_GLM_53_REASONING_EFFORTS,
|
|
25
|
+
OPENAI_GPT56_MODELS,
|
|
26
|
+
OPENAI_GPT56_PRO_MODELS,
|
|
27
|
+
OPENAI_API_GPT56_CONTEXT_WINDOWS,
|
|
28
|
+
OPENAI_API_GPT56_MAX_INPUT_TOKENS,
|
|
29
|
+
OPENAI_API_GPT56_VIRTUAL_MODELS,
|
|
30
|
+
OPENAI_API_GPT56_REASONING_EFFORTS,
|
|
31
|
+
META_MUSE_REASONING_EFFORTS,
|
|
32
|
+
META_MUSE_REASONING_EFFORT_MAP,
|
|
33
|
+
META_MUSE_CONTEXT_WINDOW,
|
|
34
|
+
META_MUSE_MODELS,
|
|
35
|
+
OPENAI_DAYBREAK_MODELS,
|
|
36
|
+
OPENAI_DAYBREAK_CONTEXT_WINDOWS,
|
|
37
|
+
OPENAI_DAYBREAK_MAX_INPUT_TOKENS,
|
|
38
|
+
OPENAI_DAYBREAK_REASONING_EFFORTS,
|
|
39
|
+
OPENROUTER_GPT56_MODELS,
|
|
40
|
+
XAI_MODELS,
|
|
41
|
+
OPENROUTER_GPT56_CONTEXT_WINDOWS,
|
|
42
|
+
THINKING_TOGGLE_EFFORTS,
|
|
43
|
+
THINKING_TOGGLE_MAP,
|
|
44
|
+
OPENCODE_GO_THINKING_TOGGLE_MODELS,
|
|
45
|
+
THINKING_BUDGET_EFFORTS,
|
|
46
|
+
QWEN38_REASONING_EFFORTS,
|
|
47
|
+
THINKING_BUDGET_MODELS,
|
|
48
|
+
OPENCODE_GO_THINKING_BUDGET_MODELS,
|
|
49
|
+
DEEPSEEK_NATIVE_THINKING_MODELS,
|
|
50
|
+
DEEPSEEK_GATEWAY_THINKING_MODELS,
|
|
51
|
+
DEEPSEEK_VISION_PREVIEW_MODEL,
|
|
52
|
+
COMMAND_CODE_MODEL_INPUT_MODALITIES,
|
|
53
|
+
deepseekThinkingEffortsFor,
|
|
54
|
+
deepseekReasoningMapFor,
|
|
55
|
+
KIMI_K3_STANDARD_CONTEXT_WINDOW,
|
|
56
|
+
KIMI_CODING_MODELS,
|
|
57
|
+
KIMI_THINKING_MODELS,
|
|
58
|
+
KIMI_CODING_NO_REASONING_MODELS,
|
|
59
|
+
KIMI_CODING_K3_REASONING_EFFORTS,
|
|
60
|
+
KIMI_CODING_K3_REASONING_EFFORT_MAP,
|
|
61
|
+
KIMI_CODING_REASONING_EFFORTS,
|
|
62
|
+
KIMI_CODING_DEFAULT_REASONING_EFFORTS,
|
|
63
|
+
KIMI_CODING_REASONING_EFFORT_MAPS,
|
|
64
|
+
KIMI_LOCKED_PARAMETER_MODELS,
|
|
65
|
+
KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS,
|
|
66
|
+
KIMI_CODING_MODEL_CONTEXT_WINDOWS,
|
|
67
|
+
KIMI_CODING_MODEL_INPUT_MODALITIES,
|
|
68
|
+
NEURALWATT_REASONING_HISTORY_MODELS,
|
|
69
|
+
UMANS_MODELS,
|
|
70
|
+
UMANS_REASONING_EFFORTS,
|
|
71
|
+
UMANS_GLM_REASONING_EFFORTS,
|
|
72
|
+
UMANS_GLM_53_REASONING_EFFORTS,
|
|
73
|
+
UMANS_TEXT_ONLY_MODELS,
|
|
74
|
+
UMANS_MODEL_CONTEXT_WINDOWS,
|
|
75
|
+
UMANS_MODEL_INPUT_MODALITIES,
|
|
76
|
+
CLINE_PASS_MODELS,
|
|
77
|
+
ORCAROUTER_MODEL_DISCOVERY,
|
|
78
|
+
ORCAROUTER_MODELS,
|
|
79
|
+
ORCAROUTER_MODEL_REASONING_EFFORTS,
|
|
80
|
+
CLINE_PASS_MODEL_CONTEXT_WINDOWS,
|
|
81
|
+
CLINE_PASS_TEXT_ONLY_MODELS,
|
|
82
|
+
CLINE_PASS_MODEL_INPUT_MODALITIES,
|
|
83
|
+
} from "./model-seeds";
|
|
84
|
+
|
|
85
|
+
export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
86
|
+
{
|
|
87
|
+
id: "openai",
|
|
88
|
+
label: "OpenAI (Codex login)",
|
|
89
|
+
adapter: "openai-responses",
|
|
90
|
+
baseUrl: "https://chatgpt.com/backend-api/codex",
|
|
91
|
+
authKind: "forward",
|
|
92
|
+
codexAccountMode: "pool",
|
|
93
|
+
supportsServiceTier: true,
|
|
94
|
+
featured: true,
|
|
95
|
+
note: "Codex login account pool (default) or Direct main-account mode via codexAccountMode",
|
|
96
|
+
},
|
|
97
|
+
{
|
|
98
|
+
id: "cursor",
|
|
99
|
+
label: "Cursor (experimental)",
|
|
100
|
+
adapter: "cursor",
|
|
101
|
+
baseUrl: "https://api2.cursor.sh",
|
|
102
|
+
authKind: "oauth",
|
|
103
|
+
featured: false,
|
|
104
|
+
dashboardPreset: true,
|
|
105
|
+
note: "Experimental Cursor bridge. Live transport and live model discovery are enabled after a standalone PKCE browser login via 'ocx login cursor'; native read/write/delete/shell/fetch execution is disabled by default and request text such as Codex sandbox markers never authorizes it. Set \"nativeLocalExec\": \"on\" on providers.cursor in ~/.opencodex/config.json (dashboard: Providers → Cursor → Edit JSON) only for a trusted local experiment where every data-plane caller is trusted. \"off\" denies all, \"codex-sandbox\" is accepted for backwards compatibility but fails closed, and legacy \"unsafeAllowNativeLocalExec\": true still means explicit operator opt-in.",
|
|
106
|
+
models: cursorModelIds(CURSOR_STATIC_MODELS),
|
|
107
|
+
liveModels: true,
|
|
108
|
+
defaultModel: "auto",
|
|
109
|
+
modelContextWindows: cursorModelContextWindows(CURSOR_STATIC_MODELS),
|
|
110
|
+
modelDisplayNames: cursorModelDisplayNames(),
|
|
111
|
+
// Cursor's Fast product is a model VARIANT, not a service_tier field, so the wire kind
|
|
112
|
+
// is cursor-variant and the request builder consumes the decision.
|
|
113
|
+
fastWire: { kind: "cursor-variant", canonicalToWire: { priority: "fast" }, foreignCallerTiers: "drop" },
|
|
114
|
+
// Deliberately NO provider-level supportsServiceTier: resolveFastPolicy short-circuits on
|
|
115
|
+
// `capability.provider === false` BEFORE consulting the per-model map, which would make
|
|
116
|
+
// these entries dead config. Absent leaves unlisted bases "unclassified", and a
|
|
117
|
+
// non-service-tier adapter cannot forward a caller tier, so they still publish no toggle.
|
|
118
|
+
modelSupportsServiceTier: Object.fromEntries(cursorFastCapableBases().map(id => [id, true])),
|
|
119
|
+
fastTierDescription: "Cursor Fast variant",
|
|
120
|
+
modelInputModalities: cursorModelInputModalities(CURSOR_STATIC_MODELS),
|
|
121
|
+
modelReasoningEfforts: cursorModelReasoningEfforts(CURSOR_STATIC_MODELS),
|
|
122
|
+
// Kimi K3 documents `max` as its API default, and its Cursor ladder has no `medium`
|
|
123
|
+
// rung — so applyReasoningLevels' medium->high->first fallback would settle the catalog
|
|
124
|
+
// default on `high`, the picker would send `high` explicitly, and the request builder's
|
|
125
|
+
// no-effort fallback to `kimi-k3-max` would never be reached. Mirrors the other K3
|
|
126
|
+
// routes (kimi, kimi-code, opencode-go).
|
|
127
|
+
modelDefaultReasoningEfforts: { "kimi-k3": "max" },
|
|
128
|
+
// Blind Cursor models (Auto routers, Composer, GLM-5.2, GLM-5.3) go through the vision sidecar;
|
|
129
|
+
// multimodal hosts (Claude/Gemini/GPT/Kimi/Grok) take native SelectedImage. The catalog
|
|
130
|
+
// still advertises image for noVision members so Codex can attach (sidecar option B).
|
|
131
|
+
noVisionModels: [...CURSOR_NO_VISION_MODELS],
|
|
132
|
+
},
|
|
133
|
+
{
|
|
134
|
+
// The canonical Cognition account provider, after absorbing `devin-cli`
|
|
135
|
+
// (devlog/_plan/260913_devin_provider_merge). The two ids were the same
|
|
136
|
+
// `devin` adapter, the same server.codeium.com api-server, and the same
|
|
137
|
+
// `devin-session-token$<JWT>` credential — only the account source
|
|
138
|
+
// differed: this entry did an Auth0 browser sign-in while `devin-cli`
|
|
139
|
+
// imported the token the installed CLI's own PKCE login had already
|
|
140
|
+
// written to credentials.toml. The merged login is import-first with a
|
|
141
|
+
// browser fallback: the CLI credential is taken when present (no browser
|
|
142
|
+
// opens), and the Auth0 flow remains because it is the only path for
|
|
143
|
+
// users without the CLI. `devin-cli` survives only as a deprecated
|
|
144
|
+
// alias; a startup migration rewrites saved provider rows, cross-config
|
|
145
|
+
// references, and auth.json slots to `devin`.
|
|
146
|
+
//
|
|
147
|
+
// `oauth` classifies the ACCOUNT, not the transport. This is not a local
|
|
148
|
+
// runtime: unlike Ollama or LM Studio it cannot answer at all until a
|
|
149
|
+
// vendor account is signed in, and `local` grouped it with things that
|
|
150
|
+
// have no account. It is also the only classification that reaches the
|
|
151
|
+
// dashboard Accounts tab, which is built from OAUTH_PROVIDERS.
|
|
152
|
+
id: "devin",
|
|
153
|
+
label: "Cognition (Devin/Windsurf)",
|
|
154
|
+
adapter: "devin",
|
|
155
|
+
baseUrl: "https://server.codeium.com",
|
|
156
|
+
authKind: "oauth",
|
|
157
|
+
featured: false,
|
|
158
|
+
// Off: `deriveProviderPresets` keys the preset catalog off this flag, so a
|
|
159
|
+
// true row would draw the provider twice — an Accounts login row and a
|
|
160
|
+
// preset tile.
|
|
161
|
+
dashboardPreset: false,
|
|
162
|
+
note: "Experimental unofficial Cognition/Devin bridge. ocx login devin first imports the credential an installed Devin CLI already holds (no browser); without one it opens Auth0 browser sign-in and exchanges the token via Cognition's RegisterUser for a long-lived API key.",
|
|
163
|
+
// Union seed of the two merged rosters: the newer devin-cli lineup first
|
|
164
|
+
// (it is the current catalog, so its default ordering wins), then the ids
|
|
165
|
+
// only the old devin entry carried. Degraded-mode seed only either way —
|
|
166
|
+
// `liveModels` discovers the account's real roster.
|
|
167
|
+
models: ["swe-2", "swe-1-7", "gpt-5-6-sol", "gpt-6-astra", "claude-opus-5", "claude-fable-5-1", "claude-sonnet-5", "glm-5-3", "kimi-k3", "gemini-3-8-flash", "grok-4-6", "swe-1-7-lightning", "gpt-5-6-luna", "gpt-5-6-terra", "claude-opus-4-8", "glm-5-2", "kimi-k2-7", "grok-4-5"],
|
|
168
|
+
liveModels: true,
|
|
169
|
+
defaultModel: "swe-2",
|
|
170
|
+
modelContextWindows: DEVIN_MODEL_CONTEXT_WINDOWS,
|
|
171
|
+
// Degraded-mode ladders only. Once a credential is present the account
|
|
172
|
+
// catalog supplies each base model its measured rungs; these two fields are
|
|
173
|
+
// what a signed-out picker and the Pi-shaped client exports fall back to.
|
|
174
|
+
modelReasoningEfforts: DEVIN_MODEL_EFFORTS,
|
|
175
|
+
reasoningEfforts: DEVIN_DEFAULT_EFFORTS,
|
|
176
|
+
},
|
|
177
|
+
{
|
|
178
|
+
id: "xai",
|
|
179
|
+
label: "xAI Grok",
|
|
180
|
+
adapter: "openai-chat",
|
|
181
|
+
baseUrl: "https://api.x.ai/v1",
|
|
182
|
+
authKind: "oauth",
|
|
183
|
+
allowKeyAuthOverride: true,
|
|
184
|
+
// Priority Processing is documented for xAI's public API-key Chat Completions and
|
|
185
|
+
// Responses endpoints. The OAuth lane is classified per-model below, not here:
|
|
186
|
+
// do not turn this into a provider-wide supportsServiceTier declaration.
|
|
187
|
+
keyAuthServiceTier: {
|
|
188
|
+
supportsServiceTier: true,
|
|
189
|
+
chatServiceTier: true,
|
|
190
|
+
},
|
|
191
|
+
// OAuth (Grok subscription gateway) service-tier capability, classified by live probe
|
|
192
|
+
// on 2026-09-13 (devlog/_fin/260913_xai_oauth_fast/020_probe-evidence.md): each listed
|
|
193
|
+
// model accepted service_tier "priority" over grok-oauth and echoed priority upstream.
|
|
194
|
+
// Key-auth already declares provider-wide support above, so this map only newly opens
|
|
195
|
+
// the OAuth lane. grok-4.20-multi-agent-0309 is deliberately absent: the gateway accepts
|
|
196
|
+
// the field but answers service_tier "default" — a live downgrade, not a fast tier.
|
|
197
|
+
// Unlisted and future-discovered ids stay unclassified.
|
|
198
|
+
modelSupportsServiceTier: {
|
|
199
|
+
"grok-4.6": true,
|
|
200
|
+
"grok-4.5": true,
|
|
201
|
+
"grok-4.3": true,
|
|
202
|
+
"grok-4.20-0309-reasoning": true,
|
|
203
|
+
"grok-4.20-0309-non-reasoning": true,
|
|
204
|
+
"grok-build-0.1": true,
|
|
205
|
+
"grok-composer-2.5-fast": true,
|
|
206
|
+
},
|
|
207
|
+
// Lets a caller-sent service_tier forward on the Chat wire (fastwire forwardCallerTier
|
|
208
|
+
// chain). Provider-wide by construction: unclassified chat-wire models then preserve a
|
|
209
|
+
// caller tier verbatim, the same contract other unclassified Responses routes already
|
|
210
|
+
// follow; --fast publication and proxy-owned fast injection stay capability-scoped by
|
|
211
|
+
// the map above. Key-auth declared the same value via keyAuthServiceTier, so the key
|
|
212
|
+
// lane is unchanged.
|
|
213
|
+
chatServiceTier: true,
|
|
214
|
+
// Shared across key and OAuth catalog rows. OAuth subscription has no
|
|
215
|
+
// per-token price, so the 2x claim is scoped to key auth.
|
|
216
|
+
fastTierDescription: "Priority processing; tier pricing applies on key auth only",
|
|
217
|
+
featured: true,
|
|
218
|
+
oauthId: "xai",
|
|
219
|
+
jawcodeBundle: "xai",
|
|
220
|
+
supportsOpenAiWebSearchToolFields: false,
|
|
221
|
+
// Live A/B on 2026-08-20: xAI rejects native custom/custom_tool_call shapes while accepting
|
|
222
|
+
// the otherwise-identical request after the custom tool is lowered to a function.
|
|
223
|
+
supportsResponsesCustomTools: false,
|
|
224
|
+
note: "Log in with your Grok account",
|
|
225
|
+
// Parallel tool calls: officially supported and default-on per docs.x.ai function-calling
|
|
226
|
+
// (verified 260709, devlog/_plan/260709_parallel_tool_calls). Streamed calls arrive whole
|
|
227
|
+
// per chunk, so the buffered parser assembles them losslessly.
|
|
228
|
+
parallelToolCalls: true,
|
|
229
|
+
// Live /v1/models discovery is the authoritative lineup (verified 260709: returns grok-4.5);
|
|
230
|
+
// the static list below is the logged-out fallback seed.
|
|
231
|
+
liveModels: true,
|
|
232
|
+
// 260709 refresh: lineup + metadata from official docs.x.ai (grok-4.5 announced 07-08);
|
|
233
|
+
// grok-composer-2.5-fast kept as account-verified (absent from public docs). Evidence:
|
|
234
|
+
// devlog/model_update/260709_model_refresh/001_xai_lineup.md.
|
|
235
|
+
// 260823: grok-4.20-multi-agent-0309 still returns 400 on Chat Completions, but works
|
|
236
|
+
// on Responses. The server reports this dated id for both it and the floating
|
|
237
|
+
// grok-4.20-multi-agent-beta-latest alias, so expose only the dated deployment id.
|
|
238
|
+
// 260813: grok-4.6 added per docs.x.ai/developers/grok-4-6. Context/vision still match
|
|
239
|
+
// grok-4.5; the reasoning ladder does not — 4.6 adds the documented xhigh rung.
|
|
240
|
+
models: XAI_MODELS,
|
|
241
|
+
// Measured only on grok-4.6 against cli-chat-proxy.grok.com: even an invalid
|
|
242
|
+
// `text.verbosity` value is accepted and low/high/omitted output length is non-monotonic.
|
|
243
|
+
// Apply the resulting opt-out to the whole xAI lineup because `text.verbosity` is an OpenAI
|
|
244
|
+
// Responses parameter absent from xAI's documented API, not because every model was probed.
|
|
245
|
+
// Keep this separate from reasoning-summary support: that bit gates Codex's
|
|
246
|
+
// entire Responses reasoning object, including reasoning.effort.
|
|
247
|
+
modelSupportsVerbosity: Object.fromEntries(XAI_MODELS.map(id => [id, false])),
|
|
248
|
+
// Provider-wide, not merely per-model: `text.verbosity` is an OpenAI Responses parameter
|
|
249
|
+
// absent from xAI's documented API, so a model discovered later has no more support for it
|
|
250
|
+
// than the seeded ones do.
|
|
251
|
+
supportsVerbosity: false,
|
|
252
|
+
defaultModel: "grok-4.5",
|
|
253
|
+
// Grok 4.6/4.5 subscription Responses callers use the native wire with the existing
|
|
254
|
+
// namespace/web-search/replay normalization. Chat remains an explicit modelAdapters
|
|
255
|
+
// opt-in. Multi-agent has no Chat wire and uses Responses under both auth modes.
|
|
256
|
+
// grok-4.6/4.5 are classified OAuth fast-tier models (modelSupportsServiceTier above),
|
|
257
|
+
// so a caller-sent service_tier:"priority" forwards on this lane — the Codex fast-toggle
|
|
258
|
+
// path. Multi-agent keeps its pin: probed 2026-09-13, the gateway downgrades its tier to
|
|
259
|
+
// "default", so forwarding a caller tier would advertise a tier it does not get.
|
|
260
|
+
modelWireDefaults: {
|
|
261
|
+
"grok-4.6": {
|
|
262
|
+
wire: "openai-responses",
|
|
263
|
+
inbound: ["responses"],
|
|
264
|
+
authModes: ["oauth"],
|
|
265
|
+
},
|
|
266
|
+
"grok-4.5": {
|
|
267
|
+
wire: "openai-responses",
|
|
268
|
+
inbound: ["responses"],
|
|
269
|
+
authModes: ["oauth"],
|
|
270
|
+
},
|
|
271
|
+
"grok-4.20-multi-agent-0309": {
|
|
272
|
+
// Even at high effort it emits no reasoning-summary deltas or encrypted replay
|
|
273
|
+
// material. Do not encode that as modelSupportsReasoningSummaries:false: through
|
|
274
|
+
// Codex #1100 that suppresses the entire reasoning object, including the effort
|
|
275
|
+
// that controls this model's agent count. An empty summary pane is harmless.
|
|
276
|
+
// Chat Completions returns 400 for this model, so every inbound uses Responses —
|
|
277
|
+
// `anthropic` included. Omitting it left providerModelWireDefault returning undefined
|
|
278
|
+
// for the Claude Messages lane, so resolveWireProtocolOverride kept xAI's provider-wide
|
|
279
|
+
// openai-chat adapter and sent this model to the wire it 400s on.
|
|
280
|
+
wire: "openai-responses",
|
|
281
|
+
inbound: ["responses", "chat", "anthropic"],
|
|
282
|
+
forwardCallerServiceTier: false,
|
|
283
|
+
},
|
|
284
|
+
},
|
|
285
|
+
// Vision lineup per docs.x.ai model-capabilities/images/understanding: the grok-4.x chat
|
|
286
|
+
// models accept image input (JPEG/PNG, URL or base64). Without this the catalog leaves
|
|
287
|
+
// inputModalities undefined, and deriveComboCatalogModel defaults an undefined member to
|
|
288
|
+
// ["text"] — so any combo containing an xAI target is advertised to Codex as text-only and
|
|
289
|
+
// the app blocks attachments client-side. grok-build-0.1 / grok-composer-2.5-fast stay out
|
|
290
|
+
// (they are already listed in noVisionModels below).
|
|
291
|
+
modelInputModalities: {
|
|
292
|
+
"grok-4.6": ["text", "image"],
|
|
293
|
+
"grok-4.5": ["text", "image"],
|
|
294
|
+
"grok-4.3": ["text", "image"],
|
|
295
|
+
"grok-4.20-multi-agent-0309": ["text", "image"],
|
|
296
|
+
"grok-4.20-0309-reasoning": ["text", "image"],
|
|
297
|
+
"grok-4.20-0309-non-reasoning": ["text", "image"],
|
|
298
|
+
},
|
|
299
|
+
noReasoningModels: ["grok-4.20-0309-non-reasoning", "grok-build-0.1", "grok-composer-2.5-fast"],
|
|
300
|
+
// Replay assistant reasoning_content for grok reasoning models: xAI documents dropped
|
|
301
|
+
// reasoning_content as the top cause of prompt-cache misses on multi-turn conversations
|
|
302
|
+
// (docs.x.ai prompt-caching/multi-turn, verified 2026-07-13 — devlog/_plan/260713_grok_caching).
|
|
303
|
+
// Models that never emit reasoning simply have no thinking parts to replay (no-op).
|
|
304
|
+
preserveReasoningContentModels: ["grok-4.6", "grok-4.5", "grok-4.3", "grok-4.20-0309-reasoning"],
|
|
305
|
+
// grok-4.5 reasoning is always-on with low/medium/high (no off tier, no xhigh).
|
|
306
|
+
// grok-4.6 adds xhigh per docs.x.ai/developers/model-capabilities/text/reasoning;
|
|
307
|
+
// multi-agent accepts the same four wire values to select 4 or 16 collaborators. xAI
|
|
308
|
+
// documents high as the 4.6 default but no multi-agent default, so do not invent one.
|
|
309
|
+
modelReasoningEfforts: {
|
|
310
|
+
"grok-4.6": ["low", "medium", "high", "xhigh"],
|
|
311
|
+
"grok-4.5": ["low", "medium", "high"],
|
|
312
|
+
"grok-4.20-multi-agent-0309": ["low", "medium", "high", "xhigh"],
|
|
313
|
+
},
|
|
314
|
+
modelDefaultReasoningEfforts: { "grok-4.6": "high" },
|
|
315
|
+
modelContextWindows: {
|
|
316
|
+
"grok-4.6": 500_000,
|
|
317
|
+
"grok-4.5": 500_000,
|
|
318
|
+
"grok-4.3": 1_000_000,
|
|
319
|
+
"grok-4.20-multi-agent-0309": 1_000_000,
|
|
320
|
+
"grok-4.20-0309-reasoning": 1_000_000,
|
|
321
|
+
"grok-4.20-0309-non-reasoning": 1_000_000,
|
|
322
|
+
"grok-build-0.1": 256_000,
|
|
323
|
+
},
|
|
324
|
+
noVisionModels: ["grok-build-0.1", "grok-composer-2.5-fast"],
|
|
325
|
+
},
|
|
326
|
+
{
|
|
327
|
+
id: "command-code",
|
|
328
|
+
label: "Command Code - Auth",
|
|
329
|
+
adapter: "command-code",
|
|
330
|
+
baseUrl: "https://api.commandcode.ai",
|
|
331
|
+
authKind: "oauth",
|
|
332
|
+
oauthId: "command-code",
|
|
333
|
+
featured: true,
|
|
334
|
+
note: "Log in with your Command Code account",
|
|
335
|
+
// OAuth needs one initial selection, but the exposed catalog is always discovered from the
|
|
336
|
+
// signed-in account. Do not add a static model list here.
|
|
337
|
+
defaultModel: "deepseek/deepseek-v4-flash",
|
|
338
|
+
liveModels: true,
|
|
339
|
+
modelDiscovery: {
|
|
340
|
+
url: "https://api.commandcode.ai/provider/v1/models",
|
|
341
|
+
maxResponseBytes: 262_144,
|
|
342
|
+
maxModels: 256,
|
|
343
|
+
},
|
|
344
|
+
// These are capability facts from official Command Code model profiles, not seeded models.
|
|
345
|
+
// Unknown/new live models deliberately do not advertise a reasoning picker.
|
|
346
|
+
reasoningEfforts: [],
|
|
347
|
+
modelReasoningEfforts: COMMAND_CODE_MODEL_REASONING_EFFORTS,
|
|
348
|
+
// The DeepSeek vision preview id is preemptive metadata — it is expected to
|
|
349
|
+
// merge into deepseek-v4-flash later.
|
|
350
|
+
modelContextWindows: {
|
|
351
|
+
[`deepseek/${DEEPSEEK_VISION_PREVIEW_MODEL}`]: 1_048_576,
|
|
352
|
+
},
|
|
353
|
+
modelInputModalities: COMMAND_CODE_MODEL_INPUT_MODALITIES,
|
|
354
|
+
defaultMaxOutputTokens: 64_000,
|
|
355
|
+
// The proprietary generate wire has no verified per-request serialization flag.
|
|
356
|
+
parallelToolCalls: false,
|
|
357
|
+
},
|
|
358
|
+
{
|
|
359
|
+
id: "orcarouter-oauth",
|
|
360
|
+
label: "OrcaRouter - Auth",
|
|
361
|
+
adapter: "openai-chat",
|
|
362
|
+
baseUrl: "https://api.orcarouter.ai/v1",
|
|
363
|
+
authKind: "oauth",
|
|
364
|
+
oauthId: "orcarouter-oauth",
|
|
365
|
+
featured: true,
|
|
366
|
+
allowBaseUrlOverride: true,
|
|
367
|
+
defaultModel: "openai/gpt-5.5",
|
|
368
|
+
models: ORCAROUTER_MODELS,
|
|
369
|
+
liveModels: true,
|
|
370
|
+
modelDiscovery: ORCAROUTER_MODEL_DISCOVERY,
|
|
371
|
+
modelReasoningEfforts: ORCAROUTER_MODEL_REASONING_EFFORTS,
|
|
372
|
+
note: "Connect your OrcaRouter account with OAuth 2.0 + PKCE; the issued API key is stored in OpenCodex's existing credential store.",
|
|
373
|
+
},
|
|
374
|
+
{
|
|
375
|
+
id: "anthropic",
|
|
376
|
+
label: "Anthropic Claude",
|
|
377
|
+
adapter: "anthropic",
|
|
378
|
+
baseUrl: "https://api.anthropic.com",
|
|
379
|
+
authKind: "oauth",
|
|
380
|
+
allowBaseUrlOverride: true,
|
|
381
|
+
featured: true,
|
|
382
|
+
oauthId: "anthropic",
|
|
383
|
+
jawcodeBundle: "anthropic",
|
|
384
|
+
note: "Log in with your Claude account",
|
|
385
|
+
models: [...ANTHROPIC_MODELS],
|
|
386
|
+
modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
|
|
387
|
+
modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
|
|
388
|
+
modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
|
|
389
|
+
// Codex omits max_output_tokens; without a provider budget the Anthropic adapter
|
|
390
|
+
// falls back to 8192, which truncates long answers with stop_reason=max_tokens.
|
|
391
|
+
defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
|
|
392
|
+
defaultModel: "claude-sonnet-5",
|
|
393
|
+
},
|
|
394
|
+
{
|
|
395
|
+
id: "anthropic-apikey",
|
|
396
|
+
label: "Anthropic (API key)",
|
|
397
|
+
adapter: "anthropic",
|
|
398
|
+
baseUrl: "https://api.anthropic.com",
|
|
399
|
+
authKind: "key",
|
|
400
|
+
featured: true,
|
|
401
|
+
dashboardUrl: "https://console.anthropic.com/settings/keys",
|
|
402
|
+
jawcodeBundle: "anthropic",
|
|
403
|
+
extraMetadataAliases: ["anthropic-key"],
|
|
404
|
+
note: "Direct Anthropic API billing — no Claude subscription",
|
|
405
|
+
models: [...ANTHROPIC_MODELS],
|
|
406
|
+
liveModels: true,
|
|
407
|
+
modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
|
|
408
|
+
modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
|
|
409
|
+
modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
|
|
410
|
+
defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
|
|
411
|
+
defaultModel: "claude-sonnet-5",
|
|
412
|
+
},
|
|
413
|
+
{
|
|
414
|
+
id: "kimi",
|
|
415
|
+
label: "Kimi",
|
|
416
|
+
adapter: "openai-chat",
|
|
417
|
+
baseUrl: "https://api.kimi.com/coding/v1",
|
|
418
|
+
authKind: "oauth",
|
|
419
|
+
modelSuffixBracketStrip: true,
|
|
420
|
+
// Kimi Code Plan documents a stable session/task prompt_cache_key as required to improve
|
|
421
|
+
// cache hit rates.
|
|
422
|
+
// The chat adapter only forwards a key already on the internal request (Codex's session key,
|
|
423
|
+
// or the one the Claude /v1/messages inbound derives); the adapter itself never invents one.
|
|
424
|
+
// Evidence: https://platform.kimi.com/docs/api/chat
|
|
425
|
+
promptCacheKey: true,
|
|
426
|
+
// Kimi's Responses endpoint rejects hook-provided context between a tool call and
|
|
427
|
+
// its matching result (#4726), the same strict shape DeepSeek exposed in #1292.
|
|
428
|
+
// The flag is inert while this preset uses the Chat wire.
|
|
429
|
+
requiresAdjacentResponsesToolResults: true,
|
|
430
|
+
featured: true,
|
|
431
|
+
oauthId: "kimi",
|
|
432
|
+
jawcodeBundle: "moonshot",
|
|
433
|
+
note: "Log in with your Kimi account",
|
|
434
|
+
models: KIMI_CODING_MODELS,
|
|
435
|
+
defaultModel: "kimi-k2.7-code",
|
|
436
|
+
modelContextWindows: KIMI_CODING_MODEL_CONTEXT_WINDOWS,
|
|
437
|
+
modelInputModalities: KIMI_CODING_MODEL_INPUT_MODALITIES,
|
|
438
|
+
// K3 accepts low/high/max; Codex aliases are normalized by the model-scoped wire map.
|
|
439
|
+
noReasoningModels: KIMI_CODING_NO_REASONING_MODELS,
|
|
440
|
+
modelReasoningEfforts: KIMI_CODING_REASONING_EFFORTS,
|
|
441
|
+
modelDefaultReasoningEfforts: KIMI_CODING_DEFAULT_REASONING_EFFORTS,
|
|
442
|
+
modelReasoningEffortMap: KIMI_CODING_REASONING_EFFORT_MAPS,
|
|
443
|
+
noTemperatureModels: KIMI_LOCKED_PARAMETER_MODELS,
|
|
444
|
+
noTopPModels: KIMI_LOCKED_PARAMETER_MODELS,
|
|
445
|
+
noPenaltyModels: KIMI_LOCKED_PARAMETER_MODELS,
|
|
446
|
+
autoToolChoiceOnlyModels: KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS,
|
|
447
|
+
preserveReasoningContentModels: KIMI_THINKING_MODELS,
|
|
448
|
+
},
|
|
449
|
+
{
|
|
450
|
+
id: "kiro",
|
|
451
|
+
label: "Kiro (AWS CodeWhisperer)",
|
|
452
|
+
adapter: "kiro",
|
|
453
|
+
baseUrl: "https://runtime.us-east-1.kiro.dev",
|
|
454
|
+
authKind: "oauth",
|
|
455
|
+
oauthId: "kiro",
|
|
456
|
+
note: "Import-first: reuses your installed and signed-in Kiro CLI session (requires `kiro-cli login`). Add account logs `kiro-cli` out, switches it through a fresh browser login, stores the account by profile ARN, and restores the previous CLI session on cancellation or failure. Experimental third-party harness — see Kiro ToS.",
|
|
457
|
+
models: KIRO_MODELS,
|
|
458
|
+
defaultModel: "kiro-auto",
|
|
459
|
+
// Kiro speaks CodeWhisperer wire, not OpenAI-style GET /models. Keep the static
|
|
460
|
+
// catalog authoritative so a spurious 2xx from runtime.../models cannot drop seeded ids
|
|
461
|
+
// (e.g. newly listed GPT-5.6 tiers) via live-discovery reconciliation.
|
|
462
|
+
liveModels: false,
|
|
463
|
+
// Per-model context metadata is maintained next to the Kiro model list.
|
|
464
|
+
modelContextWindows: KIRO_MODEL_CONTEXT_WINDOWS,
|
|
465
|
+
modelReasoningEfforts: KIRO_MODEL_REASONING_EFFORTS,
|
|
466
|
+
modelSupportsVerbosity: Object.fromEntries(KIRO_MODELS.map(id => [id, false])),
|
|
467
|
+
},
|
|
468
|
+
{
|
|
469
|
+
// Nous Portal — Nous Research subscription gateway (same backend Hermes Agent
|
|
470
|
+
// uses). OAuth is a device grant (src/oauth/nous.ts): the access token IS the
|
|
471
|
+
// per-request inference JWT (scope inference:invoke), refresh tokens are
|
|
472
|
+
// single-use and rotated on every refresh. Catalog is a mix of paid models
|
|
473
|
+
// (billed against the Portal subscription) and `:free` slugs (e.g.
|
|
474
|
+
// tencent/hy3:free, stepfun/step-3.7-flash:free, inclusionai/ling-3.0-flash:free);
|
|
475
|
+
// free-tier gating is decided live by the Portal per account, so discovery
|
|
476
|
+
// from the signed-in account is authoritative; the static seed below is the
|
|
477
|
+
// logged-out fallback and only lists free models verified on a real account
|
|
478
|
+
// (2026-08-10): the Portal free list is authoritative and currently has
|
|
479
|
+
// exactly 4 :free models: tencent/hy3:free, poolside/laguna-s-2.1:free,
|
|
480
|
+
// stepfun/step-3.7-flash:free, poolside/laguna-xs-2.1:free.
|
|
481
|
+
// inclusionai/ling-3.0-flash:free was removed from the Portal free list
|
|
482
|
+
// (404 on the inference API since 2026-08-07) and must not be seeded.
|
|
483
|
+
id: "nous",
|
|
484
|
+
label: "Nous Portal",
|
|
485
|
+
adapter: "openai-chat",
|
|
486
|
+
baseUrl: "https://inference-api.nousresearch.com/v1",
|
|
487
|
+
authKind: "oauth",
|
|
488
|
+
oauthId: "nous",
|
|
489
|
+
featured: true,
|
|
490
|
+
// Mixed free + paid provider: the free tier is per-model (the `:free`
|
|
491
|
+
// slugs), not a property of the whole provider, so freeTier stays false to
|
|
492
|
+
// avoid implying every model is free.
|
|
493
|
+
freeTier: false,
|
|
494
|
+
dashboardUrl: "https://portal.nousresearch.com",
|
|
495
|
+
defaultModel: "tencent/hy3:free",
|
|
496
|
+
liveModels: true,
|
|
497
|
+
models: ["tencent/hy3:free", "poolside/laguna-s-2.1:free", "stepfun/step-3.7-flash:free", "poolside/laguna-xs-2.1:free"],
|
|
498
|
+
modelDiscovery: {
|
|
499
|
+
// Resolves against effectiveBaseUrl (registry baseUrl .../v1) to the same
|
|
500
|
+
// canonical endpoint https://inference-api.nousresearch.com/v1/models.
|
|
501
|
+
// Nous returns a mixed paid/free catalog whose JSON can exceed 256 KiB;
|
|
502
|
+
// keep the provider-specific limit below the process-wide 4 MiB ceiling.
|
|
503
|
+
path: "models",
|
|
504
|
+
maxResponseBytes: 1_048_576,
|
|
505
|
+
maxModels: 512,
|
|
506
|
+
},
|
|
507
|
+
note: "Nous Research subscription gateway. OAuth device login with your own Portal account; mixed paid + :free models discovered live (fallback seed 2026-08-10: tencent/hy3:free, poolside/laguna-s-2.1:free, stepfun/step-3.7-flash:free, poolside/laguna-xs-2.1:free).",
|
|
508
|
+
},
|
|
509
|
+
{
|
|
510
|
+
id: "openai-apikey",
|
|
511
|
+
label: "OpenAI API",
|
|
512
|
+
adapter: "openai-responses",
|
|
513
|
+
baseUrl: "https://api.openai.com/v1",
|
|
514
|
+
authKind: "key",
|
|
515
|
+
supportsServiceTier: true,
|
|
516
|
+
featured: true,
|
|
517
|
+
dashboardUrl: "https://platform.openai.com/api-keys",
|
|
518
|
+
defaultModel: "gpt-5.5",
|
|
519
|
+
models: ["gpt-5.5", ...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS, ...OPENAI_DAYBREAK_MODELS, "gpt-6-astra"],
|
|
520
|
+
liveModels: true,
|
|
521
|
+
modelContextWindows: { ...OPENAI_API_GPT56_CONTEXT_WINDOWS, ...OPENAI_DAYBREAK_CONTEXT_WINDOWS, "gpt-6-astra": 1_050_000 },
|
|
522
|
+
modelMaxInputTokens: { ...OPENAI_API_GPT56_MAX_INPUT_TOKENS, ...OPENAI_DAYBREAK_MAX_INPUT_TOKENS, "gpt-6-astra": 922_000 },
|
|
523
|
+
modelMaxOutputTokens: { "gpt-6-astra": 128_000 },
|
|
524
|
+
modelInputModalities: Object.fromEntries(
|
|
525
|
+
["gpt-5.5", ...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS, ...OPENAI_DAYBREAK_MODELS, "gpt-6-astra"]
|
|
526
|
+
.map(id => [id, ["text", "image"]]),
|
|
527
|
+
),
|
|
528
|
+
modelReasoningEfforts: {
|
|
529
|
+
...Object.fromEntries(
|
|
530
|
+
[...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS].map(id => [id, OPENAI_API_GPT56_REASONING_EFFORTS]),
|
|
531
|
+
),
|
|
532
|
+
...OPENAI_DAYBREAK_REASONING_EFFORTS,
|
|
533
|
+
"gpt-6-astra": ["low", "medium", "high", "xhigh", "max"],
|
|
534
|
+
},
|
|
535
|
+
virtualModels: OPENAI_API_GPT56_VIRTUAL_MODELS,
|
|
536
|
+
},
|
|
537
|
+
/* [Decision Log]
|
|
538
|
+
- 목적과 의도: Reach Meta's Muse Spark models directly on Meta's own Model API, instead of only through the Command Code and OpenCode Zen resellers already in this registry.
|
|
539
|
+
- 기존 구현 및 제약 조건: Meta publishes both POST /v1/responses and POST /v1/chat/completions at https://api.meta.ai/v1, and no API key was issued for this change — every value here comes from the published spec (devlog/_plan/260903_muse_spark_plan_oauth/001).
|
|
540
|
+
- 검토한 주요 대안: register as openai-chat; use provider id "meta"; enable live discovery; wire the Muse Code subscription credential as OAuth.
|
|
541
|
+
- 선택한 방식: an openai-responses key provider under the id "meta-model", with a static two-model roster and no OAuth.
|
|
542
|
+
- 다른 대안 대신 이 방식을 선택한 이유: Meta calls Responses "the recommended default for new work ... OpenAI-compatible and exposes the full feature set", carrying reasoning replay and native input_image that Chat would forfeit. The id is "meta-model" because "meta" would capture the LIVE Command Code selector meta/muse-spark-1.3 at router.ts's provider-prefix branch, and would derive META_API_KEY — the Muse Code CLI's variable, not this API's MODEL_API_KEY.
|
|
543
|
+
- 장점, 단점 및 영향: users reach Muse Spark without a reseller; discovery stays off until an authenticated /v1/models payload is actually observed, so an unseen roster (Meta also serves image and voice families here) cannot leak into the picker.
|
|
544
|
+
*/
|
|
545
|
+
{
|
|
546
|
+
id: "meta-model",
|
|
547
|
+
label: "Meta Model API",
|
|
548
|
+
adapter: "openai-responses",
|
|
549
|
+
baseUrl: "https://api.meta.ai/v1",
|
|
550
|
+
authKind: "key",
|
|
551
|
+
dashboardUrl: "https://dev.meta.ai/docs/authentication",
|
|
552
|
+
defaultModel: "muse-spark-1.3",
|
|
553
|
+
models: META_MUSE_MODELS,
|
|
554
|
+
// Static roster: no authenticated /v1/models payload was ever observed (the only
|
|
555
|
+
// contact was an unauthenticated GET returning 401 invalid_api_key), and Meta serves
|
|
556
|
+
// non-agent families on this same base URL. Turning discovery on would publish an
|
|
557
|
+
// unseen roster into the picker.
|
|
558
|
+
liveModels: false,
|
|
559
|
+
// A user may already own a custom provider named "meta-model" pointing elsewhere;
|
|
560
|
+
// without this, registry transport canonicalization would retarget it and send their
|
|
561
|
+
// saved key to Meta.
|
|
562
|
+
preserveCustomDestination: true,
|
|
563
|
+
modelContextWindows: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_CONTEXT_WINDOW])),
|
|
564
|
+
// text+image only. Meta also documents video, audio (degraded on 1.3), and PDF, but
|
|
565
|
+
// the catalog modality enum is text/image and over-advertising poisons the exported
|
|
566
|
+
// client config (see tests/codex-integration/catalog-input-modality-enum.test.ts).
|
|
567
|
+
modelInputModalities: Object.fromEntries(META_MUSE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
|
|
568
|
+
modelReasoningEfforts: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORTS])),
|
|
569
|
+
modelReasoningEffortMap: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORT_MAP])),
|
|
570
|
+
// No defaultMaxOutputTokens: Meta publishes none. The only number in its docs
|
|
571
|
+
// (131072) appears inside a third-party config sample, and the protocol pages call
|
|
572
|
+
// the real limit "model-dependent".
|
|
573
|
+
// Meta names its variable MODEL_API_KEY, but the env var opencodex reads is derived
|
|
574
|
+
// from the provider id (META_MODEL_API_KEY). Saying only Meta's name would send a
|
|
575
|
+
// user to export a variable this proxy never reads.
|
|
576
|
+
note: "Pay-as-you-go Meta Model API. Get a key at https://dev.meta.ai (Meta calls it MODEL_API_KEY; export it here as META_MODEL_API_KEY) — a Meta developer account needs a payment method before it can serve requests, and every call is metered per token. A Muse Code subscription does NOT work here: Meta scopes that credential to the Muse Code CLI and bills any other key pay-as-you-go (dev.meta.ai/docs/muse-code/subscriptions). The Contributor tier (muse-spark-1.3-contributor) is cheap because Meta trains on your prompts — about 92% off input, 95% off output, 99% off cached input; do not send confidential material through it. Muse Spark is also reachable through resellers: command-code carries both tiers, opencode-go serves only muse-spark-1.3-contributor.",
|
|
577
|
+
},
|
|
578
|
+
/* [Decision Log]
|
|
579
|
+
- 목적과 의도: Let an operator who already signed the Muse Code CLI in reach Muse Spark with that credential, instead of provisioning a second key.
|
|
580
|
+
- 기존 구현 및 제약 조건: The CLI stores a pointer at ~/.config/muse/auth.json and the secret in the macOS Keychain (ai.meta.dev.credentials/meta). Measured: the OAuth access_token 401s on /v1/models while the sibling api_key returns 200, so the usable artifact is a static key, not a refreshable token.
|
|
581
|
+
- 검토한 주요 대안: spawn `muse login` and poll; reimplement Meta's device grant; treat it as a second key preset; ship nothing.
|
|
582
|
+
- 선택한 방식: an OAuth provider that imports the existing credential on macOS and accepts a pasted key elsewhere, validates either once, and never spawns or reimplements anything.
|
|
583
|
+
- 다른 대안 대신 이 방식을 선택한 이유: `muse login` has no non-interactive mode, so a spawned child could outlive cancellation, and polling for the pointer file is satisfied instantly by the one already on disk — reimporting the OLD account on a force-login. Reimplementing the grant would mean guessing a client id the vendor does not publish.
|
|
584
|
+
- 장점, 단점 및 영향: no new credential to provision, and the id is distinct from meta-model so neither pool contaminates the other. Meta scopes this credential to its own CLI, so the provider carries a HIGH_RISK ToS warning, a CLI-side warning before any read, and a note that says plainly what is unsupported.
|
|
585
|
+
*/
|
|
586
|
+
{
|
|
587
|
+
id: "meta-muse",
|
|
588
|
+
label: "Meta Muse Code (CLI credential)",
|
|
589
|
+
adapter: "openai-responses",
|
|
590
|
+
baseUrl: "https://api.meta.ai/v1",
|
|
591
|
+
// Meta own client sends this on every Muse Code call. We never have, so a future
|
|
592
|
+
// server-side requirement would break every Muse request with no local signal.
|
|
593
|
+
// Declared here rather than in a transport hook so it also covers model discovery
|
|
594
|
+
// (src/oauth/index.ts:1176) and still yields to a user-set header
|
|
595
|
+
// (mergeRegistryStaticHeaders, src/providers/registry.ts:3494).
|
|
596
|
+
staticHeaders: { "x-api-version": "1.0.0" },
|
|
597
|
+
authKind: "oauth",
|
|
598
|
+
oauthId: "meta-muse",
|
|
599
|
+
dashboardUrl: "https://dev.meta.ai",
|
|
600
|
+
defaultModel: "muse-spark-1.3",
|
|
601
|
+
models: META_MUSE_MODELS,
|
|
602
|
+
// Same reason as meta-model: the authenticated roster carries muse-image-1.0 and
|
|
603
|
+
// muse-voice-transcribe-1.0, which this Responses-agent provider cannot drive.
|
|
604
|
+
liveModels: false,
|
|
605
|
+
modelContextWindows: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_CONTEXT_WINDOW])),
|
|
606
|
+
modelInputModalities: Object.fromEntries(META_MUSE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
|
|
607
|
+
modelReasoningEfforts: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORTS])),
|
|
608
|
+
modelReasoningEffortMap: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORT_MAP])),
|
|
609
|
+
note: "Signs in to Meta with a browser device code on any platform, then mints the Muse Code subscription key. That grant is reimplemented from the one the Muse Code CLI performs and has NOT been exercised against Meta from OpenCodex, so treat the first login as unverified. If the Muse Code CLI is already signed in on macOS, the existing key is imported instead of starting a new grant. A pasted key from https://dev.meta.ai still works as a fallback when a device login cannot complete, and faces the same format check and live validation. A device login authenticates as Meta own Muse Code client, which is a stronger claim than reusing a key the CLI already minted. Meta scopes that credential to the Muse Code CLI, so this is an UNSUPPORTED use: Meta does not authorize subscription coverage outside its own CLI, how these calls settle is not observable from the API, and you should treat every call as billable against your account. The key, imported or pasted, is copied into OpenCodex's auth store. For an account signed in with the device login, OpenCodex refreshes Meta's subscription windows on demand from the same key endpoint the login uses, at most once every five minutes. For an imported or pasted key there is no endpoint to query them on demand, so OpenCodex reads them from streaming responses and shows the last observed value with its age; refreshing one then requires another streaming turn, and translated (non-passthrough) turns report none. Rate limits apply per team, not per key. For a supported path use the meta-model provider with your own key (export it as META_MODEL_API_KEY).",
|
|
610
|
+
},
|
|
611
|
+
{
|
|
612
|
+
id: "umans",
|
|
613
|
+
label: "Umans AI Coding Plan",
|
|
614
|
+
adapter: "anthropic",
|
|
615
|
+
baseUrl: "https://api.code.umans.ai",
|
|
616
|
+
authKind: "key",
|
|
617
|
+
featured: true,
|
|
618
|
+
dashboardUrl: "https://app.umans.ai/billing",
|
|
619
|
+
defaultModel: "umans-coder",
|
|
620
|
+
models: UMANS_MODELS,
|
|
621
|
+
modelContextWindows: UMANS_MODEL_CONTEXT_WINDOWS,
|
|
622
|
+
modelInputModalities: UMANS_MODEL_INPUT_MODALITIES,
|
|
623
|
+
note: "Coding plan via Anthropic Messages",
|
|
624
|
+
modelReasoningEfforts: {
|
|
625
|
+
"umans-coder": UMANS_REASONING_EFFORTS,
|
|
626
|
+
"umans-kimi-k2.7": UMANS_REASONING_EFFORTS,
|
|
627
|
+
"umans-flash": UMANS_REASONING_EFFORTS,
|
|
628
|
+
"umans-glm-5.3": UMANS_GLM_53_REASONING_EFFORTS,
|
|
629
|
+
"umans-glm-5.3-flash": UMANS_GLM_53_REASONING_EFFORTS,
|
|
630
|
+
"umans-glm-5.2": UMANS_GLM_REASONING_EFFORTS,
|
|
631
|
+
"umans-glm-5.1": UMANS_GLM_REASONING_EFFORTS,
|
|
632
|
+
"umans-qwen3.6-35b-a3b": UMANS_REASONING_EFFORTS,
|
|
633
|
+
},
|
|
634
|
+
noVisionModels: UMANS_TEXT_ONLY_MODELS,
|
|
635
|
+
escapeBuiltinToolNames: true,
|
|
636
|
+
},
|
|
637
|
+
{
|
|
638
|
+
id: "opencode-go", label: "opencode go", adapter: "openai-chat", baseUrl: "https://opencode.ai/zen/go/v1",
|
|
639
|
+
authKind: "key", featured: true, dashboardUrl: "https://opencode.ai/auth", defaultModel: "kimi-k2.7-code",
|
|
640
|
+
jawcodeBundle: "opencode-go", note: "GLM, DeepSeek, Kimi, Qwen, MiMo…",
|
|
641
|
+
// Zen Go can close a Chat stream after a fully assembled function call without sending
|
|
642
|
+
// finish_reason or [DONE] (#2260). The adapter still rejects incomplete argument JSON.
|
|
643
|
+
openaiChatEofTolerance: true,
|
|
644
|
+
// Go rejects reasoning.encrypted_content with previous_response_id (#3838).
|
|
645
|
+
// Use explicit replay history and the existing stateless Responses policy.
|
|
646
|
+
statelessResponses: true,
|
|
647
|
+
/* [Decision Log]
|
|
648
|
+
- 목적과 의도: Route the exact models OpenCode Go documents on the Responses endpoint — GPT 5.6 Luna, Grok 4.6, and Muse Spark Contributor (#2617).
|
|
649
|
+
- 기존 구현 및 제약 조건: The provider is mixed-wire but its provider-wide `openai-chat` adapter sent Luna to `/chat/completions`; explicit user `modelAdapters` entries must remain authoritative.
|
|
650
|
+
- 검토한 주요 대안: Change the whole provider to Responses; infer the wire from model-family names; add one registry-only exact-model default.
|
|
651
|
+
- 선택한 방식: Declare only the named models as `openai-responses` through the existing registry default mechanism; the map stays an exact-model allowlist rather than a family or provider-wide rule.
|
|
652
|
+
- 다른 대안 대신 이 방식을 선택한 이유: OpenCode Go documents sibling models on Chat or Anthropic endpoints, and an exact registry default preserves both those routes and explicit opt-out precedence.
|
|
653
|
+
- 장점, 단점 및 영향: Each listed model reaches `/responses` from every inbound surface without changing siblings; a future upstream endpoint change requires an evidence-backed registry update.
|
|
654
|
+
*/
|
|
655
|
+
modelWireDefaults: {
|
|
656
|
+
"gpt-5.6-luna": "openai-responses",
|
|
657
|
+
"grok-4.6": "openai-responses",
|
|
658
|
+
"muse-spark-1.3-contributor": "openai-responses",
|
|
659
|
+
"muse-spark-1.2-contributor": "openai-responses",
|
|
660
|
+
},
|
|
661
|
+
modelContextWindows: {
|
|
662
|
+
"kimi-k3": KIMI_K3_STANDARD_CONTEXT_WINDOW,
|
|
663
|
+
// Zen Go discovers only the gateway id, so carry DeepSeek's official 1M V4.1
|
|
664
|
+
// window here or Codex falls back to its conservative 128k routed-model default.
|
|
665
|
+
"deepseek-v4.1-flash": 1_048_576,
|
|
666
|
+
// The DeepSeek vision preview id is metadata-only here: the Go roster is
|
|
667
|
+
// discovered live, so it applies the moment the gateway serves the id.
|
|
668
|
+
[DEEPSEEK_VISION_PREVIEW_MODEL]: 1_048_576,
|
|
669
|
+
// Muse Spark Contributor serves a 1,048,576-token (1M) context window over
|
|
670
|
+
// /responses on Zen Go, matching its 1.1 sibling (Meta developer docs, verified 2026-08-28).
|
|
671
|
+
// Without this declaration the catalog falls back to 128k, capping real usable context.
|
|
672
|
+
// 1.3 ships the same window as 1.2 and is served from the same Zen Go roster.
|
|
673
|
+
"muse-spark-1.3-contributor": 1_048_576,
|
|
674
|
+
"muse-spark-1.2-contributor": 1_048_576,
|
|
675
|
+
},
|
|
676
|
+
modelInputModalities: {
|
|
677
|
+
"kimi-k3": ["text", "image"],
|
|
678
|
+
// glm-5.3-flash is a native VLM (docs.z.ai/guides/vlm/glm-5.3-flash). It is
|
|
679
|
+
// deliberately absent from this preset's noVisionModels, which is the
|
|
680
|
+
// correct NEGATIVE half, but with no positive modelInputModalities entry
|
|
681
|
+
// configuredInputModalities returns undefined and the catalog falls through
|
|
682
|
+
// to the ["text"] floor. The same model is already declared ["text","image"]
|
|
683
|
+
// on the zai and zhipu-bigmodel-coding presets, so the registry described
|
|
684
|
+
// one model two ways (#4505).
|
|
685
|
+
"glm-5.3-flash": ["text", "image"],
|
|
686
|
+
// Experimental DeepSeek vision preview — expected to merge into deepseek-v4-flash later.
|
|
687
|
+
[DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
|
|
688
|
+
// This route is text-only upstream — it is already listed in this preset's
|
|
689
|
+
// noVisionModels, which routes images through the proxy's vision sidecar and
|
|
690
|
+
// makes the catalog advertise image input on its behalf. The positive
|
|
691
|
+
// text-only declaration is what reaches an EXISTING install: derive.ts fills
|
|
692
|
+
// noVisionModels all-or-nothing, so a config persisted before this id joined
|
|
693
|
+
// the list keeps a stale list, the sidecar predicate never matches, the row
|
|
694
|
+
// carries no modality at all, and any combo containing it collapses to
|
|
695
|
+
// ["text"] (#4505). modelInputModalities IS per-key filled, so this
|
|
696
|
+
// declaration lands on old configs. It states the route's real upstream
|
|
697
|
+
// capability and keeps the sidecar explicitly distinct from native vision.
|
|
698
|
+
"deepseek-v4.1-flash": ["text"],
|
|
699
|
+
// Muse Spark Contributor is natively multimodal on Zen Go: it accepts input_image
|
|
700
|
+
// parts over /responses (probed 2026-08-26). Without this declaration the catalog
|
|
701
|
+
// advertises it text-only and the Codex app blocks image attachments client-side with
|
|
702
|
+
// "This model does not support image inputs" before the request ever reaches the proxy.
|
|
703
|
+
// 1.3 is the same-shaped successor and Command Code documents it as multimodal.
|
|
704
|
+
"muse-spark-1.3-contributor": ["text", "image"],
|
|
705
|
+
"muse-spark-1.2-contributor": ["text", "image"],
|
|
706
|
+
},
|
|
707
|
+
modelReasoningEfforts: {
|
|
708
|
+
"gpt-5.6-luna": OPENAI_API_GPT56_REASONING_EFFORTS,
|
|
709
|
+
"grok-4.6": ["low", "medium", "high", "xhigh"],
|
|
710
|
+
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
711
|
+
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
712
|
+
"glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
|
|
713
|
+
"qwen3.8-max": QWEN38_REASONING_EFFORTS,
|
|
714
|
+
"kimi-k3": KIMI_CODING_K3_REASONING_EFFORTS,
|
|
715
|
+
"kimi-k2.7-code": [],
|
|
716
|
+
"kimi-k2.7-code-highspeed": [],
|
|
717
|
+
...Object.fromEntries(OPENCODE_GO_THINKING_TOGGLE_MODELS.map(id => [id, THINKING_TOGGLE_EFFORTS])),
|
|
718
|
+
...Object.fromEntries(OPENCODE_GO_THINKING_BUDGET_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
|
|
719
|
+
...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, deepseekThinkingEffortsFor(id)])),
|
|
720
|
+
},
|
|
721
|
+
modelDefaultReasoningEfforts: { "grok-4.6": "high", "kimi-k3": "max" },
|
|
722
|
+
// glm-5.2 uses identity labels now that `max` is a native Codex level (no alias map);
|
|
723
|
+
// the thinking-toggle map is a REAL wire alias (effort -> enabled/disabled) and stays.
|
|
724
|
+
modelReasoningEffortMap: {
|
|
725
|
+
"kimi-k3": KIMI_CODING_K3_REASONING_EFFORT_MAP,
|
|
726
|
+
...Object.fromEntries(OPENCODE_GO_THINKING_TOGGLE_MODELS.map(id => [id, THINKING_TOGGLE_MAP])),
|
|
727
|
+
...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, deepseekReasoningMapFor(id)])),
|
|
728
|
+
},
|
|
729
|
+
modelSupportsReasoningSummaries: {
|
|
730
|
+
"glm-5.3": true,
|
|
731
|
+
"glm-5.3-flash": true,
|
|
732
|
+
"glm-5.2": true,
|
|
733
|
+
"glm-5.1": true,
|
|
734
|
+
"glm-5": true,
|
|
735
|
+
...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, true])),
|
|
736
|
+
},
|
|
737
|
+
thinkingToggleModels: OPENCODE_GO_THINKING_TOGGLE_MODELS,
|
|
738
|
+
/*
|
|
739
|
+
* The Go-specific list, not the shared one. The shared `THINKING_BUDGET_MODELS` also
|
|
740
|
+
* carries Neuralwatt-only ids (`qwen3.5-397b`, `qwen3.6-35b`) that this preset never
|
|
741
|
+
* gives a ladder to, so a live roster serving one of them armed the thinking-budget
|
|
742
|
+
* wire path with nothing to advertise: the catalog showed no effort control while the
|
|
743
|
+
* adapter still translated effort into `thinking_budget`.
|
|
744
|
+
*/
|
|
745
|
+
thinkingBudgetModels: OPENCODE_GO_THINKING_BUDGET_MODELS,
|
|
746
|
+
noReasoningModels: ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
|
|
747
|
+
// Text-only Zen Go models (jawcode metadata) — the vision sidecar describes images for
|
|
748
|
+
// every model listed here (and the catalog advertises image input on their behalf).
|
|
749
|
+
// Kimi K2.7 Code accepts text+image+video: do NOT list it here.
|
|
750
|
+
noVisionModels: [
|
|
751
|
+
"glm-5.3", "glm-5.2", "glm-5", "glm-5.1",
|
|
752
|
+
"deepseek-v4.1-flash", "deepseek-v4-flash",
|
|
753
|
+
"mimo-v2-pro", "mimo-v2.5-pro",
|
|
754
|
+
"minimax-m2.5", "minimax-m2.7",
|
|
755
|
+
"qwen3.7-max",
|
|
756
|
+
],
|
|
757
|
+
noTemperatureModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
|
|
758
|
+
noTopPModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
|
|
759
|
+
noPenaltyModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
|
|
760
|
+
autoToolChoiceOnlyModels: ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
|
|
761
|
+
// Issue #78: DeepSeek V4 thinking mode requires reasoning_content replay on tool-call turns.
|
|
762
|
+
preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed", ...DEEPSEEK_GATEWAY_THINKING_MODELS],
|
|
763
|
+
/*
|
|
764
|
+
* Issues #1338 / #1415: this gateway answers a `response_format` of type
|
|
765
|
+
* `json_schema` with HTTP 400 `This response_format type is unavailable now`
|
|
766
|
+
* (quoted from the upstream body as `Error from provider (Console Go)`), which
|
|
767
|
+
* breaks every Codex auto-review turn on a DeepSeek route. #1424 shipped the
|
|
768
|
+
* operator-side opt-out; operators have been applying it by hand ever since.
|
|
769
|
+
* The reported rejection is type-specific, so this narrower list downgrades the
|
|
770
|
+
* request to `json_object` instead of claiming the whole field is unavailable.
|
|
771
|
+
*/
|
|
772
|
+
noJsonSchemaModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS],
|
|
773
|
+
},
|
|
774
|
+
{
|
|
775
|
+
id: "neuralwatt",
|
|
776
|
+
label: "Neuralwatt Cloud",
|
|
777
|
+
adapter: "openai-chat",
|
|
778
|
+
baseUrl: "https://api.neuralwatt.com/v1",
|
|
779
|
+
authKind: "key",
|
|
780
|
+
dashboardUrl: "https://portal.neuralwatt.com",
|
|
781
|
+
defaultModel: "glm-5.3",
|
|
782
|
+
// 2026-07-10 live /v1/models: K2.5 rows were removed and GLM-5.2 short variants added.
|
|
783
|
+
// 260814: the glm-5.3 quartet is speculative; live discovery is authoritative and drops
|
|
784
|
+
// any id Neuralwatt has not published yet.
|
|
785
|
+
// Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md and https://api.neuralwatt.com/v1/models.
|
|
786
|
+
models: [
|
|
787
|
+
"glm-5.3", "glm-5.3-fast", "glm-5.3-short", "glm-5.3-short-fast",
|
|
788
|
+
"glm-5.3-flash",
|
|
789
|
+
"glm-5.2", "glm-5.2-fast", "glm-5.2-short", "glm-5.2-short-fast",
|
|
790
|
+
"kimi-k2.6", "kimi-k2.6-fast",
|
|
791
|
+
"kimi-k2.7-code",
|
|
792
|
+
"qwen3.5-397b", "qwen3.5-397b-fast", "qwen3.6-35b", "qwen3.6-35b-fast",
|
|
793
|
+
],
|
|
794
|
+
// Neuralwatt's /v1/models metadata is authoritative; these static hints are the offline fallback.
|
|
795
|
+
modelReasoningEfforts: {
|
|
796
|
+
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
797
|
+
"glm-5.3-fast": [],
|
|
798
|
+
"glm-5.3-short": ZAI_GLM_53_REASONING_EFFORTS,
|
|
799
|
+
"glm-5.3-short-fast": [],
|
|
800
|
+
// No `-fast`/`-short` variants are asserted for the flash tier: those suffixes
|
|
801
|
+
// encode routing Neuralwatt documents per model, and this seed has no source for them.
|
|
802
|
+
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
803
|
+
"glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
|
|
804
|
+
"glm-5.2-fast": [],
|
|
805
|
+
"glm-5.2-short": ZAI_GLM_52_REASONING_EFFORTS,
|
|
806
|
+
"glm-5.2-short-fast": [],
|
|
807
|
+
"kimi-k2.6": [],
|
|
808
|
+
"kimi-k2.6-fast": [],
|
|
809
|
+
"kimi-k2.7-code": [],
|
|
810
|
+
// Qwen3.x uses thinking_budget, NOT graded reasoning_effort; the adapter maps the five
|
|
811
|
+
// Codex picker levels onto budget fractions.
|
|
812
|
+
"qwen3.5-397b": THINKING_BUDGET_EFFORTS,
|
|
813
|
+
"qwen3.5-397b-fast": [],
|
|
814
|
+
"qwen3.6-35b": THINKING_BUDGET_EFFORTS,
|
|
815
|
+
"qwen3.6-35b-fast": [],
|
|
816
|
+
},
|
|
817
|
+
thinkingBudgetModels: THINKING_BUDGET_MODELS,
|
|
818
|
+
noReasoningModels: ["glm-5.3-fast", "glm-5.3-short-fast", "glm-5.2-fast", "glm-5.2-short-fast", "kimi-k2.6-fast", "qwen3.5-397b-fast", "qwen3.6-35b-fast"],
|
|
819
|
+
noVisionModels: ["glm-5.3", "glm-5.3-fast", "glm-5.3-short", "glm-5.3-short-fast", "glm-5.2", "glm-5.2-fast", "glm-5.2-short", "glm-5.2-short-fast", "qwen3.5-397b", "qwen3.5-397b-fast"],
|
|
820
|
+
noTemperatureModels: ["kimi-k2.7-code"],
|
|
821
|
+
noTopPModels: ["kimi-k2.7-code"],
|
|
822
|
+
noPenaltyModels: ["kimi-k2.7-code"],
|
|
823
|
+
autoToolChoiceOnlyModels: ["kimi-k2.7-code"],
|
|
824
|
+
preserveReasoningContentModels: NEURALWATT_REASONING_HISTORY_MODELS,
|
|
825
|
+
},
|
|
826
|
+
{
|
|
827
|
+
id: "openrouter",
|
|
828
|
+
label: "OpenRouter",
|
|
829
|
+
adapter: "openai-chat",
|
|
830
|
+
baseUrl: "https://openrouter.ai/api/v1",
|
|
831
|
+
authKind: "key",
|
|
832
|
+
featured: true,
|
|
833
|
+
dashboardUrl: "https://openrouter.ai/keys",
|
|
834
|
+
jawcodeBundle: "openrouter",
|
|
835
|
+
models: ["anthropic/claude-sonnet-5", ...OPENROUTER_GPT56_MODELS],
|
|
836
|
+
modelContextWindows: {
|
|
837
|
+
"anthropic/claude-sonnet-5": 1_000_000,
|
|
838
|
+
...OPENROUTER_GPT56_CONTEXT_WINDOWS,
|
|
839
|
+
},
|
|
840
|
+
// OpenRouter documents priority support for OpenAI endpoints, but not Anthropic. Keep the
|
|
841
|
+
// provider unclassified and opt in only the exact OpenAI-backed slugs we ship. These facts
|
|
842
|
+
// belong only to the canonical destination; a same-named custom gateway is unknown to us.
|
|
843
|
+
modelServiceTierCapabilityBaseUrlGuard: isCanonicalOpenRouterTarget,
|
|
844
|
+
modelSupportsServiceTier: {
|
|
845
|
+
"openai/gpt-5.6-sol": true,
|
|
846
|
+
"openai/gpt-5.6-terra": true,
|
|
847
|
+
"openai/gpt-5.6-luna": true,
|
|
848
|
+
},
|
|
849
|
+
// Deliberately no OpenRouter route pin: it bills the endpoint actually used and reports the
|
|
850
|
+
// actual service_tier. B0 confirmation therefore owns downgrade safety. Forcing `only` plus
|
|
851
|
+
// `allow_fallbacks:false` would turn a graceful priority-capacity fallback into a hard failure.
|
|
852
|
+
},
|
|
853
|
+
{
|
|
854
|
+
// Primary sources checked 2026-08-02:
|
|
855
|
+
// - docs.cline.bot/getting-started/clinepass publishes this exact catalog and explicitly
|
|
856
|
+
// authorizes using the full slugs through Cline's external API.
|
|
857
|
+
// - docs.cline.bot/api/chat-completions and /api/errors define the endpoint, reasoning delta,
|
|
858
|
+
// and choice-scoped mid-stream error contract.
|
|
859
|
+
// - Cline's official catalog source resolves per-model capabilities through OpenRouter data;
|
|
860
|
+
// the static context/modality snapshot below was cross-checked against that catalog.
|
|
861
|
+
// - cline.bot/tos identifies Cline Bot Inc. as the operator. Maintenance owner: @lidge-jun.
|
|
862
|
+
id: "cline-pass",
|
|
863
|
+
label: "ClinePass",
|
|
864
|
+
adapter: "openai-chat",
|
|
865
|
+
baseUrl: "https://api.cline.bot/api/v1",
|
|
866
|
+
authKind: "key",
|
|
867
|
+
dashboardUrl: "https://app.cline.bot",
|
|
868
|
+
defaultModel: "cline-pass/kimi-k3",
|
|
869
|
+
models: CLINE_PASS_MODELS,
|
|
870
|
+
modelContextWindows: CLINE_PASS_MODEL_CONTEXT_WINDOWS,
|
|
871
|
+
modelInputModalities: CLINE_PASS_MODEL_INPUT_MODALITIES,
|
|
872
|
+
noVisionModels: CLINE_PASS_TEXT_ONLY_MODELS,
|
|
873
|
+
// Live-probed 2026-08-13 across every static ClinePass model: the gateway accepts and
|
|
874
|
+
// validates low/medium/high/xhigh/max, and rejects an invalid sentinel. Preserve the
|
|
875
|
+
// caller's requested tier and let ClinePass own any backend-specific normalization.
|
|
876
|
+
reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
|
|
877
|
+
reasoningWireFormat: "gateway-object",
|
|
878
|
+
preserveCustomDestination: true,
|
|
879
|
+
note: "ClinePass subscription API. Uses a Cline API key and the full cline-pass/<model> upstream slug; quota is shared across the account's rolling 5-hour, weekly, and monthly limits.",
|
|
880
|
+
},
|
|
881
|
+
// Cline API (usage-billing): OpenAI-compatible Chat Completions. Model IDs follow the
|
|
882
|
+
// OpenRouter-style `provider/model` convention. Live /models discovery is key-gated (401
|
|
883
|
+
// without auth), so the static seed is the cold-start fallback. Evidence: docs.cline.bot/api/*.
|
|
884
|
+
{
|
|
885
|
+
id: "cline",
|
|
886
|
+
label: "Cline",
|
|
887
|
+
adapter: "openai-chat",
|
|
888
|
+
baseUrl: "https://api.cline.bot/api/v1",
|
|
889
|
+
authKind: "key",
|
|
890
|
+
dashboardUrl: "https://app.cline.bot",
|
|
891
|
+
liveModels: true,
|
|
892
|
+
defaultModel: "anthropic/claude-sonnet-4-6",
|
|
893
|
+
models: [
|
|
894
|
+
"anthropic/claude-sonnet-4-6",
|
|
895
|
+
"openai/gpt-4o",
|
|
896
|
+
"google/gemini-2.5-pro",
|
|
897
|
+
"deepseek/deepseek-chat",
|
|
898
|
+
"minimax/minimax-m2.5",
|
|
899
|
+
],
|
|
900
|
+
preserveCustomDestination: true,
|
|
901
|
+
note: "Cline usage-billing API: one key, 100+ models, OpenRouter-style ids. Promotional free models are IDE/CLI-only per Cline docs; minimax/minimax-m2.5 is the documented API free experimentation model.",
|
|
902
|
+
},
|
|
903
|
+
{
|
|
904
|
+
// OrcaRouter: OpenAI-compatible adaptive router (api.orcarouter.ai). The public live
|
|
905
|
+
// catalog is authoritative; model ids and input modalities are never maintained here.
|
|
906
|
+
id: "orcarouter", label: "OrcaRouter - API", adapter: "openai-chat", baseUrl: "https://api.orcarouter.ai/v1",
|
|
907
|
+
authKind: "key", dashboardUrl: "https://www.orcarouter.ai/console",
|
|
908
|
+
// The catalog is public, so a successful /models probe cannot validate a submitted key.
|
|
909
|
+
apiKeyValidation: "unknown",
|
|
910
|
+
// Standard sponsor under SPONSORS.md (agreement signed 2026-09-07). Pins the row in the
|
|
911
|
+
// picker and adds the chip; nothing about routing or defaults changes.
|
|
912
|
+
sponsor: { tier: "standard", url: "https://www.orcarouter.ai/?utm_source=opencodex&utm_medium=readme" },
|
|
913
|
+
defaultModel: "openai/gpt-5.5",
|
|
914
|
+
models: ORCAROUTER_MODELS,
|
|
915
|
+
liveModels: true,
|
|
916
|
+
modelDiscovery: ORCAROUTER_MODEL_DISCOVERY,
|
|
917
|
+
// Catalog discovery owns WHICH models exist. These entries only retain verified
|
|
918
|
+
// request-shaping facts that the upstream catalog does not currently publish.
|
|
919
|
+
modelReasoningEfforts: ORCAROUTER_MODEL_REASONING_EFFORTS,
|
|
920
|
+
note: "OpenAI-compatible adaptive router. Models and multimodal capabilities are discovered live from the public chat catalog. Use the OrcaRouter account entry for PKCE login.",
|
|
921
|
+
},
|
|
922
|
+
{
|
|
923
|
+
// PackyCode: API relay (packyapi.com) for Claude Code, Codex, Gemini and more. Codex traffic
|
|
924
|
+
// uses the OpenAI-compatible host from their Codex/Kimi Code guides (docs.packyapi.com):
|
|
925
|
+
// https://cf.api.fan/v1 — GET /v1/models answers 401 without a key, so the host is live and
|
|
926
|
+
// discovery narrows to what the key's token group allows. Model ids are bare OpenAI-style
|
|
927
|
+
// ids (the Codex token group lists gpt-5.5 / gpt-5.1-codex).
|
|
928
|
+
// Standard sponsor under SPONSORS.md; the dashboardUrl carries their affiliate code.
|
|
929
|
+
id: "packycode", label: "PackyCode", adapter: "openai-chat", baseUrl: "https://cf.api.fan/v1",
|
|
930
|
+
authKind: "key", dashboardUrl: "https://www.packyapi.com/register?aff=k5KT",
|
|
931
|
+
sponsor: { tier: "standard", url: "https://www.packyapi.com/register?aff=k5KT" },
|
|
932
|
+
defaultModel: "gpt-5.5",
|
|
933
|
+
models: ["gpt-5.5", "gpt-5.1-codex"],
|
|
934
|
+
liveModels: true,
|
|
935
|
+
// New key preset: opt into collision preservation so a row named `packycode` that a user
|
|
936
|
+
// points at a different PackyCode host keeps its own destination instead of being pulled
|
|
937
|
+
// back onto the Codex endpoint below.
|
|
938
|
+
preserveCustomDestination: true,
|
|
939
|
+
note: "API relay for Claude Code, Codex, Gemini and more. Create a Codex-group token at packyapi.com; live discovery lists what the token group allows.",
|
|
940
|
+
},
|
|
941
|
+
{
|
|
942
|
+
// BizRouter: Korean enterprise LLM gateway (api.bizrouter.ai). Model ids are
|
|
943
|
+
// vendor-namespaced (`<vendor>/<model>`) and pass through to the upstream as-is.
|
|
944
|
+
// Live-verified 2026-07-24: /v1/chat/completions accepts the `tools` field and
|
|
945
|
+
// streams, and GET /v1/models returns the per-API-key allowed catalog in the
|
|
946
|
+
// OpenAI list shape, so live model discovery narrows to what the key can use.
|
|
947
|
+
id: "bizrouter", label: "BizRouter", adapter: "openai-chat", baseUrl: "https://api.bizrouter.ai/v1",
|
|
948
|
+
authKind: "key", dashboardUrl: "https://bizrouter.ai/settings/keys",
|
|
949
|
+
defaultModel: "openai/gpt-5.6-sol",
|
|
950
|
+
models: ["openai/gpt-5.6-sol", "anthropic/claude-sonnet-5", "google/gemini-3.5-flash"],
|
|
951
|
+
note: "Korean enterprise LLM gateway. Per-key allowed models are discovered live from /v1/models. Full catalog: https://bizrouter.ai/models",
|
|
952
|
+
},
|
|
953
|
+
{ id: "groq", label: "Groq", adapter: "openai-chat", baseUrl: "https://api.groq.com/openai/v1", authKind: "key", featured: true, dashboardUrl: "https://console.groq.com/keys" },
|
|
954
|
+
// 2026-07-10 Gemini API refresh: Tier-2 ai.google.dev evidence recorded in
|
|
955
|
+
// devlog/_plan/260710_provider_hardening/001_research_frontier.md.
|
|
956
|
+
{
|
|
957
|
+
id: "google", label: "Google Gemini", adapter: "google", baseUrl: "https://generativelanguage.googleapis.com", authKind: "key", featured: true,
|
|
958
|
+
dashboardUrl: "https://aistudio.google.com/apikey", defaultModel: "gemini-3.5-flash", models: ["gemini-3.8-flash", "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro-preview", "gemini-3.7-flash"],
|
|
959
|
+
modelContextWindows: { "gemini-3.8-flash": 1_048_576, "gemini-3.6-flash": 1_048_576, "gemini-3.5-flash": 1_000_000, "gemini-3.5-flash-lite": 1_048_576, "gemini-3.7-flash": 1_048_576 },
|
|
960
|
+
modelInputModalities: { "gemini-3.8-flash": ["text", "image"], "gemini-3.6-flash": ["text", "image"], "gemini-3.5-flash-lite": ["text", "image"], "gemini-3.7-flash": ["text", "image"] },
|
|
961
|
+
modelReasoningEfforts: {
|
|
962
|
+
// 3.7 and 3.8 omit `minimal`: Google documents it as a validation error on both model
|
|
963
|
+
// pages, so advertising it hands the user a rung the API rejects. 3.5/3.6 keep theirs —
|
|
964
|
+
// their pages still list it, and this unit has no evidence to change them.
|
|
965
|
+
"gemini-3.8-flash": ["low", "medium", "high"],
|
|
966
|
+
"gemini-3.7-flash": ["low", "medium", "high"],
|
|
967
|
+
"gemini-3.6-flash": ["minimal", "low", "medium", "high"],
|
|
968
|
+
"gemini-3.5-flash": ["minimal", "low", "medium", "high"],
|
|
969
|
+
"gemini-3.1-pro-preview": ["low", "medium", "high"],
|
|
970
|
+
},
|
|
971
|
+
jawcodeBundle: "google", extraMetadataAliases: ["gemini"],
|
|
972
|
+
},
|
|
973
|
+
// 2026-07-10: defaultModel is frozen pending Vertex-specific Tier-2 evidence; Gemini API
|
|
974
|
+
// evidence from ai.google.dev does not establish Vertex publisher availability.
|
|
975
|
+
{ id: "google-vertex", label: "Google Vertex AI", adapter: "google", baseUrl: "https://aiplatform.googleapis.com", authKind: "key", dashboardUrl: "https://console.cloud.google.com/vertex-ai", defaultModel: "gemini-3-pro", googleMode: "vertex", jawcodeBundle: "google", extraMetadataAliases: ["gemini-vertex"] },
|
|
976
|
+
// Antigravity discovers models with a POST to the CCA `:fetchAvailableModels` RPC, which
|
|
977
|
+
// `buildModelsRequest` already built by hand. Declaring it here changes no request URL — the
|
|
978
|
+
// relative path resolves to the same destination — but it lets `isRegistryModelDiscoveryUrl`
|
|
979
|
+
// prove that URL, which is what admits a Clash/Surge/Mihomo TUN fake-IP answer (#4261). The
|
|
980
|
+
// path must stay RELATIVE: this row sets `allowBaseUrlOverride`, and an absolute `url` would
|
|
981
|
+
// retarget a user's custom base back to Google. A leading `./` is required because a bare
|
|
982
|
+
// `v1internal:` reads as a URL scheme and `providerModelDiscoverySpecError` rejects it.
|
|
983
|
+
{ id: "google-antigravity", alias: "agy", label: "Google Antigravity", adapter: "google", baseUrl: "https://daily-cloudcode-pa.googleapis.com", authKind: "oauth", allowBaseUrlOverride: true, dashboardUrl: "https://antigravity.google", models: ANTIGRAVITY_MODELS, liveModels: true, defaultModel: "gemini-3.8-flash", modelContextWindows: ANTIGRAVITY_MODEL_CONTEXT_WINDOWS, modelInputModalities: ANTIGRAVITY_MODEL_INPUT_MODALITIES, modelReasoningEfforts: ANTIGRAVITY_MODEL_EFFORTS, googleMode: "cloud-code-assist", showThinkingSummary: true, jawcodeBundle: "google", extraMetadataAliases: ["antigravity", "gemini-antigravity"], modelDiscovery: { path: "./v1internal:fetchAvailableModels" } },
|
|
984
|
+
{ id: "azure-openai", label: "Azure OpenAI", adapter: "azure-openai", baseUrl: "https://{resource}.openai.azure.com/openai", authKind: "key", featured: true, dashboardUrl: "https://portal.azure.com" },
|
|
985
|
+
{ id: "ollama", label: "Ollama (local)", adapter: "openai-chat", baseUrl: "http://localhost:11434/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — key usually blank" },
|
|
986
|
+
{ id: "vllm", label: "vLLM (local)", adapter: "openai-chat", baseUrl: "http://localhost:8000/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — key usually blank" },
|
|
987
|
+
{ id: "lm-studio", label: "LM Studio (local)", adapter: "openai-chat", baseUrl: "http://localhost:1234/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — no key needed" },
|
|
988
|
+
{
|
|
989
|
+
id: "deepseek",
|
|
990
|
+
label: "DeepSeek",
|
|
991
|
+
baseUrl: "https://api.deepseek.com",
|
|
992
|
+
adapter: "openai-chat",
|
|
993
|
+
authKind: "key",
|
|
994
|
+
dashboardUrl: "https://platform.deepseek.com/api_keys",
|
|
995
|
+
// Route DeepSeek's own catalog bundle so routed rebuilds restore the official
|
|
996
|
+
// context window from the vendored model-metadata bundle instead of falling
|
|
997
|
+
// back to the 128k strict-fields default (scripts/model-metadata.source.json,
|
|
998
|
+
// verified 2026-08-08).
|
|
999
|
+
jawcodeBundle: "deepseek",
|
|
1000
|
+
// deepseek-chat/deepseek-reasoner were deprecated upstream on 2026-07-24 15:59 UTC;
|
|
1001
|
+
// the current official identifier is deepseek-flash. They stay in
|
|
1002
|
+
// the list only as compatibility aliases so existing saved configs and requests
|
|
1003
|
+
// keep validating and routing (they previously mapped to v4-flash; devlog
|
|
1004
|
+
// _fin/260710_provider_hardening/002_research_cn.md). The current offerings are
|
|
1005
|
+
// V4.1-Flash — defaultModel and the model-specific wiring below use its live id.
|
|
1006
|
+
// Keep the legacy vision-preview alias; see DEEPSEEK_VISION_PREVIEW_MODEL.
|
|
1007
|
+
models: ["deepseek-chat", "deepseek-reasoner", ...DEEPSEEK_NATIVE_THINKING_MODELS, DEEPSEEK_VISION_PREVIEW_MODEL],
|
|
1008
|
+
// V4.1-Flash is the current first-party offering; `deepseek-v4-flash` now routes there
|
|
1009
|
+
// as a compatibility alias, so a new install should ask for the live id by name.
|
|
1010
|
+
defaultModel: "deepseek-flash",
|
|
1011
|
+
// Official DeepSeek Codex setup (codex-deepseek-setup.sh) advertises 1,048,576
|
|
1012
|
+
// for both V4 models; the older 1,000,000 figure was a rounded approximation.
|
|
1013
|
+
modelContextWindows: { "deepseek-flash": 1_048_576, "deepseek-v4-flash": 1_048_576, [DEEPSEEK_VISION_PREVIEW_MODEL]: 1_048_576 },
|
|
1014
|
+
modelInputModalities: {
|
|
1015
|
+
"deepseek-flash": ["text", "image"],
|
|
1016
|
+
[DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
|
|
1017
|
+
},
|
|
1018
|
+
// DeepSeek documents both V4 models as native Responses API models adapted for Codex
|
|
1019
|
+
// (model table marks Responses API ✓ for flash and pro; the /responses reference lists
|
|
1020
|
+
// both ids as accepted `model` values — verified 2026-08-13 with the V4 Pro GA,
|
|
1021
|
+
// version label DeepSeek-V4-Pro-0813).
|
|
1022
|
+
modelWireDefaults: {
|
|
1023
|
+
// Codex speaks Responses natively and DeepSeek ships a Codex-compatible
|
|
1024
|
+
// apply_patch tool on that wire, so a Responses inbound goes straight out with
|
|
1025
|
+
// no translation. Claude Code and OpenAI-compatible clients keep the
|
|
1026
|
+
// provider-wide Chat wire: DeepSeek serves Chat Completions natively too, so
|
|
1027
|
+
// translating them into Responses would add a hop onto our newest upstream path
|
|
1028
|
+
// for no gain.
|
|
1029
|
+
"deepseek-v4-flash": { wire: "openai-responses", inbound: ["responses"] },
|
|
1030
|
+
// Same Responses contract as the V4 ids it succeeds; without this row the new
|
|
1031
|
+
// default would fall back to the provider-wide Chat wire.
|
|
1032
|
+
"deepseek-flash": { wire: "openai-responses", inbound: ["responses"] },
|
|
1033
|
+
},
|
|
1034
|
+
// The #875-era bounded-JSON force (`modelResponsesUpstreamStreaming`) is retired
|
|
1035
|
+
// for this entry: the official guide documents a `response.completed` /
|
|
1036
|
+
// `response.incomplete` / `response.failed` terminal with NO `data: [DONE]`
|
|
1037
|
+
// sentinel, and live probes (2026-08-07, including the tool-result replay shape
|
|
1038
|
+
// that originally stalled) close on the terminal. The relay's terminal boundary
|
|
1039
|
+
// (src/server/relay.ts) already cuts the stream at that event and synthesizes
|
|
1040
|
+
// `[DONE]`, so forcing stream:false only delayed every byte until generation
|
|
1041
|
+
// finished (28-46 s of silence on long turns). The registry knob itself remains
|
|
1042
|
+
// for providers that need it — re-adding one line here restores the old policy.
|
|
1043
|
+
// Evidence: https://api-docs.deepseek.com/guides/responses_api/ +
|
|
1044
|
+
// devlog/_fin/260807_deepseek_responses_streaming/000_plan.md.
|
|
1045
|
+
// Current official streams normally carry a real terminal; retain a narrow grace
|
|
1046
|
+
// repair for the historical shape that closes after a complete graph without one.
|
|
1047
|
+
modelResponsesTerminalRepair: { "deepseek-flash": { graceMs: 5_000 }, "deepseek-v4-flash": { graceMs: 5_000 } },
|
|
1048
|
+
// DeepSeek's Responses route emits bare UUID item ids, which leave Codex
|
|
1049
|
+
// clients stuck on an uncommitted turn (#938). Client-facing only — raw
|
|
1050
|
+
// continuation snapshots keep the upstream ids.
|
|
1051
|
+
responsesItemIdRepair: { repairInvalidIds: true, repairMissingTerminalIds: true },
|
|
1052
|
+
// DeepSeek's Responses route is `POST /responses` with no `/v1` segment. Without
|
|
1053
|
+
// this the passthrough adapter falls back to its legacy `/v1/responses`
|
|
1054
|
+
// construction and the wire above can never route.
|
|
1055
|
+
// Evidence: https://api-docs.deepseek.com/api/create-response/
|
|
1056
|
+
responsesPath: "/responses",
|
|
1057
|
+
// DeepSeek's Responses reference does not list `service_tier`; unsupported
|
|
1058
|
+
// parameters are documented as silently ignored, but the fail-closed policy
|
|
1059
|
+
// strips the field rather than forwarding a knob the upstream never asked for.
|
|
1060
|
+
supportsServiceTier: false,
|
|
1061
|
+
// DeepSeek's Responses compatibility guide accepts plaintext reasoning items and
|
|
1062
|
+
// merges them into the adjacent assistant message, so replayed reasoning must
|
|
1063
|
+
// not be blanked the way the ChatGPT backend requires. (Whether the Responses
|
|
1064
|
+
// route REQUIRES replay on tool-call continuations is an inference from the
|
|
1065
|
+
// Chat Thinking-Mode docs, not a confirmed Responses contract.)
|
|
1066
|
+
preserveResponsesReasoningContent: true,
|
|
1067
|
+
// "The API is stateless: responses and conversations are not stored on the
|
|
1068
|
+
// server." https://api-docs.deepseek.com/api/create-response/
|
|
1069
|
+
statelessResponses: true,
|
|
1070
|
+
// DeepSeek rejects a valid Codex continuation when hook-provided developer
|
|
1071
|
+
// context splits a call from its result (#1292); parallel calls remain one
|
|
1072
|
+
// reasoning-bearing assistant batch rather than being split per pair (#1477).
|
|
1073
|
+
requiresAdjacentResponsesToolResults: true,
|
|
1074
|
+
// DeepSeek exec tool results can be present-but-empty (a script that ran without
|
|
1075
|
+
// calling text(...)); annotate them so routed models do not silently accept an
|
|
1076
|
+
// empty result or re-issue the same call.
|
|
1077
|
+
annotateEmptyToolOutputs: true,
|
|
1078
|
+
/* [Decision Log]
|
|
1079
|
+
- 목적: DeepSeek V4 thinking mode multi-turn/tool-call requests must replay prior assistant reasoning_content.
|
|
1080
|
+
- 대안 분석: Globally preserve reasoning_content for all OpenAI-compatible models; preserve it for legacy deepseek-reasoner too; mark only V4 thinking models in registry metadata.
|
|
1081
|
+
- 선택 근거: DeepSeek V4 thinking mode requires history replay, while older DeepSeek reasoner has different compatibility rules. A model-scoped registry flag fixes built-in and stale saved configs without broad provider regressions.
|
|
1082
|
+
*/
|
|
1083
|
+
modelReasoningEfforts: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, deepseekThinkingEffortsFor(id)])),
|
|
1084
|
+
modelReasoningEffortMap: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, deepseekReasoningMapFor(id)])),
|
|
1085
|
+
modelSupportsReasoningSummaries: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, true])),
|
|
1086
|
+
preserveReasoningContentModels: DEEPSEEK_NATIVE_THINKING_MODELS,
|
|
1087
|
+
// #4436: first-party deepseek-flash accepts native images on Chat and Responses.
|
|
1088
|
+
// Keep unprobed compatibility aliases on the #88 sidecar path. This must be fixed
|
|
1089
|
+
// here: router enrichment unions this list with saved config, so config cannot remove it.
|
|
1090
|
+
noVisionModels: ["deepseek-chat", "deepseek-reasoner", "deepseek-v4-flash"],
|
|
1091
|
+
},
|
|
1092
|
+
// llama-3.3-70b was deprecated by Cerebras on 2026-02-16. Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
|
|
1093
|
+
{ id: "cerebras", label: "Cerebras", baseUrl: "https://api.cerebras.ai/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://cloud.cerebras.ai/platform/apikeys", defaultModel: "gpt-oss-120b" },
|
|
1094
|
+
{
|
|
1095
|
+
// Primary sources checked 2026-08-08:
|
|
1096
|
+
// - https://chutes.ai/pricing documents the shared llm.chutes.ai/v1 OpenAI-compatible
|
|
1097
|
+
// gateway, Bearer API keys, and chat completions. Its public
|
|
1098
|
+
// https://llm.chutes.ai/v1/models response supplies supported_features for filtering.
|
|
1099
|
+
// - https://chutes.ai/terms identifies Chutes Global Corp as the platform operator, applies
|
|
1100
|
+
// to API consumers, and directs production/high-volume automated inference to PAYGO.
|
|
1101
|
+
// Maintainer: @olddonkey; no affiliation with Chutes.
|
|
1102
|
+
id: "chutes",
|
|
1103
|
+
label: "Chutes",
|
|
1104
|
+
baseUrl: "https://llm.chutes.ai/v1",
|
|
1105
|
+
adapter: "openai-chat",
|
|
1106
|
+
authKind: "key",
|
|
1107
|
+
dashboardUrl: "https://chutes.ai/auth/start",
|
|
1108
|
+
liveModels: true,
|
|
1109
|
+
preserveCustomDestination: true,
|
|
1110
|
+
// The public model catalog cannot prove that a supplied Bearer key is valid.
|
|
1111
|
+
apiKeyValidation: "unknown",
|
|
1112
|
+
// Chutes documents tool calling, but not a provider-wide parallel tool-call contract.
|
|
1113
|
+
parallelToolCalls: false,
|
|
1114
|
+
// The live catalog reports reasoning support, but not a stable effort ladder.
|
|
1115
|
+
reasoningEfforts: [],
|
|
1116
|
+
modelDiscovery: {
|
|
1117
|
+
path: "models",
|
|
1118
|
+
maxResponseBytes: 256 * 1024,
|
|
1119
|
+
maxModels: 128,
|
|
1120
|
+
filter: {
|
|
1121
|
+
// The shared LLM catalog also contains rows without native tool support. Codex needs a
|
|
1122
|
+
// complete agent loop, so admit only rows whose live metadata advertises tools.
|
|
1123
|
+
allOf: [{ path: ["supported_features"], containsAny: ["tools"] }],
|
|
1124
|
+
},
|
|
1125
|
+
},
|
|
1126
|
+
note: "Shared OpenAI-compatible LLM gateway only; live discovery exposes tool-capable rows. User-deployed custom Chute endpoints and non-LLM APIs require a custom provider.",
|
|
1127
|
+
},
|
|
1128
|
+
{
|
|
1129
|
+
id: "deepinfra",
|
|
1130
|
+
label: "DeepInfra",
|
|
1131
|
+
baseUrl: "https://api.deepinfra.com/v1/openai",
|
|
1132
|
+
adapter: "openai-chat",
|
|
1133
|
+
authKind: "key",
|
|
1134
|
+
dashboardUrl: "https://deepinfra.com/dash/api_keys",
|
|
1135
|
+
liveModels: true,
|
|
1136
|
+
preserveCustomDestination: true,
|
|
1137
|
+
modelDiscovery: {
|
|
1138
|
+
// DeepInfra documents the OpenAI model catalog outside the chat-compatible `/v1/openai`
|
|
1139
|
+
// namespace, so keep this destination registry-owned instead of deriving it from baseUrl.
|
|
1140
|
+
url: "https://api.deepinfra.com/v1/models",
|
|
1141
|
+
maxResponseBytes: 512 * 1024,
|
|
1142
|
+
maxModels: 512,
|
|
1143
|
+
filter: {
|
|
1144
|
+
allOf: [{ path: ["metadata", "tags"], containsAny: ["chat"] }],
|
|
1145
|
+
},
|
|
1146
|
+
},
|
|
1147
|
+
note: "OpenAI-compatible chat models only; live discovery excludes non-chat rows from DeepInfra's mixed model catalog.",
|
|
1148
|
+
},
|
|
1149
|
+
{
|
|
1150
|
+
id: "hyperbolic",
|
|
1151
|
+
label: "Hyperbolic",
|
|
1152
|
+
baseUrl: "https://api.hyperbolic.xyz/v1",
|
|
1153
|
+
adapter: "openai-chat",
|
|
1154
|
+
authKind: "key",
|
|
1155
|
+
dashboardUrl: "https://app.hyperbolic.ai",
|
|
1156
|
+
liveModels: true,
|
|
1157
|
+
preserveCustomDestination: true,
|
|
1158
|
+
modelDiscovery: {
|
|
1159
|
+
path: "models",
|
|
1160
|
+
maxResponseBytes: 256 * 1024,
|
|
1161
|
+
maxModels: 256,
|
|
1162
|
+
},
|
|
1163
|
+
note: "Serverless text and vision-language chat models only; Hyperbolic's separate image, audio, and GPU endpoints are out of scope.",
|
|
1164
|
+
},
|
|
1165
|
+
{
|
|
1166
|
+
// Primary sources checked 2026-08-03:
|
|
1167
|
+
// - docs.nscale.com documents the production OpenAI-compatible endpoint, bearer service
|
|
1168
|
+
// tokens, /v1/models, and a tool-calling request using this exact Llama model id.
|
|
1169
|
+
// - nscale.com/policies/terms-conditions identifies Nscale AS as the service operator and
|
|
1170
|
+
// covers customers using its public-cloud inference offering. Maintainer: @olddonkey;
|
|
1171
|
+
// no affiliation with Nscale.
|
|
1172
|
+
id: "nscale",
|
|
1173
|
+
label: "Nscale Serverless Inference",
|
|
1174
|
+
baseUrl: "https://inference.api.nscale.com/v1",
|
|
1175
|
+
adapter: "openai-chat",
|
|
1176
|
+
authKind: "key",
|
|
1177
|
+
dashboardUrl: "https://console.nscale.com",
|
|
1178
|
+
defaultModel: "meta-llama/Llama-3.1-8B-Instruct",
|
|
1179
|
+
models: ["meta-llama/Llama-3.1-8B-Instruct"],
|
|
1180
|
+
liveModels: true,
|
|
1181
|
+
preserveCustomDestination: true,
|
|
1182
|
+
// Nscale documents tools but not parallel tool calls. Keep requests serialized.
|
|
1183
|
+
parallelToolCalls: false,
|
|
1184
|
+
// The API schema accepts reasoning_effort, but does not publish per-model tiers.
|
|
1185
|
+
reasoningEfforts: [],
|
|
1186
|
+
modelDiscovery: {
|
|
1187
|
+
path: "models",
|
|
1188
|
+
maxResponseBytes: 256 * 1024,
|
|
1189
|
+
maxModels: 256,
|
|
1190
|
+
filter: {
|
|
1191
|
+
// Nscale's catalog mixes chat, image, and embedding rows without a modality field.
|
|
1192
|
+
// Admit only the exact model used in its official tool-calling API example.
|
|
1193
|
+
allOf: [{ path: ["id"], equalsAny: ["meta-llama/Llama-3.1-8B-Instruct"] }],
|
|
1194
|
+
},
|
|
1195
|
+
},
|
|
1196
|
+
note: "Serverless OpenAI-compatible inference. Live discovery admits only the tool-capable model established by Nscale's official API example; other mixed-catalog rows remain hidden pending equivalent evidence.",
|
|
1197
|
+
},
|
|
1198
|
+
{
|
|
1199
|
+
// Primary sources checked 2026-08-03:
|
|
1200
|
+
// - docs.vultr.com documents the fixed OpenAI-compatible base URL, per-subscription bearer
|
|
1201
|
+
// key, /v1/models, and states that tool calling is currently limited to kimi-k2-instruct.
|
|
1202
|
+
// - Vultr's official properties identify VULTR as a The Constant Company, LLC trademark and
|
|
1203
|
+
// document customer API integrations. Maintainer: @olddonkey; no affiliation with Vultr.
|
|
1204
|
+
id: "vultr",
|
|
1205
|
+
label: "Vultr Serverless Inference",
|
|
1206
|
+
baseUrl: "https://api.vultrinference.com/v1",
|
|
1207
|
+
adapter: "openai-chat",
|
|
1208
|
+
authKind: "key",
|
|
1209
|
+
dashboardUrl: "https://my.vultr.com",
|
|
1210
|
+
defaultModel: "kimi-k2-instruct",
|
|
1211
|
+
models: ["kimi-k2-instruct"],
|
|
1212
|
+
liveModels: true,
|
|
1213
|
+
preserveCustomDestination: true,
|
|
1214
|
+
parallelToolCalls: false,
|
|
1215
|
+
reasoningEfforts: [],
|
|
1216
|
+
modelDiscovery: {
|
|
1217
|
+
path: "models",
|
|
1218
|
+
maxResponseBytes: 256 * 1024,
|
|
1219
|
+
maxModels: 256,
|
|
1220
|
+
filter: {
|
|
1221
|
+
// Vultr explicitly limits tool calling to this model. A coding agent must not select
|
|
1222
|
+
// another chat model that cannot complete its tool loop.
|
|
1223
|
+
allOf: [{ path: ["id"], equalsAny: ["kimi-k2-instruct"] }],
|
|
1224
|
+
},
|
|
1225
|
+
},
|
|
1226
|
+
note: "Serverless Inference subscription API. Live discovery exposes only kimi-k2-instruct because Vultr documents it as the sole tool-calling model.",
|
|
1227
|
+
},
|
|
1228
|
+
];
|