@bitkyc08/opencodex 2.55.0 → 2.56.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/gui/dist/assets/{index-VuoiWj9J.js → index-D4zuyIxQ.js} +1 -1
  2. package/gui/dist/index.html +1 -1
  3. package/package.json +2 -1
  4. package/src/adapters/base.ts +21 -0
  5. package/src/adapters/cursor/transport-retry.ts +46 -1
  6. package/src/adapters/cursor.ts +4 -0
  7. package/src/adapters/kiro/adapter.ts +42 -1
  8. package/src/adapters/kiro-retry.ts +23 -4
  9. package/src/adapters/openai-chat/errors.ts +116 -0
  10. package/src/adapters/openai-chat/messages.ts +346 -0
  11. package/src/adapters/openai-chat/passthrough.ts +146 -0
  12. package/src/adapters/openai-chat/response-events.ts +117 -0
  13. package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
  14. package/src/adapters/openai-chat/tool-schema.ts +477 -0
  15. package/src/adapters/openai-chat/wire.ts +50 -0
  16. package/src/adapters/openai-chat.ts +33 -1445
  17. package/src/adapters/openai-responses/canonical-forward.ts +202 -0
  18. package/src/adapters/openai-responses/image-gen.ts +406 -0
  19. package/src/adapters/openai-responses/internal.ts +3 -0
  20. package/src/adapters/openai-responses/passthrough.ts +611 -0
  21. package/src/adapters/openai-responses/prompt-cache.ts +83 -0
  22. package/src/adapters/openai-responses/reasoning.ts +220 -0
  23. package/src/adapters/openai-responses/request-strips.ts +185 -0
  24. package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
  25. package/src/adapters/openai-responses/tool-schema.ts +293 -0
  26. package/src/adapters/openai-responses/web-search.ts +156 -0
  27. package/src/adapters/openai-responses.ts +4 -2625
  28. package/src/bridge/errors.ts +34 -0
  29. package/src/bridge/internal.ts +174 -0
  30. package/src/bridge/response-json.ts +624 -0
  31. package/src/bridge/sse.ts +1444 -0
  32. package/src/bridge.ts +5 -2204
  33. package/src/chat/inbound.ts +12 -1
  34. package/src/codex/account-lifecycle.ts +3 -0
  35. package/src/codex/account-store.ts +71 -9
  36. package/src/codex/auth-api/account-list.ts +507 -0
  37. package/src/codex/auth-api/http.ts +32 -0
  38. package/src/codex/auth-api/login-flow.ts +554 -0
  39. package/src/codex/auth-api/login-state.ts +64 -0
  40. package/src/codex/auth-api/main-account-probe.ts +331 -0
  41. package/src/codex/auth-api/pool-mode-gate.ts +274 -0
  42. package/src/codex/auth-api/pool-quota-probe.ts +512 -0
  43. package/src/codex/auth-api/reset-credit-service.ts +422 -0
  44. package/src/codex/auth-api/routes.ts +425 -0
  45. package/src/codex/auth-api/runtime-config.ts +48 -0
  46. package/src/codex/auth-api.ts +27 -3118
  47. package/src/codex/auth-context.ts +95 -28
  48. package/src/codex/catalog/auto-review.ts +507 -0
  49. package/src/codex/catalog/build-entries.ts +981 -0
  50. package/src/codex/catalog/combo-member.ts +375 -0
  51. package/src/codex/catalog/derive-entry.ts +229 -0
  52. package/src/codex/catalog/effort.ts +0 -1
  53. package/src/codex/catalog/gated-native-warn.ts +63 -0
  54. package/src/codex/catalog/gather-capture.ts +533 -0
  55. package/src/codex/catalog/model-hints.ts +691 -0
  56. package/src/codex/catalog/model-visibility.ts +304 -0
  57. package/src/codex/catalog/provider-fetch.ts +52 -2942
  58. package/src/codex/catalog/provider-models.ts +685 -0
  59. package/src/codex/catalog/restore.ts +132 -0
  60. package/src/codex/catalog/retained-sync.ts +706 -0
  61. package/src/codex/catalog/routed-gather.ts +858 -0
  62. package/src/codex/catalog/subagent-roster.ts +176 -0
  63. package/src/codex/catalog/sync.ts +52 -2698
  64. package/src/codex/inject/config-toml.ts +563 -0
  65. package/src/codex/inject/remove.ts +192 -0
  66. package/src/codex/inject/restore.ts +540 -0
  67. package/src/codex/inject/routing-classify.ts +109 -0
  68. package/src/codex/inject/routing-target.ts +125 -0
  69. package/src/codex/inject.ts +81 -1436
  70. package/src/codex/lineage.ts +458 -0
  71. package/src/codex/pool-refresh-backoff.ts +152 -0
  72. package/src/codex/routing/active-account.ts +194 -0
  73. package/src/codex/routing/cooldown-math.ts +275 -0
  74. package/src/codex/routing/health-store.ts +402 -0
  75. package/src/codex/routing/probe-lease.ts +358 -0
  76. package/src/codex/routing/selection.ts +703 -0
  77. package/src/codex/routing/thread-affinity.ts +538 -0
  78. package/src/codex/routing.ts +353 -2234
  79. package/src/codex/shim-fingerprint.ts +223 -0
  80. package/src/codex/shim-inspect.ts +175 -0
  81. package/src/codex/shim-probe.ts +367 -0
  82. package/src/codex/shim-restore-lock.ts +169 -0
  83. package/src/codex/shim-state-file.ts +151 -0
  84. package/src/codex/shim-templates.ts +265 -0
  85. package/src/codex/shim.ts +48 -1268
  86. package/src/config/diagnostics.ts +705 -0
  87. package/src/config/feature-flags.ts +55 -0
  88. package/src/config/live-reconcile.ts +403 -0
  89. package/src/config/load-degrade.ts +880 -0
  90. package/src/config/mutation-lock.ts +244 -0
  91. package/src/config/openai-tier-backup.ts +268 -0
  92. package/src/config/persist-unlocked.ts +92 -0
  93. package/src/config/proxy-env.ts +188 -0
  94. package/src/config/salvage.ts +244 -0
  95. package/src/config/schema/config-schema.ts +640 -0
  96. package/src/config/schema/leaf-validators.ts +855 -0
  97. package/src/config/warn-memo.ts +28 -0
  98. package/src/config.ts +234 -4481
  99. package/src/generated/compatibility-version.json +539 -39
  100. package/src/lib/request-execution-budget.ts +69 -20
  101. package/src/lib/spend-reservation-ledger.ts +940 -0
  102. package/src/lib/upstream-retry.ts +55 -11
  103. package/src/lib/workflow-budget.ts +553 -30
  104. package/src/providers/quota/account-cache.ts +441 -0
  105. package/src/providers/quota/antigravity.ts +295 -0
  106. package/src/providers/quota/report-cache.ts +320 -0
  107. package/src/providers/quota/vendor-probes-key.ts +1243 -0
  108. package/src/providers/quota/vendor-probes-oauth.ts +590 -0
  109. package/src/providers/quota.ts +324 -3079
  110. package/src/providers/registry/entries-core.ts +1221 -0
  111. package/src/providers/registry/entries-extended.ts +1204 -0
  112. package/src/providers/registry/model-seeds.ts +908 -0
  113. package/src/providers/registry/types.ts +352 -0
  114. package/src/providers/registry.ts +24 -3536
  115. package/src/responses/continuation-ownership.ts +29 -0
  116. package/src/responses/state/replay-fingerprint.ts +80 -0
  117. package/src/responses/state/snapshot-codec.ts +104 -0
  118. package/src/responses/state/spill-failure.ts +118 -0
  119. package/src/responses/state/spill-queue.ts +665 -0
  120. package/src/responses/state/temp-recovery.ts +257 -0
  121. package/src/responses/state.ts +82 -1143
  122. package/src/routing/identity-domains.ts +449 -0
  123. package/src/routing/probe-lease.ts +511 -0
  124. package/src/server/index/bounded-request.ts +88 -0
  125. package/src/server/index/live-sideband.ts +565 -0
  126. package/src/server/index/serve-options.ts +1766 -0
  127. package/src/server/index/startup-warnings.ts +213 -0
  128. package/src/server/index/websocket-handler.ts +335 -0
  129. package/src/server/index.ts +40 -2547
  130. package/src/server/management/route-registry.ts +26 -23
  131. package/src/server/management/shared.ts +8 -5
  132. package/src/server/management/workflow-budget-routes.ts +133 -0
  133. package/src/server/management-api.ts +12 -0
  134. package/src/server/request-log-conversation.ts +9 -7
  135. package/src/server/request-log.ts +245 -1
  136. package/src/server/responses/account-change-state.ts +233 -0
  137. package/src/server/responses/adapter-continuation.ts +514 -0
  138. package/src/server/responses/adapter-delivery.ts +214 -0
  139. package/src/server/responses/adapter-dispatch.ts +971 -0
  140. package/src/server/responses/compact.ts +59 -4
  141. package/src/server/responses/completion-policy.ts +33 -0
  142. package/src/server/responses/core-auth.ts +527 -0
  143. package/src/server/responses/core-codex-account.ts +859 -0
  144. package/src/server/responses/core-combo-failure.ts +210 -0
  145. package/src/server/responses/core-combo.ts +707 -0
  146. package/src/server/responses/core-errors.ts +152 -0
  147. package/src/server/responses/core-lifetime.ts +95 -0
  148. package/src/server/responses/core-normalize.ts +350 -0
  149. package/src/server/responses/core-opaque-recovery.ts +380 -0
  150. package/src/server/responses/core-options.ts +159 -0
  151. package/src/server/responses/core-replay.ts +225 -0
  152. package/src/server/responses/core.ts +192 -8893
  153. package/src/server/responses/passthrough-delivery.ts +856 -0
  154. package/src/server/responses/passthrough-dispatch.ts +1476 -0
  155. package/src/server/responses/passthrough-execution.ts +54 -0
  156. package/src/server/responses/request-prepare.ts +970 -0
  157. package/src/server/responses/request-send-budget.ts +164 -0
  158. package/src/server/responses/request-sidecar-auth.ts +149 -0
  159. package/src/server/responses/request-transport.ts +744 -0
  160. package/src/server/responses/response-effects.ts +157 -0
  161. package/src/server/responses/run-turn-execution.ts +448 -0
  162. package/src/server/responses/sidecar-execution.ts +469 -0
  163. package/src/server/responses-image-gen-repair.ts +1 -1
  164. package/src/server/workflow-refusal.ts +84 -0
  165. package/src/types/config.ts +30 -0
  166. package/src/usage/log.ts +146 -0
  167. package/src/usage/summary.ts +171 -21
@@ -0,0 +1,1221 @@
1
+ import { KIRO_MODELS, KIRO_MODEL_CONTEXT_WINDOWS, KIRO_MODEL_REASONING_EFFORTS } from "../kiro-models";
2
+ import { DEVIN_MODEL_CONTEXT_WINDOWS, DEVIN_MODEL_EFFORTS, DEVIN_DEFAULT_EFFORTS } from "../../adapters/devin/live-models";
3
+ import { ANTIGRAVITY_MODELS, ANTIGRAVITY_MODEL_CONTEXT_WINDOWS, ANTIGRAVITY_MODEL_EFFORTS, ANTIGRAVITY_MODEL_INPUT_MODALITIES } from "../antigravity-models";
4
+ import {
5
+ CURSOR_NO_VISION_MODELS,
6
+ CURSOR_STATIC_MODELS,
7
+ cursorModelContextWindows,
8
+ cursorModelDisplayNames,
9
+ cursorModelIds,
10
+ cursorModelInputModalities,
11
+ cursorModelReasoningEfforts,
12
+ } from "../../adapters/cursor/discovery";
13
+ import { cursorFastCapableBases } from "../../adapters/cursor/catalog";
14
+ import { COMMAND_CODE_MODEL_REASONING_EFFORTS } from "../command-code-efforts";
15
+ import { isCanonicalOpenRouterTarget } from "../openrouter-routing";
16
+ import type { ProviderRegistryEntry } from "./types";
17
+ import {
18
+ ANTHROPIC_MODELS,
19
+ ANTHROPIC_MODEL_CONTEXT_WINDOWS,
20
+ ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
21
+ ANTHROPIC_MODEL_REASONING_EFFORTS,
22
+ ZAI_GLM_52_REASONING_EFFORTS,
23
+ ZAI_GLM_53_REASONING_EFFORTS,
24
+ OPENAI_GPT56_MODELS,
25
+ OPENAI_GPT56_PRO_MODELS,
26
+ OPENAI_API_GPT56_CONTEXT_WINDOWS,
27
+ OPENAI_API_GPT56_MAX_INPUT_TOKENS,
28
+ OPENAI_API_GPT56_VIRTUAL_MODELS,
29
+ OPENAI_API_GPT56_REASONING_EFFORTS,
30
+ META_MUSE_REASONING_EFFORTS,
31
+ META_MUSE_REASONING_EFFORT_MAP,
32
+ META_MUSE_CONTEXT_WINDOW,
33
+ META_MUSE_MODELS,
34
+ OPENAI_DAYBREAK_MODELS,
35
+ OPENAI_DAYBREAK_CONTEXT_WINDOWS,
36
+ OPENAI_DAYBREAK_MAX_INPUT_TOKENS,
37
+ OPENAI_DAYBREAK_REASONING_EFFORTS,
38
+ OPENROUTER_GPT56_MODELS,
39
+ XAI_MODELS,
40
+ OPENROUTER_GPT56_CONTEXT_WINDOWS,
41
+ THINKING_TOGGLE_EFFORTS,
42
+ THINKING_TOGGLE_MAP,
43
+ OPENCODE_GO_THINKING_TOGGLE_MODELS,
44
+ THINKING_BUDGET_EFFORTS,
45
+ QWEN38_REASONING_EFFORTS,
46
+ THINKING_BUDGET_MODELS,
47
+ OPENCODE_GO_THINKING_BUDGET_MODELS,
48
+ DEEPSEEK_NATIVE_THINKING_MODELS,
49
+ DEEPSEEK_GATEWAY_THINKING_MODELS,
50
+ DEEPSEEK_VISION_PREVIEW_MODEL,
51
+ COMMAND_CODE_MODEL_INPUT_MODALITIES,
52
+ deepseekThinkingEffortsFor,
53
+ deepseekReasoningMapFor,
54
+ KIMI_K3_STANDARD_CONTEXT_WINDOW,
55
+ KIMI_CODING_MODELS,
56
+ KIMI_THINKING_MODELS,
57
+ KIMI_CODING_NO_REASONING_MODELS,
58
+ KIMI_CODING_K3_REASONING_EFFORTS,
59
+ KIMI_CODING_K3_REASONING_EFFORT_MAP,
60
+ KIMI_CODING_REASONING_EFFORTS,
61
+ KIMI_CODING_DEFAULT_REASONING_EFFORTS,
62
+ KIMI_CODING_REASONING_EFFORT_MAPS,
63
+ KIMI_LOCKED_PARAMETER_MODELS,
64
+ KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS,
65
+ KIMI_CODING_MODEL_CONTEXT_WINDOWS,
66
+ KIMI_CODING_MODEL_INPUT_MODALITIES,
67
+ NEURALWATT_REASONING_HISTORY_MODELS,
68
+ UMANS_MODELS,
69
+ UMANS_REASONING_EFFORTS,
70
+ UMANS_GLM_REASONING_EFFORTS,
71
+ UMANS_GLM_53_REASONING_EFFORTS,
72
+ UMANS_TEXT_ONLY_MODELS,
73
+ UMANS_MODEL_CONTEXT_WINDOWS,
74
+ UMANS_MODEL_INPUT_MODALITIES,
75
+ CLINE_PASS_MODELS,
76
+ ORCAROUTER_MODEL_DISCOVERY,
77
+ ORCAROUTER_MODELS,
78
+ ORCAROUTER_MODEL_REASONING_EFFORTS,
79
+ CLINE_PASS_MODEL_CONTEXT_WINDOWS,
80
+ CLINE_PASS_TEXT_ONLY_MODELS,
81
+ CLINE_PASS_MODEL_INPUT_MODALITIES,
82
+ } from "./model-seeds";
83
+
84
+ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
85
+ {
86
+ id: "openai",
87
+ label: "OpenAI (Codex login)",
88
+ adapter: "openai-responses",
89
+ baseUrl: "https://chatgpt.com/backend-api/codex",
90
+ authKind: "forward",
91
+ codexAccountMode: "pool",
92
+ supportsServiceTier: true,
93
+ featured: true,
94
+ note: "Codex login account pool (default) or Direct main-account mode via codexAccountMode",
95
+ },
96
+ {
97
+ id: "cursor",
98
+ label: "Cursor (experimental)",
99
+ adapter: "cursor",
100
+ baseUrl: "https://api2.cursor.sh",
101
+ authKind: "oauth",
102
+ featured: false,
103
+ dashboardPreset: true,
104
+ note: "Experimental Cursor bridge. Live transport and live model discovery are enabled after a standalone PKCE browser login via 'ocx login cursor'; native read/write/delete/shell/fetch execution is disabled by default and request text such as Codex sandbox markers never authorizes it. Set \"nativeLocalExec\": \"on\" on providers.cursor in ~/.opencodex/config.json (dashboard: Providers → Cursor → Edit JSON) only for a trusted local experiment where every data-plane caller is trusted. \"off\" denies all, \"codex-sandbox\" is accepted for backwards compatibility but fails closed, and legacy \"unsafeAllowNativeLocalExec\": true still means explicit operator opt-in.",
105
+ models: cursorModelIds(CURSOR_STATIC_MODELS),
106
+ liveModels: true,
107
+ defaultModel: "auto",
108
+ modelContextWindows: cursorModelContextWindows(CURSOR_STATIC_MODELS),
109
+ modelDisplayNames: cursorModelDisplayNames(),
110
+ // Cursor's Fast product is a model VARIANT, not a service_tier field, so the wire kind
111
+ // is cursor-variant and the request builder consumes the decision.
112
+ fastWire: { kind: "cursor-variant", canonicalToWire: { priority: "fast" }, foreignCallerTiers: "drop" },
113
+ // Deliberately NO provider-level supportsServiceTier: resolveFastPolicy short-circuits on
114
+ // `capability.provider === false` BEFORE consulting the per-model map, which would make
115
+ // these entries dead config. Absent leaves unlisted bases "unclassified", and a
116
+ // non-service-tier adapter cannot forward a caller tier, so they still publish no toggle.
117
+ modelSupportsServiceTier: Object.fromEntries(cursorFastCapableBases().map(id => [id, true])),
118
+ fastTierDescription: "Cursor Fast variant",
119
+ modelInputModalities: cursorModelInputModalities(CURSOR_STATIC_MODELS),
120
+ modelReasoningEfforts: cursorModelReasoningEfforts(CURSOR_STATIC_MODELS),
121
+ // Kimi K3 documents `max` as its API default, and its Cursor ladder has no `medium`
122
+ // rung — so applyReasoningLevels' medium->high->first fallback would settle the catalog
123
+ // default on `high`, the picker would send `high` explicitly, and the request builder's
124
+ // no-effort fallback to `kimi-k3-max` would never be reached. Mirrors the other K3
125
+ // routes (kimi, kimi-code, opencode-go).
126
+ modelDefaultReasoningEfforts: { "kimi-k3": "max" },
127
+ // Blind Cursor models (Auto routers, Composer, GLM-5.2, GLM-5.3) go through the vision sidecar;
128
+ // multimodal hosts (Claude/Gemini/GPT/Kimi/Grok) take native SelectedImage. The catalog
129
+ // still advertises image for noVision members so Codex can attach (sidecar option B).
130
+ noVisionModels: [...CURSOR_NO_VISION_MODELS],
131
+ },
132
+ {
133
+ // The canonical Cognition account provider, after absorbing `devin-cli`
134
+ // (devlog/_plan/260913_devin_provider_merge). The two ids were the same
135
+ // `devin` adapter, the same server.codeium.com api-server, and the same
136
+ // `devin-session-token$<JWT>` credential — only the account source
137
+ // differed: this entry did an Auth0 browser sign-in while `devin-cli`
138
+ // imported the token the installed CLI's own PKCE login had already
139
+ // written to credentials.toml. The merged login is import-first with a
140
+ // browser fallback: the CLI credential is taken when present (no browser
141
+ // opens), and the Auth0 flow remains because it is the only path for
142
+ // users without the CLI. `devin-cli` survives only as a deprecated
143
+ // alias; a startup migration rewrites saved provider rows, cross-config
144
+ // references, and auth.json slots to `devin`.
145
+ //
146
+ // `oauth` classifies the ACCOUNT, not the transport. This is not a local
147
+ // runtime: unlike Ollama or LM Studio it cannot answer at all until a
148
+ // vendor account is signed in, and `local` grouped it with things that
149
+ // have no account. It is also the only classification that reaches the
150
+ // dashboard Accounts tab, which is built from OAUTH_PROVIDERS.
151
+ id: "devin",
152
+ label: "Cognition (Devin/Windsurf)",
153
+ adapter: "devin",
154
+ baseUrl: "https://server.codeium.com",
155
+ authKind: "oauth",
156
+ featured: false,
157
+ // Off: `deriveProviderPresets` keys the preset catalog off this flag, so a
158
+ // true row would draw the provider twice — an Accounts login row and a
159
+ // preset tile.
160
+ dashboardPreset: false,
161
+ note: "Experimental unofficial Cognition/Devin bridge. ocx login devin first imports the credential an installed Devin CLI already holds (no browser); without one it opens Auth0 browser sign-in and exchanges the token via Cognition's RegisterUser for a long-lived API key.",
162
+ // Union seed of the two merged rosters: the newer devin-cli lineup first
163
+ // (it is the current catalog, so its default ordering wins), then the ids
164
+ // only the old devin entry carried. Degraded-mode seed only either way —
165
+ // `liveModels` discovers the account's real roster.
166
+ models: ["swe-2", "swe-1-7", "gpt-5-6-sol", "gpt-6-astra", "claude-opus-5", "claude-fable-5-1", "claude-sonnet-5", "glm-5-3", "kimi-k3", "gemini-3-8-flash", "grok-4-6", "swe-1-7-lightning", "gpt-5-6-luna", "gpt-5-6-terra", "claude-opus-4-8", "glm-5-2", "kimi-k2-7", "grok-4-5"],
167
+ liveModels: true,
168
+ defaultModel: "swe-2",
169
+ modelContextWindows: DEVIN_MODEL_CONTEXT_WINDOWS,
170
+ // Degraded-mode ladders only. Once a credential is present the account
171
+ // catalog supplies each base model its measured rungs; these two fields are
172
+ // what a signed-out picker and the Pi-shaped client exports fall back to.
173
+ modelReasoningEfforts: DEVIN_MODEL_EFFORTS,
174
+ reasoningEfforts: DEVIN_DEFAULT_EFFORTS,
175
+ },
176
+ {
177
+ id: "xai",
178
+ label: "xAI Grok",
179
+ adapter: "openai-chat",
180
+ baseUrl: "https://api.x.ai/v1",
181
+ authKind: "oauth",
182
+ allowKeyAuthOverride: true,
183
+ // Priority Processing is documented for xAI's public API-key Chat Completions and
184
+ // Responses endpoints. The OAuth lane is classified per-model below, not here:
185
+ // do not turn this into a provider-wide supportsServiceTier declaration.
186
+ keyAuthServiceTier: {
187
+ supportsServiceTier: true,
188
+ chatServiceTier: true,
189
+ },
190
+ // OAuth (Grok subscription gateway) service-tier capability, classified by live probe
191
+ // on 2026-09-13 (devlog/_fin/260913_xai_oauth_fast/020_probe-evidence.md): each listed
192
+ // model accepted service_tier "priority" over grok-oauth and echoed priority upstream.
193
+ // Key-auth already declares provider-wide support above, so this map only newly opens
194
+ // the OAuth lane. grok-4.20-multi-agent-0309 is deliberately absent: the gateway accepts
195
+ // the field but answers service_tier "default" — a live downgrade, not a fast tier.
196
+ // Unlisted and future-discovered ids stay unclassified.
197
+ modelSupportsServiceTier: {
198
+ "grok-4.6": true,
199
+ "grok-4.5": true,
200
+ "grok-4.3": true,
201
+ "grok-4.20-0309-reasoning": true,
202
+ "grok-4.20-0309-non-reasoning": true,
203
+ "grok-build-0.1": true,
204
+ "grok-composer-2.5-fast": true,
205
+ },
206
+ // Lets a caller-sent service_tier forward on the Chat wire (fastwire forwardCallerTier
207
+ // chain). Provider-wide by construction: unclassified chat-wire models then preserve a
208
+ // caller tier verbatim, the same contract other unclassified Responses routes already
209
+ // follow; --fast publication and proxy-owned fast injection stay capability-scoped by
210
+ // the map above. Key-auth declared the same value via keyAuthServiceTier, so the key
211
+ // lane is unchanged.
212
+ chatServiceTier: true,
213
+ // Shared across key and OAuth catalog rows. OAuth subscription has no
214
+ // per-token price, so the 2x claim is scoped to key auth.
215
+ fastTierDescription: "Priority processing; tier pricing applies on key auth only",
216
+ featured: true,
217
+ oauthId: "xai",
218
+ jawcodeBundle: "xai",
219
+ supportsOpenAiWebSearchToolFields: false,
220
+ // Live A/B on 2026-08-20: xAI rejects native custom/custom_tool_call shapes while accepting
221
+ // the otherwise-identical request after the custom tool is lowered to a function.
222
+ supportsResponsesCustomTools: false,
223
+ note: "Log in with your Grok account",
224
+ // Parallel tool calls: officially supported and default-on per docs.x.ai function-calling
225
+ // (verified 260709, devlog/_plan/260709_parallel_tool_calls). Streamed calls arrive whole
226
+ // per chunk, so the buffered parser assembles them losslessly.
227
+ parallelToolCalls: true,
228
+ // Live /v1/models discovery is the authoritative lineup (verified 260709: returns grok-4.5);
229
+ // the static list below is the logged-out fallback seed.
230
+ liveModels: true,
231
+ // 260709 refresh: lineup + metadata from official docs.x.ai (grok-4.5 announced 07-08);
232
+ // grok-composer-2.5-fast kept as account-verified (absent from public docs). Evidence:
233
+ // devlog/model_update/260709_model_refresh/001_xai_lineup.md.
234
+ // 260823: grok-4.20-multi-agent-0309 still returns 400 on Chat Completions, but works
235
+ // on Responses. The server reports this dated id for both it and the floating
236
+ // grok-4.20-multi-agent-beta-latest alias, so expose only the dated deployment id.
237
+ // 260813: grok-4.6 added per docs.x.ai/developers/grok-4-6. Context/vision still match
238
+ // grok-4.5; the reasoning ladder does not — 4.6 adds the documented xhigh rung.
239
+ models: XAI_MODELS,
240
+ // Measured only on grok-4.6 against cli-chat-proxy.grok.com: even an invalid
241
+ // `text.verbosity` value is accepted and low/high/omitted output length is non-monotonic.
242
+ // Apply the resulting opt-out to the whole xAI lineup because `text.verbosity` is an OpenAI
243
+ // Responses parameter absent from xAI's documented API, not because every model was probed.
244
+ // Keep this separate from reasoning-summary support: that bit gates Codex's
245
+ // entire Responses reasoning object, including reasoning.effort.
246
+ modelSupportsVerbosity: Object.fromEntries(XAI_MODELS.map(id => [id, false])),
247
+ // Provider-wide, not merely per-model: `text.verbosity` is an OpenAI Responses parameter
248
+ // absent from xAI's documented API, so a model discovered later has no more support for it
249
+ // than the seeded ones do.
250
+ supportsVerbosity: false,
251
+ defaultModel: "grok-4.5",
252
+ // Grok 4.6/4.5 subscription Responses callers use the native wire with the existing
253
+ // namespace/web-search/replay normalization. Chat remains an explicit modelAdapters
254
+ // opt-in. Multi-agent has no Chat wire and uses Responses under both auth modes.
255
+ // grok-4.6/4.5 are classified OAuth fast-tier models (modelSupportsServiceTier above),
256
+ // so a caller-sent service_tier:"priority" forwards on this lane — the Codex fast-toggle
257
+ // path. Multi-agent keeps its pin: probed 2026-09-13, the gateway downgrades its tier to
258
+ // "default", so forwarding a caller tier would advertise a tier it does not get.
259
+ modelWireDefaults: {
260
+ "grok-4.6": {
261
+ wire: "openai-responses",
262
+ inbound: ["responses"],
263
+ authModes: ["oauth"],
264
+ },
265
+ "grok-4.5": {
266
+ wire: "openai-responses",
267
+ inbound: ["responses"],
268
+ authModes: ["oauth"],
269
+ },
270
+ "grok-4.20-multi-agent-0309": {
271
+ // Even at high effort it emits no reasoning-summary deltas or encrypted replay
272
+ // material. Do not encode that as modelSupportsReasoningSummaries:false: through
273
+ // Codex #1100 that suppresses the entire reasoning object, including the effort
274
+ // that controls this model's agent count. An empty summary pane is harmless.
275
+ // Chat Completions returns 400 for this model, so every inbound uses Responses —
276
+ // `anthropic` included. Omitting it left providerModelWireDefault returning undefined
277
+ // for the Claude Messages lane, so resolveWireProtocolOverride kept xAI's provider-wide
278
+ // openai-chat adapter and sent this model to the wire it 400s on.
279
+ wire: "openai-responses",
280
+ inbound: ["responses", "chat", "anthropic"],
281
+ forwardCallerServiceTier: false,
282
+ },
283
+ },
284
+ // Vision lineup per docs.x.ai model-capabilities/images/understanding: the grok-4.x chat
285
+ // models accept image input (JPEG/PNG, URL or base64). Without this the catalog leaves
286
+ // inputModalities undefined, and deriveComboCatalogModel defaults an undefined member to
287
+ // ["text"] — so any combo containing an xAI target is advertised to Codex as text-only and
288
+ // the app blocks attachments client-side. grok-build-0.1 / grok-composer-2.5-fast stay out
289
+ // (they are already listed in noVisionModels below).
290
+ modelInputModalities: {
291
+ "grok-4.6": ["text", "image"],
292
+ "grok-4.5": ["text", "image"],
293
+ "grok-4.3": ["text", "image"],
294
+ "grok-4.20-multi-agent-0309": ["text", "image"],
295
+ "grok-4.20-0309-reasoning": ["text", "image"],
296
+ "grok-4.20-0309-non-reasoning": ["text", "image"],
297
+ },
298
+ noReasoningModels: ["grok-4.20-0309-non-reasoning", "grok-build-0.1", "grok-composer-2.5-fast"],
299
+ // Replay assistant reasoning_content for grok reasoning models: xAI documents dropped
300
+ // reasoning_content as the top cause of prompt-cache misses on multi-turn conversations
301
+ // (docs.x.ai prompt-caching/multi-turn, verified 2026-07-13 — devlog/_plan/260713_grok_caching).
302
+ // Models that never emit reasoning simply have no thinking parts to replay (no-op).
303
+ preserveReasoningContentModels: ["grok-4.6", "grok-4.5", "grok-4.3", "grok-4.20-0309-reasoning"],
304
+ // grok-4.5 reasoning is always-on with low/medium/high (no off tier, no xhigh).
305
+ // grok-4.6 adds xhigh per docs.x.ai/developers/model-capabilities/text/reasoning;
306
+ // multi-agent accepts the same four wire values to select 4 or 16 collaborators. xAI
307
+ // documents high as the 4.6 default but no multi-agent default, so do not invent one.
308
+ modelReasoningEfforts: {
309
+ "grok-4.6": ["low", "medium", "high", "xhigh"],
310
+ "grok-4.5": ["low", "medium", "high"],
311
+ "grok-4.20-multi-agent-0309": ["low", "medium", "high", "xhigh"],
312
+ },
313
+ modelDefaultReasoningEfforts: { "grok-4.6": "high" },
314
+ modelContextWindows: {
315
+ "grok-4.6": 500_000,
316
+ "grok-4.5": 500_000,
317
+ "grok-4.3": 1_000_000,
318
+ "grok-4.20-multi-agent-0309": 1_000_000,
319
+ "grok-4.20-0309-reasoning": 1_000_000,
320
+ "grok-4.20-0309-non-reasoning": 1_000_000,
321
+ "grok-build-0.1": 256_000,
322
+ },
323
+ noVisionModels: ["grok-build-0.1", "grok-composer-2.5-fast"],
324
+ },
325
+ {
326
+ id: "command-code",
327
+ label: "Command Code - Auth",
328
+ adapter: "command-code",
329
+ baseUrl: "https://api.commandcode.ai",
330
+ authKind: "oauth",
331
+ oauthId: "command-code",
332
+ featured: true,
333
+ note: "Log in with your Command Code account",
334
+ // OAuth needs one initial selection, but the exposed catalog is always discovered from the
335
+ // signed-in account. Do not add a static model list here.
336
+ defaultModel: "deepseek/deepseek-v4-flash",
337
+ liveModels: true,
338
+ modelDiscovery: {
339
+ url: "https://api.commandcode.ai/provider/v1/models",
340
+ maxResponseBytes: 262_144,
341
+ maxModels: 256,
342
+ },
343
+ // These are capability facts from official Command Code model profiles, not seeded models.
344
+ // Unknown/new live models deliberately do not advertise a reasoning picker.
345
+ reasoningEfforts: [],
346
+ modelReasoningEfforts: COMMAND_CODE_MODEL_REASONING_EFFORTS,
347
+ // The DeepSeek vision preview id is preemptive metadata — it is expected to
348
+ // merge into deepseek-v4-flash later.
349
+ modelContextWindows: {
350
+ [`deepseek/${DEEPSEEK_VISION_PREVIEW_MODEL}`]: 1_048_576,
351
+ },
352
+ modelInputModalities: COMMAND_CODE_MODEL_INPUT_MODALITIES,
353
+ defaultMaxOutputTokens: 64_000,
354
+ // The proprietary generate wire has no verified per-request serialization flag.
355
+ parallelToolCalls: false,
356
+ },
357
+ {
358
+ id: "orcarouter-oauth",
359
+ label: "OrcaRouter - Auth",
360
+ adapter: "openai-chat",
361
+ baseUrl: "https://api.orcarouter.ai/v1",
362
+ authKind: "oauth",
363
+ oauthId: "orcarouter-oauth",
364
+ featured: true,
365
+ allowBaseUrlOverride: true,
366
+ defaultModel: "openai/gpt-5.5",
367
+ models: ORCAROUTER_MODELS,
368
+ liveModels: true,
369
+ modelDiscovery: ORCAROUTER_MODEL_DISCOVERY,
370
+ modelReasoningEfforts: ORCAROUTER_MODEL_REASONING_EFFORTS,
371
+ note: "Connect your OrcaRouter account with OAuth 2.0 + PKCE; the issued API key is stored in OpenCodex's existing credential store.",
372
+ },
373
+ {
374
+ id: "anthropic",
375
+ label: "Anthropic Claude",
376
+ adapter: "anthropic",
377
+ baseUrl: "https://api.anthropic.com",
378
+ authKind: "oauth",
379
+ allowBaseUrlOverride: true,
380
+ featured: true,
381
+ oauthId: "anthropic",
382
+ jawcodeBundle: "anthropic",
383
+ note: "Log in with your Claude account",
384
+ models: [...ANTHROPIC_MODELS],
385
+ modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
386
+ modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
387
+ // Codex omits max_output_tokens; without a provider budget the Anthropic adapter
388
+ // falls back to 8192, which truncates long answers with stop_reason=max_tokens.
389
+ defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
390
+ defaultModel: "claude-sonnet-5",
391
+ },
392
+ {
393
+ id: "anthropic-apikey",
394
+ label: "Anthropic (API key)",
395
+ adapter: "anthropic",
396
+ baseUrl: "https://api.anthropic.com",
397
+ authKind: "key",
398
+ featured: true,
399
+ dashboardUrl: "https://console.anthropic.com/settings/keys",
400
+ jawcodeBundle: "anthropic",
401
+ extraMetadataAliases: ["anthropic-key"],
402
+ note: "Direct Anthropic API billing — no Claude subscription",
403
+ models: [...ANTHROPIC_MODELS],
404
+ liveModels: true,
405
+ modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
406
+ modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
407
+ defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
408
+ defaultModel: "claude-sonnet-5",
409
+ },
410
+ {
411
+ id: "kimi",
412
+ label: "Kimi",
413
+ adapter: "openai-chat",
414
+ baseUrl: "https://api.kimi.com/coding/v1",
415
+ authKind: "oauth",
416
+ modelSuffixBracketStrip: true,
417
+ // Kimi Code Plan documents a stable session/task prompt_cache_key as required to improve
418
+ // cache hit rates.
419
+ // The chat adapter only forwards a key already on the internal request (Codex's session key,
420
+ // or the one the Claude /v1/messages inbound derives); the adapter itself never invents one.
421
+ // Evidence: https://platform.kimi.com/docs/api/chat
422
+ promptCacheKey: true,
423
+ featured: true,
424
+ oauthId: "kimi",
425
+ jawcodeBundle: "moonshot",
426
+ note: "Log in with your Kimi account",
427
+ models: KIMI_CODING_MODELS,
428
+ defaultModel: "kimi-k2.7-code",
429
+ modelContextWindows: KIMI_CODING_MODEL_CONTEXT_WINDOWS,
430
+ modelInputModalities: KIMI_CODING_MODEL_INPUT_MODALITIES,
431
+ // K3 accepts low/high/max; Codex aliases are normalized by the model-scoped wire map.
432
+ noReasoningModels: KIMI_CODING_NO_REASONING_MODELS,
433
+ modelReasoningEfforts: KIMI_CODING_REASONING_EFFORTS,
434
+ modelDefaultReasoningEfforts: KIMI_CODING_DEFAULT_REASONING_EFFORTS,
435
+ modelReasoningEffortMap: KIMI_CODING_REASONING_EFFORT_MAPS,
436
+ noTemperatureModels: KIMI_LOCKED_PARAMETER_MODELS,
437
+ noTopPModels: KIMI_LOCKED_PARAMETER_MODELS,
438
+ noPenaltyModels: KIMI_LOCKED_PARAMETER_MODELS,
439
+ autoToolChoiceOnlyModels: KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS,
440
+ preserveReasoningContentModels: KIMI_THINKING_MODELS,
441
+ },
442
+ {
443
+ id: "kiro",
444
+ label: "Kiro (AWS CodeWhisperer)",
445
+ adapter: "kiro",
446
+ baseUrl: "https://runtime.us-east-1.kiro.dev",
447
+ authKind: "oauth",
448
+ oauthId: "kiro",
449
+ note: "Import-first: reuses your installed and signed-in Kiro CLI session (requires `kiro-cli login`). Add account logs `kiro-cli` out, switches it through a fresh browser login, stores the account by profile ARN, and restores the previous CLI session on cancellation or failure. Experimental third-party harness — see Kiro ToS.",
450
+ models: KIRO_MODELS,
451
+ defaultModel: "kiro-auto",
452
+ // Kiro speaks CodeWhisperer wire, not OpenAI-style GET /models. Keep the static
453
+ // catalog authoritative so a spurious 2xx from runtime.../models cannot drop seeded ids
454
+ // (e.g. newly listed GPT-5.6 tiers) via live-discovery reconciliation.
455
+ liveModels: false,
456
+ // Per-model context metadata is maintained next to the Kiro model list.
457
+ modelContextWindows: KIRO_MODEL_CONTEXT_WINDOWS,
458
+ modelReasoningEfforts: KIRO_MODEL_REASONING_EFFORTS,
459
+ modelSupportsVerbosity: Object.fromEntries(KIRO_MODELS.map(id => [id, false])),
460
+ },
461
+ {
462
+ // Nous Portal — Nous Research subscription gateway (same backend Hermes Agent
463
+ // uses). OAuth is a device grant (src/oauth/nous.ts): the access token IS the
464
+ // per-request inference JWT (scope inference:invoke), refresh tokens are
465
+ // single-use and rotated on every refresh. Catalog is a mix of paid models
466
+ // (billed against the Portal subscription) and `:free` slugs (e.g.
467
+ // tencent/hy3:free, stepfun/step-3.7-flash:free, inclusionai/ling-3.0-flash:free);
468
+ // free-tier gating is decided live by the Portal per account, so discovery
469
+ // from the signed-in account is authoritative; the static seed below is the
470
+ // logged-out fallback and only lists free models verified on a real account
471
+ // (2026-08-10): the Portal free list is authoritative and currently has
472
+ // exactly 4 :free models: tencent/hy3:free, poolside/laguna-s-2.1:free,
473
+ // stepfun/step-3.7-flash:free, poolside/laguna-xs-2.1:free.
474
+ // inclusionai/ling-3.0-flash:free was removed from the Portal free list
475
+ // (404 on the inference API since 2026-08-07) and must not be seeded.
476
+ id: "nous",
477
+ label: "Nous Portal",
478
+ adapter: "openai-chat",
479
+ baseUrl: "https://inference-api.nousresearch.com/v1",
480
+ authKind: "oauth",
481
+ oauthId: "nous",
482
+ featured: true,
483
+ // Mixed free + paid provider: the free tier is per-model (the `:free`
484
+ // slugs), not a property of the whole provider, so freeTier stays false to
485
+ // avoid implying every model is free.
486
+ freeTier: false,
487
+ dashboardUrl: "https://portal.nousresearch.com",
488
+ defaultModel: "tencent/hy3:free",
489
+ liveModels: true,
490
+ models: ["tencent/hy3:free", "poolside/laguna-s-2.1:free", "stepfun/step-3.7-flash:free", "poolside/laguna-xs-2.1:free"],
491
+ modelDiscovery: {
492
+ // Resolves against effectiveBaseUrl (registry baseUrl .../v1) to the same
493
+ // canonical endpoint https://inference-api.nousresearch.com/v1/models.
494
+ // Nous returns a mixed paid/free catalog whose JSON can exceed 256 KiB;
495
+ // keep the provider-specific limit below the process-wide 4 MiB ceiling.
496
+ path: "models",
497
+ maxResponseBytes: 1_048_576,
498
+ maxModels: 512,
499
+ },
500
+ note: "Nous Research subscription gateway. OAuth device login with your own Portal account; mixed paid + :free models discovered live (fallback seed 2026-08-10: tencent/hy3:free, poolside/laguna-s-2.1:free, stepfun/step-3.7-flash:free, poolside/laguna-xs-2.1:free).",
501
+ },
502
+ {
503
+ id: "openai-apikey",
504
+ label: "OpenAI API",
505
+ adapter: "openai-responses",
506
+ baseUrl: "https://api.openai.com/v1",
507
+ authKind: "key",
508
+ supportsServiceTier: true,
509
+ featured: true,
510
+ dashboardUrl: "https://platform.openai.com/api-keys",
511
+ defaultModel: "gpt-5.5",
512
+ models: ["gpt-5.5", ...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS, ...OPENAI_DAYBREAK_MODELS, "gpt-6-astra"],
513
+ liveModels: true,
514
+ modelContextWindows: { ...OPENAI_API_GPT56_CONTEXT_WINDOWS, ...OPENAI_DAYBREAK_CONTEXT_WINDOWS, "gpt-6-astra": 1_050_000 },
515
+ modelMaxInputTokens: { ...OPENAI_API_GPT56_MAX_INPUT_TOKENS, ...OPENAI_DAYBREAK_MAX_INPUT_TOKENS, "gpt-6-astra": 922_000 },
516
+ modelMaxOutputTokens: { "gpt-6-astra": 128_000 },
517
+ modelInputModalities: Object.fromEntries(
518
+ ["gpt-5.5", ...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS, ...OPENAI_DAYBREAK_MODELS, "gpt-6-astra"]
519
+ .map(id => [id, ["text", "image"]]),
520
+ ),
521
+ modelReasoningEfforts: {
522
+ ...Object.fromEntries(
523
+ [...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS].map(id => [id, OPENAI_API_GPT56_REASONING_EFFORTS]),
524
+ ),
525
+ ...OPENAI_DAYBREAK_REASONING_EFFORTS,
526
+ "gpt-6-astra": ["low", "medium", "high", "xhigh", "max"],
527
+ },
528
+ virtualModels: OPENAI_API_GPT56_VIRTUAL_MODELS,
529
+ },
530
+ /* [Decision Log]
531
+ - 목적과 의도: Reach Meta's Muse Spark models directly on Meta's own Model API, instead of only through the Command Code and OpenCode Zen resellers already in this registry.
532
+ - 기존 구현 및 제약 조건: Meta publishes both POST /v1/responses and POST /v1/chat/completions at https://api.meta.ai/v1, and no API key was issued for this change — every value here comes from the published spec (devlog/_plan/260903_muse_spark_plan_oauth/001).
533
+ - 검토한 주요 대안: register as openai-chat; use provider id "meta"; enable live discovery; wire the Muse Code subscription credential as OAuth.
534
+ - 선택한 방식: an openai-responses key provider under the id "meta-model", with a static two-model roster and no OAuth.
535
+ - 다른 대안 대신 이 방식을 선택한 이유: Meta calls Responses "the recommended default for new work ... OpenAI-compatible and exposes the full feature set", carrying reasoning replay and native input_image that Chat would forfeit. The id is "meta-model" because "meta" would capture the LIVE Command Code selector meta/muse-spark-1.3 at router.ts's provider-prefix branch, and would derive META_API_KEY — the Muse Code CLI's variable, not this API's MODEL_API_KEY.
536
+ - 장점, 단점 및 영향: users reach Muse Spark without a reseller; discovery stays off until an authenticated /v1/models payload is actually observed, so an unseen roster (Meta also serves image and voice families here) cannot leak into the picker.
537
+ */
538
+ {
539
+ id: "meta-model",
540
+ label: "Meta Model API",
541
+ adapter: "openai-responses",
542
+ baseUrl: "https://api.meta.ai/v1",
543
+ authKind: "key",
544
+ dashboardUrl: "https://dev.meta.ai/docs/authentication",
545
+ defaultModel: "muse-spark-1.3",
546
+ models: META_MUSE_MODELS,
547
+ // Static roster: no authenticated /v1/models payload was ever observed (the only
548
+ // contact was an unauthenticated GET returning 401 invalid_api_key), and Meta serves
549
+ // non-agent families on this same base URL. Turning discovery on would publish an
550
+ // unseen roster into the picker.
551
+ liveModels: false,
552
+ // A user may already own a custom provider named "meta-model" pointing elsewhere;
553
+ // without this, registry transport canonicalization would retarget it and send their
554
+ // saved key to Meta.
555
+ preserveCustomDestination: true,
556
+ modelContextWindows: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_CONTEXT_WINDOW])),
557
+ // text+image only. Meta also documents video, audio (degraded on 1.3), and PDF, but
558
+ // the catalog modality enum is text/image and over-advertising poisons the exported
559
+ // client config (see tests/codex-integration/catalog-input-modality-enum.test.ts).
560
+ modelInputModalities: Object.fromEntries(META_MUSE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
561
+ modelReasoningEfforts: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORTS])),
562
+ modelReasoningEffortMap: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORT_MAP])),
563
+ // No defaultMaxOutputTokens: Meta publishes none. The only number in its docs
564
+ // (131072) appears inside a third-party config sample, and the protocol pages call
565
+ // the real limit "model-dependent".
566
+ // Meta names its variable MODEL_API_KEY, but the env var opencodex reads is derived
567
+ // from the provider id (META_MODEL_API_KEY). Saying only Meta's name would send a
568
+ // user to export a variable this proxy never reads.
569
+ note: "Pay-as-you-go Meta Model API. Get a key at https://dev.meta.ai (Meta calls it MODEL_API_KEY; export it here as META_MODEL_API_KEY) — a Meta developer account needs a payment method before it can serve requests, and every call is metered per token. A Muse Code subscription does NOT work here: Meta scopes that credential to the Muse Code CLI and bills any other key pay-as-you-go (dev.meta.ai/docs/muse-code/subscriptions). The Contributor tier (muse-spark-1.3-contributor) is cheap because Meta trains on your prompts — about 92% off input, 95% off output, 99% off cached input; do not send confidential material through it. Muse Spark is also reachable through resellers: command-code carries both tiers, opencode-go serves only muse-spark-1.3-contributor.",
570
+ },
571
+ /* [Decision Log]
572
+ - 목적과 의도: Let an operator who already signed the Muse Code CLI in reach Muse Spark with that credential, instead of provisioning a second key.
573
+ - 기존 구현 및 제약 조건: The CLI stores a pointer at ~/.config/muse/auth.json and the secret in the macOS Keychain (ai.meta.dev.credentials/meta). Measured: the OAuth access_token 401s on /v1/models while the sibling api_key returns 200, so the usable artifact is a static key, not a refreshable token.
574
+ - 검토한 주요 대안: spawn `muse login` and poll; reimplement Meta's device grant; treat it as a second key preset; ship nothing.
575
+ - 선택한 방식: an OAuth provider that imports the existing credential on macOS and accepts a pasted key elsewhere, validates either once, and never spawns or reimplements anything.
576
+ - 다른 대안 대신 이 방식을 선택한 이유: `muse login` has no non-interactive mode, so a spawned child could outlive cancellation, and polling for the pointer file is satisfied instantly by the one already on disk — reimporting the OLD account on a force-login. Reimplementing the grant would mean guessing a client id the vendor does not publish.
577
+ - 장점, 단점 및 영향: no new credential to provision, and the id is distinct from meta-model so neither pool contaminates the other. Meta scopes this credential to its own CLI, so the provider carries a HIGH_RISK ToS warning, a CLI-side warning before any read, and a note that says plainly what is unsupported.
578
+ */
579
+ {
580
+ id: "meta-muse",
581
+ label: "Meta Muse Code (CLI credential)",
582
+ adapter: "openai-responses",
583
+ baseUrl: "https://api.meta.ai/v1",
584
+ // Meta own client sends this on every Muse Code call. We never have, so a future
585
+ // server-side requirement would break every Muse request with no local signal.
586
+ // Declared here rather than in a transport hook so it also covers model discovery
587
+ // (src/oauth/index.ts:1176) and still yields to a user-set header
588
+ // (mergeRegistryStaticHeaders, src/providers/registry.ts:3494).
589
+ staticHeaders: { "x-api-version": "1.0.0" },
590
+ authKind: "oauth",
591
+ oauthId: "meta-muse",
592
+ dashboardUrl: "https://dev.meta.ai",
593
+ defaultModel: "muse-spark-1.3",
594
+ models: META_MUSE_MODELS,
595
+ // Same reason as meta-model: the authenticated roster carries muse-image-1.0 and
596
+ // muse-voice-transcribe-1.0, which this Responses-agent provider cannot drive.
597
+ liveModels: false,
598
+ modelContextWindows: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_CONTEXT_WINDOW])),
599
+ modelInputModalities: Object.fromEntries(META_MUSE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
600
+ modelReasoningEfforts: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORTS])),
601
+ modelReasoningEffortMap: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORT_MAP])),
602
+ note: "Signs in to Meta with a browser device code on any platform, then mints the Muse Code subscription key. That grant is reimplemented from the one the Muse Code CLI performs and has NOT been exercised against Meta from OpenCodex, so treat the first login as unverified. If the Muse Code CLI is already signed in on macOS, the existing key is imported instead of starting a new grant. A pasted key from https://dev.meta.ai still works as a fallback when a device login cannot complete, and faces the same format check and live validation. A device login authenticates as Meta own Muse Code client, which is a stronger claim than reusing a key the CLI already minted. Meta scopes that credential to the Muse Code CLI, so this is an UNSUPPORTED use: Meta does not authorize subscription coverage outside its own CLI, how these calls settle is not observable from the API, and you should treat every call as billable against your account. The key, imported or pasted, is copied into OpenCodex's auth store. For an account signed in with the device login, OpenCodex refreshes Meta's subscription windows on demand from the same key endpoint the login uses, at most once every five minutes. For an imported or pasted key there is no endpoint to query them on demand, so OpenCodex reads them from streaming responses and shows the last observed value with its age; refreshing one then requires another streaming turn, and translated (non-passthrough) turns report none. Rate limits apply per team, not per key. For a supported path use the meta-model provider with your own key (export it as META_MODEL_API_KEY).",
603
+ },
604
+ {
605
+ id: "umans",
606
+ label: "Umans AI Coding Plan",
607
+ adapter: "anthropic",
608
+ baseUrl: "https://api.code.umans.ai",
609
+ authKind: "key",
610
+ featured: true,
611
+ dashboardUrl: "https://app.umans.ai/billing",
612
+ defaultModel: "umans-coder",
613
+ models: UMANS_MODELS,
614
+ modelContextWindows: UMANS_MODEL_CONTEXT_WINDOWS,
615
+ modelInputModalities: UMANS_MODEL_INPUT_MODALITIES,
616
+ note: "Coding plan via Anthropic Messages",
617
+ modelReasoningEfforts: {
618
+ "umans-coder": UMANS_REASONING_EFFORTS,
619
+ "umans-kimi-k2.7": UMANS_REASONING_EFFORTS,
620
+ "umans-flash": UMANS_REASONING_EFFORTS,
621
+ "umans-glm-5.3": UMANS_GLM_53_REASONING_EFFORTS,
622
+ "umans-glm-5.3-flash": UMANS_GLM_53_REASONING_EFFORTS,
623
+ "umans-glm-5.2": UMANS_GLM_REASONING_EFFORTS,
624
+ "umans-glm-5.1": UMANS_GLM_REASONING_EFFORTS,
625
+ "umans-qwen3.6-35b-a3b": UMANS_REASONING_EFFORTS,
626
+ },
627
+ noVisionModels: UMANS_TEXT_ONLY_MODELS,
628
+ escapeBuiltinToolNames: true,
629
+ },
630
+ {
631
+ id: "opencode-go", label: "opencode go", adapter: "openai-chat", baseUrl: "https://opencode.ai/zen/go/v1",
632
+ authKind: "key", featured: true, dashboardUrl: "https://opencode.ai/auth", defaultModel: "kimi-k2.7-code",
633
+ jawcodeBundle: "opencode-go", note: "GLM, DeepSeek, Kimi, Qwen, MiMo…",
634
+ // Zen Go can close a Chat stream after a fully assembled function call without sending
635
+ // finish_reason or [DONE] (#2260). The adapter still rejects incomplete argument JSON.
636
+ openaiChatEofTolerance: true,
637
+ // Go rejects reasoning.encrypted_content with previous_response_id (#3838).
638
+ // Use explicit replay history and the existing stateless Responses policy.
639
+ statelessResponses: true,
640
+ /* [Decision Log]
641
+ - 목적과 의도: Route the exact models OpenCode Go documents on the Responses endpoint — GPT 5.6 Luna, Grok 4.6, and Muse Spark Contributor (#2617).
642
+ - 기존 구현 및 제약 조건: The provider is mixed-wire but its provider-wide `openai-chat` adapter sent Luna to `/chat/completions`; explicit user `modelAdapters` entries must remain authoritative.
643
+ - 검토한 주요 대안: Change the whole provider to Responses; infer the wire from model-family names; add one registry-only exact-model default.
644
+ - 선택한 방식: Declare only the named models as `openai-responses` through the existing registry default mechanism; the map stays an exact-model allowlist rather than a family or provider-wide rule.
645
+ - 다른 대안 대신 이 방식을 선택한 이유: OpenCode Go documents sibling models on Chat or Anthropic endpoints, and an exact registry default preserves both those routes and explicit opt-out precedence.
646
+ - 장점, 단점 및 영향: Each listed model reaches `/responses` from every inbound surface without changing siblings; a future upstream endpoint change requires an evidence-backed registry update.
647
+ */
648
+ modelWireDefaults: {
649
+ "gpt-5.6-luna": "openai-responses",
650
+ "grok-4.6": "openai-responses",
651
+ "muse-spark-1.3-contributor": "openai-responses",
652
+ "muse-spark-1.2-contributor": "openai-responses",
653
+ },
654
+ modelContextWindows: {
655
+ "kimi-k3": KIMI_K3_STANDARD_CONTEXT_WINDOW,
656
+ // Zen Go discovers only the gateway id, so carry DeepSeek's official 1M V4.1
657
+ // window here or Codex falls back to its conservative 128k routed-model default.
658
+ "deepseek-v4.1-flash": 1_048_576,
659
+ // The DeepSeek vision preview id is metadata-only here: the Go roster is
660
+ // discovered live, so it applies the moment the gateway serves the id.
661
+ [DEEPSEEK_VISION_PREVIEW_MODEL]: 1_048_576,
662
+ // Muse Spark Contributor serves a 1,048,576-token (1M) context window over
663
+ // /responses on Zen Go, matching its 1.1 sibling (Meta developer docs, verified 2026-08-28).
664
+ // Without this declaration the catalog falls back to 128k, capping real usable context.
665
+ // 1.3 ships the same window as 1.2 and is served from the same Zen Go roster.
666
+ "muse-spark-1.3-contributor": 1_048_576,
667
+ "muse-spark-1.2-contributor": 1_048_576,
668
+ },
669
+ modelInputModalities: {
670
+ "kimi-k3": ["text", "image"],
671
+ // glm-5.3-flash is a native VLM (docs.z.ai/guides/vlm/glm-5.3-flash). It is
672
+ // deliberately absent from this preset's noVisionModels, which is the
673
+ // correct NEGATIVE half, but with no positive modelInputModalities entry
674
+ // configuredInputModalities returns undefined and the catalog falls through
675
+ // to the ["text"] floor. The same model is already declared ["text","image"]
676
+ // on the zai and zhipu-bigmodel-coding presets, so the registry described
677
+ // one model two ways (#4505).
678
+ "glm-5.3-flash": ["text", "image"],
679
+ // Experimental DeepSeek vision preview — expected to merge into deepseek-v4-flash later.
680
+ [DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
681
+ // This route is text-only upstream — it is already listed in this preset's
682
+ // noVisionModels, which routes images through the proxy's vision sidecar and
683
+ // makes the catalog advertise image input on its behalf. The positive
684
+ // text-only declaration is what reaches an EXISTING install: derive.ts fills
685
+ // noVisionModels all-or-nothing, so a config persisted before this id joined
686
+ // the list keeps a stale list, the sidecar predicate never matches, the row
687
+ // carries no modality at all, and any combo containing it collapses to
688
+ // ["text"] (#4505). modelInputModalities IS per-key filled, so this
689
+ // declaration lands on old configs. It states the route's real upstream
690
+ // capability and keeps the sidecar explicitly distinct from native vision.
691
+ "deepseek-v4.1-flash": ["text"],
692
+ // Muse Spark Contributor is natively multimodal on Zen Go: it accepts input_image
693
+ // parts over /responses (probed 2026-08-26). Without this declaration the catalog
694
+ // advertises it text-only and the Codex app blocks image attachments client-side with
695
+ // "This model does not support image inputs" before the request ever reaches the proxy.
696
+ // 1.3 is the same-shaped successor and Command Code documents it as multimodal.
697
+ "muse-spark-1.3-contributor": ["text", "image"],
698
+ "muse-spark-1.2-contributor": ["text", "image"],
699
+ },
700
+ modelReasoningEfforts: {
701
+ "gpt-5.6-luna": OPENAI_API_GPT56_REASONING_EFFORTS,
702
+ "grok-4.6": ["low", "medium", "high", "xhigh"],
703
+ "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
704
+ "glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
705
+ "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
706
+ "qwen3.8-max": QWEN38_REASONING_EFFORTS,
707
+ "kimi-k3": KIMI_CODING_K3_REASONING_EFFORTS,
708
+ "kimi-k2.7-code": [],
709
+ "kimi-k2.7-code-highspeed": [],
710
+ ...Object.fromEntries(OPENCODE_GO_THINKING_TOGGLE_MODELS.map(id => [id, THINKING_TOGGLE_EFFORTS])),
711
+ ...Object.fromEntries(OPENCODE_GO_THINKING_BUDGET_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
712
+ ...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, deepseekThinkingEffortsFor(id)])),
713
+ },
714
+ modelDefaultReasoningEfforts: { "grok-4.6": "high", "kimi-k3": "max" },
715
+ // glm-5.2 uses identity labels now that `max` is a native Codex level (no alias map);
716
+ // the thinking-toggle map is a REAL wire alias (effort -> enabled/disabled) and stays.
717
+ modelReasoningEffortMap: {
718
+ "kimi-k3": KIMI_CODING_K3_REASONING_EFFORT_MAP,
719
+ ...Object.fromEntries(OPENCODE_GO_THINKING_TOGGLE_MODELS.map(id => [id, THINKING_TOGGLE_MAP])),
720
+ ...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, deepseekReasoningMapFor(id)])),
721
+ },
722
+ modelSupportsReasoningSummaries: {
723
+ "glm-5.3": true,
724
+ "glm-5.3-flash": true,
725
+ "glm-5.2": true,
726
+ "glm-5.1": true,
727
+ "glm-5": true,
728
+ ...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, true])),
729
+ },
730
+ thinkingToggleModels: OPENCODE_GO_THINKING_TOGGLE_MODELS,
731
+ /*
732
+ * The Go-specific list, not the shared one. The shared `THINKING_BUDGET_MODELS` also
733
+ * carries Neuralwatt-only ids (`qwen3.5-397b`, `qwen3.6-35b`) that this preset never
734
+ * gives a ladder to, so a live roster serving one of them armed the thinking-budget
735
+ * wire path with nothing to advertise: the catalog showed no effort control while the
736
+ * adapter still translated effort into `thinking_budget`.
737
+ */
738
+ thinkingBudgetModels: OPENCODE_GO_THINKING_BUDGET_MODELS,
739
+ noReasoningModels: ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
740
+ // Text-only Zen Go models (jawcode metadata) — the vision sidecar describes images for
741
+ // every model listed here (and the catalog advertises image input on their behalf).
742
+ // Kimi K2.7 Code accepts text+image+video: do NOT list it here.
743
+ noVisionModels: [
744
+ "glm-5.3", "glm-5.2", "glm-5", "glm-5.1",
745
+ "deepseek-v4.1-flash", "deepseek-v4-flash",
746
+ "mimo-v2-pro", "mimo-v2.5-pro",
747
+ "minimax-m2.5", "minimax-m2.7",
748
+ "qwen3.7-max",
749
+ ],
750
+ noTemperatureModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
751
+ noTopPModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
752
+ noPenaltyModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
753
+ autoToolChoiceOnlyModels: ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
754
+ // Issue #78: DeepSeek V4 thinking mode requires reasoning_content replay on tool-call turns.
755
+ preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed", ...DEEPSEEK_GATEWAY_THINKING_MODELS],
756
+ /*
757
+ * Issues #1338 / #1415: this gateway answers a `response_format` of type
758
+ * `json_schema` with HTTP 400 `This response_format type is unavailable now`
759
+ * (quoted from the upstream body as `Error from provider (Console Go)`), which
760
+ * breaks every Codex auto-review turn on a DeepSeek route. #1424 shipped the
761
+ * operator-side opt-out; operators have been applying it by hand ever since.
762
+ * The reported rejection is type-specific, so this narrower list downgrades the
763
+ * request to `json_object` instead of claiming the whole field is unavailable.
764
+ */
765
+ noJsonSchemaModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS],
766
+ },
767
+ {
768
+ id: "neuralwatt",
769
+ label: "Neuralwatt Cloud",
770
+ adapter: "openai-chat",
771
+ baseUrl: "https://api.neuralwatt.com/v1",
772
+ authKind: "key",
773
+ dashboardUrl: "https://portal.neuralwatt.com",
774
+ defaultModel: "glm-5.3",
775
+ // 2026-07-10 live /v1/models: K2.5 rows were removed and GLM-5.2 short variants added.
776
+ // 260814: the glm-5.3 quartet is speculative; live discovery is authoritative and drops
777
+ // any id Neuralwatt has not published yet.
778
+ // Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md and https://api.neuralwatt.com/v1/models.
779
+ models: [
780
+ "glm-5.3", "glm-5.3-fast", "glm-5.3-short", "glm-5.3-short-fast",
781
+ "glm-5.3-flash",
782
+ "glm-5.2", "glm-5.2-fast", "glm-5.2-short", "glm-5.2-short-fast",
783
+ "kimi-k2.6", "kimi-k2.6-fast",
784
+ "kimi-k2.7-code",
785
+ "qwen3.5-397b", "qwen3.5-397b-fast", "qwen3.6-35b", "qwen3.6-35b-fast",
786
+ ],
787
+ // Neuralwatt's /v1/models metadata is authoritative; these static hints are the offline fallback.
788
+ modelReasoningEfforts: {
789
+ "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
790
+ "glm-5.3-fast": [],
791
+ "glm-5.3-short": ZAI_GLM_53_REASONING_EFFORTS,
792
+ "glm-5.3-short-fast": [],
793
+ // No `-fast`/`-short` variants are asserted for the flash tier: those suffixes
794
+ // encode routing Neuralwatt documents per model, and this seed has no source for them.
795
+ "glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
796
+ "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
797
+ "glm-5.2-fast": [],
798
+ "glm-5.2-short": ZAI_GLM_52_REASONING_EFFORTS,
799
+ "glm-5.2-short-fast": [],
800
+ "kimi-k2.6": [],
801
+ "kimi-k2.6-fast": [],
802
+ "kimi-k2.7-code": [],
803
+ // Qwen3.x uses thinking_budget, NOT graded reasoning_effort; the adapter maps the five
804
+ // Codex picker levels onto budget fractions.
805
+ "qwen3.5-397b": THINKING_BUDGET_EFFORTS,
806
+ "qwen3.5-397b-fast": [],
807
+ "qwen3.6-35b": THINKING_BUDGET_EFFORTS,
808
+ "qwen3.6-35b-fast": [],
809
+ },
810
+ thinkingBudgetModels: THINKING_BUDGET_MODELS,
811
+ noReasoningModels: ["glm-5.3-fast", "glm-5.3-short-fast", "glm-5.2-fast", "glm-5.2-short-fast", "kimi-k2.6-fast", "qwen3.5-397b-fast", "qwen3.6-35b-fast"],
812
+ noVisionModels: ["glm-5.3", "glm-5.3-fast", "glm-5.3-short", "glm-5.3-short-fast", "glm-5.2", "glm-5.2-fast", "glm-5.2-short", "glm-5.2-short-fast", "qwen3.5-397b", "qwen3.5-397b-fast"],
813
+ noTemperatureModels: ["kimi-k2.7-code"],
814
+ noTopPModels: ["kimi-k2.7-code"],
815
+ noPenaltyModels: ["kimi-k2.7-code"],
816
+ autoToolChoiceOnlyModels: ["kimi-k2.7-code"],
817
+ preserveReasoningContentModels: NEURALWATT_REASONING_HISTORY_MODELS,
818
+ },
819
+ {
820
+ id: "openrouter",
821
+ label: "OpenRouter",
822
+ adapter: "openai-chat",
823
+ baseUrl: "https://openrouter.ai/api/v1",
824
+ authKind: "key",
825
+ featured: true,
826
+ dashboardUrl: "https://openrouter.ai/keys",
827
+ jawcodeBundle: "openrouter",
828
+ models: ["anthropic/claude-sonnet-5", ...OPENROUTER_GPT56_MODELS],
829
+ modelContextWindows: {
830
+ "anthropic/claude-sonnet-5": 1_000_000,
831
+ ...OPENROUTER_GPT56_CONTEXT_WINDOWS,
832
+ },
833
+ // OpenRouter documents priority support for OpenAI endpoints, but not Anthropic. Keep the
834
+ // provider unclassified and opt in only the exact OpenAI-backed slugs we ship. These facts
835
+ // belong only to the canonical destination; a same-named custom gateway is unknown to us.
836
+ modelServiceTierCapabilityBaseUrlGuard: isCanonicalOpenRouterTarget,
837
+ modelSupportsServiceTier: {
838
+ "openai/gpt-5.6-sol": true,
839
+ "openai/gpt-5.6-terra": true,
840
+ "openai/gpt-5.6-luna": true,
841
+ },
842
+ // Deliberately no OpenRouter route pin: it bills the endpoint actually used and reports the
843
+ // actual service_tier. B0 confirmation therefore owns downgrade safety. Forcing `only` plus
844
+ // `allow_fallbacks:false` would turn a graceful priority-capacity fallback into a hard failure.
845
+ },
846
+ {
847
+ // Primary sources checked 2026-08-02:
848
+ // - docs.cline.bot/getting-started/clinepass publishes this exact catalog and explicitly
849
+ // authorizes using the full slugs through Cline's external API.
850
+ // - docs.cline.bot/api/chat-completions and /api/errors define the endpoint, reasoning delta,
851
+ // and choice-scoped mid-stream error contract.
852
+ // - Cline's official catalog source resolves per-model capabilities through OpenRouter data;
853
+ // the static context/modality snapshot below was cross-checked against that catalog.
854
+ // - cline.bot/tos identifies Cline Bot Inc. as the operator. Maintenance owner: @lidge-jun.
855
+ id: "cline-pass",
856
+ label: "ClinePass",
857
+ adapter: "openai-chat",
858
+ baseUrl: "https://api.cline.bot/api/v1",
859
+ authKind: "key",
860
+ dashboardUrl: "https://app.cline.bot",
861
+ defaultModel: "cline-pass/kimi-k3",
862
+ models: CLINE_PASS_MODELS,
863
+ modelContextWindows: CLINE_PASS_MODEL_CONTEXT_WINDOWS,
864
+ modelInputModalities: CLINE_PASS_MODEL_INPUT_MODALITIES,
865
+ noVisionModels: CLINE_PASS_TEXT_ONLY_MODELS,
866
+ // Live-probed 2026-08-13 across every static ClinePass model: the gateway accepts and
867
+ // validates low/medium/high/xhigh/max, and rejects an invalid sentinel. Preserve the
868
+ // caller's requested tier and let ClinePass own any backend-specific normalization.
869
+ reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
870
+ reasoningWireFormat: "gateway-object",
871
+ preserveCustomDestination: true,
872
+ note: "ClinePass subscription API. Uses a Cline API key and the full cline-pass/<model> upstream slug; quota is shared across the account's rolling 5-hour, weekly, and monthly limits.",
873
+ },
874
+ // Cline API (usage-billing): OpenAI-compatible Chat Completions. Model IDs follow the
875
+ // OpenRouter-style `provider/model` convention. Live /models discovery is key-gated (401
876
+ // without auth), so the static seed is the cold-start fallback. Evidence: docs.cline.bot/api/*.
877
+ {
878
+ id: "cline",
879
+ label: "Cline",
880
+ adapter: "openai-chat",
881
+ baseUrl: "https://api.cline.bot/api/v1",
882
+ authKind: "key",
883
+ dashboardUrl: "https://app.cline.bot",
884
+ liveModels: true,
885
+ defaultModel: "anthropic/claude-sonnet-4-6",
886
+ models: [
887
+ "anthropic/claude-sonnet-4-6",
888
+ "openai/gpt-4o",
889
+ "google/gemini-2.5-pro",
890
+ "deepseek/deepseek-chat",
891
+ "minimax/minimax-m2.5",
892
+ ],
893
+ preserveCustomDestination: true,
894
+ note: "Cline usage-billing API: one key, 100+ models, OpenRouter-style ids. Promotional free models are IDE/CLI-only per Cline docs; minimax/minimax-m2.5 is the documented API free experimentation model.",
895
+ },
896
+ {
897
+ // OrcaRouter: OpenAI-compatible adaptive router (api.orcarouter.ai). The public live
898
+ // catalog is authoritative; model ids and input modalities are never maintained here.
899
+ id: "orcarouter", label: "OrcaRouter - API", adapter: "openai-chat", baseUrl: "https://api.orcarouter.ai/v1",
900
+ authKind: "key", dashboardUrl: "https://www.orcarouter.ai/console",
901
+ // The catalog is public, so a successful /models probe cannot validate a submitted key.
902
+ apiKeyValidation: "unknown",
903
+ // Standard sponsor under SPONSORS.md (agreement signed 2026-09-07). Pins the row in the
904
+ // picker and adds the chip; nothing about routing or defaults changes.
905
+ sponsor: { tier: "standard", url: "https://www.orcarouter.ai/?utm_source=opencodex&utm_medium=readme" },
906
+ defaultModel: "openai/gpt-5.5",
907
+ models: ORCAROUTER_MODELS,
908
+ liveModels: true,
909
+ modelDiscovery: ORCAROUTER_MODEL_DISCOVERY,
910
+ // Catalog discovery owns WHICH models exist. These entries only retain verified
911
+ // request-shaping facts that the upstream catalog does not currently publish.
912
+ modelReasoningEfforts: ORCAROUTER_MODEL_REASONING_EFFORTS,
913
+ note: "OpenAI-compatible adaptive router. Models and multimodal capabilities are discovered live from the public chat catalog. Use the OrcaRouter account entry for PKCE login.",
914
+ },
915
+ {
916
+ // PackyCode: API relay (packyapi.com) for Claude Code, Codex, Gemini and more. Codex traffic
917
+ // uses the OpenAI-compatible host from their Codex/Kimi Code guides (docs.packyapi.com):
918
+ // https://cf.api.fan/v1 — GET /v1/models answers 401 without a key, so the host is live and
919
+ // discovery narrows to what the key's token group allows. Model ids are bare OpenAI-style
920
+ // ids (the Codex token group lists gpt-5.5 / gpt-5.1-codex).
921
+ // Standard sponsor under SPONSORS.md; the dashboardUrl carries their affiliate code.
922
+ id: "packycode", label: "PackyCode", adapter: "openai-chat", baseUrl: "https://cf.api.fan/v1",
923
+ authKind: "key", dashboardUrl: "https://www.packyapi.com/register?aff=k5KT",
924
+ sponsor: { tier: "standard", url: "https://www.packyapi.com/register?aff=k5KT" },
925
+ defaultModel: "gpt-5.5",
926
+ models: ["gpt-5.5", "gpt-5.1-codex"],
927
+ liveModels: true,
928
+ // New key preset: opt into collision preservation so a row named `packycode` that a user
929
+ // points at a different PackyCode host keeps its own destination instead of being pulled
930
+ // back onto the Codex endpoint below.
931
+ preserveCustomDestination: true,
932
+ note: "API relay for Claude Code, Codex, Gemini and more. Create a Codex-group token at packyapi.com; live discovery lists what the token group allows.",
933
+ },
934
+ {
935
+ // BizRouter: Korean enterprise LLM gateway (api.bizrouter.ai). Model ids are
936
+ // vendor-namespaced (`<vendor>/<model>`) and pass through to the upstream as-is.
937
+ // Live-verified 2026-07-24: /v1/chat/completions accepts the `tools` field and
938
+ // streams, and GET /v1/models returns the per-API-key allowed catalog in the
939
+ // OpenAI list shape, so live model discovery narrows to what the key can use.
940
+ id: "bizrouter", label: "BizRouter", adapter: "openai-chat", baseUrl: "https://api.bizrouter.ai/v1",
941
+ authKind: "key", dashboardUrl: "https://bizrouter.ai/settings/keys",
942
+ defaultModel: "openai/gpt-5.6-sol",
943
+ models: ["openai/gpt-5.6-sol", "anthropic/claude-sonnet-5", "google/gemini-3.5-flash"],
944
+ note: "Korean enterprise LLM gateway. Per-key allowed models are discovered live from /v1/models. Full catalog: https://bizrouter.ai/models",
945
+ },
946
+ { id: "groq", label: "Groq", adapter: "openai-chat", baseUrl: "https://api.groq.com/openai/v1", authKind: "key", featured: true, dashboardUrl: "https://console.groq.com/keys" },
947
+ // 2026-07-10 Gemini API refresh: Tier-2 ai.google.dev evidence recorded in
948
+ // devlog/_plan/260710_provider_hardening/001_research_frontier.md.
949
+ {
950
+ id: "google", label: "Google Gemini", adapter: "google", baseUrl: "https://generativelanguage.googleapis.com", authKind: "key", featured: true,
951
+ dashboardUrl: "https://aistudio.google.com/apikey", defaultModel: "gemini-3.5-flash", models: ["gemini-3.8-flash", "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro-preview", "gemini-3.7-flash"],
952
+ modelContextWindows: { "gemini-3.8-flash": 1_048_576, "gemini-3.6-flash": 1_048_576, "gemini-3.5-flash": 1_000_000, "gemini-3.5-flash-lite": 1_048_576, "gemini-3.7-flash": 1_048_576 },
953
+ modelInputModalities: { "gemini-3.8-flash": ["text", "image"], "gemini-3.6-flash": ["text", "image"], "gemini-3.5-flash-lite": ["text", "image"], "gemini-3.7-flash": ["text", "image"] },
954
+ modelReasoningEfforts: {
955
+ // 3.7 and 3.8 omit `minimal`: Google documents it as a validation error on both model
956
+ // pages, so advertising it hands the user a rung the API rejects. 3.5/3.6 keep theirs —
957
+ // their pages still list it, and this unit has no evidence to change them.
958
+ "gemini-3.8-flash": ["low", "medium", "high"],
959
+ "gemini-3.7-flash": ["low", "medium", "high"],
960
+ "gemini-3.6-flash": ["minimal", "low", "medium", "high"],
961
+ "gemini-3.5-flash": ["minimal", "low", "medium", "high"],
962
+ "gemini-3.1-pro-preview": ["low", "medium", "high"],
963
+ },
964
+ jawcodeBundle: "google", extraMetadataAliases: ["gemini"],
965
+ },
966
+ // 2026-07-10: defaultModel is frozen pending Vertex-specific Tier-2 evidence; Gemini API
967
+ // evidence from ai.google.dev does not establish Vertex publisher availability.
968
+ { id: "google-vertex", label: "Google Vertex AI", adapter: "google", baseUrl: "https://aiplatform.googleapis.com", authKind: "key", dashboardUrl: "https://console.cloud.google.com/vertex-ai", defaultModel: "gemini-3-pro", googleMode: "vertex", jawcodeBundle: "google", extraMetadataAliases: ["gemini-vertex"] },
969
+ // Antigravity discovers models with a POST to the CCA `:fetchAvailableModels` RPC, which
970
+ // `buildModelsRequest` already built by hand. Declaring it here changes no request URL — the
971
+ // relative path resolves to the same destination — but it lets `isRegistryModelDiscoveryUrl`
972
+ // prove that URL, which is what admits a Clash/Surge/Mihomo TUN fake-IP answer (#4261). The
973
+ // path must stay RELATIVE: this row sets `allowBaseUrlOverride`, and an absolute `url` would
974
+ // retarget a user's custom base back to Google. A leading `./` is required because a bare
975
+ // `v1internal:` reads as a URL scheme and `providerModelDiscoverySpecError` rejects it.
976
+ { id: "google-antigravity", alias: "agy", label: "Google Antigravity", adapter: "google", baseUrl: "https://daily-cloudcode-pa.googleapis.com", authKind: "oauth", allowBaseUrlOverride: true, dashboardUrl: "https://antigravity.google", models: ANTIGRAVITY_MODELS, liveModels: true, defaultModel: "gemini-3.8-flash", modelContextWindows: ANTIGRAVITY_MODEL_CONTEXT_WINDOWS, modelInputModalities: ANTIGRAVITY_MODEL_INPUT_MODALITIES, modelReasoningEfforts: ANTIGRAVITY_MODEL_EFFORTS, googleMode: "cloud-code-assist", showThinkingSummary: true, jawcodeBundle: "google", extraMetadataAliases: ["antigravity", "gemini-antigravity"], modelDiscovery: { path: "./v1internal:fetchAvailableModels" } },
977
+ { id: "azure-openai", label: "Azure OpenAI", adapter: "azure-openai", baseUrl: "https://{resource}.openai.azure.com/openai", authKind: "key", featured: true, dashboardUrl: "https://portal.azure.com" },
978
+ { id: "ollama", label: "Ollama (local)", adapter: "openai-chat", baseUrl: "http://localhost:11434/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — key usually blank" },
979
+ { id: "vllm", label: "vLLM (local)", adapter: "openai-chat", baseUrl: "http://localhost:8000/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — key usually blank" },
980
+ { id: "lm-studio", label: "LM Studio (local)", adapter: "openai-chat", baseUrl: "http://localhost:1234/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — no key needed" },
981
+ {
982
+ id: "deepseek",
983
+ label: "DeepSeek",
984
+ baseUrl: "https://api.deepseek.com",
985
+ adapter: "openai-chat",
986
+ authKind: "key",
987
+ dashboardUrl: "https://platform.deepseek.com/api_keys",
988
+ // Route DeepSeek's own catalog bundle so routed rebuilds restore the official
989
+ // context window from the vendored model-metadata bundle instead of falling
990
+ // back to the 128k strict-fields default (scripts/model-metadata.source.json,
991
+ // verified 2026-08-08).
992
+ jawcodeBundle: "deepseek",
993
+ // deepseek-chat/deepseek-reasoner were deprecated upstream on 2026-07-24 15:59 UTC;
994
+ // the current official identifier is deepseek-flash. They stay in
995
+ // the list only as compatibility aliases so existing saved configs and requests
996
+ // keep validating and routing (they previously mapped to v4-flash; devlog
997
+ // _fin/260710_provider_hardening/002_research_cn.md). The current offerings are
998
+ // V4.1-Flash — defaultModel and the model-specific wiring below use its live id.
999
+ // Keep the legacy vision-preview alias; see DEEPSEEK_VISION_PREVIEW_MODEL.
1000
+ models: ["deepseek-chat", "deepseek-reasoner", ...DEEPSEEK_NATIVE_THINKING_MODELS, DEEPSEEK_VISION_PREVIEW_MODEL],
1001
+ // V4.1-Flash is the current first-party offering; `deepseek-v4-flash` now routes there
1002
+ // as a compatibility alias, so a new install should ask for the live id by name.
1003
+ defaultModel: "deepseek-flash",
1004
+ // Official DeepSeek Codex setup (codex-deepseek-setup.sh) advertises 1,048,576
1005
+ // for both V4 models; the older 1,000,000 figure was a rounded approximation.
1006
+ modelContextWindows: { "deepseek-flash": 1_048_576, "deepseek-v4-flash": 1_048_576, [DEEPSEEK_VISION_PREVIEW_MODEL]: 1_048_576 },
1007
+ modelInputModalities: {
1008
+ "deepseek-flash": ["text", "image"],
1009
+ [DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
1010
+ },
1011
+ // DeepSeek documents both V4 models as native Responses API models adapted for Codex
1012
+ // (model table marks Responses API ✓ for flash and pro; the /responses reference lists
1013
+ // both ids as accepted `model` values — verified 2026-08-13 with the V4 Pro GA,
1014
+ // version label DeepSeek-V4-Pro-0813).
1015
+ modelWireDefaults: {
1016
+ // Codex speaks Responses natively and DeepSeek ships a Codex-compatible
1017
+ // apply_patch tool on that wire, so a Responses inbound goes straight out with
1018
+ // no translation. Claude Code and OpenAI-compatible clients keep the
1019
+ // provider-wide Chat wire: DeepSeek serves Chat Completions natively too, so
1020
+ // translating them into Responses would add a hop onto our newest upstream path
1021
+ // for no gain.
1022
+ "deepseek-v4-flash": { wire: "openai-responses", inbound: ["responses"] },
1023
+ // Same Responses contract as the V4 ids it succeeds; without this row the new
1024
+ // default would fall back to the provider-wide Chat wire.
1025
+ "deepseek-flash": { wire: "openai-responses", inbound: ["responses"] },
1026
+ },
1027
+ // The #875-era bounded-JSON force (`modelResponsesUpstreamStreaming`) is retired
1028
+ // for this entry: the official guide documents a `response.completed` /
1029
+ // `response.incomplete` / `response.failed` terminal with NO `data: [DONE]`
1030
+ // sentinel, and live probes (2026-08-07, including the tool-result replay shape
1031
+ // that originally stalled) close on the terminal. The relay's terminal boundary
1032
+ // (src/server/relay.ts) already cuts the stream at that event and synthesizes
1033
+ // `[DONE]`, so forcing stream:false only delayed every byte until generation
1034
+ // finished (28-46 s of silence on long turns). The registry knob itself remains
1035
+ // for providers that need it — re-adding one line here restores the old policy.
1036
+ // Evidence: https://api-docs.deepseek.com/guides/responses_api/ +
1037
+ // devlog/_fin/260807_deepseek_responses_streaming/000_plan.md.
1038
+ // Current official streams normally carry a real terminal; retain a narrow grace
1039
+ // repair for the historical shape that closes after a complete graph without one.
1040
+ modelResponsesTerminalRepair: { "deepseek-flash": { graceMs: 5_000 }, "deepseek-v4-flash": { graceMs: 5_000 } },
1041
+ // DeepSeek's Responses route emits bare UUID item ids, which leave Codex
1042
+ // clients stuck on an uncommitted turn (#938). Client-facing only — raw
1043
+ // continuation snapshots keep the upstream ids.
1044
+ responsesItemIdRepair: { repairInvalidIds: true, repairMissingTerminalIds: true },
1045
+ // DeepSeek's Responses route is `POST /responses` with no `/v1` segment. Without
1046
+ // this the passthrough adapter falls back to its legacy `/v1/responses`
1047
+ // construction and the wire above can never route.
1048
+ // Evidence: https://api-docs.deepseek.com/api/create-response/
1049
+ responsesPath: "/responses",
1050
+ // DeepSeek's Responses reference does not list `service_tier`; unsupported
1051
+ // parameters are documented as silently ignored, but the fail-closed policy
1052
+ // strips the field rather than forwarding a knob the upstream never asked for.
1053
+ supportsServiceTier: false,
1054
+ // DeepSeek's Responses compatibility guide accepts plaintext reasoning items and
1055
+ // merges them into the adjacent assistant message, so replayed reasoning must
1056
+ // not be blanked the way the ChatGPT backend requires. (Whether the Responses
1057
+ // route REQUIRES replay on tool-call continuations is an inference from the
1058
+ // Chat Thinking-Mode docs, not a confirmed Responses contract.)
1059
+ preserveResponsesReasoningContent: true,
1060
+ // "The API is stateless: responses and conversations are not stored on the
1061
+ // server." https://api-docs.deepseek.com/api/create-response/
1062
+ statelessResponses: true,
1063
+ // DeepSeek rejects a valid Codex continuation when hook-provided developer
1064
+ // context splits a call from its result (#1292); parallel calls remain one
1065
+ // reasoning-bearing assistant batch rather than being split per pair (#1477).
1066
+ requiresAdjacentResponsesToolResults: true,
1067
+ // DeepSeek exec tool results can be present-but-empty (a script that ran without
1068
+ // calling text(...)); annotate them so routed models do not silently accept an
1069
+ // empty result or re-issue the same call.
1070
+ annotateEmptyToolOutputs: true,
1071
+ /* [Decision Log]
1072
+ - 목적: DeepSeek V4 thinking mode multi-turn/tool-call requests must replay prior assistant reasoning_content.
1073
+ - 대안 분석: Globally preserve reasoning_content for all OpenAI-compatible models; preserve it for legacy deepseek-reasoner too; mark only V4 thinking models in registry metadata.
1074
+ - 선택 근거: DeepSeek V4 thinking mode requires history replay, while older DeepSeek reasoner has different compatibility rules. A model-scoped registry flag fixes built-in and stale saved configs without broad provider regressions.
1075
+ */
1076
+ modelReasoningEfforts: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, deepseekThinkingEffortsFor(id)])),
1077
+ modelReasoningEffortMap: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, deepseekReasoningMapFor(id)])),
1078
+ modelSupportsReasoningSummaries: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, true])),
1079
+ preserveReasoningContentModels: DEEPSEEK_NATIVE_THINKING_MODELS,
1080
+ // #4436: first-party deepseek-flash accepts native images on Chat and Responses.
1081
+ // Keep unprobed compatibility aliases on the #88 sidecar path. This must be fixed
1082
+ // here: router enrichment unions this list with saved config, so config cannot remove it.
1083
+ noVisionModels: ["deepseek-chat", "deepseek-reasoner", "deepseek-v4-flash"],
1084
+ },
1085
+ // llama-3.3-70b was deprecated by Cerebras on 2026-02-16. Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
1086
+ { id: "cerebras", label: "Cerebras", baseUrl: "https://api.cerebras.ai/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://cloud.cerebras.ai/platform/apikeys", defaultModel: "gpt-oss-120b" },
1087
+ {
1088
+ // Primary sources checked 2026-08-08:
1089
+ // - https://chutes.ai/pricing documents the shared llm.chutes.ai/v1 OpenAI-compatible
1090
+ // gateway, Bearer API keys, and chat completions. Its public
1091
+ // https://llm.chutes.ai/v1/models response supplies supported_features for filtering.
1092
+ // - https://chutes.ai/terms identifies Chutes Global Corp as the platform operator, applies
1093
+ // to API consumers, and directs production/high-volume automated inference to PAYGO.
1094
+ // Maintainer: @olddonkey; no affiliation with Chutes.
1095
+ id: "chutes",
1096
+ label: "Chutes",
1097
+ baseUrl: "https://llm.chutes.ai/v1",
1098
+ adapter: "openai-chat",
1099
+ authKind: "key",
1100
+ dashboardUrl: "https://chutes.ai/auth/start",
1101
+ liveModels: true,
1102
+ preserveCustomDestination: true,
1103
+ // The public model catalog cannot prove that a supplied Bearer key is valid.
1104
+ apiKeyValidation: "unknown",
1105
+ // Chutes documents tool calling, but not a provider-wide parallel tool-call contract.
1106
+ parallelToolCalls: false,
1107
+ // The live catalog reports reasoning support, but not a stable effort ladder.
1108
+ reasoningEfforts: [],
1109
+ modelDiscovery: {
1110
+ path: "models",
1111
+ maxResponseBytes: 256 * 1024,
1112
+ maxModels: 128,
1113
+ filter: {
1114
+ // The shared LLM catalog also contains rows without native tool support. Codex needs a
1115
+ // complete agent loop, so admit only rows whose live metadata advertises tools.
1116
+ allOf: [{ path: ["supported_features"], containsAny: ["tools"] }],
1117
+ },
1118
+ },
1119
+ note: "Shared OpenAI-compatible LLM gateway only; live discovery exposes tool-capable rows. User-deployed custom Chute endpoints and non-LLM APIs require a custom provider.",
1120
+ },
1121
+ {
1122
+ id: "deepinfra",
1123
+ label: "DeepInfra",
1124
+ baseUrl: "https://api.deepinfra.com/v1/openai",
1125
+ adapter: "openai-chat",
1126
+ authKind: "key",
1127
+ dashboardUrl: "https://deepinfra.com/dash/api_keys",
1128
+ liveModels: true,
1129
+ preserveCustomDestination: true,
1130
+ modelDiscovery: {
1131
+ // DeepInfra documents the OpenAI model catalog outside the chat-compatible `/v1/openai`
1132
+ // namespace, so keep this destination registry-owned instead of deriving it from baseUrl.
1133
+ url: "https://api.deepinfra.com/v1/models",
1134
+ maxResponseBytes: 512 * 1024,
1135
+ maxModels: 512,
1136
+ filter: {
1137
+ allOf: [{ path: ["metadata", "tags"], containsAny: ["chat"] }],
1138
+ },
1139
+ },
1140
+ note: "OpenAI-compatible chat models only; live discovery excludes non-chat rows from DeepInfra's mixed model catalog.",
1141
+ },
1142
+ {
1143
+ id: "hyperbolic",
1144
+ label: "Hyperbolic",
1145
+ baseUrl: "https://api.hyperbolic.xyz/v1",
1146
+ adapter: "openai-chat",
1147
+ authKind: "key",
1148
+ dashboardUrl: "https://app.hyperbolic.ai",
1149
+ liveModels: true,
1150
+ preserveCustomDestination: true,
1151
+ modelDiscovery: {
1152
+ path: "models",
1153
+ maxResponseBytes: 256 * 1024,
1154
+ maxModels: 256,
1155
+ },
1156
+ note: "Serverless text and vision-language chat models only; Hyperbolic's separate image, audio, and GPU endpoints are out of scope.",
1157
+ },
1158
+ {
1159
+ // Primary sources checked 2026-08-03:
1160
+ // - docs.nscale.com documents the production OpenAI-compatible endpoint, bearer service
1161
+ // tokens, /v1/models, and a tool-calling request using this exact Llama model id.
1162
+ // - nscale.com/policies/terms-conditions identifies Nscale AS as the service operator and
1163
+ // covers customers using its public-cloud inference offering. Maintainer: @olddonkey;
1164
+ // no affiliation with Nscale.
1165
+ id: "nscale",
1166
+ label: "Nscale Serverless Inference",
1167
+ baseUrl: "https://inference.api.nscale.com/v1",
1168
+ adapter: "openai-chat",
1169
+ authKind: "key",
1170
+ dashboardUrl: "https://console.nscale.com",
1171
+ defaultModel: "meta-llama/Llama-3.1-8B-Instruct",
1172
+ models: ["meta-llama/Llama-3.1-8B-Instruct"],
1173
+ liveModels: true,
1174
+ preserveCustomDestination: true,
1175
+ // Nscale documents tools but not parallel tool calls. Keep requests serialized.
1176
+ parallelToolCalls: false,
1177
+ // The API schema accepts reasoning_effort, but does not publish per-model tiers.
1178
+ reasoningEfforts: [],
1179
+ modelDiscovery: {
1180
+ path: "models",
1181
+ maxResponseBytes: 256 * 1024,
1182
+ maxModels: 256,
1183
+ filter: {
1184
+ // Nscale's catalog mixes chat, image, and embedding rows without a modality field.
1185
+ // Admit only the exact model used in its official tool-calling API example.
1186
+ allOf: [{ path: ["id"], equalsAny: ["meta-llama/Llama-3.1-8B-Instruct"] }],
1187
+ },
1188
+ },
1189
+ note: "Serverless OpenAI-compatible inference. Live discovery admits only the tool-capable model established by Nscale's official API example; other mixed-catalog rows remain hidden pending equivalent evidence.",
1190
+ },
1191
+ {
1192
+ // Primary sources checked 2026-08-03:
1193
+ // - docs.vultr.com documents the fixed OpenAI-compatible base URL, per-subscription bearer
1194
+ // key, /v1/models, and states that tool calling is currently limited to kimi-k2-instruct.
1195
+ // - Vultr's official properties identify VULTR as a The Constant Company, LLC trademark and
1196
+ // document customer API integrations. Maintainer: @olddonkey; no affiliation with Vultr.
1197
+ id: "vultr",
1198
+ label: "Vultr Serverless Inference",
1199
+ baseUrl: "https://api.vultrinference.com/v1",
1200
+ adapter: "openai-chat",
1201
+ authKind: "key",
1202
+ dashboardUrl: "https://my.vultr.com",
1203
+ defaultModel: "kimi-k2-instruct",
1204
+ models: ["kimi-k2-instruct"],
1205
+ liveModels: true,
1206
+ preserveCustomDestination: true,
1207
+ parallelToolCalls: false,
1208
+ reasoningEfforts: [],
1209
+ modelDiscovery: {
1210
+ path: "models",
1211
+ maxResponseBytes: 256 * 1024,
1212
+ maxModels: 256,
1213
+ filter: {
1214
+ // Vultr explicitly limits tool calling to this model. A coding agent must not select
1215
+ // another chat model that cannot complete its tool loop.
1216
+ allOf: [{ path: ["id"], equalsAny: ["kimi-k2-instruct"] }],
1217
+ },
1218
+ },
1219
+ note: "Serverless Inference subscription API. Live discovery exposes only kimi-k2-instruct because Vultr documents it as the sole tool-calling model.",
1220
+ },
1221
+ ];