@bitkyc08/opencodex 2.55.0 → 2.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (256) hide show
  1. package/bin/ocx.mjs +10 -0
  2. package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
  3. package/gui/dist/assets/{index-VuoiWj9J.js → index-Cz7CLdif.js} +21 -21
  4. package/gui/dist/index.html +2 -2
  5. package/package.json +4 -3
  6. package/src/adapters/base.ts +21 -0
  7. package/src/adapters/codebuddy/adapter.ts +2 -1
  8. package/src/adapters/codebuddy/scaffold-guard.ts +248 -0
  9. package/src/adapters/command-code.ts +1 -1
  10. package/src/adapters/cursor/envelope-echo.ts +8 -2
  11. package/src/adapters/cursor/transport-retry.ts +46 -1
  12. package/src/adapters/cursor.ts +4 -0
  13. package/src/adapters/google.ts +7 -7
  14. package/src/adapters/kiro/adapter.ts +42 -1
  15. package/src/adapters/kiro/payload.ts +17 -3
  16. package/src/adapters/kiro/reasoning.ts +70 -7
  17. package/src/adapters/kiro/stream.ts +8 -2
  18. package/src/adapters/kiro/wire.ts +2 -1
  19. package/src/adapters/kiro-events.ts +21 -13
  20. package/src/adapters/kiro-retry.ts +23 -4
  21. package/src/adapters/openai-chat/errors.ts +116 -0
  22. package/src/adapters/openai-chat/messages.ts +346 -0
  23. package/src/adapters/openai-chat/passthrough.ts +146 -0
  24. package/src/adapters/openai-chat/response-events.ts +117 -0
  25. package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
  26. package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
  27. package/src/adapters/openai-chat/tool-schema.ts +495 -0
  28. package/src/adapters/openai-chat/wire.ts +50 -0
  29. package/src/adapters/openai-chat.ts +40 -1452
  30. package/src/adapters/openai-responses/canonical-forward.ts +202 -0
  31. package/src/adapters/openai-responses/image-gen.ts +406 -0
  32. package/src/adapters/openai-responses/internal.ts +3 -0
  33. package/src/adapters/openai-responses/passthrough.ts +642 -0
  34. package/src/adapters/openai-responses/prompt-cache.ts +83 -0
  35. package/src/adapters/openai-responses/reasoning.ts +220 -0
  36. package/src/adapters/openai-responses/request-strips.ts +185 -0
  37. package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
  38. package/src/adapters/openai-responses/tool-schema.ts +293 -0
  39. package/src/adapters/openai-responses/web-search.ts +156 -0
  40. package/src/adapters/openai-responses.ts +4 -2625
  41. package/src/bridge/errors.ts +58 -0
  42. package/src/bridge/internal.ts +174 -0
  43. package/src/bridge/response-json.ts +630 -0
  44. package/src/bridge/sse.ts +1462 -0
  45. package/src/bridge.ts +5 -2204
  46. package/src/chat/inbound.ts +12 -1
  47. package/src/claude/desktop-profile.ts +66 -9
  48. package/src/claude/outbound.ts +18 -0
  49. package/src/cli/account-main.ts +1 -1
  50. package/src/cli/capabilities.ts +2 -2
  51. package/src/cli/combo.ts +10 -1
  52. package/src/cli/index.ts +48 -5
  53. package/src/cli/registry.ts +2 -1
  54. package/src/cli/system-command.ts +4 -4
  55. package/src/clients/config-export.ts +7 -3
  56. package/src/codex/account-label.ts +14 -3
  57. package/src/codex/account-lifecycle.ts +3 -0
  58. package/src/codex/account-store.ts +184 -35
  59. package/src/codex/account-usability.ts +21 -0
  60. package/src/codex/auth-api/account-list.ts +507 -0
  61. package/src/codex/auth-api/http.ts +32 -0
  62. package/src/codex/auth-api/login-flow.ts +566 -0
  63. package/src/codex/auth-api/login-state.ts +64 -0
  64. package/src/codex/auth-api/main-account-probe.ts +331 -0
  65. package/src/codex/auth-api/pool-mode-gate.ts +274 -0
  66. package/src/codex/auth-api/pool-quota-probe.ts +512 -0
  67. package/src/codex/auth-api/reset-credit-service.ts +431 -0
  68. package/src/codex/auth-api/routes.ts +425 -0
  69. package/src/codex/auth-api/runtime-config.ts +48 -0
  70. package/src/codex/auth-api.ts +27 -3118
  71. package/src/codex/auth-context.ts +252 -35
  72. package/src/codex/catalog/aggregation.ts +80 -1
  73. package/src/codex/catalog/auto-review.ts +507 -0
  74. package/src/codex/catalog/build-entries.ts +981 -0
  75. package/src/codex/catalog/combo-member.ts +375 -0
  76. package/src/codex/catalog/derive-entry.ts +229 -0
  77. package/src/codex/catalog/effort.ts +0 -1
  78. package/src/codex/catalog/gated-native-warn.ts +63 -0
  79. package/src/codex/catalog/gather-capture.ts +533 -0
  80. package/src/codex/catalog/model-hints.ts +691 -0
  81. package/src/codex/catalog/model-visibility.ts +305 -0
  82. package/src/codex/catalog/provider-fetch.ts +52 -2942
  83. package/src/codex/catalog/provider-models.ts +685 -0
  84. package/src/codex/catalog/remote.ts +30 -0
  85. package/src/codex/catalog/restore.ts +132 -0
  86. package/src/codex/catalog/retained-sync.ts +714 -0
  87. package/src/codex/catalog/routed-gather.ts +895 -0
  88. package/src/codex/catalog/subagent-roster.ts +176 -0
  89. package/src/codex/catalog/sync.ts +52 -2698
  90. package/src/codex/cli-install-provenance.ts +7 -1
  91. package/src/codex/convergence.ts +7 -2
  92. package/src/codex/desktop-app/types.ts +11 -2
  93. package/src/codex/desktop-app/windows.ts +5 -5
  94. package/src/codex/inject/config-toml.ts +563 -0
  95. package/src/codex/inject/remove.ts +192 -0
  96. package/src/codex/inject/restore.ts +567 -0
  97. package/src/codex/inject/routing-classify.ts +109 -0
  98. package/src/codex/inject/routing-target.ts +125 -0
  99. package/src/codex/inject.ts +89 -1444
  100. package/src/codex/lineage.ts +458 -0
  101. package/src/codex/model-entitlements.ts +152 -15
  102. package/src/codex/pool-refresh-backoff.ts +161 -0
  103. package/src/codex/quota-rejection.ts +104 -15
  104. package/src/codex/routing/active-account.ts +194 -0
  105. package/src/codex/routing/cache-affinity.ts +70 -0
  106. package/src/codex/routing/cooldown-math.ts +285 -0
  107. package/src/codex/routing/health-store.ts +402 -0
  108. package/src/codex/routing/probe-lease.ts +358 -0
  109. package/src/codex/routing/selection.ts +780 -0
  110. package/src/codex/routing/thread-affinity.ts +586 -0
  111. package/src/codex/routing/transient-hold-dispatch.ts +141 -0
  112. package/src/codex/routing.ts +370 -2271
  113. package/src/codex/shim-fingerprint.ts +223 -0
  114. package/src/codex/shim-inspect.ts +175 -0
  115. package/src/codex/shim-probe.ts +367 -0
  116. package/src/codex/shim-restore-lock.ts +169 -0
  117. package/src/codex/shim-state-file.ts +151 -0
  118. package/src/codex/shim-templates.ts +265 -0
  119. package/src/codex/shim.ts +48 -1268
  120. package/src/codex/warmup.ts +1 -1
  121. package/src/combos/failover.ts +85 -0
  122. package/src/combos/request.ts +17 -10
  123. package/src/combos/types.ts +23 -2
  124. package/src/config/diagnostics.ts +705 -0
  125. package/src/config/feature-flags.ts +55 -0
  126. package/src/config/live-reconcile.ts +403 -0
  127. package/src/config/load-degrade.ts +880 -0
  128. package/src/config/mutation-lock.ts +244 -0
  129. package/src/config/openai-tier-backup.ts +268 -0
  130. package/src/config/pending-teardown.ts +31 -0
  131. package/src/config/persist-unlocked.ts +92 -0
  132. package/src/config/proxy-env.ts +188 -0
  133. package/src/config/salvage.ts +244 -0
  134. package/src/config/schema/config-schema.ts +640 -0
  135. package/src/config/schema/leaf-validators.ts +855 -0
  136. package/src/config/warn-memo.ts +28 -0
  137. package/src/config.ts +234 -4481
  138. package/src/generated/compatibility-version.json +649 -121
  139. package/src/images/loop.ts +1 -1
  140. package/src/lib/errors.ts +17 -0
  141. package/src/lib/request-execution-budget.ts +198 -23
  142. package/src/lib/spend-reservation-ledger.ts +958 -0
  143. package/src/lib/state-store-registrations.ts +6 -2
  144. package/src/lib/test-home-guard.ts +85 -1
  145. package/src/lib/upstream-retry.ts +132 -21
  146. package/src/lib/windows-elevation.ts +76 -14
  147. package/src/lib/workflow-budget.ts +553 -30
  148. package/src/oauth/index.ts +2 -2
  149. package/src/oauth/key-providers.ts +2 -2
  150. package/src/providers/kiro-models.ts +4 -3
  151. package/src/providers/label.ts +19 -1
  152. package/src/providers/model-discovery.ts +16 -0
  153. package/src/providers/quota/account-cache.ts +441 -0
  154. package/src/providers/quota/antigravity.ts +295 -0
  155. package/src/providers/quota/report-cache.ts +320 -0
  156. package/src/providers/quota/vendor-probes-key.ts +1243 -0
  157. package/src/providers/quota/vendor-probes-oauth.ts +590 -0
  158. package/src/providers/quota.ts +324 -3079
  159. package/src/providers/registry/entries-core.ts +1228 -0
  160. package/src/providers/registry/entries-extended.ts +1213 -0
  161. package/src/providers/registry/model-seeds.ts +912 -0
  162. package/src/providers/registry/types.ts +352 -0
  163. package/src/providers/registry.ts +24 -3536
  164. package/src/responses/continuation-ownership.ts +29 -0
  165. package/src/responses/reasoning-envelope.ts +6 -3
  166. package/src/responses/state/replay-fingerprint.ts +80 -0
  167. package/src/responses/state/snapshot-codec.ts +104 -0
  168. package/src/responses/state/spill-failure.ts +118 -0
  169. package/src/responses/state/spill-queue.ts +665 -0
  170. package/src/responses/state/temp-recovery.ts +257 -0
  171. package/src/responses/state.ts +82 -1143
  172. package/src/routing/identity-domains.ts +456 -0
  173. package/src/routing/probe-lease.ts +613 -0
  174. package/src/server/chat-completions.ts +3 -1
  175. package/src/server/chat-native.ts +37 -9
  176. package/src/server/index/bounded-request.ts +88 -0
  177. package/src/server/index/live-sideband.ts +601 -0
  178. package/src/server/index/serve-options.ts +1766 -0
  179. package/src/server/index/startup-warnings.ts +213 -0
  180. package/src/server/index/websocket-handler.ts +339 -0
  181. package/src/server/index.ts +45 -2552
  182. package/src/server/inspection-tee.ts +107 -0
  183. package/src/server/live.ts +46 -1
  184. package/src/server/management/combo-routes.ts +10 -1
  185. package/src/server/management/route-registry.ts +26 -23
  186. package/src/server/management/shared.ts +8 -5
  187. package/src/server/management/workflow-budget-routes.ts +133 -0
  188. package/src/server/management-api.ts +12 -0
  189. package/src/server/relay-eager.ts +2 -0
  190. package/src/server/relay.ts +14 -19
  191. package/src/server/request-log-conversation.ts +9 -7
  192. package/src/server/request-log.ts +372 -4
  193. package/src/server/response-log-body.ts +153 -0
  194. package/src/server/responses/account-change-state.ts +307 -0
  195. package/src/server/responses/adapter-continuation.ts +540 -0
  196. package/src/server/responses/adapter-delivery.ts +208 -0
  197. package/src/server/responses/adapter-dispatch.ts +1042 -0
  198. package/src/server/responses/codex-ws-wire.ts +5 -0
  199. package/src/server/responses/collaboration.ts +74 -4
  200. package/src/server/responses/combo-session-recall.ts +68 -8
  201. package/src/server/responses/compact.ts +113 -17
  202. package/src/server/responses/completion-policy.ts +33 -0
  203. package/src/server/responses/core-auth.ts +529 -0
  204. package/src/server/responses/core-codex-account.ts +907 -0
  205. package/src/server/responses/core-combo-failure.ts +210 -0
  206. package/src/server/responses/core-combo.ts +787 -0
  207. package/src/server/responses/core-errors.ts +170 -0
  208. package/src/server/responses/core-lifetime.ts +95 -0
  209. package/src/server/responses/core-normalize.ts +350 -0
  210. package/src/server/responses/core-opaque-recovery.ts +380 -0
  211. package/src/server/responses/core-options.ts +159 -0
  212. package/src/server/responses/core-replay.ts +298 -0
  213. package/src/server/responses/core.ts +192 -8893
  214. package/src/server/responses/encrypted-payload.ts +0 -1
  215. package/src/server/responses/input-admission.ts +126 -6
  216. package/src/server/responses/passthrough-delivery.ts +869 -0
  217. package/src/server/responses/passthrough-dispatch.ts +1494 -0
  218. package/src/server/responses/passthrough-error.ts +38 -2
  219. package/src/server/responses/passthrough-execution.ts +54 -0
  220. package/src/server/responses/request-prepare.ts +1080 -0
  221. package/src/server/responses/request-send-budget.ts +259 -0
  222. package/src/server/responses/request-sidecar-auth.ts +149 -0
  223. package/src/server/responses/request-spend.ts +147 -0
  224. package/src/server/responses/request-transport.ts +803 -0
  225. package/src/server/responses/response-effects.ts +157 -0
  226. package/src/server/responses/run-turn-execution.ts +476 -0
  227. package/src/server/responses/sidecar-execution.ts +463 -0
  228. package/src/server/responses/terminal-guard.ts +65 -4
  229. package/src/server/responses-image-gen-repair.ts +1 -1
  230. package/src/server/responses-undeclared-tool-guard.ts +9 -5
  231. package/src/server/workflow-refusal.ts +84 -0
  232. package/src/service/windows-ops.ts +210 -16
  233. package/src/service/windows-scheduler.ts +28 -21
  234. package/src/service.ts +1 -1
  235. package/src/types/config.ts +34 -1
  236. package/src/types/request.ts +8 -5
  237. package/src/types/tools.ts +24 -0
  238. package/src/types.ts +2 -0
  239. package/src/update/index.ts +10 -0
  240. package/src/update/stop-contract.d.mts +1 -0
  241. package/src/update/stop-contract.mjs +19 -0
  242. package/src/update/stop-decision.d.mts +1 -1
  243. package/src/update/stop-decision.mjs +12 -3
  244. package/src/usage/log.ts +147 -1
  245. package/src/usage/summary.ts +171 -21
  246. package/src/vision/anthropic-describe.ts +1 -1
  247. package/src/vision/describe.ts +5 -5
  248. package/src/web-search/anthropic-executor.ts +1 -1
  249. package/src/web-search/exa-executor.ts +1 -1
  250. package/src/web-search/executor.ts +1 -1
  251. package/src/web-search/gemini-executor.ts +1 -1
  252. package/src/web-search/loop.ts +1 -1
  253. package/src/web-search/ollama-executor.ts +1 -1
  254. package/src/web-search/parse.ts +67 -14
  255. package/src/web-search/passthrough-bridge.ts +64 -31
  256. package/src/web-search/xai-executor.ts +1 -1
@@ -0,0 +1,1228 @@
1
+ import { KIRO_MODELS, KIRO_MODEL_CONTEXT_WINDOWS, KIRO_MODEL_REASONING_EFFORTS } from "../kiro-models";
2
+ import { DEVIN_MODEL_CONTEXT_WINDOWS, DEVIN_MODEL_EFFORTS, DEVIN_DEFAULT_EFFORTS } from "../../adapters/devin/live-models";
3
+ import { ANTIGRAVITY_MODELS, ANTIGRAVITY_MODEL_CONTEXT_WINDOWS, ANTIGRAVITY_MODEL_EFFORTS, ANTIGRAVITY_MODEL_INPUT_MODALITIES } from "../antigravity-models";
4
+ import {
5
+ CURSOR_NO_VISION_MODELS,
6
+ CURSOR_STATIC_MODELS,
7
+ cursorModelContextWindows,
8
+ cursorModelDisplayNames,
9
+ cursorModelIds,
10
+ cursorModelInputModalities,
11
+ cursorModelReasoningEfforts,
12
+ } from "../../adapters/cursor/discovery";
13
+ import { cursorFastCapableBases } from "../../adapters/cursor/catalog";
14
+ import { COMMAND_CODE_MODEL_REASONING_EFFORTS } from "../command-code-efforts";
15
+ import { isCanonicalOpenRouterTarget } from "../openrouter-routing";
16
+ import type { ProviderRegistryEntry } from "./types";
17
+ import {
18
+ ANTHROPIC_MODELS,
19
+ ANTHROPIC_MODEL_CONTEXT_WINDOWS,
20
+ ANTHROPIC_MODEL_INPUT_MODALITIES,
21
+ ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
22
+ ANTHROPIC_MODEL_REASONING_EFFORTS,
23
+ ZAI_GLM_52_REASONING_EFFORTS,
24
+ ZAI_GLM_53_REASONING_EFFORTS,
25
+ OPENAI_GPT56_MODELS,
26
+ OPENAI_GPT56_PRO_MODELS,
27
+ OPENAI_API_GPT56_CONTEXT_WINDOWS,
28
+ OPENAI_API_GPT56_MAX_INPUT_TOKENS,
29
+ OPENAI_API_GPT56_VIRTUAL_MODELS,
30
+ OPENAI_API_GPT56_REASONING_EFFORTS,
31
+ META_MUSE_REASONING_EFFORTS,
32
+ META_MUSE_REASONING_EFFORT_MAP,
33
+ META_MUSE_CONTEXT_WINDOW,
34
+ META_MUSE_MODELS,
35
+ OPENAI_DAYBREAK_MODELS,
36
+ OPENAI_DAYBREAK_CONTEXT_WINDOWS,
37
+ OPENAI_DAYBREAK_MAX_INPUT_TOKENS,
38
+ OPENAI_DAYBREAK_REASONING_EFFORTS,
39
+ OPENROUTER_GPT56_MODELS,
40
+ XAI_MODELS,
41
+ OPENROUTER_GPT56_CONTEXT_WINDOWS,
42
+ THINKING_TOGGLE_EFFORTS,
43
+ THINKING_TOGGLE_MAP,
44
+ OPENCODE_GO_THINKING_TOGGLE_MODELS,
45
+ THINKING_BUDGET_EFFORTS,
46
+ QWEN38_REASONING_EFFORTS,
47
+ THINKING_BUDGET_MODELS,
48
+ OPENCODE_GO_THINKING_BUDGET_MODELS,
49
+ DEEPSEEK_NATIVE_THINKING_MODELS,
50
+ DEEPSEEK_GATEWAY_THINKING_MODELS,
51
+ DEEPSEEK_VISION_PREVIEW_MODEL,
52
+ COMMAND_CODE_MODEL_INPUT_MODALITIES,
53
+ deepseekThinkingEffortsFor,
54
+ deepseekReasoningMapFor,
55
+ KIMI_K3_STANDARD_CONTEXT_WINDOW,
56
+ KIMI_CODING_MODELS,
57
+ KIMI_THINKING_MODELS,
58
+ KIMI_CODING_NO_REASONING_MODELS,
59
+ KIMI_CODING_K3_REASONING_EFFORTS,
60
+ KIMI_CODING_K3_REASONING_EFFORT_MAP,
61
+ KIMI_CODING_REASONING_EFFORTS,
62
+ KIMI_CODING_DEFAULT_REASONING_EFFORTS,
63
+ KIMI_CODING_REASONING_EFFORT_MAPS,
64
+ KIMI_LOCKED_PARAMETER_MODELS,
65
+ KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS,
66
+ KIMI_CODING_MODEL_CONTEXT_WINDOWS,
67
+ KIMI_CODING_MODEL_INPUT_MODALITIES,
68
+ NEURALWATT_REASONING_HISTORY_MODELS,
69
+ UMANS_MODELS,
70
+ UMANS_REASONING_EFFORTS,
71
+ UMANS_GLM_REASONING_EFFORTS,
72
+ UMANS_GLM_53_REASONING_EFFORTS,
73
+ UMANS_TEXT_ONLY_MODELS,
74
+ UMANS_MODEL_CONTEXT_WINDOWS,
75
+ UMANS_MODEL_INPUT_MODALITIES,
76
+ CLINE_PASS_MODELS,
77
+ ORCAROUTER_MODEL_DISCOVERY,
78
+ ORCAROUTER_MODELS,
79
+ ORCAROUTER_MODEL_REASONING_EFFORTS,
80
+ CLINE_PASS_MODEL_CONTEXT_WINDOWS,
81
+ CLINE_PASS_TEXT_ONLY_MODELS,
82
+ CLINE_PASS_MODEL_INPUT_MODALITIES,
83
+ } from "./model-seeds";
84
+
85
+ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
86
+ {
87
+ id: "openai",
88
+ label: "OpenAI (Codex login)",
89
+ adapter: "openai-responses",
90
+ baseUrl: "https://chatgpt.com/backend-api/codex",
91
+ authKind: "forward",
92
+ codexAccountMode: "pool",
93
+ supportsServiceTier: true,
94
+ featured: true,
95
+ note: "Codex login account pool (default) or Direct main-account mode via codexAccountMode",
96
+ },
97
+ {
98
+ id: "cursor",
99
+ label: "Cursor (experimental)",
100
+ adapter: "cursor",
101
+ baseUrl: "https://api2.cursor.sh",
102
+ authKind: "oauth",
103
+ featured: false,
104
+ dashboardPreset: true,
105
+ note: "Experimental Cursor bridge. Live transport and live model discovery are enabled after a standalone PKCE browser login via 'ocx login cursor'; native read/write/delete/shell/fetch execution is disabled by default and request text such as Codex sandbox markers never authorizes it. Set \"nativeLocalExec\": \"on\" on providers.cursor in ~/.opencodex/config.json (dashboard: Providers → Cursor → Edit JSON) only for a trusted local experiment where every data-plane caller is trusted. \"off\" denies all, \"codex-sandbox\" is accepted for backwards compatibility but fails closed, and legacy \"unsafeAllowNativeLocalExec\": true still means explicit operator opt-in.",
106
+ models: cursorModelIds(CURSOR_STATIC_MODELS),
107
+ liveModels: true,
108
+ defaultModel: "auto",
109
+ modelContextWindows: cursorModelContextWindows(CURSOR_STATIC_MODELS),
110
+ modelDisplayNames: cursorModelDisplayNames(),
111
+ // Cursor's Fast product is a model VARIANT, not a service_tier field, so the wire kind
112
+ // is cursor-variant and the request builder consumes the decision.
113
+ fastWire: { kind: "cursor-variant", canonicalToWire: { priority: "fast" }, foreignCallerTiers: "drop" },
114
+ // Deliberately NO provider-level supportsServiceTier: resolveFastPolicy short-circuits on
115
+ // `capability.provider === false` BEFORE consulting the per-model map, which would make
116
+ // these entries dead config. Absent leaves unlisted bases "unclassified", and a
117
+ // non-service-tier adapter cannot forward a caller tier, so they still publish no toggle.
118
+ modelSupportsServiceTier: Object.fromEntries(cursorFastCapableBases().map(id => [id, true])),
119
+ fastTierDescription: "Cursor Fast variant",
120
+ modelInputModalities: cursorModelInputModalities(CURSOR_STATIC_MODELS),
121
+ modelReasoningEfforts: cursorModelReasoningEfforts(CURSOR_STATIC_MODELS),
122
+ // Kimi K3 documents `max` as its API default, and its Cursor ladder has no `medium`
123
+ // rung — so applyReasoningLevels' medium->high->first fallback would settle the catalog
124
+ // default on `high`, the picker would send `high` explicitly, and the request builder's
125
+ // no-effort fallback to `kimi-k3-max` would never be reached. Mirrors the other K3
126
+ // routes (kimi, kimi-code, opencode-go).
127
+ modelDefaultReasoningEfforts: { "kimi-k3": "max" },
128
+ // Blind Cursor models (Auto routers, Composer, GLM-5.2, GLM-5.3) go through the vision sidecar;
129
+ // multimodal hosts (Claude/Gemini/GPT/Kimi/Grok) take native SelectedImage. The catalog
130
+ // still advertises image for noVision members so Codex can attach (sidecar option B).
131
+ noVisionModels: [...CURSOR_NO_VISION_MODELS],
132
+ },
133
+ {
134
+ // The canonical Cognition account provider, after absorbing `devin-cli`
135
+ // (devlog/_plan/260913_devin_provider_merge). The two ids were the same
136
+ // `devin` adapter, the same server.codeium.com api-server, and the same
137
+ // `devin-session-token$<JWT>` credential — only the account source
138
+ // differed: this entry did an Auth0 browser sign-in while `devin-cli`
139
+ // imported the token the installed CLI's own PKCE login had already
140
+ // written to credentials.toml. The merged login is import-first with a
141
+ // browser fallback: the CLI credential is taken when present (no browser
142
+ // opens), and the Auth0 flow remains because it is the only path for
143
+ // users without the CLI. `devin-cli` survives only as a deprecated
144
+ // alias; a startup migration rewrites saved provider rows, cross-config
145
+ // references, and auth.json slots to `devin`.
146
+ //
147
+ // `oauth` classifies the ACCOUNT, not the transport. This is not a local
148
+ // runtime: unlike Ollama or LM Studio it cannot answer at all until a
149
+ // vendor account is signed in, and `local` grouped it with things that
150
+ // have no account. It is also the only classification that reaches the
151
+ // dashboard Accounts tab, which is built from OAUTH_PROVIDERS.
152
+ id: "devin",
153
+ label: "Cognition (Devin/Windsurf)",
154
+ adapter: "devin",
155
+ baseUrl: "https://server.codeium.com",
156
+ authKind: "oauth",
157
+ featured: false,
158
+ // Off: `deriveProviderPresets` keys the preset catalog off this flag, so a
159
+ // true row would draw the provider twice — an Accounts login row and a
160
+ // preset tile.
161
+ dashboardPreset: false,
162
+ note: "Experimental unofficial Cognition/Devin bridge. ocx login devin first imports the credential an installed Devin CLI already holds (no browser); without one it opens Auth0 browser sign-in and exchanges the token via Cognition's RegisterUser for a long-lived API key.",
163
+ // Union seed of the two merged rosters: the newer devin-cli lineup first
164
+ // (it is the current catalog, so its default ordering wins), then the ids
165
+ // only the old devin entry carried. Degraded-mode seed only either way —
166
+ // `liveModels` discovers the account's real roster.
167
+ models: ["swe-2", "swe-1-7", "gpt-5-6-sol", "gpt-6-astra", "claude-opus-5", "claude-fable-5-1", "claude-sonnet-5", "glm-5-3", "kimi-k3", "gemini-3-8-flash", "grok-4-6", "swe-1-7-lightning", "gpt-5-6-luna", "gpt-5-6-terra", "claude-opus-4-8", "glm-5-2", "kimi-k2-7", "grok-4-5"],
168
+ liveModels: true,
169
+ defaultModel: "swe-2",
170
+ modelContextWindows: DEVIN_MODEL_CONTEXT_WINDOWS,
171
+ // Degraded-mode ladders only. Once a credential is present the account
172
+ // catalog supplies each base model its measured rungs; these two fields are
173
+ // what a signed-out picker and the Pi-shaped client exports fall back to.
174
+ modelReasoningEfforts: DEVIN_MODEL_EFFORTS,
175
+ reasoningEfforts: DEVIN_DEFAULT_EFFORTS,
176
+ },
177
+ {
178
+ id: "xai",
179
+ label: "xAI Grok",
180
+ adapter: "openai-chat",
181
+ baseUrl: "https://api.x.ai/v1",
182
+ authKind: "oauth",
183
+ allowKeyAuthOverride: true,
184
+ // Priority Processing is documented for xAI's public API-key Chat Completions and
185
+ // Responses endpoints. The OAuth lane is classified per-model below, not here:
186
+ // do not turn this into a provider-wide supportsServiceTier declaration.
187
+ keyAuthServiceTier: {
188
+ supportsServiceTier: true,
189
+ chatServiceTier: true,
190
+ },
191
+ // OAuth (Grok subscription gateway) service-tier capability, classified by live probe
192
+ // on 2026-09-13 (devlog/_fin/260913_xai_oauth_fast/020_probe-evidence.md): each listed
193
+ // model accepted service_tier "priority" over grok-oauth and echoed priority upstream.
194
+ // Key-auth already declares provider-wide support above, so this map only newly opens
195
+ // the OAuth lane. grok-4.20-multi-agent-0309 is deliberately absent: the gateway accepts
196
+ // the field but answers service_tier "default" — a live downgrade, not a fast tier.
197
+ // Unlisted and future-discovered ids stay unclassified.
198
+ modelSupportsServiceTier: {
199
+ "grok-4.6": true,
200
+ "grok-4.5": true,
201
+ "grok-4.3": true,
202
+ "grok-4.20-0309-reasoning": true,
203
+ "grok-4.20-0309-non-reasoning": true,
204
+ "grok-build-0.1": true,
205
+ "grok-composer-2.5-fast": true,
206
+ },
207
+ // Lets a caller-sent service_tier forward on the Chat wire (fastwire forwardCallerTier
208
+ // chain). Provider-wide by construction: unclassified chat-wire models then preserve a
209
+ // caller tier verbatim, the same contract other unclassified Responses routes already
210
+ // follow; --fast publication and proxy-owned fast injection stay capability-scoped by
211
+ // the map above. Key-auth declared the same value via keyAuthServiceTier, so the key
212
+ // lane is unchanged.
213
+ chatServiceTier: true,
214
+ // Shared across key and OAuth catalog rows. OAuth subscription has no
215
+ // per-token price, so the 2x claim is scoped to key auth.
216
+ fastTierDescription: "Priority processing; tier pricing applies on key auth only",
217
+ featured: true,
218
+ oauthId: "xai",
219
+ jawcodeBundle: "xai",
220
+ supportsOpenAiWebSearchToolFields: false,
221
+ // Live A/B on 2026-08-20: xAI rejects native custom/custom_tool_call shapes while accepting
222
+ // the otherwise-identical request after the custom tool is lowered to a function.
223
+ supportsResponsesCustomTools: false,
224
+ note: "Log in with your Grok account",
225
+ // Parallel tool calls: officially supported and default-on per docs.x.ai function-calling
226
+ // (verified 260709, devlog/_plan/260709_parallel_tool_calls). Streamed calls arrive whole
227
+ // per chunk, so the buffered parser assembles them losslessly.
228
+ parallelToolCalls: true,
229
+ // Live /v1/models discovery is the authoritative lineup (verified 260709: returns grok-4.5);
230
+ // the static list below is the logged-out fallback seed.
231
+ liveModels: true,
232
+ // 260709 refresh: lineup + metadata from official docs.x.ai (grok-4.5 announced 07-08);
233
+ // grok-composer-2.5-fast kept as account-verified (absent from public docs). Evidence:
234
+ // devlog/model_update/260709_model_refresh/001_xai_lineup.md.
235
+ // 260823: grok-4.20-multi-agent-0309 still returns 400 on Chat Completions, but works
236
+ // on Responses. The server reports this dated id for both it and the floating
237
+ // grok-4.20-multi-agent-beta-latest alias, so expose only the dated deployment id.
238
+ // 260813: grok-4.6 added per docs.x.ai/developers/grok-4-6. Context/vision still match
239
+ // grok-4.5; the reasoning ladder does not — 4.6 adds the documented xhigh rung.
240
+ models: XAI_MODELS,
241
+ // Measured only on grok-4.6 against cli-chat-proxy.grok.com: even an invalid
242
+ // `text.verbosity` value is accepted and low/high/omitted output length is non-monotonic.
243
+ // Apply the resulting opt-out to the whole xAI lineup because `text.verbosity` is an OpenAI
244
+ // Responses parameter absent from xAI's documented API, not because every model was probed.
245
+ // Keep this separate from reasoning-summary support: that bit gates Codex's
246
+ // entire Responses reasoning object, including reasoning.effort.
247
+ modelSupportsVerbosity: Object.fromEntries(XAI_MODELS.map(id => [id, false])),
248
+ // Provider-wide, not merely per-model: `text.verbosity` is an OpenAI Responses parameter
249
+ // absent from xAI's documented API, so a model discovered later has no more support for it
250
+ // than the seeded ones do.
251
+ supportsVerbosity: false,
252
+ defaultModel: "grok-4.5",
253
+ // Grok 4.6/4.5 subscription Responses callers use the native wire with the existing
254
+ // namespace/web-search/replay normalization. Chat remains an explicit modelAdapters
255
+ // opt-in. Multi-agent has no Chat wire and uses Responses under both auth modes.
256
+ // grok-4.6/4.5 are classified OAuth fast-tier models (modelSupportsServiceTier above),
257
+ // so a caller-sent service_tier:"priority" forwards on this lane — the Codex fast-toggle
258
+ // path. Multi-agent keeps its pin: probed 2026-09-13, the gateway downgrades its tier to
259
+ // "default", so forwarding a caller tier would advertise a tier it does not get.
260
+ modelWireDefaults: {
261
+ "grok-4.6": {
262
+ wire: "openai-responses",
263
+ inbound: ["responses"],
264
+ authModes: ["oauth"],
265
+ },
266
+ "grok-4.5": {
267
+ wire: "openai-responses",
268
+ inbound: ["responses"],
269
+ authModes: ["oauth"],
270
+ },
271
+ "grok-4.20-multi-agent-0309": {
272
+ // Even at high effort it emits no reasoning-summary deltas or encrypted replay
273
+ // material. Do not encode that as modelSupportsReasoningSummaries:false: through
274
+ // Codex #1100 that suppresses the entire reasoning object, including the effort
275
+ // that controls this model's agent count. An empty summary pane is harmless.
276
+ // Chat Completions returns 400 for this model, so every inbound uses Responses —
277
+ // `anthropic` included. Omitting it left providerModelWireDefault returning undefined
278
+ // for the Claude Messages lane, so resolveWireProtocolOverride kept xAI's provider-wide
279
+ // openai-chat adapter and sent this model to the wire it 400s on.
280
+ wire: "openai-responses",
281
+ inbound: ["responses", "chat", "anthropic"],
282
+ forwardCallerServiceTier: false,
283
+ },
284
+ },
285
+ // Vision lineup per docs.x.ai model-capabilities/images/understanding: the grok-4.x chat
286
+ // models accept image input (JPEG/PNG, URL or base64). Without this the catalog leaves
287
+ // inputModalities undefined, and deriveComboCatalogModel defaults an undefined member to
288
+ // ["text"] — so any combo containing an xAI target is advertised to Codex as text-only and
289
+ // the app blocks attachments client-side. grok-build-0.1 / grok-composer-2.5-fast stay out
290
+ // (they are already listed in noVisionModels below).
291
+ modelInputModalities: {
292
+ "grok-4.6": ["text", "image"],
293
+ "grok-4.5": ["text", "image"],
294
+ "grok-4.3": ["text", "image"],
295
+ "grok-4.20-multi-agent-0309": ["text", "image"],
296
+ "grok-4.20-0309-reasoning": ["text", "image"],
297
+ "grok-4.20-0309-non-reasoning": ["text", "image"],
298
+ },
299
+ noReasoningModels: ["grok-4.20-0309-non-reasoning", "grok-build-0.1", "grok-composer-2.5-fast"],
300
+ // Replay assistant reasoning_content for grok reasoning models: xAI documents dropped
301
+ // reasoning_content as the top cause of prompt-cache misses on multi-turn conversations
302
+ // (docs.x.ai prompt-caching/multi-turn, verified 2026-07-13 — devlog/_plan/260713_grok_caching).
303
+ // Models that never emit reasoning simply have no thinking parts to replay (no-op).
304
+ preserveReasoningContentModels: ["grok-4.6", "grok-4.5", "grok-4.3", "grok-4.20-0309-reasoning"],
305
+ // grok-4.5 reasoning is always-on with low/medium/high (no off tier, no xhigh).
306
+ // grok-4.6 adds xhigh per docs.x.ai/developers/model-capabilities/text/reasoning;
307
+ // multi-agent accepts the same four wire values to select 4 or 16 collaborators. xAI
308
+ // documents high as the 4.6 default but no multi-agent default, so do not invent one.
309
+ modelReasoningEfforts: {
310
+ "grok-4.6": ["low", "medium", "high", "xhigh"],
311
+ "grok-4.5": ["low", "medium", "high"],
312
+ "grok-4.20-multi-agent-0309": ["low", "medium", "high", "xhigh"],
313
+ },
314
+ modelDefaultReasoningEfforts: { "grok-4.6": "high" },
315
+ modelContextWindows: {
316
+ "grok-4.6": 500_000,
317
+ "grok-4.5": 500_000,
318
+ "grok-4.3": 1_000_000,
319
+ "grok-4.20-multi-agent-0309": 1_000_000,
320
+ "grok-4.20-0309-reasoning": 1_000_000,
321
+ "grok-4.20-0309-non-reasoning": 1_000_000,
322
+ "grok-build-0.1": 256_000,
323
+ },
324
+ noVisionModels: ["grok-build-0.1", "grok-composer-2.5-fast"],
325
+ },
326
+ {
327
+ id: "command-code",
328
+ label: "Command Code - Auth",
329
+ adapter: "command-code",
330
+ baseUrl: "https://api.commandcode.ai",
331
+ authKind: "oauth",
332
+ oauthId: "command-code",
333
+ featured: true,
334
+ note: "Log in with your Command Code account",
335
+ // OAuth needs one initial selection, but the exposed catalog is always discovered from the
336
+ // signed-in account. Do not add a static model list here.
337
+ defaultModel: "deepseek/deepseek-v4-flash",
338
+ liveModels: true,
339
+ modelDiscovery: {
340
+ url: "https://api.commandcode.ai/provider/v1/models",
341
+ maxResponseBytes: 262_144,
342
+ maxModels: 256,
343
+ },
344
+ // These are capability facts from official Command Code model profiles, not seeded models.
345
+ // Unknown/new live models deliberately do not advertise a reasoning picker.
346
+ reasoningEfforts: [],
347
+ modelReasoningEfforts: COMMAND_CODE_MODEL_REASONING_EFFORTS,
348
+ // The DeepSeek vision preview id is preemptive metadata — it is expected to
349
+ // merge into deepseek-v4-flash later.
350
+ modelContextWindows: {
351
+ [`deepseek/${DEEPSEEK_VISION_PREVIEW_MODEL}`]: 1_048_576,
352
+ },
353
+ modelInputModalities: COMMAND_CODE_MODEL_INPUT_MODALITIES,
354
+ defaultMaxOutputTokens: 64_000,
355
+ // The proprietary generate wire has no verified per-request serialization flag.
356
+ parallelToolCalls: false,
357
+ },
358
+ {
359
+ id: "orcarouter-oauth",
360
+ label: "OrcaRouter - Auth",
361
+ adapter: "openai-chat",
362
+ baseUrl: "https://api.orcarouter.ai/v1",
363
+ authKind: "oauth",
364
+ oauthId: "orcarouter-oauth",
365
+ featured: true,
366
+ allowBaseUrlOverride: true,
367
+ defaultModel: "openai/gpt-5.5",
368
+ models: ORCAROUTER_MODELS,
369
+ liveModels: true,
370
+ modelDiscovery: ORCAROUTER_MODEL_DISCOVERY,
371
+ modelReasoningEfforts: ORCAROUTER_MODEL_REASONING_EFFORTS,
372
+ note: "Connect your OrcaRouter account with OAuth 2.0 + PKCE; the issued API key is stored in OpenCodex's existing credential store.",
373
+ },
374
+ {
375
+ id: "anthropic",
376
+ label: "Anthropic Claude",
377
+ adapter: "anthropic",
378
+ baseUrl: "https://api.anthropic.com",
379
+ authKind: "oauth",
380
+ allowBaseUrlOverride: true,
381
+ featured: true,
382
+ oauthId: "anthropic",
383
+ jawcodeBundle: "anthropic",
384
+ note: "Log in with your Claude account",
385
+ models: [...ANTHROPIC_MODELS],
386
+ modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
387
+ modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
388
+ modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
389
+ // Codex omits max_output_tokens; without a provider budget the Anthropic adapter
390
+ // falls back to 8192, which truncates long answers with stop_reason=max_tokens.
391
+ defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
392
+ defaultModel: "claude-sonnet-5",
393
+ },
394
+ {
395
+ id: "anthropic-apikey",
396
+ label: "Anthropic (API key)",
397
+ adapter: "anthropic",
398
+ baseUrl: "https://api.anthropic.com",
399
+ authKind: "key",
400
+ featured: true,
401
+ dashboardUrl: "https://console.anthropic.com/settings/keys",
402
+ jawcodeBundle: "anthropic",
403
+ extraMetadataAliases: ["anthropic-key"],
404
+ note: "Direct Anthropic API billing — no Claude subscription",
405
+ models: [...ANTHROPIC_MODELS],
406
+ liveModels: true,
407
+ modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
408
+ modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
409
+ modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
410
+ defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
411
+ defaultModel: "claude-sonnet-5",
412
+ },
413
+ {
414
+ id: "kimi",
415
+ label: "Kimi",
416
+ adapter: "openai-chat",
417
+ baseUrl: "https://api.kimi.com/coding/v1",
418
+ authKind: "oauth",
419
+ modelSuffixBracketStrip: true,
420
+ // Kimi Code Plan documents a stable session/task prompt_cache_key as required to improve
421
+ // cache hit rates.
422
+ // The chat adapter only forwards a key already on the internal request (Codex's session key,
423
+ // or the one the Claude /v1/messages inbound derives); the adapter itself never invents one.
424
+ // Evidence: https://platform.kimi.com/docs/api/chat
425
+ promptCacheKey: true,
426
+ // Kimi's Responses endpoint rejects hook-provided context between a tool call and
427
+ // its matching result (#4726), the same strict shape DeepSeek exposed in #1292.
428
+ // The flag is inert while this preset uses the Chat wire.
429
+ requiresAdjacentResponsesToolResults: true,
430
+ featured: true,
431
+ oauthId: "kimi",
432
+ jawcodeBundle: "moonshot",
433
+ note: "Log in with your Kimi account",
434
+ models: KIMI_CODING_MODELS,
435
+ defaultModel: "kimi-k2.7-code",
436
+ modelContextWindows: KIMI_CODING_MODEL_CONTEXT_WINDOWS,
437
+ modelInputModalities: KIMI_CODING_MODEL_INPUT_MODALITIES,
438
+ // K3 accepts low/high/max; Codex aliases are normalized by the model-scoped wire map.
439
+ noReasoningModels: KIMI_CODING_NO_REASONING_MODELS,
440
+ modelReasoningEfforts: KIMI_CODING_REASONING_EFFORTS,
441
+ modelDefaultReasoningEfforts: KIMI_CODING_DEFAULT_REASONING_EFFORTS,
442
+ modelReasoningEffortMap: KIMI_CODING_REASONING_EFFORT_MAPS,
443
+ noTemperatureModels: KIMI_LOCKED_PARAMETER_MODELS,
444
+ noTopPModels: KIMI_LOCKED_PARAMETER_MODELS,
445
+ noPenaltyModels: KIMI_LOCKED_PARAMETER_MODELS,
446
+ autoToolChoiceOnlyModels: KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS,
447
+ preserveReasoningContentModels: KIMI_THINKING_MODELS,
448
+ },
449
+ {
450
+ id: "kiro",
451
+ label: "Kiro (AWS CodeWhisperer)",
452
+ adapter: "kiro",
453
+ baseUrl: "https://runtime.us-east-1.kiro.dev",
454
+ authKind: "oauth",
455
+ oauthId: "kiro",
456
+ note: "Import-first: reuses your installed and signed-in Kiro CLI session (requires `kiro-cli login`). Add account logs `kiro-cli` out, switches it through a fresh browser login, stores the account by profile ARN, and restores the previous CLI session on cancellation or failure. Experimental third-party harness — see Kiro ToS.",
457
+ models: KIRO_MODELS,
458
+ defaultModel: "kiro-auto",
459
+ // Kiro speaks CodeWhisperer wire, not OpenAI-style GET /models. Keep the static
460
+ // catalog authoritative so a spurious 2xx from runtime.../models cannot drop seeded ids
461
+ // (e.g. newly listed GPT-5.6 tiers) via live-discovery reconciliation.
462
+ liveModels: false,
463
+ // Per-model context metadata is maintained next to the Kiro model list.
464
+ modelContextWindows: KIRO_MODEL_CONTEXT_WINDOWS,
465
+ modelReasoningEfforts: KIRO_MODEL_REASONING_EFFORTS,
466
+ modelSupportsVerbosity: Object.fromEntries(KIRO_MODELS.map(id => [id, false])),
467
+ },
468
+ {
469
+ // Nous Portal — Nous Research subscription gateway (same backend Hermes Agent
470
+ // uses). OAuth is a device grant (src/oauth/nous.ts): the access token IS the
471
+ // per-request inference JWT (scope inference:invoke), refresh tokens are
472
+ // single-use and rotated on every refresh. Catalog is a mix of paid models
473
+ // (billed against the Portal subscription) and `:free` slugs (e.g.
474
+ // tencent/hy3:free, stepfun/step-3.7-flash:free, inclusionai/ling-3.0-flash:free);
475
+ // free-tier gating is decided live by the Portal per account, so discovery
476
+ // from the signed-in account is authoritative; the static seed below is the
477
+ // logged-out fallback and only lists free models verified on a real account
478
+ // (2026-08-10): the Portal free list is authoritative and currently has
479
+ // exactly 4 :free models: tencent/hy3:free, poolside/laguna-s-2.1:free,
480
+ // stepfun/step-3.7-flash:free, poolside/laguna-xs-2.1:free.
481
+ // inclusionai/ling-3.0-flash:free was removed from the Portal free list
482
+ // (404 on the inference API since 2026-08-07) and must not be seeded.
483
+ id: "nous",
484
+ label: "Nous Portal",
485
+ adapter: "openai-chat",
486
+ baseUrl: "https://inference-api.nousresearch.com/v1",
487
+ authKind: "oauth",
488
+ oauthId: "nous",
489
+ featured: true,
490
+ // Mixed free + paid provider: the free tier is per-model (the `:free`
491
+ // slugs), not a property of the whole provider, so freeTier stays false to
492
+ // avoid implying every model is free.
493
+ freeTier: false,
494
+ dashboardUrl: "https://portal.nousresearch.com",
495
+ defaultModel: "tencent/hy3:free",
496
+ liveModels: true,
497
+ models: ["tencent/hy3:free", "poolside/laguna-s-2.1:free", "stepfun/step-3.7-flash:free", "poolside/laguna-xs-2.1:free"],
498
+ modelDiscovery: {
499
+ // Resolves against effectiveBaseUrl (registry baseUrl .../v1) to the same
500
+ // canonical endpoint https://inference-api.nousresearch.com/v1/models.
501
+ // Nous returns a mixed paid/free catalog whose JSON can exceed 256 KiB;
502
+ // keep the provider-specific limit below the process-wide 4 MiB ceiling.
503
+ path: "models",
504
+ maxResponseBytes: 1_048_576,
505
+ maxModels: 512,
506
+ },
507
+ note: "Nous Research subscription gateway. OAuth device login with your own Portal account; mixed paid + :free models discovered live (fallback seed 2026-08-10: tencent/hy3:free, poolside/laguna-s-2.1:free, stepfun/step-3.7-flash:free, poolside/laguna-xs-2.1:free).",
508
+ },
509
+ {
510
+ id: "openai-apikey",
511
+ label: "OpenAI API",
512
+ adapter: "openai-responses",
513
+ baseUrl: "https://api.openai.com/v1",
514
+ authKind: "key",
515
+ supportsServiceTier: true,
516
+ featured: true,
517
+ dashboardUrl: "https://platform.openai.com/api-keys",
518
+ defaultModel: "gpt-5.5",
519
+ models: ["gpt-5.5", ...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS, ...OPENAI_DAYBREAK_MODELS, "gpt-6-astra"],
520
+ liveModels: true,
521
+ modelContextWindows: { ...OPENAI_API_GPT56_CONTEXT_WINDOWS, ...OPENAI_DAYBREAK_CONTEXT_WINDOWS, "gpt-6-astra": 1_050_000 },
522
+ modelMaxInputTokens: { ...OPENAI_API_GPT56_MAX_INPUT_TOKENS, ...OPENAI_DAYBREAK_MAX_INPUT_TOKENS, "gpt-6-astra": 922_000 },
523
+ modelMaxOutputTokens: { "gpt-6-astra": 128_000 },
524
+ modelInputModalities: Object.fromEntries(
525
+ ["gpt-5.5", ...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS, ...OPENAI_DAYBREAK_MODELS, "gpt-6-astra"]
526
+ .map(id => [id, ["text", "image"]]),
527
+ ),
528
+ modelReasoningEfforts: {
529
+ ...Object.fromEntries(
530
+ [...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS].map(id => [id, OPENAI_API_GPT56_REASONING_EFFORTS]),
531
+ ),
532
+ ...OPENAI_DAYBREAK_REASONING_EFFORTS,
533
+ "gpt-6-astra": ["low", "medium", "high", "xhigh", "max"],
534
+ },
535
+ virtualModels: OPENAI_API_GPT56_VIRTUAL_MODELS,
536
+ },
537
+ /* [Decision Log]
538
+ - 목적과 의도: Reach Meta's Muse Spark models directly on Meta's own Model API, instead of only through the Command Code and OpenCode Zen resellers already in this registry.
539
+ - 기존 구현 및 제약 조건: Meta publishes both POST /v1/responses and POST /v1/chat/completions at https://api.meta.ai/v1, and no API key was issued for this change — every value here comes from the published spec (devlog/_plan/260903_muse_spark_plan_oauth/001).
540
+ - 검토한 주요 대안: register as openai-chat; use provider id "meta"; enable live discovery; wire the Muse Code subscription credential as OAuth.
541
+ - 선택한 방식: an openai-responses key provider under the id "meta-model", with a static two-model roster and no OAuth.
542
+ - 다른 대안 대신 이 방식을 선택한 이유: Meta calls Responses "the recommended default for new work ... OpenAI-compatible and exposes the full feature set", carrying reasoning replay and native input_image that Chat would forfeit. The id is "meta-model" because "meta" would capture the LIVE Command Code selector meta/muse-spark-1.3 at router.ts's provider-prefix branch, and would derive META_API_KEY — the Muse Code CLI's variable, not this API's MODEL_API_KEY.
543
+ - 장점, 단점 및 영향: users reach Muse Spark without a reseller; discovery stays off until an authenticated /v1/models payload is actually observed, so an unseen roster (Meta also serves image and voice families here) cannot leak into the picker.
544
+ */
545
+ {
546
+ id: "meta-model",
547
+ label: "Meta Model API",
548
+ adapter: "openai-responses",
549
+ baseUrl: "https://api.meta.ai/v1",
550
+ authKind: "key",
551
+ dashboardUrl: "https://dev.meta.ai/docs/authentication",
552
+ defaultModel: "muse-spark-1.3",
553
+ models: META_MUSE_MODELS,
554
+ // Static roster: no authenticated /v1/models payload was ever observed (the only
555
+ // contact was an unauthenticated GET returning 401 invalid_api_key), and Meta serves
556
+ // non-agent families on this same base URL. Turning discovery on would publish an
557
+ // unseen roster into the picker.
558
+ liveModels: false,
559
+ // A user may already own a custom provider named "meta-model" pointing elsewhere;
560
+ // without this, registry transport canonicalization would retarget it and send their
561
+ // saved key to Meta.
562
+ preserveCustomDestination: true,
563
+ modelContextWindows: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_CONTEXT_WINDOW])),
564
+ // text+image only. Meta also documents video, audio (degraded on 1.3), and PDF, but
565
+ // the catalog modality enum is text/image and over-advertising poisons the exported
566
+ // client config (see tests/codex-integration/catalog-input-modality-enum.test.ts).
567
+ modelInputModalities: Object.fromEntries(META_MUSE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
568
+ modelReasoningEfforts: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORTS])),
569
+ modelReasoningEffortMap: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORT_MAP])),
570
+ // No defaultMaxOutputTokens: Meta publishes none. The only number in its docs
571
+ // (131072) appears inside a third-party config sample, and the protocol pages call
572
+ // the real limit "model-dependent".
573
+ // Meta names its variable MODEL_API_KEY, but the env var opencodex reads is derived
574
+ // from the provider id (META_MODEL_API_KEY). Saying only Meta's name would send a
575
+ // user to export a variable this proxy never reads.
576
+ note: "Pay-as-you-go Meta Model API. Get a key at https://dev.meta.ai (Meta calls it MODEL_API_KEY; export it here as META_MODEL_API_KEY) — a Meta developer account needs a payment method before it can serve requests, and every call is metered per token. A Muse Code subscription does NOT work here: Meta scopes that credential to the Muse Code CLI and bills any other key pay-as-you-go (dev.meta.ai/docs/muse-code/subscriptions). The Contributor tier (muse-spark-1.3-contributor) is cheap because Meta trains on your prompts — about 92% off input, 95% off output, 99% off cached input; do not send confidential material through it. Muse Spark is also reachable through resellers: command-code carries both tiers, opencode-go serves only muse-spark-1.3-contributor.",
577
+ },
578
+ /* [Decision Log]
579
+ - 목적과 의도: Let an operator who already signed the Muse Code CLI in reach Muse Spark with that credential, instead of provisioning a second key.
580
+ - 기존 구현 및 제약 조건: The CLI stores a pointer at ~/.config/muse/auth.json and the secret in the macOS Keychain (ai.meta.dev.credentials/meta). Measured: the OAuth access_token 401s on /v1/models while the sibling api_key returns 200, so the usable artifact is a static key, not a refreshable token.
581
+ - 검토한 주요 대안: spawn `muse login` and poll; reimplement Meta's device grant; treat it as a second key preset; ship nothing.
582
+ - 선택한 방식: an OAuth provider that imports the existing credential on macOS and accepts a pasted key elsewhere, validates either once, and never spawns or reimplements anything.
583
+ - 다른 대안 대신 이 방식을 선택한 이유: `muse login` has no non-interactive mode, so a spawned child could outlive cancellation, and polling for the pointer file is satisfied instantly by the one already on disk — reimporting the OLD account on a force-login. Reimplementing the grant would mean guessing a client id the vendor does not publish.
584
+ - 장점, 단점 및 영향: no new credential to provision, and the id is distinct from meta-model so neither pool contaminates the other. Meta scopes this credential to its own CLI, so the provider carries a HIGH_RISK ToS warning, a CLI-side warning before any read, and a note that says plainly what is unsupported.
585
+ */
586
+ {
587
+ id: "meta-muse",
588
+ label: "Meta Muse Code (CLI credential)",
589
+ adapter: "openai-responses",
590
+ baseUrl: "https://api.meta.ai/v1",
591
+ // Meta own client sends this on every Muse Code call. We never have, so a future
592
+ // server-side requirement would break every Muse request with no local signal.
593
+ // Declared here rather than in a transport hook so it also covers model discovery
594
+ // (src/oauth/index.ts:1176) and still yields to a user-set header
595
+ // (mergeRegistryStaticHeaders, src/providers/registry.ts:3494).
596
+ staticHeaders: { "x-api-version": "1.0.0" },
597
+ authKind: "oauth",
598
+ oauthId: "meta-muse",
599
+ dashboardUrl: "https://dev.meta.ai",
600
+ defaultModel: "muse-spark-1.3",
601
+ models: META_MUSE_MODELS,
602
+ // Same reason as meta-model: the authenticated roster carries muse-image-1.0 and
603
+ // muse-voice-transcribe-1.0, which this Responses-agent provider cannot drive.
604
+ liveModels: false,
605
+ modelContextWindows: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_CONTEXT_WINDOW])),
606
+ modelInputModalities: Object.fromEntries(META_MUSE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
607
+ modelReasoningEfforts: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORTS])),
608
+ modelReasoningEffortMap: Object.fromEntries(META_MUSE_MODELS.map(id => [id, META_MUSE_REASONING_EFFORT_MAP])),
609
+ note: "Signs in to Meta with a browser device code on any platform, then mints the Muse Code subscription key. That grant is reimplemented from the one the Muse Code CLI performs and has NOT been exercised against Meta from OpenCodex, so treat the first login as unverified. If the Muse Code CLI is already signed in on macOS, the existing key is imported instead of starting a new grant. A pasted key from https://dev.meta.ai still works as a fallback when a device login cannot complete, and faces the same format check and live validation. A device login authenticates as Meta own Muse Code client, which is a stronger claim than reusing a key the CLI already minted. Meta scopes that credential to the Muse Code CLI, so this is an UNSUPPORTED use: Meta does not authorize subscription coverage outside its own CLI, how these calls settle is not observable from the API, and you should treat every call as billable against your account. The key, imported or pasted, is copied into OpenCodex's auth store. For an account signed in with the device login, OpenCodex refreshes Meta's subscription windows on demand from the same key endpoint the login uses, at most once every five minutes. For an imported or pasted key there is no endpoint to query them on demand, so OpenCodex reads them from streaming responses and shows the last observed value with its age; refreshing one then requires another streaming turn, and translated (non-passthrough) turns report none. Rate limits apply per team, not per key. For a supported path use the meta-model provider with your own key (export it as META_MODEL_API_KEY).",
610
+ },
611
+ {
612
+ id: "umans",
613
+ label: "Umans AI Coding Plan",
614
+ adapter: "anthropic",
615
+ baseUrl: "https://api.code.umans.ai",
616
+ authKind: "key",
617
+ featured: true,
618
+ dashboardUrl: "https://app.umans.ai/billing",
619
+ defaultModel: "umans-coder",
620
+ models: UMANS_MODELS,
621
+ modelContextWindows: UMANS_MODEL_CONTEXT_WINDOWS,
622
+ modelInputModalities: UMANS_MODEL_INPUT_MODALITIES,
623
+ note: "Coding plan via Anthropic Messages",
624
+ modelReasoningEfforts: {
625
+ "umans-coder": UMANS_REASONING_EFFORTS,
626
+ "umans-kimi-k2.7": UMANS_REASONING_EFFORTS,
627
+ "umans-flash": UMANS_REASONING_EFFORTS,
628
+ "umans-glm-5.3": UMANS_GLM_53_REASONING_EFFORTS,
629
+ "umans-glm-5.3-flash": UMANS_GLM_53_REASONING_EFFORTS,
630
+ "umans-glm-5.2": UMANS_GLM_REASONING_EFFORTS,
631
+ "umans-glm-5.1": UMANS_GLM_REASONING_EFFORTS,
632
+ "umans-qwen3.6-35b-a3b": UMANS_REASONING_EFFORTS,
633
+ },
634
+ noVisionModels: UMANS_TEXT_ONLY_MODELS,
635
+ escapeBuiltinToolNames: true,
636
+ },
637
+ {
638
+ id: "opencode-go", label: "opencode go", adapter: "openai-chat", baseUrl: "https://opencode.ai/zen/go/v1",
639
+ authKind: "key", featured: true, dashboardUrl: "https://opencode.ai/auth", defaultModel: "kimi-k2.7-code",
640
+ jawcodeBundle: "opencode-go", note: "GLM, DeepSeek, Kimi, Qwen, MiMo…",
641
+ // Zen Go can close a Chat stream after a fully assembled function call without sending
642
+ // finish_reason or [DONE] (#2260). The adapter still rejects incomplete argument JSON.
643
+ openaiChatEofTolerance: true,
644
+ // Go rejects reasoning.encrypted_content with previous_response_id (#3838).
645
+ // Use explicit replay history and the existing stateless Responses policy.
646
+ statelessResponses: true,
647
+ /* [Decision Log]
648
+ - 목적과 의도: Route the exact models OpenCode Go documents on the Responses endpoint — GPT 5.6 Luna, Grok 4.6, and Muse Spark Contributor (#2617).
649
+ - 기존 구현 및 제약 조건: The provider is mixed-wire but its provider-wide `openai-chat` adapter sent Luna to `/chat/completions`; explicit user `modelAdapters` entries must remain authoritative.
650
+ - 검토한 주요 대안: Change the whole provider to Responses; infer the wire from model-family names; add one registry-only exact-model default.
651
+ - 선택한 방식: Declare only the named models as `openai-responses` through the existing registry default mechanism; the map stays an exact-model allowlist rather than a family or provider-wide rule.
652
+ - 다른 대안 대신 이 방식을 선택한 이유: OpenCode Go documents sibling models on Chat or Anthropic endpoints, and an exact registry default preserves both those routes and explicit opt-out precedence.
653
+ - 장점, 단점 및 영향: Each listed model reaches `/responses` from every inbound surface without changing siblings; a future upstream endpoint change requires an evidence-backed registry update.
654
+ */
655
+ modelWireDefaults: {
656
+ "gpt-5.6-luna": "openai-responses",
657
+ "grok-4.6": "openai-responses",
658
+ "muse-spark-1.3-contributor": "openai-responses",
659
+ "muse-spark-1.2-contributor": "openai-responses",
660
+ },
661
+ modelContextWindows: {
662
+ "kimi-k3": KIMI_K3_STANDARD_CONTEXT_WINDOW,
663
+ // Zen Go discovers only the gateway id, so carry DeepSeek's official 1M V4.1
664
+ // window here or Codex falls back to its conservative 128k routed-model default.
665
+ "deepseek-v4.1-flash": 1_048_576,
666
+ // The DeepSeek vision preview id is metadata-only here: the Go roster is
667
+ // discovered live, so it applies the moment the gateway serves the id.
668
+ [DEEPSEEK_VISION_PREVIEW_MODEL]: 1_048_576,
669
+ // Muse Spark Contributor serves a 1,048,576-token (1M) context window over
670
+ // /responses on Zen Go, matching its 1.1 sibling (Meta developer docs, verified 2026-08-28).
671
+ // Without this declaration the catalog falls back to 128k, capping real usable context.
672
+ // 1.3 ships the same window as 1.2 and is served from the same Zen Go roster.
673
+ "muse-spark-1.3-contributor": 1_048_576,
674
+ "muse-spark-1.2-contributor": 1_048_576,
675
+ },
676
+ modelInputModalities: {
677
+ "kimi-k3": ["text", "image"],
678
+ // glm-5.3-flash is a native VLM (docs.z.ai/guides/vlm/glm-5.3-flash). It is
679
+ // deliberately absent from this preset's noVisionModels, which is the
680
+ // correct NEGATIVE half, but with no positive modelInputModalities entry
681
+ // configuredInputModalities returns undefined and the catalog falls through
682
+ // to the ["text"] floor. The same model is already declared ["text","image"]
683
+ // on the zai and zhipu-bigmodel-coding presets, so the registry described
684
+ // one model two ways (#4505).
685
+ "glm-5.3-flash": ["text", "image"],
686
+ // Experimental DeepSeek vision preview — expected to merge into deepseek-v4-flash later.
687
+ [DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
688
+ // This route is text-only upstream — it is already listed in this preset's
689
+ // noVisionModels, which routes images through the proxy's vision sidecar and
690
+ // makes the catalog advertise image input on its behalf. The positive
691
+ // text-only declaration is what reaches an EXISTING install: derive.ts fills
692
+ // noVisionModels all-or-nothing, so a config persisted before this id joined
693
+ // the list keeps a stale list, the sidecar predicate never matches, the row
694
+ // carries no modality at all, and any combo containing it collapses to
695
+ // ["text"] (#4505). modelInputModalities IS per-key filled, so this
696
+ // declaration lands on old configs. It states the route's real upstream
697
+ // capability and keeps the sidecar explicitly distinct from native vision.
698
+ "deepseek-v4.1-flash": ["text"],
699
+ // Muse Spark Contributor is natively multimodal on Zen Go: it accepts input_image
700
+ // parts over /responses (probed 2026-08-26). Without this declaration the catalog
701
+ // advertises it text-only and the Codex app blocks image attachments client-side with
702
+ // "This model does not support image inputs" before the request ever reaches the proxy.
703
+ // 1.3 is the same-shaped successor and Command Code documents it as multimodal.
704
+ "muse-spark-1.3-contributor": ["text", "image"],
705
+ "muse-spark-1.2-contributor": ["text", "image"],
706
+ },
707
+ modelReasoningEfforts: {
708
+ "gpt-5.6-luna": OPENAI_API_GPT56_REASONING_EFFORTS,
709
+ "grok-4.6": ["low", "medium", "high", "xhigh"],
710
+ "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
711
+ "glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
712
+ "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
713
+ "qwen3.8-max": QWEN38_REASONING_EFFORTS,
714
+ "kimi-k3": KIMI_CODING_K3_REASONING_EFFORTS,
715
+ "kimi-k2.7-code": [],
716
+ "kimi-k2.7-code-highspeed": [],
717
+ ...Object.fromEntries(OPENCODE_GO_THINKING_TOGGLE_MODELS.map(id => [id, THINKING_TOGGLE_EFFORTS])),
718
+ ...Object.fromEntries(OPENCODE_GO_THINKING_BUDGET_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
719
+ ...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, deepseekThinkingEffortsFor(id)])),
720
+ },
721
+ modelDefaultReasoningEfforts: { "grok-4.6": "high", "kimi-k3": "max" },
722
+ // glm-5.2 uses identity labels now that `max` is a native Codex level (no alias map);
723
+ // the thinking-toggle map is a REAL wire alias (effort -> enabled/disabled) and stays.
724
+ modelReasoningEffortMap: {
725
+ "kimi-k3": KIMI_CODING_K3_REASONING_EFFORT_MAP,
726
+ ...Object.fromEntries(OPENCODE_GO_THINKING_TOGGLE_MODELS.map(id => [id, THINKING_TOGGLE_MAP])),
727
+ ...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, deepseekReasoningMapFor(id)])),
728
+ },
729
+ modelSupportsReasoningSummaries: {
730
+ "glm-5.3": true,
731
+ "glm-5.3-flash": true,
732
+ "glm-5.2": true,
733
+ "glm-5.1": true,
734
+ "glm-5": true,
735
+ ...Object.fromEntries(DEEPSEEK_GATEWAY_THINKING_MODELS.map(id => [id, true])),
736
+ },
737
+ thinkingToggleModels: OPENCODE_GO_THINKING_TOGGLE_MODELS,
738
+ /*
739
+ * The Go-specific list, not the shared one. The shared `THINKING_BUDGET_MODELS` also
740
+ * carries Neuralwatt-only ids (`qwen3.5-397b`, `qwen3.6-35b`) that this preset never
741
+ * gives a ladder to, so a live roster serving one of them armed the thinking-budget
742
+ * wire path with nothing to advertise: the catalog showed no effort control while the
743
+ * adapter still translated effort into `thinking_budget`.
744
+ */
745
+ thinkingBudgetModels: OPENCODE_GO_THINKING_BUDGET_MODELS,
746
+ noReasoningModels: ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
747
+ // Text-only Zen Go models (jawcode metadata) — the vision sidecar describes images for
748
+ // every model listed here (and the catalog advertises image input on their behalf).
749
+ // Kimi K2.7 Code accepts text+image+video: do NOT list it here.
750
+ noVisionModels: [
751
+ "glm-5.3", "glm-5.2", "glm-5", "glm-5.1",
752
+ "deepseek-v4.1-flash", "deepseek-v4-flash",
753
+ "mimo-v2-pro", "mimo-v2.5-pro",
754
+ "minimax-m2.5", "minimax-m2.7",
755
+ "qwen3.7-max",
756
+ ],
757
+ noTemperatureModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
758
+ noTopPModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
759
+ noPenaltyModels: ["kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
760
+ autoToolChoiceOnlyModels: ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"],
761
+ // Issue #78: DeepSeek V4 thinking mode requires reasoning_content replay on tool-call turns.
762
+ preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed", ...DEEPSEEK_GATEWAY_THINKING_MODELS],
763
+ /*
764
+ * Issues #1338 / #1415: this gateway answers a `response_format` of type
765
+ * `json_schema` with HTTP 400 `This response_format type is unavailable now`
766
+ * (quoted from the upstream body as `Error from provider (Console Go)`), which
767
+ * breaks every Codex auto-review turn on a DeepSeek route. #1424 shipped the
768
+ * operator-side opt-out; operators have been applying it by hand ever since.
769
+ * The reported rejection is type-specific, so this narrower list downgrades the
770
+ * request to `json_object` instead of claiming the whole field is unavailable.
771
+ */
772
+ noJsonSchemaModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS],
773
+ },
774
+ {
775
+ id: "neuralwatt",
776
+ label: "Neuralwatt Cloud",
777
+ adapter: "openai-chat",
778
+ baseUrl: "https://api.neuralwatt.com/v1",
779
+ authKind: "key",
780
+ dashboardUrl: "https://portal.neuralwatt.com",
781
+ defaultModel: "glm-5.3",
782
+ // 2026-07-10 live /v1/models: K2.5 rows were removed and GLM-5.2 short variants added.
783
+ // 260814: the glm-5.3 quartet is speculative; live discovery is authoritative and drops
784
+ // any id Neuralwatt has not published yet.
785
+ // Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md and https://api.neuralwatt.com/v1/models.
786
+ models: [
787
+ "glm-5.3", "glm-5.3-fast", "glm-5.3-short", "glm-5.3-short-fast",
788
+ "glm-5.3-flash",
789
+ "glm-5.2", "glm-5.2-fast", "glm-5.2-short", "glm-5.2-short-fast",
790
+ "kimi-k2.6", "kimi-k2.6-fast",
791
+ "kimi-k2.7-code",
792
+ "qwen3.5-397b", "qwen3.5-397b-fast", "qwen3.6-35b", "qwen3.6-35b-fast",
793
+ ],
794
+ // Neuralwatt's /v1/models metadata is authoritative; these static hints are the offline fallback.
795
+ modelReasoningEfforts: {
796
+ "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
797
+ "glm-5.3-fast": [],
798
+ "glm-5.3-short": ZAI_GLM_53_REASONING_EFFORTS,
799
+ "glm-5.3-short-fast": [],
800
+ // No `-fast`/`-short` variants are asserted for the flash tier: those suffixes
801
+ // encode routing Neuralwatt documents per model, and this seed has no source for them.
802
+ "glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
803
+ "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
804
+ "glm-5.2-fast": [],
805
+ "glm-5.2-short": ZAI_GLM_52_REASONING_EFFORTS,
806
+ "glm-5.2-short-fast": [],
807
+ "kimi-k2.6": [],
808
+ "kimi-k2.6-fast": [],
809
+ "kimi-k2.7-code": [],
810
+ // Qwen3.x uses thinking_budget, NOT graded reasoning_effort; the adapter maps the five
811
+ // Codex picker levels onto budget fractions.
812
+ "qwen3.5-397b": THINKING_BUDGET_EFFORTS,
813
+ "qwen3.5-397b-fast": [],
814
+ "qwen3.6-35b": THINKING_BUDGET_EFFORTS,
815
+ "qwen3.6-35b-fast": [],
816
+ },
817
+ thinkingBudgetModels: THINKING_BUDGET_MODELS,
818
+ noReasoningModels: ["glm-5.3-fast", "glm-5.3-short-fast", "glm-5.2-fast", "glm-5.2-short-fast", "kimi-k2.6-fast", "qwen3.5-397b-fast", "qwen3.6-35b-fast"],
819
+ noVisionModels: ["glm-5.3", "glm-5.3-fast", "glm-5.3-short", "glm-5.3-short-fast", "glm-5.2", "glm-5.2-fast", "glm-5.2-short", "glm-5.2-short-fast", "qwen3.5-397b", "qwen3.5-397b-fast"],
820
+ noTemperatureModels: ["kimi-k2.7-code"],
821
+ noTopPModels: ["kimi-k2.7-code"],
822
+ noPenaltyModels: ["kimi-k2.7-code"],
823
+ autoToolChoiceOnlyModels: ["kimi-k2.7-code"],
824
+ preserveReasoningContentModels: NEURALWATT_REASONING_HISTORY_MODELS,
825
+ },
826
+ {
827
+ id: "openrouter",
828
+ label: "OpenRouter",
829
+ adapter: "openai-chat",
830
+ baseUrl: "https://openrouter.ai/api/v1",
831
+ authKind: "key",
832
+ featured: true,
833
+ dashboardUrl: "https://openrouter.ai/keys",
834
+ jawcodeBundle: "openrouter",
835
+ models: ["anthropic/claude-sonnet-5", ...OPENROUTER_GPT56_MODELS],
836
+ modelContextWindows: {
837
+ "anthropic/claude-sonnet-5": 1_000_000,
838
+ ...OPENROUTER_GPT56_CONTEXT_WINDOWS,
839
+ },
840
+ // OpenRouter documents priority support for OpenAI endpoints, but not Anthropic. Keep the
841
+ // provider unclassified and opt in only the exact OpenAI-backed slugs we ship. These facts
842
+ // belong only to the canonical destination; a same-named custom gateway is unknown to us.
843
+ modelServiceTierCapabilityBaseUrlGuard: isCanonicalOpenRouterTarget,
844
+ modelSupportsServiceTier: {
845
+ "openai/gpt-5.6-sol": true,
846
+ "openai/gpt-5.6-terra": true,
847
+ "openai/gpt-5.6-luna": true,
848
+ },
849
+ // Deliberately no OpenRouter route pin: it bills the endpoint actually used and reports the
850
+ // actual service_tier. B0 confirmation therefore owns downgrade safety. Forcing `only` plus
851
+ // `allow_fallbacks:false` would turn a graceful priority-capacity fallback into a hard failure.
852
+ },
853
+ {
854
+ // Primary sources checked 2026-08-02:
855
+ // - docs.cline.bot/getting-started/clinepass publishes this exact catalog and explicitly
856
+ // authorizes using the full slugs through Cline's external API.
857
+ // - docs.cline.bot/api/chat-completions and /api/errors define the endpoint, reasoning delta,
858
+ // and choice-scoped mid-stream error contract.
859
+ // - Cline's official catalog source resolves per-model capabilities through OpenRouter data;
860
+ // the static context/modality snapshot below was cross-checked against that catalog.
861
+ // - cline.bot/tos identifies Cline Bot Inc. as the operator. Maintenance owner: @lidge-jun.
862
+ id: "cline-pass",
863
+ label: "ClinePass",
864
+ adapter: "openai-chat",
865
+ baseUrl: "https://api.cline.bot/api/v1",
866
+ authKind: "key",
867
+ dashboardUrl: "https://app.cline.bot",
868
+ defaultModel: "cline-pass/kimi-k3",
869
+ models: CLINE_PASS_MODELS,
870
+ modelContextWindows: CLINE_PASS_MODEL_CONTEXT_WINDOWS,
871
+ modelInputModalities: CLINE_PASS_MODEL_INPUT_MODALITIES,
872
+ noVisionModels: CLINE_PASS_TEXT_ONLY_MODELS,
873
+ // Live-probed 2026-08-13 across every static ClinePass model: the gateway accepts and
874
+ // validates low/medium/high/xhigh/max, and rejects an invalid sentinel. Preserve the
875
+ // caller's requested tier and let ClinePass own any backend-specific normalization.
876
+ reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
877
+ reasoningWireFormat: "gateway-object",
878
+ preserveCustomDestination: true,
879
+ note: "ClinePass subscription API. Uses a Cline API key and the full cline-pass/<model> upstream slug; quota is shared across the account's rolling 5-hour, weekly, and monthly limits.",
880
+ },
881
+ // Cline API (usage-billing): OpenAI-compatible Chat Completions. Model IDs follow the
882
+ // OpenRouter-style `provider/model` convention. Live /models discovery is key-gated (401
883
+ // without auth), so the static seed is the cold-start fallback. Evidence: docs.cline.bot/api/*.
884
+ {
885
+ id: "cline",
886
+ label: "Cline",
887
+ adapter: "openai-chat",
888
+ baseUrl: "https://api.cline.bot/api/v1",
889
+ authKind: "key",
890
+ dashboardUrl: "https://app.cline.bot",
891
+ liveModels: true,
892
+ defaultModel: "anthropic/claude-sonnet-4-6",
893
+ models: [
894
+ "anthropic/claude-sonnet-4-6",
895
+ "openai/gpt-4o",
896
+ "google/gemini-2.5-pro",
897
+ "deepseek/deepseek-chat",
898
+ "minimax/minimax-m2.5",
899
+ ],
900
+ preserveCustomDestination: true,
901
+ note: "Cline usage-billing API: one key, 100+ models, OpenRouter-style ids. Promotional free models are IDE/CLI-only per Cline docs; minimax/minimax-m2.5 is the documented API free experimentation model.",
902
+ },
903
+ {
904
+ // OrcaRouter: OpenAI-compatible adaptive router (api.orcarouter.ai). The public live
905
+ // catalog is authoritative; model ids and input modalities are never maintained here.
906
+ id: "orcarouter", label: "OrcaRouter - API", adapter: "openai-chat", baseUrl: "https://api.orcarouter.ai/v1",
907
+ authKind: "key", dashboardUrl: "https://www.orcarouter.ai/console",
908
+ // The catalog is public, so a successful /models probe cannot validate a submitted key.
909
+ apiKeyValidation: "unknown",
910
+ // Standard sponsor under SPONSORS.md (agreement signed 2026-09-07). Pins the row in the
911
+ // picker and adds the chip; nothing about routing or defaults changes.
912
+ sponsor: { tier: "standard", url: "https://www.orcarouter.ai/?utm_source=opencodex&utm_medium=readme" },
913
+ defaultModel: "openai/gpt-5.5",
914
+ models: ORCAROUTER_MODELS,
915
+ liveModels: true,
916
+ modelDiscovery: ORCAROUTER_MODEL_DISCOVERY,
917
+ // Catalog discovery owns WHICH models exist. These entries only retain verified
918
+ // request-shaping facts that the upstream catalog does not currently publish.
919
+ modelReasoningEfforts: ORCAROUTER_MODEL_REASONING_EFFORTS,
920
+ note: "OpenAI-compatible adaptive router. Models and multimodal capabilities are discovered live from the public chat catalog. Use the OrcaRouter account entry for PKCE login.",
921
+ },
922
+ {
923
+ // PackyCode: API relay (packyapi.com) for Claude Code, Codex, Gemini and more. Codex traffic
924
+ // uses the OpenAI-compatible host from their Codex/Kimi Code guides (docs.packyapi.com):
925
+ // https://cf.api.fan/v1 — GET /v1/models answers 401 without a key, so the host is live and
926
+ // discovery narrows to what the key's token group allows. Model ids are bare OpenAI-style
927
+ // ids (the Codex token group lists gpt-5.5 / gpt-5.1-codex).
928
+ // Standard sponsor under SPONSORS.md; the dashboardUrl carries their affiliate code.
929
+ id: "packycode", label: "PackyCode", adapter: "openai-chat", baseUrl: "https://cf.api.fan/v1",
930
+ authKind: "key", dashboardUrl: "https://www.packyapi.com/register?aff=k5KT",
931
+ sponsor: { tier: "standard", url: "https://www.packyapi.com/register?aff=k5KT" },
932
+ defaultModel: "gpt-5.5",
933
+ models: ["gpt-5.5", "gpt-5.1-codex"],
934
+ liveModels: true,
935
+ // New key preset: opt into collision preservation so a row named `packycode` that a user
936
+ // points at a different PackyCode host keeps its own destination instead of being pulled
937
+ // back onto the Codex endpoint below.
938
+ preserveCustomDestination: true,
939
+ note: "API relay for Claude Code, Codex, Gemini and more. Create a Codex-group token at packyapi.com; live discovery lists what the token group allows.",
940
+ },
941
+ {
942
+ // BizRouter: Korean enterprise LLM gateway (api.bizrouter.ai). Model ids are
943
+ // vendor-namespaced (`<vendor>/<model>`) and pass through to the upstream as-is.
944
+ // Live-verified 2026-07-24: /v1/chat/completions accepts the `tools` field and
945
+ // streams, and GET /v1/models returns the per-API-key allowed catalog in the
946
+ // OpenAI list shape, so live model discovery narrows to what the key can use.
947
+ id: "bizrouter", label: "BizRouter", adapter: "openai-chat", baseUrl: "https://api.bizrouter.ai/v1",
948
+ authKind: "key", dashboardUrl: "https://bizrouter.ai/settings/keys",
949
+ defaultModel: "openai/gpt-5.6-sol",
950
+ models: ["openai/gpt-5.6-sol", "anthropic/claude-sonnet-5", "google/gemini-3.5-flash"],
951
+ note: "Korean enterprise LLM gateway. Per-key allowed models are discovered live from /v1/models. Full catalog: https://bizrouter.ai/models",
952
+ },
953
+ { id: "groq", label: "Groq", adapter: "openai-chat", baseUrl: "https://api.groq.com/openai/v1", authKind: "key", featured: true, dashboardUrl: "https://console.groq.com/keys" },
954
+ // 2026-07-10 Gemini API refresh: Tier-2 ai.google.dev evidence recorded in
955
+ // devlog/_plan/260710_provider_hardening/001_research_frontier.md.
956
+ {
957
+ id: "google", label: "Google Gemini", adapter: "google", baseUrl: "https://generativelanguage.googleapis.com", authKind: "key", featured: true,
958
+ dashboardUrl: "https://aistudio.google.com/apikey", defaultModel: "gemini-3.5-flash", models: ["gemini-3.8-flash", "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro-preview", "gemini-3.7-flash"],
959
+ modelContextWindows: { "gemini-3.8-flash": 1_048_576, "gemini-3.6-flash": 1_048_576, "gemini-3.5-flash": 1_000_000, "gemini-3.5-flash-lite": 1_048_576, "gemini-3.7-flash": 1_048_576 },
960
+ modelInputModalities: { "gemini-3.8-flash": ["text", "image"], "gemini-3.6-flash": ["text", "image"], "gemini-3.5-flash-lite": ["text", "image"], "gemini-3.7-flash": ["text", "image"] },
961
+ modelReasoningEfforts: {
962
+ // 3.7 and 3.8 omit `minimal`: Google documents it as a validation error on both model
963
+ // pages, so advertising it hands the user a rung the API rejects. 3.5/3.6 keep theirs —
964
+ // their pages still list it, and this unit has no evidence to change them.
965
+ "gemini-3.8-flash": ["low", "medium", "high"],
966
+ "gemini-3.7-flash": ["low", "medium", "high"],
967
+ "gemini-3.6-flash": ["minimal", "low", "medium", "high"],
968
+ "gemini-3.5-flash": ["minimal", "low", "medium", "high"],
969
+ "gemini-3.1-pro-preview": ["low", "medium", "high"],
970
+ },
971
+ jawcodeBundle: "google", extraMetadataAliases: ["gemini"],
972
+ },
973
+ // 2026-07-10: defaultModel is frozen pending Vertex-specific Tier-2 evidence; Gemini API
974
+ // evidence from ai.google.dev does not establish Vertex publisher availability.
975
+ { id: "google-vertex", label: "Google Vertex AI", adapter: "google", baseUrl: "https://aiplatform.googleapis.com", authKind: "key", dashboardUrl: "https://console.cloud.google.com/vertex-ai", defaultModel: "gemini-3-pro", googleMode: "vertex", jawcodeBundle: "google", extraMetadataAliases: ["gemini-vertex"] },
976
+ // Antigravity discovers models with a POST to the CCA `:fetchAvailableModels` RPC, which
977
+ // `buildModelsRequest` already built by hand. Declaring it here changes no request URL — the
978
+ // relative path resolves to the same destination — but it lets `isRegistryModelDiscoveryUrl`
979
+ // prove that URL, which is what admits a Clash/Surge/Mihomo TUN fake-IP answer (#4261). The
980
+ // path must stay RELATIVE: this row sets `allowBaseUrlOverride`, and an absolute `url` would
981
+ // retarget a user's custom base back to Google. A leading `./` is required because a bare
982
+ // `v1internal:` reads as a URL scheme and `providerModelDiscoverySpecError` rejects it.
983
+ { id: "google-antigravity", alias: "agy", label: "Google Antigravity", adapter: "google", baseUrl: "https://daily-cloudcode-pa.googleapis.com", authKind: "oauth", allowBaseUrlOverride: true, dashboardUrl: "https://antigravity.google", models: ANTIGRAVITY_MODELS, liveModels: true, defaultModel: "gemini-3.8-flash", modelContextWindows: ANTIGRAVITY_MODEL_CONTEXT_WINDOWS, modelInputModalities: ANTIGRAVITY_MODEL_INPUT_MODALITIES, modelReasoningEfforts: ANTIGRAVITY_MODEL_EFFORTS, googleMode: "cloud-code-assist", showThinkingSummary: true, jawcodeBundle: "google", extraMetadataAliases: ["antigravity", "gemini-antigravity"], modelDiscovery: { path: "./v1internal:fetchAvailableModels" } },
984
+ { id: "azure-openai", label: "Azure OpenAI", adapter: "azure-openai", baseUrl: "https://{resource}.openai.azure.com/openai", authKind: "key", featured: true, dashboardUrl: "https://portal.azure.com" },
985
+ { id: "ollama", label: "Ollama (local)", adapter: "openai-chat", baseUrl: "http://localhost:11434/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — key usually blank" },
986
+ { id: "vllm", label: "vLLM (local)", adapter: "openai-chat", baseUrl: "http://localhost:8000/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — key usually blank" },
987
+ { id: "lm-studio", label: "LM Studio (local)", adapter: "openai-chat", baseUrl: "http://localhost:1234/v1", authKind: "local", allowPrivateNetworkByDefault: true, allowBaseUrlOverride: true, featured: true, note: "Local — no key needed" },
988
+ {
989
+ id: "deepseek",
990
+ label: "DeepSeek",
991
+ baseUrl: "https://api.deepseek.com",
992
+ adapter: "openai-chat",
993
+ authKind: "key",
994
+ dashboardUrl: "https://platform.deepseek.com/api_keys",
995
+ // Route DeepSeek's own catalog bundle so routed rebuilds restore the official
996
+ // context window from the vendored model-metadata bundle instead of falling
997
+ // back to the 128k strict-fields default (scripts/model-metadata.source.json,
998
+ // verified 2026-08-08).
999
+ jawcodeBundle: "deepseek",
1000
+ // deepseek-chat/deepseek-reasoner were deprecated upstream on 2026-07-24 15:59 UTC;
1001
+ // the current official identifier is deepseek-flash. They stay in
1002
+ // the list only as compatibility aliases so existing saved configs and requests
1003
+ // keep validating and routing (they previously mapped to v4-flash; devlog
1004
+ // _fin/260710_provider_hardening/002_research_cn.md). The current offerings are
1005
+ // V4.1-Flash — defaultModel and the model-specific wiring below use its live id.
1006
+ // Keep the legacy vision-preview alias; see DEEPSEEK_VISION_PREVIEW_MODEL.
1007
+ models: ["deepseek-chat", "deepseek-reasoner", ...DEEPSEEK_NATIVE_THINKING_MODELS, DEEPSEEK_VISION_PREVIEW_MODEL],
1008
+ // V4.1-Flash is the current first-party offering; `deepseek-v4-flash` now routes there
1009
+ // as a compatibility alias, so a new install should ask for the live id by name.
1010
+ defaultModel: "deepseek-flash",
1011
+ // Official DeepSeek Codex setup (codex-deepseek-setup.sh) advertises 1,048,576
1012
+ // for both V4 models; the older 1,000,000 figure was a rounded approximation.
1013
+ modelContextWindows: { "deepseek-flash": 1_048_576, "deepseek-v4-flash": 1_048_576, [DEEPSEEK_VISION_PREVIEW_MODEL]: 1_048_576 },
1014
+ modelInputModalities: {
1015
+ "deepseek-flash": ["text", "image"],
1016
+ [DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
1017
+ },
1018
+ // DeepSeek documents both V4 models as native Responses API models adapted for Codex
1019
+ // (model table marks Responses API ✓ for flash and pro; the /responses reference lists
1020
+ // both ids as accepted `model` values — verified 2026-08-13 with the V4 Pro GA,
1021
+ // version label DeepSeek-V4-Pro-0813).
1022
+ modelWireDefaults: {
1023
+ // Codex speaks Responses natively and DeepSeek ships a Codex-compatible
1024
+ // apply_patch tool on that wire, so a Responses inbound goes straight out with
1025
+ // no translation. Claude Code and OpenAI-compatible clients keep the
1026
+ // provider-wide Chat wire: DeepSeek serves Chat Completions natively too, so
1027
+ // translating them into Responses would add a hop onto our newest upstream path
1028
+ // for no gain.
1029
+ "deepseek-v4-flash": { wire: "openai-responses", inbound: ["responses"] },
1030
+ // Same Responses contract as the V4 ids it succeeds; without this row the new
1031
+ // default would fall back to the provider-wide Chat wire.
1032
+ "deepseek-flash": { wire: "openai-responses", inbound: ["responses"] },
1033
+ },
1034
+ // The #875-era bounded-JSON force (`modelResponsesUpstreamStreaming`) is retired
1035
+ // for this entry: the official guide documents a `response.completed` /
1036
+ // `response.incomplete` / `response.failed` terminal with NO `data: [DONE]`
1037
+ // sentinel, and live probes (2026-08-07, including the tool-result replay shape
1038
+ // that originally stalled) close on the terminal. The relay's terminal boundary
1039
+ // (src/server/relay.ts) already cuts the stream at that event and synthesizes
1040
+ // `[DONE]`, so forcing stream:false only delayed every byte until generation
1041
+ // finished (28-46 s of silence on long turns). The registry knob itself remains
1042
+ // for providers that need it — re-adding one line here restores the old policy.
1043
+ // Evidence: https://api-docs.deepseek.com/guides/responses_api/ +
1044
+ // devlog/_fin/260807_deepseek_responses_streaming/000_plan.md.
1045
+ // Current official streams normally carry a real terminal; retain a narrow grace
1046
+ // repair for the historical shape that closes after a complete graph without one.
1047
+ modelResponsesTerminalRepair: { "deepseek-flash": { graceMs: 5_000 }, "deepseek-v4-flash": { graceMs: 5_000 } },
1048
+ // DeepSeek's Responses route emits bare UUID item ids, which leave Codex
1049
+ // clients stuck on an uncommitted turn (#938). Client-facing only — raw
1050
+ // continuation snapshots keep the upstream ids.
1051
+ responsesItemIdRepair: { repairInvalidIds: true, repairMissingTerminalIds: true },
1052
+ // DeepSeek's Responses route is `POST /responses` with no `/v1` segment. Without
1053
+ // this the passthrough adapter falls back to its legacy `/v1/responses`
1054
+ // construction and the wire above can never route.
1055
+ // Evidence: https://api-docs.deepseek.com/api/create-response/
1056
+ responsesPath: "/responses",
1057
+ // DeepSeek's Responses reference does not list `service_tier`; unsupported
1058
+ // parameters are documented as silently ignored, but the fail-closed policy
1059
+ // strips the field rather than forwarding a knob the upstream never asked for.
1060
+ supportsServiceTier: false,
1061
+ // DeepSeek's Responses compatibility guide accepts plaintext reasoning items and
1062
+ // merges them into the adjacent assistant message, so replayed reasoning must
1063
+ // not be blanked the way the ChatGPT backend requires. (Whether the Responses
1064
+ // route REQUIRES replay on tool-call continuations is an inference from the
1065
+ // Chat Thinking-Mode docs, not a confirmed Responses contract.)
1066
+ preserveResponsesReasoningContent: true,
1067
+ // "The API is stateless: responses and conversations are not stored on the
1068
+ // server." https://api-docs.deepseek.com/api/create-response/
1069
+ statelessResponses: true,
1070
+ // DeepSeek rejects a valid Codex continuation when hook-provided developer
1071
+ // context splits a call from its result (#1292); parallel calls remain one
1072
+ // reasoning-bearing assistant batch rather than being split per pair (#1477).
1073
+ requiresAdjacentResponsesToolResults: true,
1074
+ // DeepSeek exec tool results can be present-but-empty (a script that ran without
1075
+ // calling text(...)); annotate them so routed models do not silently accept an
1076
+ // empty result or re-issue the same call.
1077
+ annotateEmptyToolOutputs: true,
1078
+ /* [Decision Log]
1079
+ - 목적: DeepSeek V4 thinking mode multi-turn/tool-call requests must replay prior assistant reasoning_content.
1080
+ - 대안 분석: Globally preserve reasoning_content for all OpenAI-compatible models; preserve it for legacy deepseek-reasoner too; mark only V4 thinking models in registry metadata.
1081
+ - 선택 근거: DeepSeek V4 thinking mode requires history replay, while older DeepSeek reasoner has different compatibility rules. A model-scoped registry flag fixes built-in and stale saved configs without broad provider regressions.
1082
+ */
1083
+ modelReasoningEfforts: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, deepseekThinkingEffortsFor(id)])),
1084
+ modelReasoningEffortMap: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, deepseekReasoningMapFor(id)])),
1085
+ modelSupportsReasoningSummaries: Object.fromEntries(DEEPSEEK_NATIVE_THINKING_MODELS.map(id => [id, true])),
1086
+ preserveReasoningContentModels: DEEPSEEK_NATIVE_THINKING_MODELS,
1087
+ // #4436: first-party deepseek-flash accepts native images on Chat and Responses.
1088
+ // Keep unprobed compatibility aliases on the #88 sidecar path. This must be fixed
1089
+ // here: router enrichment unions this list with saved config, so config cannot remove it.
1090
+ noVisionModels: ["deepseek-chat", "deepseek-reasoner", "deepseek-v4-flash"],
1091
+ },
1092
+ // llama-3.3-70b was deprecated by Cerebras on 2026-02-16. Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
1093
+ { id: "cerebras", label: "Cerebras", baseUrl: "https://api.cerebras.ai/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://cloud.cerebras.ai/platform/apikeys", defaultModel: "gpt-oss-120b" },
1094
+ {
1095
+ // Primary sources checked 2026-08-08:
1096
+ // - https://chutes.ai/pricing documents the shared llm.chutes.ai/v1 OpenAI-compatible
1097
+ // gateway, Bearer API keys, and chat completions. Its public
1098
+ // https://llm.chutes.ai/v1/models response supplies supported_features for filtering.
1099
+ // - https://chutes.ai/terms identifies Chutes Global Corp as the platform operator, applies
1100
+ // to API consumers, and directs production/high-volume automated inference to PAYGO.
1101
+ // Maintainer: @olddonkey; no affiliation with Chutes.
1102
+ id: "chutes",
1103
+ label: "Chutes",
1104
+ baseUrl: "https://llm.chutes.ai/v1",
1105
+ adapter: "openai-chat",
1106
+ authKind: "key",
1107
+ dashboardUrl: "https://chutes.ai/auth/start",
1108
+ liveModels: true,
1109
+ preserveCustomDestination: true,
1110
+ // The public model catalog cannot prove that a supplied Bearer key is valid.
1111
+ apiKeyValidation: "unknown",
1112
+ // Chutes documents tool calling, but not a provider-wide parallel tool-call contract.
1113
+ parallelToolCalls: false,
1114
+ // The live catalog reports reasoning support, but not a stable effort ladder.
1115
+ reasoningEfforts: [],
1116
+ modelDiscovery: {
1117
+ path: "models",
1118
+ maxResponseBytes: 256 * 1024,
1119
+ maxModels: 128,
1120
+ filter: {
1121
+ // The shared LLM catalog also contains rows without native tool support. Codex needs a
1122
+ // complete agent loop, so admit only rows whose live metadata advertises tools.
1123
+ allOf: [{ path: ["supported_features"], containsAny: ["tools"] }],
1124
+ },
1125
+ },
1126
+ note: "Shared OpenAI-compatible LLM gateway only; live discovery exposes tool-capable rows. User-deployed custom Chute endpoints and non-LLM APIs require a custom provider.",
1127
+ },
1128
+ {
1129
+ id: "deepinfra",
1130
+ label: "DeepInfra",
1131
+ baseUrl: "https://api.deepinfra.com/v1/openai",
1132
+ adapter: "openai-chat",
1133
+ authKind: "key",
1134
+ dashboardUrl: "https://deepinfra.com/dash/api_keys",
1135
+ liveModels: true,
1136
+ preserveCustomDestination: true,
1137
+ modelDiscovery: {
1138
+ // DeepInfra documents the OpenAI model catalog outside the chat-compatible `/v1/openai`
1139
+ // namespace, so keep this destination registry-owned instead of deriving it from baseUrl.
1140
+ url: "https://api.deepinfra.com/v1/models",
1141
+ maxResponseBytes: 512 * 1024,
1142
+ maxModels: 512,
1143
+ filter: {
1144
+ allOf: [{ path: ["metadata", "tags"], containsAny: ["chat"] }],
1145
+ },
1146
+ },
1147
+ note: "OpenAI-compatible chat models only; live discovery excludes non-chat rows from DeepInfra's mixed model catalog.",
1148
+ },
1149
+ {
1150
+ id: "hyperbolic",
1151
+ label: "Hyperbolic",
1152
+ baseUrl: "https://api.hyperbolic.xyz/v1",
1153
+ adapter: "openai-chat",
1154
+ authKind: "key",
1155
+ dashboardUrl: "https://app.hyperbolic.ai",
1156
+ liveModels: true,
1157
+ preserveCustomDestination: true,
1158
+ modelDiscovery: {
1159
+ path: "models",
1160
+ maxResponseBytes: 256 * 1024,
1161
+ maxModels: 256,
1162
+ },
1163
+ note: "Serverless text and vision-language chat models only; Hyperbolic's separate image, audio, and GPU endpoints are out of scope.",
1164
+ },
1165
+ {
1166
+ // Primary sources checked 2026-08-03:
1167
+ // - docs.nscale.com documents the production OpenAI-compatible endpoint, bearer service
1168
+ // tokens, /v1/models, and a tool-calling request using this exact Llama model id.
1169
+ // - nscale.com/policies/terms-conditions identifies Nscale AS as the service operator and
1170
+ // covers customers using its public-cloud inference offering. Maintainer: @olddonkey;
1171
+ // no affiliation with Nscale.
1172
+ id: "nscale",
1173
+ label: "Nscale Serverless Inference",
1174
+ baseUrl: "https://inference.api.nscale.com/v1",
1175
+ adapter: "openai-chat",
1176
+ authKind: "key",
1177
+ dashboardUrl: "https://console.nscale.com",
1178
+ defaultModel: "meta-llama/Llama-3.1-8B-Instruct",
1179
+ models: ["meta-llama/Llama-3.1-8B-Instruct"],
1180
+ liveModels: true,
1181
+ preserveCustomDestination: true,
1182
+ // Nscale documents tools but not parallel tool calls. Keep requests serialized.
1183
+ parallelToolCalls: false,
1184
+ // The API schema accepts reasoning_effort, but does not publish per-model tiers.
1185
+ reasoningEfforts: [],
1186
+ modelDiscovery: {
1187
+ path: "models",
1188
+ maxResponseBytes: 256 * 1024,
1189
+ maxModels: 256,
1190
+ filter: {
1191
+ // Nscale's catalog mixes chat, image, and embedding rows without a modality field.
1192
+ // Admit only the exact model used in its official tool-calling API example.
1193
+ allOf: [{ path: ["id"], equalsAny: ["meta-llama/Llama-3.1-8B-Instruct"] }],
1194
+ },
1195
+ },
1196
+ note: "Serverless OpenAI-compatible inference. Live discovery admits only the tool-capable model established by Nscale's official API example; other mixed-catalog rows remain hidden pending equivalent evidence.",
1197
+ },
1198
+ {
1199
+ // Primary sources checked 2026-08-03:
1200
+ // - docs.vultr.com documents the fixed OpenAI-compatible base URL, per-subscription bearer
1201
+ // key, /v1/models, and states that tool calling is currently limited to kimi-k2-instruct.
1202
+ // - Vultr's official properties identify VULTR as a The Constant Company, LLC trademark and
1203
+ // document customer API integrations. Maintainer: @olddonkey; no affiliation with Vultr.
1204
+ id: "vultr",
1205
+ label: "Vultr Serverless Inference",
1206
+ baseUrl: "https://api.vultrinference.com/v1",
1207
+ adapter: "openai-chat",
1208
+ authKind: "key",
1209
+ dashboardUrl: "https://my.vultr.com",
1210
+ defaultModel: "kimi-k2-instruct",
1211
+ models: ["kimi-k2-instruct"],
1212
+ liveModels: true,
1213
+ preserveCustomDestination: true,
1214
+ parallelToolCalls: false,
1215
+ reasoningEfforts: [],
1216
+ modelDiscovery: {
1217
+ path: "models",
1218
+ maxResponseBytes: 256 * 1024,
1219
+ maxModels: 256,
1220
+ filter: {
1221
+ // Vultr explicitly limits tool calling to this model. A coding agent must not select
1222
+ // another chat model that cannot complete its tool loop.
1223
+ allOf: [{ path: ["id"], equalsAny: ["kimi-k2-instruct"] }],
1224
+ },
1225
+ },
1226
+ note: "Serverless Inference subscription API. Live discovery exposes only kimi-k2-instruct because Vultr documents it as the sole tool-calling model.",
1227
+ },
1228
+ ];