@bitkyc08/opencodex 2.55.0 → 2.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (256) hide show
  1. package/bin/ocx.mjs +10 -0
  2. package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
  3. package/gui/dist/assets/{index-VuoiWj9J.js → index-Cz7CLdif.js} +21 -21
  4. package/gui/dist/index.html +2 -2
  5. package/package.json +4 -3
  6. package/src/adapters/base.ts +21 -0
  7. package/src/adapters/codebuddy/adapter.ts +2 -1
  8. package/src/adapters/codebuddy/scaffold-guard.ts +248 -0
  9. package/src/adapters/command-code.ts +1 -1
  10. package/src/adapters/cursor/envelope-echo.ts +8 -2
  11. package/src/adapters/cursor/transport-retry.ts +46 -1
  12. package/src/adapters/cursor.ts +4 -0
  13. package/src/adapters/google.ts +7 -7
  14. package/src/adapters/kiro/adapter.ts +42 -1
  15. package/src/adapters/kiro/payload.ts +17 -3
  16. package/src/adapters/kiro/reasoning.ts +70 -7
  17. package/src/adapters/kiro/stream.ts +8 -2
  18. package/src/adapters/kiro/wire.ts +2 -1
  19. package/src/adapters/kiro-events.ts +21 -13
  20. package/src/adapters/kiro-retry.ts +23 -4
  21. package/src/adapters/openai-chat/errors.ts +116 -0
  22. package/src/adapters/openai-chat/messages.ts +346 -0
  23. package/src/adapters/openai-chat/passthrough.ts +146 -0
  24. package/src/adapters/openai-chat/response-events.ts +117 -0
  25. package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
  26. package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
  27. package/src/adapters/openai-chat/tool-schema.ts +495 -0
  28. package/src/adapters/openai-chat/wire.ts +50 -0
  29. package/src/adapters/openai-chat.ts +40 -1452
  30. package/src/adapters/openai-responses/canonical-forward.ts +202 -0
  31. package/src/adapters/openai-responses/image-gen.ts +406 -0
  32. package/src/adapters/openai-responses/internal.ts +3 -0
  33. package/src/adapters/openai-responses/passthrough.ts +642 -0
  34. package/src/adapters/openai-responses/prompt-cache.ts +83 -0
  35. package/src/adapters/openai-responses/reasoning.ts +220 -0
  36. package/src/adapters/openai-responses/request-strips.ts +185 -0
  37. package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
  38. package/src/adapters/openai-responses/tool-schema.ts +293 -0
  39. package/src/adapters/openai-responses/web-search.ts +156 -0
  40. package/src/adapters/openai-responses.ts +4 -2625
  41. package/src/bridge/errors.ts +58 -0
  42. package/src/bridge/internal.ts +174 -0
  43. package/src/bridge/response-json.ts +630 -0
  44. package/src/bridge/sse.ts +1462 -0
  45. package/src/bridge.ts +5 -2204
  46. package/src/chat/inbound.ts +12 -1
  47. package/src/claude/desktop-profile.ts +66 -9
  48. package/src/claude/outbound.ts +18 -0
  49. package/src/cli/account-main.ts +1 -1
  50. package/src/cli/capabilities.ts +2 -2
  51. package/src/cli/combo.ts +10 -1
  52. package/src/cli/index.ts +48 -5
  53. package/src/cli/registry.ts +2 -1
  54. package/src/cli/system-command.ts +4 -4
  55. package/src/clients/config-export.ts +7 -3
  56. package/src/codex/account-label.ts +14 -3
  57. package/src/codex/account-lifecycle.ts +3 -0
  58. package/src/codex/account-store.ts +184 -35
  59. package/src/codex/account-usability.ts +21 -0
  60. package/src/codex/auth-api/account-list.ts +507 -0
  61. package/src/codex/auth-api/http.ts +32 -0
  62. package/src/codex/auth-api/login-flow.ts +566 -0
  63. package/src/codex/auth-api/login-state.ts +64 -0
  64. package/src/codex/auth-api/main-account-probe.ts +331 -0
  65. package/src/codex/auth-api/pool-mode-gate.ts +274 -0
  66. package/src/codex/auth-api/pool-quota-probe.ts +512 -0
  67. package/src/codex/auth-api/reset-credit-service.ts +431 -0
  68. package/src/codex/auth-api/routes.ts +425 -0
  69. package/src/codex/auth-api/runtime-config.ts +48 -0
  70. package/src/codex/auth-api.ts +27 -3118
  71. package/src/codex/auth-context.ts +252 -35
  72. package/src/codex/catalog/aggregation.ts +80 -1
  73. package/src/codex/catalog/auto-review.ts +507 -0
  74. package/src/codex/catalog/build-entries.ts +981 -0
  75. package/src/codex/catalog/combo-member.ts +375 -0
  76. package/src/codex/catalog/derive-entry.ts +229 -0
  77. package/src/codex/catalog/effort.ts +0 -1
  78. package/src/codex/catalog/gated-native-warn.ts +63 -0
  79. package/src/codex/catalog/gather-capture.ts +533 -0
  80. package/src/codex/catalog/model-hints.ts +691 -0
  81. package/src/codex/catalog/model-visibility.ts +305 -0
  82. package/src/codex/catalog/provider-fetch.ts +52 -2942
  83. package/src/codex/catalog/provider-models.ts +685 -0
  84. package/src/codex/catalog/remote.ts +30 -0
  85. package/src/codex/catalog/restore.ts +132 -0
  86. package/src/codex/catalog/retained-sync.ts +714 -0
  87. package/src/codex/catalog/routed-gather.ts +895 -0
  88. package/src/codex/catalog/subagent-roster.ts +176 -0
  89. package/src/codex/catalog/sync.ts +52 -2698
  90. package/src/codex/cli-install-provenance.ts +7 -1
  91. package/src/codex/convergence.ts +7 -2
  92. package/src/codex/desktop-app/types.ts +11 -2
  93. package/src/codex/desktop-app/windows.ts +5 -5
  94. package/src/codex/inject/config-toml.ts +563 -0
  95. package/src/codex/inject/remove.ts +192 -0
  96. package/src/codex/inject/restore.ts +567 -0
  97. package/src/codex/inject/routing-classify.ts +109 -0
  98. package/src/codex/inject/routing-target.ts +125 -0
  99. package/src/codex/inject.ts +89 -1444
  100. package/src/codex/lineage.ts +458 -0
  101. package/src/codex/model-entitlements.ts +152 -15
  102. package/src/codex/pool-refresh-backoff.ts +161 -0
  103. package/src/codex/quota-rejection.ts +104 -15
  104. package/src/codex/routing/active-account.ts +194 -0
  105. package/src/codex/routing/cache-affinity.ts +70 -0
  106. package/src/codex/routing/cooldown-math.ts +285 -0
  107. package/src/codex/routing/health-store.ts +402 -0
  108. package/src/codex/routing/probe-lease.ts +358 -0
  109. package/src/codex/routing/selection.ts +780 -0
  110. package/src/codex/routing/thread-affinity.ts +586 -0
  111. package/src/codex/routing/transient-hold-dispatch.ts +141 -0
  112. package/src/codex/routing.ts +370 -2271
  113. package/src/codex/shim-fingerprint.ts +223 -0
  114. package/src/codex/shim-inspect.ts +175 -0
  115. package/src/codex/shim-probe.ts +367 -0
  116. package/src/codex/shim-restore-lock.ts +169 -0
  117. package/src/codex/shim-state-file.ts +151 -0
  118. package/src/codex/shim-templates.ts +265 -0
  119. package/src/codex/shim.ts +48 -1268
  120. package/src/codex/warmup.ts +1 -1
  121. package/src/combos/failover.ts +85 -0
  122. package/src/combos/request.ts +17 -10
  123. package/src/combos/types.ts +23 -2
  124. package/src/config/diagnostics.ts +705 -0
  125. package/src/config/feature-flags.ts +55 -0
  126. package/src/config/live-reconcile.ts +403 -0
  127. package/src/config/load-degrade.ts +880 -0
  128. package/src/config/mutation-lock.ts +244 -0
  129. package/src/config/openai-tier-backup.ts +268 -0
  130. package/src/config/pending-teardown.ts +31 -0
  131. package/src/config/persist-unlocked.ts +92 -0
  132. package/src/config/proxy-env.ts +188 -0
  133. package/src/config/salvage.ts +244 -0
  134. package/src/config/schema/config-schema.ts +640 -0
  135. package/src/config/schema/leaf-validators.ts +855 -0
  136. package/src/config/warn-memo.ts +28 -0
  137. package/src/config.ts +234 -4481
  138. package/src/generated/compatibility-version.json +649 -121
  139. package/src/images/loop.ts +1 -1
  140. package/src/lib/errors.ts +17 -0
  141. package/src/lib/request-execution-budget.ts +198 -23
  142. package/src/lib/spend-reservation-ledger.ts +958 -0
  143. package/src/lib/state-store-registrations.ts +6 -2
  144. package/src/lib/test-home-guard.ts +85 -1
  145. package/src/lib/upstream-retry.ts +132 -21
  146. package/src/lib/windows-elevation.ts +76 -14
  147. package/src/lib/workflow-budget.ts +553 -30
  148. package/src/oauth/index.ts +2 -2
  149. package/src/oauth/key-providers.ts +2 -2
  150. package/src/providers/kiro-models.ts +4 -3
  151. package/src/providers/label.ts +19 -1
  152. package/src/providers/model-discovery.ts +16 -0
  153. package/src/providers/quota/account-cache.ts +441 -0
  154. package/src/providers/quota/antigravity.ts +295 -0
  155. package/src/providers/quota/report-cache.ts +320 -0
  156. package/src/providers/quota/vendor-probes-key.ts +1243 -0
  157. package/src/providers/quota/vendor-probes-oauth.ts +590 -0
  158. package/src/providers/quota.ts +324 -3079
  159. package/src/providers/registry/entries-core.ts +1228 -0
  160. package/src/providers/registry/entries-extended.ts +1213 -0
  161. package/src/providers/registry/model-seeds.ts +912 -0
  162. package/src/providers/registry/types.ts +352 -0
  163. package/src/providers/registry.ts +24 -3536
  164. package/src/responses/continuation-ownership.ts +29 -0
  165. package/src/responses/reasoning-envelope.ts +6 -3
  166. package/src/responses/state/replay-fingerprint.ts +80 -0
  167. package/src/responses/state/snapshot-codec.ts +104 -0
  168. package/src/responses/state/spill-failure.ts +118 -0
  169. package/src/responses/state/spill-queue.ts +665 -0
  170. package/src/responses/state/temp-recovery.ts +257 -0
  171. package/src/responses/state.ts +82 -1143
  172. package/src/routing/identity-domains.ts +456 -0
  173. package/src/routing/probe-lease.ts +613 -0
  174. package/src/server/chat-completions.ts +3 -1
  175. package/src/server/chat-native.ts +37 -9
  176. package/src/server/index/bounded-request.ts +88 -0
  177. package/src/server/index/live-sideband.ts +601 -0
  178. package/src/server/index/serve-options.ts +1766 -0
  179. package/src/server/index/startup-warnings.ts +213 -0
  180. package/src/server/index/websocket-handler.ts +339 -0
  181. package/src/server/index.ts +45 -2552
  182. package/src/server/inspection-tee.ts +107 -0
  183. package/src/server/live.ts +46 -1
  184. package/src/server/management/combo-routes.ts +10 -1
  185. package/src/server/management/route-registry.ts +26 -23
  186. package/src/server/management/shared.ts +8 -5
  187. package/src/server/management/workflow-budget-routes.ts +133 -0
  188. package/src/server/management-api.ts +12 -0
  189. package/src/server/relay-eager.ts +2 -0
  190. package/src/server/relay.ts +14 -19
  191. package/src/server/request-log-conversation.ts +9 -7
  192. package/src/server/request-log.ts +372 -4
  193. package/src/server/response-log-body.ts +153 -0
  194. package/src/server/responses/account-change-state.ts +307 -0
  195. package/src/server/responses/adapter-continuation.ts +540 -0
  196. package/src/server/responses/adapter-delivery.ts +208 -0
  197. package/src/server/responses/adapter-dispatch.ts +1042 -0
  198. package/src/server/responses/codex-ws-wire.ts +5 -0
  199. package/src/server/responses/collaboration.ts +74 -4
  200. package/src/server/responses/combo-session-recall.ts +68 -8
  201. package/src/server/responses/compact.ts +113 -17
  202. package/src/server/responses/completion-policy.ts +33 -0
  203. package/src/server/responses/core-auth.ts +529 -0
  204. package/src/server/responses/core-codex-account.ts +907 -0
  205. package/src/server/responses/core-combo-failure.ts +210 -0
  206. package/src/server/responses/core-combo.ts +787 -0
  207. package/src/server/responses/core-errors.ts +170 -0
  208. package/src/server/responses/core-lifetime.ts +95 -0
  209. package/src/server/responses/core-normalize.ts +350 -0
  210. package/src/server/responses/core-opaque-recovery.ts +380 -0
  211. package/src/server/responses/core-options.ts +159 -0
  212. package/src/server/responses/core-replay.ts +298 -0
  213. package/src/server/responses/core.ts +192 -8893
  214. package/src/server/responses/encrypted-payload.ts +0 -1
  215. package/src/server/responses/input-admission.ts +126 -6
  216. package/src/server/responses/passthrough-delivery.ts +869 -0
  217. package/src/server/responses/passthrough-dispatch.ts +1494 -0
  218. package/src/server/responses/passthrough-error.ts +38 -2
  219. package/src/server/responses/passthrough-execution.ts +54 -0
  220. package/src/server/responses/request-prepare.ts +1080 -0
  221. package/src/server/responses/request-send-budget.ts +259 -0
  222. package/src/server/responses/request-sidecar-auth.ts +149 -0
  223. package/src/server/responses/request-spend.ts +147 -0
  224. package/src/server/responses/request-transport.ts +803 -0
  225. package/src/server/responses/response-effects.ts +157 -0
  226. package/src/server/responses/run-turn-execution.ts +476 -0
  227. package/src/server/responses/sidecar-execution.ts +463 -0
  228. package/src/server/responses/terminal-guard.ts +65 -4
  229. package/src/server/responses-image-gen-repair.ts +1 -1
  230. package/src/server/responses-undeclared-tool-guard.ts +9 -5
  231. package/src/server/workflow-refusal.ts +84 -0
  232. package/src/service/windows-ops.ts +210 -16
  233. package/src/service/windows-scheduler.ts +28 -21
  234. package/src/service.ts +1 -1
  235. package/src/types/config.ts +34 -1
  236. package/src/types/request.ts +8 -5
  237. package/src/types/tools.ts +24 -0
  238. package/src/types.ts +2 -0
  239. package/src/update/index.ts +10 -0
  240. package/src/update/stop-contract.d.mts +1 -0
  241. package/src/update/stop-contract.mjs +19 -0
  242. package/src/update/stop-decision.d.mts +1 -1
  243. package/src/update/stop-decision.mjs +12 -3
  244. package/src/usage/log.ts +147 -1
  245. package/src/usage/summary.ts +171 -21
  246. package/src/vision/anthropic-describe.ts +1 -1
  247. package/src/vision/describe.ts +5 -5
  248. package/src/web-search/anthropic-executor.ts +1 -1
  249. package/src/web-search/exa-executor.ts +1 -1
  250. package/src/web-search/executor.ts +1 -1
  251. package/src/web-search/gemini-executor.ts +1 -1
  252. package/src/web-search/loop.ts +1 -1
  253. package/src/web-search/ollama-executor.ts +1 -1
  254. package/src/web-search/parse.ts +67 -14
  255. package/src/web-search/passthrough-bridge.ts +64 -31
  256. package/src/web-search/xai-executor.ts +1 -1
@@ -0,0 +1,912 @@
1
+ import type { ProviderModelDiscoverySpec } from "./types";
2
+
3
+ // Shared between the OAuth (Claude account) and API-key Anthropic entries so both expose the
4
+ // same static model seed.
5
+ // 260710 context refresh: Tier-2 evidence in
6
+ // devlog/_plan/260710_provider_hardening/001_research_frontier.md.
7
+ // 260902 Claude Fable 5.1 (`claude-fable-5-1`): 1M context / 128K output / adaptive thinking
8
+ // always on, per the official models overview and pricing page (platform.claude.com).
9
+ export const ANTHROPIC_MODELS = ["claude-fable-5-1", "claude-fable-5", "claude-sonnet-5", "claude-opus-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-sonnet-4-6", "claude-haiku-4-5"];
10
+ export const ANTHROPIC_MODEL_CONTEXT_WINDOWS: Record<string, number> = { "claude-fable-5-1": 1_000_000, "claude-sonnet-5": 1_000_000, "claude-fable-5": 1_000_000, "claude-opus-5": 1_000_000, "claude-opus-4-8": 1_000_000, "claude-opus-4-7": 1_000_000, "claude-opus-4-6": 1_000_000, "claude-sonnet-4-6": 1_000_000, "claude-haiku-4-5": 200_000 };
11
+ // All seeded Claude models support vision: https://platform.claude.com/docs/en/models/overview
12
+ export const ANTHROPIC_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
13
+ ANTHROPIC_MODELS.map(id => [id, ["text", "image"]]),
14
+ );
15
+ // Every current Claude family accepts at least 64k output tokens (Haiku 4.5 / Sonnet 4.x
16
+ // through Opus 5 and Fable 5). Anthropic caps max_tokens per model server-side, so a
17
+ // larger request never over-allocates; it only stops the 8192 truncation.
18
+ export const ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS = 64_000;
19
+ /**
20
+ * The effort rungs opencodex exposes for native Anthropic models. Without this the
21
+ * providers advertised no ladder at all, so every client that keys its effort control off
22
+ * `reasoningEfforts` — Aside and the rest of the Pi-shaped exports — wrote these models
23
+ * with no control, while the SAME Claude models routed through `cursor` or
24
+ * `google-antigravity` had one.
25
+ *
26
+ * This is an opencodex ladder, not a claim that each model takes `output_config.effort`.
27
+ * The adapter serves two wire shapes (src/adapters/anthropic.ts): adaptive families
28
+ * (fable, sonnet >= 5, opus >= 4.7) send the effort directly, while opus 4.6, sonnet 4.6
29
+ * and haiku 4.5 take the legacy path where `reasoningBudget` TRANSLATES each rung into
30
+ * `thinking.budget_tokens`. Anthropic documents `low|medium|high|max` for the 4.6 models
31
+ * and no effort parameter at all for haiku 4.5; the budget translation is what makes five
32
+ * rungs meaningful there, and it clamps below `max_tokens` so none of them 400.
33
+ *
34
+ * Deliberately excluded, each because advertising it would offer a control that does not
35
+ * do what it says:
36
+ * - `minimal`: `adaptiveEffort` rewrites it to `low` (the adaptive wire 400s on it), so
37
+ * it is not a distinct setting.
38
+ * - `none`: only sonnet >= 5 accepts an explicit thinking disable
39
+ * (`EXPLICIT_THINKING_DISABLE_FAMILY_MINIMUMS`); Fable rejects one outright.
40
+ * - `ultra`: not an Anthropic concept, and it is degraded to `max` at the request
41
+ * boundary anyway (src/responses/parser.ts).
42
+ */
43
+ export const ANTHROPIC_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
44
+ export const ANTHROPIC_MODEL_REASONING_EFFORTS: Record<string, string[]> = Object.fromEntries(
45
+ ANTHROPIC_MODELS.map(id => [id, [...ANTHROPIC_REASONING_EFFORTS]]),
46
+ );
47
+
48
+ // 260814 GLM-5.3 is registered pre-emptively alongside 5.2 everywhere 5.2 appears. Z.AI's
49
+ // devpack "How to Switch Models" page (docs.z.ai/devpack/latest-model) lists glm-5.3 and
50
+ // glm-5.3[1m] as Coding Plan ids on the unchanged endpoints; the capability and pricing
51
+ // tables were not published yet, so every 5.3 row mirrors its 5.2 sibling until they settle.
52
+ // The non-Z.AI providers below are speculative on purpose: they carry 5.2 today and are
53
+ // expected to pick 5.3 up on their usual lag. Providers whose live /v1/models discovery is
54
+ // enabled self-correct on the next successful fetch; static ones need a follow-up refresh.
55
+ // Every 5.3 family member, so the effort ladder, the default effort and the output
56
+ // cap are derived in ONE place. `glm-5.3-flash` was seeded into the model list and
57
+ // the context map by hand and left out of this constant, which meant it advertised
58
+ // a 1M context with a null effort ladder, no default effort and no output cap while
59
+ // its siblings carried three tiers, a `max` default and 131072 tokens. A member
60
+ // added to the list but not to the family is a model whose metadata silently
61
+ // disappears.
62
+ export const ZAI_GLM_53_MODELS = ["glm-5.3", "glm-5.3[1m]", "glm-5.3-flash"];
63
+ export const ZAI_GLM_52_MODELS = ["glm-5.2", "glm-5.2[1m]"];
64
+ export const ZAI_GLM_5X_MODELS = [...ZAI_GLM_53_MODELS, ...ZAI_GLM_52_MODELS];
65
+ /**
66
+ * The 5.x rows whose images the PROXY has to describe, which is NOT the same set as
67
+ * the 5.x rows themselves.
68
+ *
69
+ * `glm-5.3-flash` is a native VLM (docs.z.ai/guides/vlm/glm-5.3-flash), so listing it
70
+ * in `noVisionModels` sent an image through the vision sidecar and handed the model a
71
+ * text description of a picture it could have read itself - no error, worse answer,
72
+ * extra call. The correction commit fixed the Alibaba entries and left the eight
73
+ * providers that reach this constant behind.
74
+ *
75
+ * Kept separate from ZAI_GLM_5X_MODELS rather than filtered at each use site: that
76
+ * constant also drives `modelSupportsReasoningSummaries` and
77
+ * `preserveReasoningContentModels`, where flash DOES belong.
78
+ */
79
+ export const ZAI_GLM_5X_SIDECAR_VISION_MODELS = ZAI_GLM_5X_MODELS.filter(id => id !== "glm-5.3-flash");
80
+ /**
81
+ * Positive input-modality declaration for the Chat-path GLM rows.
82
+ *
83
+ * `noVisionModels` already keeps Flash out of the vision sidecar, but that is a NEGATIVE
84
+ * statement: it stops a detour without telling the catalog what the model can read. With
85
+ * no `modelInputModalities` entry, `configuredInputModalities` returns undefined and the
86
+ * catalog falls through to the `["text"]` floor, so every client export (ZCode, Pi, OMP)
87
+ * listed a native VLM as text-only and its picker refused to attach an image.
88
+ *
89
+ * The Responses sibling row below already declares this positively, so the same model was
90
+ * described two different ways in one registry.
91
+ *
92
+ * Authoritative source: `GET https://api.z.ai/api/v1/models` returns `input_modalities:
93
+ * ["text"]` for glm-5.3 and `["text", "image"]` for glm-5.3-flash (captured in
94
+ * devlog/_plan/260912_zcode_protocol_and_catalog/evidence/zai-responses-models.json).
95
+ * docs.z.ai/devpack/latest-model says the same in prose: "GLM-5.3 is a text-only model...
96
+ * GLM-5.3-FLASH is a multimodal model". Upstream also lists video and file for Flash;
97
+ * neither the internal vocabulary nor the export vocabulary can express them, so `image`
98
+ * is where this stops.
99
+ */
100
+ export const ZAI_GLM_5X_INPUT_MODALITIES: Record<string, string[]> = {
101
+ ...Object.fromEntries(ZAI_GLM_5X_SIDECAR_VISION_MODELS.map(id => [id, ["text"]])),
102
+ "glm-5.3-flash": ["text", "image"],
103
+ };
104
+ export const ZAI_GLM_52_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
105
+ /**
106
+ * GLM-5.3 does NOT share 5.2's five-tier ladder. docs.z.ai/devpack/latest-model folds every
107
+ * incoming effort into three effective tiers — low/minimal/light -> low, medium/high -> high,
108
+ * xhigh/max/ultra -> max — with max as both the default and the unknown-value fallback.
109
+ * Advertising five levels would publish two picker rows that are indistinguishable on the wire,
110
+ * so only the effective tiers are exposed (same treatment Cursor and Baseten already give GLM).
111
+ */
112
+ export const ZAI_GLM_53_REASONING_EFFORTS = ["low", "high", "max"];
113
+ /** Per-model ladders for the Coding Plan rows: 5.3 gets its three effective tiers, 5.2 keeps five. */
114
+ export const ZAI_GLM_5X_REASONING_EFFORTS: Record<string, string[]> = {
115
+ ...Object.fromEntries(ZAI_GLM_53_MODELS.map(id => [id, ZAI_GLM_53_REASONING_EFFORTS])),
116
+ ...Object.fromEntries(ZAI_GLM_52_MODELS.map(id => [id, ZAI_GLM_52_REASONING_EFFORTS])),
117
+ };
118
+ // 260710 MiniMax models and context windows: Tier-2 evidence in
119
+ // devlog/_plan/260710_provider_hardening/002_research_cn.md.
120
+ export const MINIMAX_MODELS = [
121
+ "MiniMax-M3",
122
+ "MiniMax-M2.7", "MiniMax-M2.7-highspeed",
123
+ "MiniMax-M2.5", "MiniMax-M2.5-highspeed",
124
+ "MiniMax-M2.1", "MiniMax-M2.1-highspeed",
125
+ "MiniMax-M2",
126
+ ];
127
+ export const MINIMAX_MODEL_CONTEXT_WINDOWS: Record<string, number> = Object.fromEntries(
128
+ MINIMAX_MODELS.map(id => [id, id === "MiniMax-M3" ? 1_000_000 : 204_800]),
129
+ );
130
+ export const MINIMAX_M3_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
131
+ export const MINIMAX_M3_REASONING_EFFORT_MAP: Record<string, string> = {
132
+ none: "disabled",
133
+ minimal: "disabled",
134
+ low: "disabled",
135
+ medium: "adaptive",
136
+ high: "adaptive",
137
+ xhigh: "adaptive",
138
+ max: "adaptive",
139
+ };
140
+ export const OPENAI_GPT56_MODELS = ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"];
141
+ export const OPENAI_GPT56_PRO_MODELS = ["gpt-5.6-sol-pro", "gpt-5.6-terra-pro", "gpt-5.6-luna-pro"];
142
+ export const OPENAI_API_GPT56_CONTEXT_WINDOW = 1_050_000;
143
+ export const OPENAI_API_GPT56_CONTEXT_WINDOWS: Record<string, number> = {
144
+ ...Object.fromEntries([...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS].map(id => [id, OPENAI_API_GPT56_CONTEXT_WINDOW])),
145
+ "gpt-5.5": OPENAI_API_GPT56_CONTEXT_WINDOW,
146
+ };
147
+ export const OPENAI_API_GPT56_MAX_INPUT_TOKENS: Record<string, number> = {
148
+ ...Object.fromEntries([...OPENAI_GPT56_MODELS, ...OPENAI_GPT56_PRO_MODELS].map(id => [id, 922_000])),
149
+ "gpt-5.5": 922_000,
150
+ };
151
+ export const OPENAI_API_GPT56_VIRTUAL_MODELS: Record<string, { wireModelId: string; reasoningMode: "pro" }> = {
152
+ "gpt-5.6-sol-pro": { wireModelId: "gpt-5.6-sol", reasoningMode: "pro" },
153
+ "gpt-5.6-terra-pro": { wireModelId: "gpt-5.6-terra", reasoningMode: "pro" },
154
+ "gpt-5.6-luna-pro": { wireModelId: "gpt-5.6-luna", reasoningMode: "pro" },
155
+ };
156
+ export const OPENAI_API_GPT56_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
157
+ /*
158
+ * Meta Model API (https://api.meta.ai/v1) — published ladder, deliberately NOT the
159
+ * house set. dev.meta.ai/docs/reasoning lists "none", "minimal", "low", "medium",
160
+ * "high", "xhigh" and then excludes "none" for this family: "not supported by Muse
161
+ * Spark and returns HTTP 400". "max" and "ultra" are absent from the vendor's list
162
+ * entirely, so appending one by family resemblance would invent a wire value.
163
+ *
164
+ * Corroborated on a second surface: an unauthenticated OpenCode Zen probe of
165
+ * muse-spark-1.3-contributor-free (2026-09-03) accepted minimal..xhigh, rejected
166
+ * max/ultra with `unknown variant`, and rejected none with "does not support none
167
+ * with this model".
168
+ */
169
+ export const META_MUSE_REASONING_EFFORTS = ["minimal", "low", "medium", "high", "xhigh"];
170
+ /*
171
+ * Identity wire map. `requestToCodexEffort` (src/reasoning-effort.ts) rewrites
172
+ * `minimal` to `low` unless a model-scoped wire map says otherwise, so without this
173
+ * the picker would advertise an effort the wire never sends — and a registry-array
174
+ * assertion would pass while the request body was wrong. Identity because Meta's
175
+ * values ARE the Codex names.
176
+ */
177
+ export const META_MUSE_REASONING_EFFORT_MAP: Record<string, string> = Object.fromEntries(
178
+ META_MUSE_REASONING_EFFORTS.map(effort => [effort, effort]),
179
+ );
180
+ /** Both Muse Spark 1.3 tiers publish a 1,048,576-token window (dev.meta.ai/docs/models). */
181
+ export const META_MUSE_CONTEXT_WINDOW = 1_048_576;
182
+ export const META_MUSE_MODELS = ["muse-spark-1.3", "muse-spark-1.3-contributor"];
183
+ /**
184
+ * Daybreak program aliases. These `-latest` ids are the stable contract: OpenAI repoints
185
+ * them at newer snapshots over time (red -> gpt-5.6-cyber, blue -> gpt-5.6-sol as of
186
+ * 2026-08-11), so registering the ALIAS inherits future model swaps while a pinned
187
+ * snapshot id would silently go stale. Snapshot ids are deliberately absent here.
188
+ * Responses-only per both published endpoint tables (`v1/chat/completions` is marked
189
+ * Not supported) — never add these to a chat-completions provider. Access needs separate
190
+ * Daybreak approval and provisioning, so neither is ever a default.
191
+ * Verified 2026-08-11: developers.openai.com/api/docs/models/daybreak-red-latest.md
192
+ * and .../daybreak-blue-latest.md
193
+ */
194
+ export const OPENAI_DAYBREAK_MODELS = ["daybreak-red-latest", "daybreak-blue-latest"];
195
+ export const OPENAI_DAYBREAK_CONTEXT_WINDOWS: Record<string, number> = {
196
+ "daybreak-red-latest": 400_000,
197
+ "daybreak-blue-latest": 1_050_000,
198
+ };
199
+ export const OPENAI_DAYBREAK_MAX_INPUT_TOKENS: Record<string, number> = {
200
+ "daybreak-red-latest": 272_000,
201
+ "daybreak-blue-latest": 922_000,
202
+ };
203
+ /**
204
+ * Neither Daybreak page publishes a reasoning-effort ladder. An explicit empty array means
205
+ * "expose no effort control"; OMITTING the key would instead fall back to the full routed
206
+ * ladder (`configuredReasoningEfforts` returns undefined -> `applyReasoningLevels` uses
207
+ * ROUTED_REASONING_LEVELS), which would advertise efforts the models never documented.
208
+ * `noReasoningModels` is wrong here: both pages document reasoning-token support, so these
209
+ * are reasoning models with no *selectable* ladder.
210
+ */
211
+ export const OPENAI_DAYBREAK_REASONING_EFFORTS: Record<string, string[]> = Object.fromEntries(
212
+ OPENAI_DAYBREAK_MODELS.map(id => [id, [] as string[]]),
213
+ );
214
+ export const OPENROUTER_GPT56_MODELS = OPENAI_GPT56_MODELS.map(id => `openai/${id}`);
215
+ export const XAI_MODELS = [
216
+ "grok-4.6",
217
+ "grok-4.5",
218
+ "grok-4.3",
219
+ "grok-4.20-multi-agent-0309",
220
+ "grok-4.20-0309-reasoning",
221
+ "grok-4.20-0309-non-reasoning",
222
+ "grok-build-0.1",
223
+ "grok-composer-2.5-fast",
224
+ ];
225
+ // OpenRouter's live /endpoints routes report 1,050,000; keep this separate from the
226
+ // unverified OpenAI API-key seed. Evidence: devlog/_plan/260710_provider_hardening/003_research_aggregators.md.
227
+ export const OPENROUTER_GPT56_CONTEXT_WINDOW = 1_050_000;
228
+ export const OPENROUTER_GPT56_CONTEXT_WINDOWS = {
229
+ "openai/gpt-5.6-sol": OPENROUTER_GPT56_CONTEXT_WINDOW,
230
+ "openai/gpt-5.6-terra": OPENROUTER_GPT56_CONTEXT_WINDOW,
231
+ "openai/gpt-5.6-luna": OPENROUTER_GPT56_CONTEXT_WINDOW,
232
+ };
233
+
234
+ /**
235
+ * Vendor thinking-toggle models (MiMo v2.x, GLM 5/5.1 on Zen Go): the wire knob is
236
+ * `thinking: {type: enabled|disabled}` — a binary. Advertise the full Codex picker ladder
237
+ * and map efforts onto the toggle. Zen Go
238
+ * pass-through probed live 2026-07-07 (glm-5.2 toggle verified; mimo/minimax accept shape).
239
+ */
240
+ export const THINKING_TOGGLE_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
241
+ export const THINKING_TOGGLE_MAP: Record<string, string> = {
242
+ none: "disabled",
243
+ minimal: "disabled",
244
+ low: "disabled",
245
+ medium: "enabled",
246
+ high: "enabled",
247
+ xhigh: "enabled",
248
+ max: "enabled",
249
+ };
250
+ export const OPENCODE_GO_THINKING_TOGGLE_MODELS = [
251
+ "mimo-v2.5", "mimo-v2.5-pro", "glm-5", "glm-5.1",
252
+ ];
253
+ /**
254
+ * Zhipu's domestic BigModel platform. Text families first, then the vision member: modalities are
255
+ * declared per model because `noVisionModels` means the opposite of "text only" here — it routes
256
+ * images through the proxy's vision sidecar (src/codex/catalog/provider-fetch.ts), a claim nobody
257
+ * has verified for BigModel-hosted GLM.
258
+ */
259
+ // `glm-5.3-flash` is deliberately absent: it is a native VLM
260
+ // (docs.z.ai/guides/vlm/glm-5.3-flash), unlike glm-5.3 itself.
261
+ export const ZHIPU_BIGMODEL_TEXT_MODELS = ["glm-4.6", "glm-4.7", "glm-4.7-flash", "glm-5", "glm-5.1", "glm-5.2", "glm-5.3"];
262
+ export const ZHIPU_BIGMODEL_MODELS = [...ZHIPU_BIGMODEL_TEXT_MODELS, "glm-4.6v"];
263
+ export const ZHIPU_BIGMODEL_INPUT_MODALITIES: Record<string, string[]> = {
264
+ ...Object.fromEntries(ZHIPU_BIGMODEL_TEXT_MODELS.map(id => [id, ["text"]])),
265
+ "glm-4.6v": ["text", "image"],
266
+ };
267
+ export const ZHIPU_BIGMODEL_THINKING_TOGGLE_MODELS = ["glm-4.6", "glm-4.7", "glm-5", "glm-5.1", "glm-5.2", "glm-5.3", "glm-5.3-flash"];
268
+ export const THINKING_BUDGET_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
269
+ // Qwen3.8-Max is the first Qwen3.x model with official direct `reasoning_effort` support.
270
+ // Evidence: https://qwen.ai/blog?id=qwen3.8
271
+ export const QWEN38_REASONING_EFFORTS = ["low", "medium", "xhigh"];
272
+ export const THINKING_BUDGET_MODELS = [
273
+ "qwen3.5-397b", "qwen3.6-35b",
274
+ "qwen3.5-plus", "qwen3.6-plus", "qwen3.7-max", "qwen3.7-plus",
275
+ ];
276
+ export const OPENCODE_GO_THINKING_BUDGET_MODELS = ["qwen3.5-plus", "qwen3.6-plus", "qwen3.7-max", "qwen3.7-plus"];
277
+ /*
278
+ * DeepSeek moved the whole V4 name set on 2026-09-10. V4.1-Flash ships as deepseek-flash
279
+ * on the first-party API; deepseek-v4-flash and the vision preview retire as models but
280
+ * keep routing there as compatibility aliases, and deepseek-v4-pro follows from
281
+ * 2026-09-14 04:00 UTC. Evidence: https://api-docs.deepseek.com/news/news260910/.
282
+ *
283
+ * The spelling differs by who serves it, so one shared list cannot express it: the
284
+ * first-party API answers to deepseek-flash, while the Zen gateway exposes the route as
285
+ * deepseek-v4.1-flash (issue #4253, PR #4258). Vendor-hosted rosters (Volcengine plan
286
+ * snapshots, Alibaba) publish on their own schedule and keep the legacy set until they say
287
+ * otherwise - a first-party retirement notice does not end their deployment.
288
+ */
289
+ export const DEEPSEEK_V4_LEGACY_MODELS = ["deepseek-v4-flash"];
290
+ /*
291
+ * `deepseek-v4-pro` is deliberately absent from both live sets. DeepSeek retires it from
292
+ * 2026-09-14 04:00 UTC and routes its requests to V4.1-Flash until a V4.1 Pro exists, so a
293
+ * row here would advertise a Pro context window and Pro pricing for a route that serves
294
+ * Flash. The retirement is followed through every roster in this file, including the
295
+ * vendor-hosted ones; providers that discover their models live are handled by
296
+ * `ROUTED_MODEL_COMPATIBILITY_EXCLUSIONS` because deleting a row there removes the
297
+ * model's capabilities rather than the model.
298
+ */
299
+ export const DEEPSEEK_NATIVE_THINKING_MODELS = ["deepseek-flash", "deepseek-v4-flash"];
300
+ export const DEEPSEEK_GATEWAY_THINKING_MODELS = ["deepseek-v4.1-flash", "deepseek-v4-flash"];
301
+ /*
302
+ * DeepSeek's legacy vision preview id (released 2026-08-21). First-party probes
303
+ * in #4436 resolve it to image-capable `deepseek-flash`; retain the existing
304
+ * declarations because gateway support is specific to each served identifier.
305
+ */
306
+ export const DEEPSEEK_VISION_PREVIEW_MODEL = "deepseek-v4-flash-vision-exp";
307
+ /**
308
+ * CommandCode routes verified to accept image input end-to-end (#2406).
309
+ *
310
+ * Verified-negative and therefore deliberately ABSENT: deepseek/deepseek-v4-flash,
311
+ * zai-org/GLM-5.2, zai-org/GLM-5.3, xai/grok-4.6. Those
312
+ * routes accept the request and drop the image, which is worse than declining it — the
313
+ * model answers about an image it never saw. Do not add an id here on family resemblance;
314
+ * capability intersection trusts this map.
315
+ */
316
+ export const COMMAND_CODE_IMAGE_MODELS = [
317
+ `deepseek/${DEEPSEEK_VISION_PREVIEW_MODEL}`,
318
+ "gpt-5.6-luna",
319
+ "gpt-5.6-sol",
320
+ "MiniMaxAI/MiniMax-M3",
321
+ "moonshotai/Kimi-K3",
322
+ "meta/muse-spark-1.3",
323
+ "meta/muse-spark-1.3-contributor",
324
+ "meta/muse-spark-1.2",
325
+ "meta/muse-spark-1.2-contributor",
326
+ // Native Z.AI VLM (docs.z.ai/guides/vlm/glm-5.3-flash). This exact id is already
327
+ // classified as natively vision-capable in NVIDIA_NIM_VISION_MODELS in this file;
328
+ // it is not one of the verified-negative ids the header names (those are
329
+ // deepseek/deepseek-v4-flash, zai-org/GLM-5.2, zai-org/GLM-5.3, xai/grok-4.6 —
330
+ // different ids). Adding it on the shared GLM-5.3 prefix would be the family-
331
+ // resemblance mistake the header forbids; the VLM docs are the evidence (#4505).
332
+ "z-ai/glm-5.3-flash",
333
+ ] as const;
334
+ /**
335
+ * Native image stays sourced from COMMAND_CODE_IMAGE_MODELS. Text-only routes
336
+ * sit beside that list so the catalog can still advertise sidecar coverage
337
+ * without claiming the gateway itself accepts a picture.
338
+ *
339
+ * The gateway-prefixed DeepSeek V4.1 Flash route has no verified native image
340
+ * support, so declaring it image-capable would hand it a picture it drops. A
341
+ * positive text-only declaration makes it a vision-sidecar consumer
342
+ * (src/vision/eligibility.ts), so the catalog advertises image input on its
343
+ * behalf and the four-target combo in #4505 intersects to ["text","image"]
344
+ * instead of ["text"] — without claiming native vision. modelInputModalities
345
+ * is per-key filled, so this reaches an existing install even when
346
+ * noVisionModels was persisted before the id joined that list.
347
+ */
348
+ export const COMMAND_CODE_TEXT_ONLY_MODELS = [
349
+ "deepseek/deepseek-v4.1-flash",
350
+ ] as const;
351
+ export const COMMAND_CODE_MODEL_INPUT_MODALITIES: Record<string, ["text"] | ["text", "image"]> = {
352
+ ...Object.fromEntries(COMMAND_CODE_IMAGE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
353
+ ...Object.fromEntries(COMMAND_CODE_TEXT_ONLY_MODELS.map(id => [id, ["text"] as ["text"]])),
354
+ };
355
+ export const OPENCODE_FREE_DEEPSEEK_MODELS = ["deepseek-v4-flash-free"];
356
+ /*
357
+ * Zen free models that reject `image_url` upstream (#1043, and the reproducible
358
+ * half of #1024).
359
+ *
360
+ * Zen publishes NO modality metadata — its `/v1/models` returns only id, object,
361
+ * created, owned_by — so this list is measured, not derived. Each id was probed
362
+ * once against https://opencode.ai/zen/v1 on 2026-08-05 with a text control first
363
+ * and then a 1x1 PNG; the six below failed the image request, four of them with
364
+ * `[404] No endpoints found that support image input` and `big-pickle` with the
365
+ * exact deserialize error quoted in #1043.
366
+ *
367
+ * `mimo-v2.5-free` and `longcat-2.0-free` ACCEPT images. They remain absent
368
+ * from the blind list and are recorded separately as positive input-modality evidence,
369
+ * so capability-positive dispatch can forward images without relying on blacklist absence.
370
+ *
371
+ * Zen's roster is discovered live while this list is static, so it is a dated
372
+ * exception list, not a capability model. Re-probe before extending it.
373
+ * Evidence: devlog/_fin/260805_bug_fix_stack/002_zen_modality_probe.md
374
+ */
375
+ export const OPENCODE_ZEN_TEXT_ONLY_MODELS = [
376
+ "big-pickle",
377
+ "nemotron-3-ultra-free",
378
+ "ling-3.0-flash-free",
379
+ "north-mini-code-free",
380
+ "laguna-s-2.1-free",
381
+ "deepseek-v4-flash-free",
382
+ ];
383
+ export const OPENCODE_ZEN_IMAGE_MODELS = ["mimo-v2.5-free", "longcat-2.0-free"] as const;
384
+ /*
385
+ * DeepSeek's Codex ladder is low/high/max. With the V4 Pro GA release
386
+ * (DeepSeek-V4-Pro-0813) the official thinking-mode table is IDENTICAL for both
387
+ * V4 models (api-docs.deepseek.com/guides/thinking_mode, verified 2026-08-13):
388
+ *
389
+ * requested | v4-flash | v4-pro
390
+ * low | low | low
391
+ * medium | high | high
392
+ * high | high | high
393
+ * xhigh | high | high
394
+ * max | max | max
395
+ *
396
+ * Before GA, Pro silently upgraded low->high and mapped xhigh->max (#1057-era
397
+ * table); the page's footnote about an early-August Pro mapping update landed
398
+ * with this GA, so Pro now advertises the same three real tiers as Flash.
399
+ *
400
+ * Two standing notes (#1057):
401
+ *
402
+ * - `xhigh` is a COMPATIBILITY ALIAS, not a native tier. It stays in the wire maps
403
+ * so existing requests and saved configs keep working, but it is not advertised.
404
+ * - `medium` has no row in the vendor table — mapping it to `high` is OUR
405
+ * compatibility choice for clients that only speak the OpenAI ladder.
406
+ */
407
+ export const DEEPSEEK_FLASH_THINKING_EFFORTS = ["low", "high", "max"];
408
+ export const DEEPSEEK_PRO_THINKING_EFFORTS = ["low", "high", "max"];
409
+ export const DEEPSEEK_PRO_REASONING_MAP: Record<string, string> = {
410
+ low: "low",
411
+ medium: "high",
412
+ high: "high",
413
+ xhigh: "high",
414
+ max: "max",
415
+ };
416
+ export const DEEPSEEK_FLASH_REASONING_MAP: Record<string, string> = {
417
+ low: "low",
418
+ medium: "high",
419
+ high: "high",
420
+ xhigh: "high",
421
+ max: "max",
422
+ };
423
+ /**
424
+ * Flash-versus-Pro classification for DeepSeek V4 model ids, including prefixed
425
+ * (`deepseek/deepseek-v4.1-flash`) and suffixed (`deepseek-v4-flash-free`) forms.
426
+ * `tests/providers/provider-registry-parity.test.ts` enumerates every id the registry
427
+ * actually passes here, so a future id this substring test would misread cannot
428
+ * land silently.
429
+ */
430
+ export const isDeepseekFlashModel = (modelId: string): boolean =>
431
+ modelId.toLowerCase().includes("flash");
432
+ export const deepseekThinkingEffortsFor = (modelId: string): string[] =>
433
+ isDeepseekFlashModel(modelId) ? DEEPSEEK_FLASH_THINKING_EFFORTS : DEEPSEEK_PRO_THINKING_EFFORTS;
434
+ export const deepseekReasoningMapFor = (modelId: string): Record<string, string> =>
435
+ isDeepseekFlashModel(modelId) ? DEEPSEEK_FLASH_REASONING_MAP : DEEPSEEK_PRO_REASONING_MAP;
436
+ // 260719 Alibaba Token Plan Personal Edition (China/Beijing). Keep it distinct from
437
+ // Coding Plan: the products use different exact allowlists and different base URLs.
438
+ // Evidence: https://help.aliyun.com/en/model-studio/token-plan-personal-overview
439
+ // https://help.aliyun.com/en/model-studio/token-plan-quickstart
440
+ export const ALIBABA_TOKEN_PLAN_MODELS = [
441
+ "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
442
+ "glm-5.3", "glm-5.3-flash", "glm-5.2",
443
+ ];
444
+ export const ALIBABA_TOKEN_PLAN_QWEN_MODELS = [
445
+ "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
446
+ ];
447
+ export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
448
+ "qwen3.8-max": ["text", "image"],
449
+ "qwen3.7-max": ["text", "image"],
450
+ "qwen3.7-plus": ["text", "image"],
451
+ "qwen3.6-flash": ["text", "image"],
452
+ "glm-5.3": ["text"],
453
+ "glm-5.3-flash": ["text", "image"],
454
+ "glm-5.2": ["text"],
455
+ };
456
+
457
+ // 260721 Alibaba Token Plan International (ap-southeast-1 / Singapore, hardened 260721).
458
+ // Multi-vendor lineup distinct from Beijing — includes DeepSeek V4 flash, Kimi K2.7, MiniMax.
459
+ // Evidence: https://www.alibabacloud.com/help/en/model-studio/token-plan-overview
460
+ // https://qwencloud.com/pricing/token-plan (qwen3.8 metadata)
461
+ export const ALIBABA_INTL_TOKEN_PLAN_MODELS = [
462
+ "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
463
+ "deepseek-v4-flash", "deepseek-v3.2",
464
+ "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5",
465
+ "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5",
466
+ "MiniMax-M2.5",
467
+ ];
468
+ export const ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS = [
469
+ "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
470
+ ];
471
+
472
+ // 260722 Tencent Cloud Coding Plan. The plan's model set is explicitly dynamic; these are the
473
+ // current documented ids and live discovery remains enabled so successful /models responses win.
474
+ // Tencent marks every Coding Plan model as text-only input and restricts plan keys to interactive
475
+ // coding tools (not custom application backends or non-interactive batch automation).
476
+ // Evidence: https://cloud.tencent.cn/document/product/1823/130092
477
+ export const TENCENT_CODING_PLAN_MODELS = ["tc-code-latest", "glm-5", "kimi-k2.5", "minimax-m2.5"];
478
+ // Volcengine's authenticated /api/v3/models catalog mixes chat models with embedding,
479
+ // image, video, and 3D generation resources. Keep the Codex-facing presets scoped to
480
+ // models documented for text/agent or Coding Plan use.
481
+ //
482
+ // Maintenance owner: @lidge-jun. Verified 2026-08-01 against the vendor's own docs —
483
+ // endpoints https://docs.volcengine.com/docs/82379/1528783 (Coding Plan) and
484
+ // https://docs.volcengine.com/docs/82379/2165245 (Agent Plan); Codex CLI integration
485
+ // https://www.volcengine.com/docs/82379/2556056; supported clients
486
+ // https://www.volcengine.com/docs/82379/2188957; terms https://www.volcengine.com/docs/6256/64903
487
+ // (北京火山引擎科技有限公司). Plan quota is restricted to supported AI coding tools and misuse
488
+ // is documented as grounds for suspension — see the `note` on both Plan entries.
489
+ // Report a break by opening an issue tagging the owner; the three things that rot first are the
490
+ // static catalogs (liveModels:false cannot self-heal), the base URLs, and those Plan terms.
491
+ // Full evidence ledger: devlog/_fin/260801_pr611_volcengine_evidence/000_evidence_ledger.md
492
+ export const VOLCENGINE_ARK_MODELS = [
493
+ "doubao-seed-2-1-pro-260628",
494
+ "doubao-seed-2-1-turbo-260628",
495
+ "doubao-seed-evolving",
496
+ "deepseek-v4-flash-260425",
497
+ "deepseek-v3-2-251201",
498
+ // No glm-5-3 row: Ark pins date-stamped snapshot ids (glm-5-2-260617) that cannot be
499
+ // guessed ahead of the vendor publishing them. Add it once /api/v3/models lists one.
500
+ "glm-5-2-260617",
501
+ "glm-4-7-251222",
502
+ ];
503
+ export const VOLCENGINE_DOUBAO_THINKING_MODELS = [
504
+ "doubao-seed-2-1-pro-260628",
505
+ "doubao-seed-2-1-turbo-260628",
506
+ "doubao-seed-evolving",
507
+ ];
508
+ export const VOLCENGINE_CODING_PLAN_MODELS = [
509
+ "ark-code-latest",
510
+ "doubao-seed-2.0-code",
511
+ "deepseek-v4-flash",
512
+ "glm-5.3",
513
+ "glm-5.3-flash",
514
+ "glm-5.2",
515
+ "kimi-k2.6",
516
+ "minimax-m3",
517
+ ];
518
+ export const VOLCENGINE_AGENT_PLAN_MODELS = [
519
+ "deepseek-v4-flash",
520
+ "glm-5.3",
521
+ "glm-5.3-flash",
522
+ "glm-5.2",
523
+ "kimi-k2.6",
524
+ "minimax-m3",
525
+ "doubao-seed-2.0-pro",
526
+ ];
527
+ export const VOLCENGINE_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
528
+ "kimi-k2.6": ["text", "image"],
529
+ "minimax-m3": ["text", "image"],
530
+ // Native VLM (docs.z.ai/guides/vlm/glm-5.3-flash), so it is declared here and left
531
+ // out of the text-only list below.
532
+ "glm-5.3-flash": ["text", "image"],
533
+ };
534
+ // Every other Plan model is text-only. Declaring this explicitly keeps the vision
535
+ // sidecar from advertising image input for models that cannot accept it — the same
536
+ // treatment tencent-coding-plan gives its (entirely text-only) plan catalog.
537
+ export const VOLCENGINE_PLAN_TEXT_ONLY_MODELS = [
538
+ "ark-code-latest",
539
+ "doubao-seed-2.0-code",
540
+ "deepseek-v4-flash",
541
+ "glm-5.3",
542
+ "glm-5.2",
543
+ "doubao-seed-2.0-pro",
544
+ ];
545
+ export const ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
546
+ "qwen3.8-max": ["text", "image"],
547
+ "qwen3.7-max": ["text", "image"],
548
+ "qwen3.7-plus": ["text", "image"],
549
+ "qwen3.6-plus": ["text", "image"],
550
+ "qwen3.6-flash": ["text", "image"],
551
+ "deepseek-v4-flash": ["text"],
552
+ "deepseek-v3.2": ["text"],
553
+ "kimi-k2.7-code": ["text", "image"],
554
+ "kimi-k2.6": ["text", "image"],
555
+ "kimi-k2.5": ["text", "image"],
556
+ "glm-5.3": ["text"],
557
+ "glm-5.3-flash": ["text", "image"],
558
+ "glm-5.2": ["text"],
559
+ "glm-5.1": ["text"],
560
+ "glm-5": ["text"],
561
+ "MiniMax-M2.5": ["text"],
562
+ };
563
+
564
+ // 260717 Kimi K3: the subscription endpoint uses one upstream id (`k3`) for both
565
+ // entitlement tiers. Bare `k3` advertises the Moderato 256K ceiling; the local `[1m]`
566
+ // alias advertises Allegretto's 1M ceiling and is stripped before the upstream request.
567
+ // The separately billed Moonshot API uses `kimi-k3`.
568
+ // Evidence: https://www.kimi.com/code/docs/en/kimi-code/models.html
569
+ // https://www.kimi.com/code/docs/en/kimi-code/error-reference.html
570
+ export const KIMI_K3_STANDARD_CONTEXT_WINDOW = 262_144;
571
+ export const KIMI_K3_1M_CONTEXT_WINDOW = 1_048_576;
572
+ export const KIMI_CODING_K3_MODELS = ["k3", "k3[1m]"];
573
+ export const KIMI_LEGACY_API_MODELS = ["kimi-k2.7-code", "kimi-k2.7-code-highspeed", "kimi-k2.6", "kimi-k2.5"];
574
+ export const KIMI_API_MODELS = ["kimi-k3", ...KIMI_LEGACY_API_MODELS];
575
+ export const KIMI_CODING_MODELS = [...KIMI_CODING_K3_MODELS, ...KIMI_LEGACY_API_MODELS, "kimi-for-coding"];
576
+ export const KIMI_THINKING_MODELS = KIMI_CODING_MODELS;
577
+ export const KIMI_CODING_NO_REASONING_MODELS = KIMI_CODING_MODELS.filter(id => !KIMI_CODING_K3_MODELS.includes(id));
578
+ export const KIMI_API_NO_REASONING_MODELS = KIMI_API_MODELS.filter(id => id !== "kimi-k3");
579
+ export const KIMI_CODING_K3_REASONING_EFFORTS = ["low", "high", "max"];
580
+ export const KIMI_CODING_K3_REASONING_EFFORT_MAP: Record<string, string> = {
581
+ none: "none",
582
+ low: "low",
583
+ medium: "high",
584
+ high: "high",
585
+ xhigh: "max",
586
+ max: "max",
587
+ };
588
+ export const KIMI_CODING_REASONING_EFFORTS = Object.fromEntries(
589
+ KIMI_CODING_MODELS.map(id => [id, KIMI_CODING_K3_MODELS.includes(id) ? KIMI_CODING_K3_REASONING_EFFORTS : []]),
590
+ );
591
+ export const KIMI_CODING_DEFAULT_REASONING_EFFORTS = Object.fromEntries(
592
+ KIMI_CODING_K3_MODELS.map(id => [id, "max"]),
593
+ );
594
+ export const KIMI_CODING_REASONING_EFFORT_MAPS = Object.fromEntries(
595
+ KIMI_CODING_K3_MODELS.map(id => [id, KIMI_CODING_K3_REASONING_EFFORT_MAP]),
596
+ );
597
+ export const KIMI_API_REASONING_EFFORTS = Object.fromEntries(
598
+ KIMI_API_MODELS.map(id => [id, id === "kimi-k3" ? ["max"] : []]),
599
+ );
600
+ export const KIMI_LOCKED_PARAMETER_MODELS = KIMI_CODING_MODELS;
601
+ export const KIMI_AUTO_TOOL_CHOICE_ONLY_MODELS = ["kimi-k2.7-code", "kimi-k2.7-code-highspeed", "kimi-for-coding"];
602
+ export const KIMI_API_MODEL_CONTEXT_WINDOWS: Record<string, number> = Object.fromEntries(
603
+ KIMI_API_MODELS.map(id => [id, id === "kimi-k3" ? KIMI_K3_1M_CONTEXT_WINDOW : 262_144]),
604
+ );
605
+ export const KIMI_API_MODEL_INPUT_MODALITIES = { "kimi-k3": ["text", "image"] };
606
+
607
+ // 260715 NVIDIA NIM kimi family (issue #126): documented served ids on integrate
608
+ // chat/completions per docs.api.nvidia.com/nim/reference/llm-apis; live /v1/models
609
+ // currently lists only kimi-k2.6 but the list is dynamic, so carry the documented family.
610
+ export const NVIDIA_NIM_KIMI_THINKING_MODELS = [
611
+ "moonshotai/kimi-k2.6", "moonshotai/kimi-k2.5", "moonshotai/kimi-k2-thinking",
612
+ ];
613
+ export const NVIDIA_NIM_KIMI_MODELS = [
614
+ ...NVIDIA_NIM_KIMI_THINKING_MODELS,
615
+ "moonshotai/kimi-k2-instruct", "moonshotai/kimi-k2-instruct-0905",
616
+ ];
617
+ /**
618
+ * 260804 issue #956: NIM publishes no input-modality metadata on `/v1/models`, so the
619
+ * registry is the only source of truth for which models can see images.
620
+ *
621
+ * Two lists, both verified per-model against NVIDIA documentation on 2026-08-04
622
+ * (build.nvidia.com model pages and docs.api.nvidia.com/nim/reference/*). Evidence and
623
+ * the per-id audit: devlog/_fin/260804_stack7_service_vision/011_nim_id_audit.md.
624
+ *
625
+ * Read `noVisionModels` carefully — it lists models that CANNOT see images, which is
626
+ * what routes them through the proxy's vision sidecar (src/vision/index.ts) and makes the
627
+ * catalog advertise image input for them. Membership is wrong in BOTH directions:
628
+ * - a text-only model missing from it keeps issue #956 (images blocked or rejected);
629
+ * - a vision model wrongly IN it gets its image silently replaced by another model's
630
+ * text description — no error, worse answers, extra cost.
631
+ *
632
+ * A new NIM id must be classified DELIBERATELY against its NVIDIA page, never assumed
633
+ * from its name: `google/gemma-4-31b-it` carries no vision marker yet accepts images,
634
+ * `-vl` also appears on embedding/reranking models, and `google/codegemma-7b` is
635
+ * text-only while `google/codegemma-1.1-7b` has no current page at all. An unclassified
636
+ * id is intentionally left alone rather than defaulted, because NIM serves non-chat
637
+ * endpoints (embeddings, rerankers, guards, OCR) that reach the same code path.
638
+ */
639
+ export const NVIDIA_NIM_VISION_MODELS = [
640
+ "meta/llama-3.2-11b-vision-instruct", "meta/llama-3.2-90b-vision-instruct",
641
+ "nvidia/llama-3.1-nemotron-nano-vl-8b-v1", "nvidia/nemotron-nano-12b-v2-vl",
642
+ "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "nvidia/cosmos3-nano-reasoner",
643
+ "nvidia/ising-calibration-1.5-31b", "nvidia/ising-calibration-1-35b-a3b",
644
+ "google/gemma-4-31b-it", "google/diffusiongemma-26b-a4b-it",
645
+ "minimaxai/minimax-m3", "moonshotai/kimi-k2.6", "moonshotai/kimi-k2.5",
646
+ "stepfun-ai/step-3.7-flash", "thinkingmachines/inkling",
647
+ "mistralai/mistral-medium-3.5-128b",
648
+ "z-ai/glm-5.3-flash",
649
+ ];
650
+ /**
651
+ * The catalog advertises image input only for `noVisionModels` members, so a natively
652
+ * vision-capable model would otherwise be published as text-only and the Codex app would
653
+ * block attachments before the native path ever runs.
654
+ */
655
+ export const NVIDIA_NIM_VISION_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
656
+ NVIDIA_NIM_VISION_MODELS.map(id => [id, ["text", "image"]]),
657
+ );
658
+ /**
659
+ * Text-only NIM chat models — 26 ids, each carrying an explicit `Input Modalities: Text`
660
+ * (or equivalent) on its NVIDIA page. PR #964 proposed ~64; six of those are natively
661
+ * image-capable and live in NVIDIA_NIM_VISION_MODELS above, and 32 more had no current
662
+ * NVIDIA page and were dropped rather than assumed.
663
+ *
664
+ * kimi-k2-thinking and kimi-k2-instruct are text-only while k2.5/k2.6 are not — vision
665
+ * and reasoning are independent axes, so all four stay in NVIDIA_NIM_KIMI_MODELS for
666
+ * reasoning suppression regardless of which list they appear in here.
667
+ */
668
+ export const NVIDIA_NIM_NO_VISION_MODELS = [
669
+ "deepseek-ai/deepseek-v4-flash",
670
+ "google/codegemma-7b",
671
+ "meta/llama-3.1-70b-instruct", "meta/llama-3.1-8b-instruct",
672
+ "meta/llama-3.2-1b-instruct", "meta/llama-3.2-3b-instruct",
673
+ "meta/llama-3.3-70b-instruct", "meta/llama2-70b",
674
+ "mistralai/mistral-7b-instruct-v0.3", "mistralai/mistral-nemotron",
675
+ "moonshotai/kimi-k2-thinking", "moonshotai/kimi-k2-instruct",
676
+ "nvidia/llama-3.1-nemotron-nano-8b-v1", "nvidia/llama-3.1-nemotron-ultra-253b-v1",
677
+ "nvidia/llama-3.3-nemotron-super-49b-v1", "nvidia/llama-3.3-nemotron-super-49b-v1.5",
678
+ "nvidia/nemotron-3-nano-30b-a3b", "nvidia/nemotron-3-super-120b-a12b",
679
+ "nvidia/nemotron-3-ultra-550b-a55b", "nvidia/nemotron-mini-4b-instruct",
680
+ "nvidia/nvidia-nemotron-nano-9b-v2",
681
+ "openai/gpt-oss-120b", "openai/gpt-oss-20b",
682
+ // z-ai/glm-5.3-flash belongs in NVIDIA_NIM_VISION_MODELS, not here: Z.AI documents
683
+ // it under docs.z.ai/guides/vlm/. The header above says an id must be classified
684
+ // deliberately rather than assumed from its name, and inheriting glm-5.3's
685
+ // text-only verdict because of the shared prefix is exactly that mistake.
686
+ "poolside/laguna-xs-2.1", "z-ai/glm-5.3", "z-ai/glm-5.2",
687
+ ];
688
+ export const KIMI_CODING_MODEL_CONTEXT_WINDOWS: Record<string, number> = Object.fromEntries(
689
+ KIMI_CODING_MODELS.map(id => [id, id === "k3[1m]" ? KIMI_K3_1M_CONTEXT_WINDOW : KIMI_K3_STANDARD_CONTEXT_WINDOW]),
690
+ );
691
+ export const KIMI_CODING_MODEL_INPUT_MODALITIES = Object.fromEntries(
692
+ KIMI_CODING_K3_MODELS.map(id => [id, ["text", "image"]]),
693
+ );
694
+ export const NEURALWATT_REASONING_HISTORY_MODELS = [
695
+ "glm-5.3", "glm-5.3-short", "glm-5.3-flash",
696
+ "glm-5.2", "glm-5.2-short",
697
+ "kimi-k2.6", "kimi-k2.7-code",
698
+ "qwen3.5-397b", "qwen3.6-35b",
699
+ ];
700
+
701
+ // 260728 Baseten Model APIs: `/v1/models` owns the live lineup, while these hints
702
+ // describe only capabilities that Baseten documents per slug. Unlisted live models
703
+ // intentionally inherit the empty provider ladder instead of being advertised with
704
+ // opencodex's generic reasoning defaults. Audio is omitted because the current proxy
705
+ // request model does not carry OpenAI `audio_url` parts.
706
+ // Evidence: https://docs.baseten.co/inference/model-apis/reasoning
707
+ // https://docs.baseten.co/inference/model-apis/vision
708
+ export const BASETEN_FULL_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
709
+ export const BASETEN_MODEL_REASONING_EFFORTS: Record<string, string[]> = {
710
+ "thinkingmachines/inkling": BASETEN_FULL_REASONING_EFFORTS,
711
+ "openai/gpt-oss-120b": BASETEN_FULL_REASONING_EFFORTS,
712
+ "moonshotai/Kimi-K3": ["low", "high", "max"],
713
+ // 260814: GLM-5.3 honours low/high/max upstream, unlike 5.2's high/max on Baseten.
714
+ "zai-org/GLM-5.3": ["low", "high", "max"],
715
+ "zai-org/GLM-5.3-Fast": ["low", "high", "max"],
716
+ "zai-org/GLM-5.2": ["high", "max"],
717
+ "zai-org/GLM-5.2-Fast": ["high", "max"],
718
+ };
719
+ export const BASETEN_MODEL_REASONING_EFFORT_MAP: Record<string, Record<string, string>> = {
720
+ "thinkingmachines/inkling": { none: "none", minimal: "minimal" },
721
+ "openai/gpt-oss-120b": { none: "none", minimal: "minimal" },
722
+ "moonshotai/Kimi-K3": { none: "none" },
723
+ "zai-org/GLM-5.3": { none: "none" },
724
+ "zai-org/GLM-5.3-Fast": { none: "none" },
725
+ "zai-org/GLM-5.2": { none: "none" },
726
+ "zai-org/GLM-5.2-Fast": { none: "none" },
727
+ };
728
+ export const BASETEN_MODEL_DEFAULT_REASONING_EFFORTS: Record<string, string> = {
729
+ "thinkingmachines/inkling": "high",
730
+ "openai/gpt-oss-120b": "medium",
731
+ "moonshotai/Kimi-K3": "max",
732
+ };
733
+ export const BASETEN_MODEL_INPUT_MODALITIES: Record<string, string[]> = {
734
+ "thinkingmachines/inkling": ["text", "image"],
735
+ "moonshotai/Kimi-K2.6": ["text", "image"],
736
+ "moonshotai/Kimi-K2.7-Code": ["text", "image"],
737
+ "moonshotai/Kimi-K3": ["text", "image"],
738
+ };
739
+
740
+ // 260801 DigitalOcean and Scaleway expose OpenAI-shaped `/v1/models` rows with only
741
+ // id/object/created/owned_by, while their shared serverless catalogs also contain
742
+ // non-chat and endpoint-specific models. Fail closed by intersecting live discovery
743
+ // with ids that the providers' current first-party model tables establish for Chat
744
+ // Completions. A newly listed id therefore needs a docs-backed registry refresh before
745
+ // it can enter the Codex catalog.
746
+ // Evidence: https://docs.digitalocean.com/products/inference/details/models/
747
+ // https://docs.digitalocean.com/reference/api/reference/serverless-inference/
748
+ // https://www.scaleway.com/en/docs/generative-apis/reference-content/supported-models/
749
+ export const DIGITALOCEAN_CHAT_COMPLETION_MODELS = [
750
+ "arcee-trinity-large-thinking",
751
+ "openai-gpt-5.6-sol",
752
+ "openai-gpt-5.6-terra",
753
+ "openai-gpt-5.6-luna",
754
+ "qwen3-coder-flash",
755
+ "qwen3.5-397b-a17b",
756
+ "deepseek-4-flash",
757
+ "deepseek-3.2",
758
+ "gemma-4-31B-it",
759
+ "minimax-m2.5",
760
+ "kimi-k3",
761
+ "kimi-k2.6",
762
+ "kimi-k2.5",
763
+ "llama3.3-70b-instruct",
764
+ "llama-4-maverick",
765
+ "mistral-3-14B",
766
+ "nemotron-3-ultra-550b",
767
+ "nvidia-nemotron-3-super-120b",
768
+ "nemotron-3-nano-omni",
769
+ "nemotron-nano-12b-v2-vl",
770
+ "mimo-v2.5-pro",
771
+ "glm-5.3",
772
+ "glm-5.3-flash",
773
+ "glm-5.2",
774
+ "glm-5.1",
775
+ "glm-5",
776
+ // The API reference uses this native slash id in its Chat Completions example.
777
+ "meta-llama/Meta-Llama-3.1-8B-Instruct",
778
+ ] as const;
779
+ export const SCALEWAY_SERVERLESS_CHAT_MODELS = [
780
+ "glm-5.3",
781
+ "glm-5.3-flash",
782
+ "glm-5.2",
783
+ // gpt-oss-120b is intentionally omitted: Scaleway requires Responses API for tool calling,
784
+ // while this preset routes Codex agent tools through Chat Completions.
785
+ "qwen3.6-35b-a3b",
786
+ "qwen3.5-397b-a17b",
787
+ "qwen3-235b-a22b-instruct-2507",
788
+ "qwen3-coder-30b-a3b-instruct",
789
+ "gemma-4-26b-a4b-it",
790
+ "llama-3.3-70b-instruct",
791
+ "mistral-medium-3.5-128b",
792
+ "mistral-small-3.2-24b-instruct-2506",
793
+ "pixtral-12b-2409",
794
+ ] as const;
795
+ export const SCALEWAY_MODEL_INPUT_MODALITIES: Record<string, string[]> = {
796
+ "pixtral-12b-2409": ["text", "image"],
797
+ };
798
+ export const UMANS_MODELS = [
799
+ "umans-coder",
800
+ "umans-kimi-k2.7",
801
+ "umans-flash",
802
+ "umans-glm-5.3",
803
+ "umans-glm-5.3-flash",
804
+ "umans-glm-5.2",
805
+ "umans-glm-5.1",
806
+ "umans-qwen3.6-35b-a3b",
807
+ ];
808
+ export const UMANS_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
809
+ export const UMANS_GLM_REASONING_EFFORTS = ["high", "xhigh", "max"];
810
+ // 260814: Z.AI folds GLM-5.3 efforts into low/high/max, so `low` is a real tier here and
811
+ // `xhigh` is not distinct from `max` (docs.z.ai/devpack/latest-model).
812
+ export const UMANS_GLM_53_REASONING_EFFORTS = ["low", "high", "max"];
813
+ // `umans-glm-5.3-flash` is NOT here: Z.AI documents glm-5.3-flash under
814
+ // docs.z.ai/guides/vlm/, so it takes images natively and does not need the proxy's
815
+ // vision sidecar. The seeding pass classified it from the family name and a later
816
+ // pass corrected only some of the providers; this is one it missed.
817
+ export const UMANS_TEXT_ONLY_MODELS = ["umans-glm-5.3", "umans-glm-5.2", "umans-glm-5.1"];
818
+ export const UMANS_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
819
+ "umans-coder": 262_144,
820
+ "umans-kimi-k2.7": 262_144,
821
+ "umans-flash": 262_144,
822
+ "umans-glm-5.3": 405_504,
823
+ // Mirrors the sibling this provider already carries. Umans has not published a
824
+ // separate window for the flash tier; asserting a different number would be a guess.
825
+ "umans-glm-5.3-flash": 405_504,
826
+ "umans-glm-5.2": 405_504,
827
+ "umans-glm-5.1": 202_752,
828
+ "umans-qwen3.6-35b-a3b": 262_144,
829
+ };
830
+ export const UMANS_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
831
+ UMANS_MODELS.map(id => [id, UMANS_TEXT_ONLY_MODELS.includes(id) ? ["text"] : ["text", "image"]]),
832
+ );
833
+ export const CLINE_PASS_MODELS = [
834
+ "cline-pass/glm-5.3",
835
+ "cline-pass/glm-5.3-flash",
836
+ "cline-pass/glm-5.2",
837
+ "cline-pass/kimi-k3",
838
+ "cline-pass/kimi-k2.7-code",
839
+ "cline-pass/kimi-k2.6",
840
+ "cline-pass/deepseek-v4-flash",
841
+ "cline-pass/mimo-v2.5",
842
+ "cline-pass/mimo-v2.5-pro",
843
+ "cline-pass/minimax-m3",
844
+ "cline-pass/qwen3.8-max",
845
+ "cline-pass/qwen3.7-max",
846
+ "cline-pass/qwen3.7-plus",
847
+ ];
848
+
849
+ export const ORCAROUTER_MODEL_DISCOVERY: ProviderModelDiscoverySpec = {
850
+ path: "models",
851
+ query: { capability: "chat" },
852
+ maxResponseBytes: 512 * 1024,
853
+ maxModels: 512,
854
+ filter: {
855
+ anyOf: [{
856
+ path: ["supported_endpoint_types"],
857
+ containsAny: ["openai", "openai-response", "anthropic", "gemini"],
858
+ caseInsensitive: true,
859
+ }],
860
+ noneOf: [{
861
+ path: ["supported_endpoint_types"],
862
+ containsAny: ["image-generation", "openai-video", "jina-rerank"],
863
+ caseInsensitive: true,
864
+ }],
865
+ },
866
+ };
867
+ // Preserve the previously verified cold-start catalog. Live discovery remains authoritative
868
+ // when it succeeds, but a temporary catalog outage must not erase the provider's known-good
869
+ // selectors from the picker. `orcarouter/auto` is intentionally retained here even though the
870
+ // public catalog did not enumerate it at the latest verification (2026-09-07).
871
+ export const ORCAROUTER_MODELS = [
872
+ "openai/gpt-5.5",
873
+ "anthropic/claude-opus-4.8",
874
+ "google/gemini-3.5-flash",
875
+ "orcarouter/auto",
876
+ ];
877
+ export const ORCAROUTER_MODEL_REASONING_EFFORTS = {
878
+ // Live /models currently exposes ids and modalities, not the accepted reasoning ladder.
879
+ "openai/gpt-5.5": ["low", "medium", "high", "xhigh"],
880
+ };
881
+ export const CLINE_PASS_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
882
+ "cline-pass/glm-5.3": 1_048_576,
883
+ "cline-pass/glm-5.3-flash": 1_048_576,
884
+ "cline-pass/glm-5.2": 1_048_576,
885
+ "cline-pass/kimi-k3": 1_048_576,
886
+ "cline-pass/kimi-k2.7-code": 262_144,
887
+ "cline-pass/kimi-k2.6": 262_144,
888
+ "cline-pass/deepseek-v4-flash": 1_048_576,
889
+ "cline-pass/mimo-v2.5": 1_050_000,
890
+ "cline-pass/mimo-v2.5-pro": 1_050_000,
891
+ "cline-pass/minimax-m3": 1_048_576,
892
+ "cline-pass/qwen3.7-max": 1_000_000,
893
+ "cline-pass/qwen3.7-plus": 1_000_000,
894
+ };
895
+ export const CLINE_PASS_IMAGE_MODELS = new Set([
896
+ "cline-pass/kimi-k3",
897
+ "cline-pass/kimi-k2.7-code",
898
+ "cline-pass/kimi-k2.6",
899
+ "cline-pass/mimo-v2.5",
900
+ "cline-pass/minimax-m3",
901
+ "cline-pass/qwen3.7-plus",
902
+ // Native VLM (docs.z.ai/guides/vlm/), so its images do not go through the proxy's
903
+ // sidecar. Adding it here moves it out of CLINE_PASS_TEXT_ONLY_MODELS and flips its
904
+ // declared modalities to ["text", "image"] in one edit, because both are derived
905
+ // from this set.
906
+ "cline-pass/glm-5.3-flash",
907
+ ]);
908
+ export const CLINE_PASS_MODALITY_KNOWN_MODELS = CLINE_PASS_MODELS.filter(id => id !== "cline-pass/qwen3.8-max");
909
+ export const CLINE_PASS_TEXT_ONLY_MODELS = CLINE_PASS_MODALITY_KNOWN_MODELS.filter(id => !CLINE_PASS_IMAGE_MODELS.has(id));
910
+ export const CLINE_PASS_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
911
+ CLINE_PASS_MODALITY_KNOWN_MODELS.map(id => [id, CLINE_PASS_IMAGE_MODELS.has(id) ? ["text", "image"] : ["text"]]),
912
+ );