@bitkyc08/opencodex 2.57.0 → 2.59.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (241) hide show
  1. package/README.md +28 -10
  2. package/gui/dist/assets/index-C5IebErG.js +136 -0
  3. package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
  4. package/gui/dist/index.html +2 -2
  5. package/gui/dist/provider-icons/crusoe.svg +1 -0
  6. package/gui/dist/provider-icons/opper.svg +3 -0
  7. package/package.json +2 -2
  8. package/src/adapters/base.ts +11 -1
  9. package/src/adapters/codebuddy/scaffold-guard.ts +5 -4
  10. package/src/adapters/command-code.ts +13 -4
  11. package/src/adapters/cursor/catalog.ts +11 -0
  12. package/src/adapters/cursor/cursor-errors.ts +15 -0
  13. package/src/adapters/cursor/discovery.ts +65 -1
  14. package/src/adapters/cursor/effort-map.ts +16 -2
  15. package/src/adapters/cursor/envelope-echo.ts +55 -2
  16. package/src/adapters/cursor/live-transport.ts +5 -1
  17. package/src/adapters/cursor/message-mapper.ts +3 -2
  18. package/src/adapters/cursor/protobuf-events.ts +110 -11
  19. package/src/adapters/cursor/protobuf-request.ts +27 -6
  20. package/src/adapters/cursor/request-builder.ts +14 -3
  21. package/src/adapters/cursor/text-toolcall.ts +230 -0
  22. package/src/adapters/cursor/thread-continuity.ts +141 -0
  23. package/src/adapters/cursor/tool-guidance.ts +5 -4
  24. package/src/adapters/cursor/types.ts +5 -0
  25. package/src/adapters/cursor.ts +97 -6
  26. package/src/adapters/devin/cloud-direct/chat.ts +11 -2
  27. package/src/adapters/devin/cloud-direct/index.ts +7 -0
  28. package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
  29. package/src/adapters/devin.ts +75 -13
  30. package/src/adapters/google-antigravity-wire.ts +29 -2
  31. package/src/adapters/google-http.ts +45 -13
  32. package/src/adapters/google.ts +23 -4
  33. package/src/adapters/mimo-free.ts +32 -17
  34. package/src/adapters/ollama-native.ts +42 -8
  35. package/src/adapters/openai-chat/response-events.ts +61 -0
  36. package/src/adapters/openai-chat.ts +5 -10
  37. package/src/adapters/openai-responses/passthrough.ts +40 -5
  38. package/src/adapters/openai-responses/request-strips.ts +43 -0
  39. package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
  40. package/src/adapters/openai-responses/tool-schema.ts +19 -7
  41. package/src/adapters/physical-send.ts +50 -0
  42. package/src/adapters/responses-tool-schema.ts +76 -46
  43. package/src/adapters/run-turn-queue.ts +17 -4
  44. package/src/bridge/response-json.ts +2 -2
  45. package/src/bridge/sse.ts +166 -25
  46. package/src/claude/context-windows.ts +22 -0
  47. package/src/claude/outbound.ts +46 -5
  48. package/src/cli/account-api.ts +4 -3
  49. package/src/cli/account-extended.ts +22 -2
  50. package/src/cli/account-orca-import.ts +63 -0
  51. package/src/cli/account.ts +32 -4
  52. package/src/cli/capabilities.ts +40 -0
  53. package/src/cli/claude.ts +29 -1
  54. package/src/cli/codex-cli-update.ts +97 -2
  55. package/src/cli/config-command.ts +35 -18
  56. package/src/cli/dispatch.ts +71 -4
  57. package/src/cli/doctor.ts +197 -2
  58. package/src/cli/help.ts +4 -1
  59. package/src/cli/index.ts +132 -22
  60. package/src/cli/models-runtime.ts +33 -4
  61. package/src/cli/registry.ts +11 -1
  62. package/src/cli/runtime-api.ts +44 -0
  63. package/src/cli/start-args.ts +94 -0
  64. package/src/cli/system-command.ts +72 -1
  65. package/src/cli/uninstall-client-state.ts +12 -0
  66. package/src/client/machine-api.ts +4 -3
  67. package/src/client/machine-listener.ts +14 -1
  68. package/src/clients/config-export/constants.ts +2 -3
  69. package/src/clients/config-export.ts +5 -5
  70. package/src/codex/account-store.ts +81 -5
  71. package/src/codex/auth-api/pool-quota-probe.ts +14 -3
  72. package/src/codex/auth-api/routes.ts +17 -2
  73. package/src/codex/auth-context.ts +58 -20
  74. package/src/codex/catalog/build-entries.ts +25 -4
  75. package/src/codex/catalog/derive-entry.ts +8 -1
  76. package/src/codex/catalog/effort.ts +10 -6
  77. package/src/codex/catalog/gather-capture.ts +1 -0
  78. package/src/codex/catalog/model-hints.ts +37 -5
  79. package/src/codex/catalog/parsing.ts +83 -5
  80. package/src/codex/catalog/reserve-warn.ts +96 -0
  81. package/src/codex/catalog/retained-sync.ts +19 -0
  82. package/src/codex/catalog/routed-gather.ts +42 -3
  83. package/src/codex/cli-installation-identity.ts +210 -0
  84. package/src/codex/cli-installation-targets.ts +158 -0
  85. package/src/codex/convergence.ts +5 -0
  86. package/src/codex/desktop-switches.ts +145 -0
  87. package/src/codex/history-job.ts +5 -1
  88. package/src/codex/history-provider.ts +37 -5
  89. package/src/codex/history-state-open.ts +105 -0
  90. package/src/codex/history-worker.ts +14 -1
  91. package/src/codex/inject/config-toml.ts +44 -2
  92. package/src/codex/inject/remove.ts +145 -7
  93. package/src/codex/inject/restore.ts +204 -32
  94. package/src/codex/inject.ts +6 -9
  95. package/src/codex/lineage.ts +83 -32
  96. package/src/codex/loopback-target.ts +40 -0
  97. package/src/codex/main-account-hard-lock.ts +2 -1
  98. package/src/codex/main-account.ts +10 -3
  99. package/src/codex/main-device-reauth.ts +17 -9
  100. package/src/codex/model-entitlements.ts +60 -1
  101. package/src/codex/native-profile-startup.ts +64 -20
  102. package/src/codex/observed-model-denials.ts +137 -0
  103. package/src/codex/orca-auth-source.ts +94 -0
  104. package/src/codex/orca-import.ts +219 -0
  105. package/src/codex/prompt-text-probe.ts +282 -12
  106. package/src/codex/quota-401-recovery.ts +12 -0
  107. package/src/codex/quota-types.ts +65 -0
  108. package/src/codex/quota.ts +24 -19
  109. package/src/codex/routing/cooldown-math.ts +8 -47
  110. package/src/codex/routing/pin-drain.ts +57 -0
  111. package/src/codex/routing.ts +13 -15
  112. package/src/codex/subagent-model-fallback.ts +94 -0
  113. package/src/codex/windows-installation-files.ts +224 -0
  114. package/src/combos/failover.ts +122 -5
  115. package/src/config/atomic-write.ts +83 -8
  116. package/src/config/diagnostics.ts +21 -0
  117. package/src/config/load-degrade.ts +15 -0
  118. package/src/config/pending-teardown.ts +8 -0
  119. package/src/config/process-state.ts +36 -3
  120. package/src/config/provider-relative-send-path.ts +16 -0
  121. package/src/config/proxy-env.ts +23 -5
  122. package/src/config/schema/config-schema.ts +23 -0
  123. package/src/config/schema/leaf-validators.ts +65 -17
  124. package/src/generated/compatibility-version.json +337 -201
  125. package/src/generated/model-metadata.ts +1 -1
  126. package/src/lib/bounded-body.ts +4 -2
  127. package/src/lib/bounded-subprocess.ts +62 -10
  128. package/src/lib/destination-policy.ts +48 -6
  129. package/src/lib/errors.ts +3 -15
  130. package/src/lib/local-destinations.ts +32 -5
  131. package/src/lib/provider-outbound.ts +3 -3
  132. package/src/lib/proxy-env.ts +70 -3
  133. package/src/lib/request-execution-budget.ts +11 -3
  134. package/src/lib/response-body-inactivity.ts +193 -0
  135. package/src/lib/retry-delay.ts +69 -0
  136. package/src/lib/socks5-fetch.ts +631 -0
  137. package/src/lib/spend-reservation-ledger.ts +115 -9
  138. package/src/lib/windows-secret-acl.ts +151 -15
  139. package/src/lib/windows-user-principal.ts +5 -1
  140. package/src/lib/workflow-budget.ts +145 -8
  141. package/src/oauth/account-quota-rank.ts +72 -15
  142. package/src/oauth/generic-account-failover.ts +40 -27
  143. package/src/oauth/orcarouter.ts +15 -2
  144. package/src/oauth/store.ts +8 -0
  145. package/src/providers/codex-capacity.ts +9 -0
  146. package/src/providers/derive.ts +6 -0
  147. package/src/providers/devin-provider-merge-migration.ts +33 -12
  148. package/src/providers/free-directory.ts +20 -2
  149. package/src/providers/key-failover.ts +261 -7
  150. package/src/providers/model-discovery.ts +19 -7
  151. package/src/providers/model-rename-migration.ts +1 -0
  152. package/src/providers/openai-sidecar.ts +4 -0
  153. package/src/providers/opencode-go-transport.ts +14 -5
  154. package/src/providers/quota/report-cache.ts +3 -0
  155. package/src/providers/registry/entries-core.ts +11 -0
  156. package/src/providers/registry/entries-extended.ts +146 -28
  157. package/src/providers/registry/model-seeds.ts +136 -29
  158. package/src/providers/registry/types.ts +9 -0
  159. package/src/responses/apply-patch-envelope.ts +44 -11
  160. package/src/responses/bridge-search-replay-cache.ts +152 -0
  161. package/src/responses/code-mode-helper-compat.ts +26 -16
  162. package/src/responses/custom-tool-compat.ts +1 -1
  163. package/src/responses/hosted-tool-policy.ts +85 -2
  164. package/src/responses/schema.ts +9 -2
  165. package/src/responses/spill-store.ts +17 -0
  166. package/src/responses/state/body-policy.ts +25 -0
  167. package/src/responses/state/spill-queue.ts +8 -6
  168. package/src/responses/state.ts +3 -22
  169. package/src/router.ts +4 -0
  170. package/src/server/auth-cors.ts +27 -0
  171. package/src/server/chat-completions.ts +9 -4
  172. package/src/server/chat-native-sse.ts +26 -9
  173. package/src/server/chat-native.ts +10 -4
  174. package/src/server/claude-messages.ts +24 -2
  175. package/src/server/gui-static.ts +36 -2
  176. package/src/server/inbound-body-admission.ts +187 -0
  177. package/src/server/index/websocket-handler.ts +48 -1
  178. package/src/server/index.ts +15 -19
  179. package/src/server/management/api-access.ts +3 -4
  180. package/src/server/management/config-routes.ts +57 -10
  181. package/src/server/management/provider-capability-config.ts +35 -7
  182. package/src/server/management/provider-routes.ts +70 -18
  183. package/src/server/models-capabilities.ts +24 -3
  184. package/src/server/proxy-liveness.ts +97 -2
  185. package/src/server/relay.ts +17 -24
  186. package/src/server/request-log.ts +25 -1
  187. package/src/server/responses/adapter-continuation.ts +71 -27
  188. package/src/server/responses/adapter-delivery.ts +39 -8
  189. package/src/server/responses/adapter-dispatch.ts +52 -24
  190. package/src/server/responses/codex-ws-exchange.ts +65 -4
  191. package/src/server/responses/combo-stream-preflight.ts +68 -5
  192. package/src/server/responses/compact.ts +60 -11
  193. package/src/server/responses/core-codex-account.ts +83 -22
  194. package/src/server/responses/core-combo.ts +26 -0
  195. package/src/server/responses/core-normalize.ts +12 -5
  196. package/src/server/responses/core-options.ts +3 -0
  197. package/src/server/responses/fetch-helpers.ts +72 -3
  198. package/src/server/responses/native-injection-protocol.ts +42 -0
  199. package/src/server/responses/native-injection-replay.ts +105 -0
  200. package/src/server/responses/native-injection.ts +242 -0
  201. package/src/server/responses/native-response-control.ts +56 -0
  202. package/src/server/responses/native-response-json.ts +14 -0
  203. package/src/server/responses/native-response-output.ts +37 -0
  204. package/src/server/responses/native-steering-log.ts +44 -0
  205. package/src/server/responses/native-steering-policy.ts +49 -0
  206. package/src/server/responses/native-steering-replay.ts +126 -0
  207. package/src/server/responses/native-steering-settings.ts +76 -0
  208. package/src/server/responses/native-steering.ts +400 -0
  209. package/src/server/responses/native-tool-results.ts +130 -0
  210. package/src/server/responses/passthrough-delivery.ts +21 -1
  211. package/src/server/responses/passthrough-dispatch.ts +146 -49
  212. package/src/server/responses/passthrough-execution.ts +11 -1
  213. package/src/server/responses/request-prepare.ts +70 -0
  214. package/src/server/responses/request-send-budget.ts +84 -7
  215. package/src/server/responses/request-sidecar-auth.ts +16 -8
  216. package/src/server/responses/request-spend.ts +38 -9
  217. package/src/server/responses/request-transport.ts +13 -10
  218. package/src/server/responses/run-turn-execution.ts +20 -5
  219. package/src/server/responses/sidecar-execution.ts +2 -0
  220. package/src/server/responses/ws-upstream.ts +23 -2
  221. package/src/server/responses-custom-tool-repair.ts +2 -2
  222. package/src/server/sse-frame-buffer.ts +12 -10
  223. package/src/server/sse-payload-rewrite.ts +36 -9
  224. package/src/server/stop-teardown.ts +8 -1
  225. package/src/server/system-env-shell.ts +5 -1
  226. package/src/server/system-env.ts +7 -1
  227. package/src/server/workflow-refusal.ts +56 -2
  228. package/src/server/ws-bridge.ts +16 -1
  229. package/src/service/cli.ts +29 -7
  230. package/src/service/guards.ts +10 -0
  231. package/src/service/health.ts +43 -0
  232. package/src/service/state.ts +7 -2
  233. package/src/types/accounts.ts +4 -0
  234. package/src/types/config.ts +104 -3
  235. package/src/types/provider.ts +32 -0
  236. package/src/types/request.ts +7 -1
  237. package/src/types/wire.ts +9 -1
  238. package/src/usage/expected-prices.ts +28 -0
  239. package/src/usage/log.ts +87 -4
  240. package/src/web-search/passthrough-bridge.ts +39 -5
  241. package/gui/dist/assets/index-Cz7CLdif.js +0 -128
@@ -282,6 +282,17 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
282
282
  forwardCallerServiceTier: false,
283
283
  },
284
284
  },
285
+ // Grok 4.6/4.5 OAuth Responses replays Codex tool history. After a mid-stream 502/reset,
286
+ // the client can resend a function_call without a matching output, or with hook-injected
287
+ // developer context between the pair. Google already synthesizes a missing tool_result
288
+ // (#2199). xAI's Responses parser does not, so the next turns 400 and the thread snowballs.
289
+ // Reuse the existing adjacency capability (Kimi #4726, DeepSeek #1292). Do not set
290
+ // statelessResponses: xAI stores responses for 30 days and documents previous_response_id.
291
+ // https://docs.x.ai/developers/model-capabilities/text/comparison
292
+ requiresAdjacentResponsesToolResults: true,
293
+ // The dangling half of the same failure: a call whose output never arrived. Kimi accepts that
294
+ // shape, so this is a second capability rather than a widening of the one above.
295
+ requiresPairedResponsesToolResults: true,
285
296
  // Vision lineup per docs.x.ai model-capabilities/images/understanding: the grok-4.x chat
286
297
  // models accept image input (JPEG/PNG, URL or base64). Without this the catalog leaves
287
298
  // inputModalities undefined, and deriveComboCatalogModel defaults an undefined member to
@@ -56,6 +56,11 @@ import {
56
56
  ALIBABA_TOKEN_PLAN_MODELS,
57
57
  ALIBABA_TOKEN_PLAN_QWEN_MODELS,
58
58
  ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
59
+ ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
60
+ ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
61
+ ALIBABA_TOKEN_PLAN_NO_VISION,
62
+ ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
63
+ QWEN38_FAMILY,
59
64
  ALIBABA_INTL_TOKEN_PLAN_MODELS,
60
65
  ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS,
61
66
  TENCENT_CODING_PLAN_MODELS,
@@ -93,6 +98,10 @@ import {
93
98
  DIGITALOCEAN_CHAT_COMPLETION_MODELS,
94
99
  SCALEWAY_SERVERLESS_CHAT_MODELS,
95
100
  SCALEWAY_MODEL_INPUT_MODALITIES,
101
+ OPPER_MODELS,
102
+ OPPER_MODEL_CONTEXT_WINDOWS,
103
+ OPPER_MODEL_MAX_OUTPUT_TOKENS,
104
+ OPPER_MODEL_INPUT_MODALITIES,
96
105
  } from "./model-seeds";
97
106
 
98
107
  export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
@@ -214,6 +223,72 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
214
223
  },
215
224
  note: "Shared Token Factory text-output inference only; live discovery excludes embedding and image-generation rows.",
216
225
  },
226
+ {
227
+ // Primary sources checked 2026-09-11:
228
+ // - https://docs.crusoecloud.com/quickstart/getting-started-with-serverless-inference documents
229
+ // the fixed OpenAI-compatible host https://api.inference.crusoecloud.com/v1, Bearer API keys
230
+ // created in the Cloud console (Intelligence Foundry > Inference > Create API Key), and an
231
+ // OpenAI SDK chat.completions example against meta-llama/Llama-3.3-70B-Instruct.
232
+ // - https://docs.crusoecloud.com/serverless-inference/available-models lists the served models
233
+ // with slash-delimited ids; https://docs.crusoecloud.com/serverless-inference/rate-limits
234
+ // documents per-project, per-model TPM/RPM limits (429 when exceeded, 503 under shared load).
235
+ // - GET /v1/models rejects unauthenticated requests with 401 {"errors":["Authentication failed"]},
236
+ // so a successful authenticated list response is evidence that the supplied key is valid.
237
+ // An authenticated capture on 2026-09-12 returned 18 rows shaped like OpenRouter's catalog
238
+ // (`is_public`, `type`, `context_length`, `architecture.modality` of "text" or "multimodal",
239
+ // `tags`, `pricing`, `supported_parameters`); 17 were public serverless models and one was an
240
+ // account-private dedicated deployment with empty `type`/`modality`. `type` is blank on one
241
+ // public model, so the filter keys on `is_public` plus `architecture.modality` instead.
242
+ // - https://legal.crusoe.ai/ hosts the Crusoe Cloud Platform Terms of Service v1.10 (effective
243
+ // 2026-08-10), which name Crusoe Technologies LLC as the contracting entity, and the Service
244
+ // Specific Terms v5.0 (effective 2026-07-14), whose Crusoe Intelligence Foundry Terms cover the
245
+ // Managed Inference Service reached through the Crusoe API.
246
+ // - https://models.dev/api.json (provider "crusoe") records openai/gpt-oss-120b as the one served
247
+ // model with a low/medium/high reasoning_effort ladder; the other reasoning models expose an
248
+ // on/off toggle only.
249
+ // Maintainer: @acheamponge, who works at Crusoe (affiliation disclosed) and also maintains the
250
+ // models.dev crusoe entry.
251
+ id: "crusoe",
252
+ label: "Crusoe",
253
+ baseUrl: "https://api.inference.crusoecloud.com/v1",
254
+ adapter: "openai-chat",
255
+ authKind: "key",
256
+ dashboardUrl: "https://console.crusoecloud.com",
257
+ liveModels: true,
258
+ preserveCustomDestination: true,
259
+ // The getting-started guide documents tools through the OpenAI SDK but no provider-wide
260
+ // parallel tool-call contract.
261
+ parallelToolCalls: false,
262
+ // Only gpt-oss-120b has a real effort ladder; toggle-style reasoning models must not be promoted
263
+ // to Codex's full fallback ladder.
264
+ reasoningEfforts: [],
265
+ modelReasoningEfforts: { "openai/gpt-oss-120b": ["low", "medium", "high"] },
266
+ directReasoningEffortModels: ["openai/gpt-oss-120b"],
267
+ // The catalog reports `architecture.modality: "multimodal"` without an input list. Four rows
268
+ // also carry the explicit "image text to text" tag; yutori/n2 instead reports multimodal
269
+ // type/modality plus browser/computer-use tags. Those five captured rows are classified here.
270
+ modelInputModalities: {
271
+ "google/gemma-4-31b-it": ["text", "image"],
272
+ "moonshotai/Kimi-K2.6": ["text", "image"],
273
+ "nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B": ["text", "image"],
274
+ "yutori/n2": ["text", "image"],
275
+ "zai-org/GLM-5.3-Flash": ["text", "image"],
276
+ },
277
+ modelDiscovery: {
278
+ path: "models",
279
+ maxResponseBytes: 256 * 1024,
280
+ maxModels: 256,
281
+ filter: {
282
+ // Keep public serverless rows whose architecture produces text; account-private
283
+ // deployments (blank modality) and any embedding or media rows fail closed.
284
+ allOf: [
285
+ { path: ["is_public"], equalsAny: [true] },
286
+ { path: ["architecture", "modality"], equalsAny: ["text", "multimodal"] },
287
+ ],
288
+ },
289
+ },
290
+ note: "Public Serverless Inference chat models on the shared OpenAI-compatible host; account-private and self-serve dedicated deployments are excluded from discovery and out of scope.",
291
+ },
217
292
  {
218
293
  id: "digitalocean",
219
294
  label: "DigitalOcean Serverless Inference",
@@ -427,6 +502,7 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
427
502
  // model_access_denied, which is why the Chat path cannot simply hang off the new base.
428
503
  responsesPath: "/api/v1/responses",
429
504
  chatCompletionsPath: "/api/coding/paas/v4/chat/completions",
505
+ modelDiscovery: { path: "/api/v1/models", envelopeKey: "models", idField: "slug" },
430
506
  // The address this row occupied before the move. A saved custom provider still pointing
431
507
  // at the Chat endpoint keeps receiving this row's metadata (#1100).
432
508
  destinationAliases: [{ baseUrl: "https://api.z.ai/api/coding/paas/v4", adapter: "openai-chat" }],
@@ -724,22 +800,36 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
724
800
  liveModels: false,
725
801
  note: "Token Plan Personal Edition · China (Beijing)",
726
802
  modelInputModalities: ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
727
- modelContextWindows: {
728
- "qwen3.8-max": 983_616, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
729
- "qwen3.6-flash": 1_000_000, "glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000,
730
- },
803
+ modelContextWindows: ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
804
+ modelMaxOutputTokens: ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
731
805
  modelReasoningEfforts: {
732
806
  ...Object.fromEntries(ALIBABA_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
733
- "qwen3.8-max": QWEN38_REASONING_EFFORTS,
734
- "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
735
- "glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
807
+ ...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
736
808
  "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
809
+ "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
810
+ "deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
811
+ "deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
812
+ "deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
813
+ "deepseek-v4.1-flash": deepseekThinkingEffortsFor("deepseek-v4.1-flash"),
814
+ },
815
+ modelReasoningEffortMap: {
816
+ "deepseek-v4-pro": deepseekReasoningMapFor("deepseek-v4-pro"),
817
+ "deepseek-v4-pro-0813": deepseekReasoningMapFor("deepseek-v4-pro-0813"),
818
+ "deepseek-v4-flash-0731": deepseekReasoningMapFor("deepseek-v4-flash-0731"),
819
+ "deepseek-v4.1-flash": deepseekReasoningMapFor("deepseek-v4.1-flash"),
737
820
  },
738
- modelDefaultReasoningEfforts: { "qwen3.8-max": "xhigh" },
739
- directReasoningEffortModels: ["qwen3.8-max"],
740
- thinkingBudgetModels: ALIBABA_TOKEN_PLAN_QWEN_MODELS.filter(id => id !== "qwen3.8-max"),
741
- preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash"],
742
- noVisionModels: ["glm-5.3", "glm-5.2"],
821
+ // Probed 260915 on the plan gateway: json_object returns valid JSON, strict
822
+ // json_schema is rejected 400 ("This response_format type is unavailable now")
823
+ // in both thinking modes, so requests downgrade to json_object rather than
824
+ // sending a schema the gateway refuses.
825
+ noJsonSchemaModels: ["deepseek-v4.1-flash"],
826
+ modelDefaultReasoningEfforts: Object.fromEntries(QWEN38_FAMILY.map(id => [id, "xhigh"])),
827
+ directReasoningEffortModels: QWEN38_FAMILY,
828
+ thinkingBudgetModels: ALIBABA_TOKEN_PLAN_QWEN_MODELS.filter(id => !QWEN38_FAMILY.includes(id)),
829
+ preserveReasoningContentModels: ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
830
+ noVisionModels: ALIBABA_TOKEN_PLAN_NO_VISION,
831
+ // The gateway accepts prompt_cache_key on every Token Plan chat model (probed 260902).
832
+ promptCacheKey: true,
743
833
  },
744
834
  {
745
835
  id: "alibaba-token-plan-intl",
@@ -756,31 +846,35 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
756
846
  note: "Token Plan Team Edition · Singapore (ap-southeast-1)",
757
847
  metadataModelIdNormalize: "case-insensitive",
758
848
  modelInputModalities: ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES,
759
- modelContextWindows: {
760
- "qwen3.8-max": 983_616,
761
- "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000, "qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
762
- "deepseek-v4-flash": 1_000_000, "deepseek-v3.2": 131_072,
763
- "kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
764
- "glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000, "glm-5.1": 1_000_000, "glm-5": 1_000_000,
765
- "MiniMax-M2.5": 204_800,
766
- },
849
+ modelContextWindows: ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
850
+ modelMaxOutputTokens: ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
767
851
  modelReasoningEfforts: {
768
852
  ...Object.fromEntries(ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
769
- "qwen3.8-max": QWEN38_REASONING_EFFORTS,
770
- "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
771
- "glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
853
+ ...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
772
854
  "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
855
+ "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
856
+ "deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
857
+ "deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
773
858
  "deepseek-v4-flash": deepseekThinkingEffortsFor("deepseek-v4-flash"),
859
+ "deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
860
+ "deepseek-v4.1-flash": deepseekThinkingEffortsFor("deepseek-v4.1-flash"),
774
861
  },
775
862
  modelReasoningEffortMap: {
863
+ "deepseek-v4-pro": deepseekReasoningMapFor("deepseek-v4-pro"),
864
+ "deepseek-v4-pro-0813": deepseekReasoningMapFor("deepseek-v4-pro-0813"),
776
865
  "deepseek-v4-flash": deepseekReasoningMapFor("deepseek-v4-flash"),
866
+ "deepseek-v4-flash-0731": deepseekReasoningMapFor("deepseek-v4-flash-0731"),
867
+ "deepseek-v4.1-flash": deepseekReasoningMapFor("deepseek-v4.1-flash"),
777
868
  },
778
- directReasoningEffortModels: ["qwen3.8-max"],
779
- thinkingBudgetModels: ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.filter(id => id !== "qwen3.8-max"),
780
- preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "deepseek-v4-flash", "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash"],
781
- noVisionModels: ["deepseek-v4-flash", "deepseek-v3.2", "glm-5.3", "glm-5.2", "glm-5.1", "glm-5", "MiniMax-M2.5"],
869
+ // Same 260915 json_schema rejection probe as the Beijing entry.
870
+ noJsonSchemaModels: ["deepseek-v4.1-flash"],
871
+ directReasoningEffortModels: QWEN38_FAMILY,
872
+ thinkingBudgetModels: ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.filter(id => !QWEN38_FAMILY.includes(id)),
873
+ preserveReasoningContentModels: ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
874
+ noVisionModels: ALIBABA_TOKEN_PLAN_NO_VISION,
782
875
  noReasoningModels: ["kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "deepseek-v3.2", "glm-5.1", "glm-5", "MiniMax-M2.5"],
783
- modelDefaultReasoningEfforts: { "qwen3.8-max": "xhigh" },
876
+ modelDefaultReasoningEfforts: Object.fromEntries(QWEN38_FAMILY.map(id => [id, "xhigh"])),
877
+ promptCacheKey: true,
784
878
  },
785
879
  // NEEDS_HUMAN 2026-07-10: kept for config compatibility, but this is a dashboard URL,
786
880
  // no /models endpoint is documented, and tools are silently ignored upstream per docs.parallel.ai.
@@ -935,6 +1029,30 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
935
1029
  noJsonSchemaModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS, ...OPENCODE_FREE_DEEPSEEK_MODELS],
936
1030
  },
937
1031
  { id: "vercel-ai-gateway", label: "Vercel AI Gateway", baseUrl: "https://ai-gateway.vercel.sh/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://vercel.com/dashboard" },
1032
+ {
1033
+ // Opper: EU-hosted AI gateway (Opper AI AB, Stockholm). One OpenAI-compatible endpoint and one
1034
+ // key in front of 30+ upstream providers. Seeded ids are Opper *pools* (bare names such as
1035
+ // `claude-sonnet-4-6`): the gateway chooses the provider/region per request, and a
1036
+ // `vendor/model` id (`anthropic/claude-sonnet-4-6`, `aws/claude-sonnet-4-6-eu`) pins one route.
1037
+ // The original provider author reported on 2026-09-08 that GET /v3/compat/models answers 401
1038
+ // without a key, so the default discovery URL doubles as key validation. Windows, output caps
1039
+ // and modalities live in model-seeds.ts (smallest value / shared modality across each pool's
1040
+ // members); live discovery owns which models exist.
1041
+ id: "opper",
1042
+ label: "Opper",
1043
+ adapter: "openai-chat",
1044
+ baseUrl: "https://api.opper.ai/v3/compat",
1045
+ authKind: "key",
1046
+ dashboardUrl: "https://platform.opper.ai",
1047
+ liveModels: true,
1048
+ preserveCustomDestination: true,
1049
+ defaultModel: "claude-sonnet-4-6",
1050
+ models: OPPER_MODELS,
1051
+ modelContextWindows: OPPER_MODEL_CONTEXT_WINDOWS,
1052
+ modelMaxOutputTokens: OPPER_MODEL_MAX_OUTPUT_TOKENS,
1053
+ modelInputModalities: OPPER_MODEL_INPUT_MODALITIES,
1054
+ note: "EU-hosted AI gateway: one OpenAI-compatible endpoint and one key in front of 30+ providers. Bare model ids are pools (claude-sonnet-4-6, gpt-5.5) and Opper picks the route per request; vendor/model ids (anthropic/claude-sonnet-4-6) pin one provider. The catalogue is discovered live from /v3/compat/models with your key; the public list is at opper.ai/models. Token rates are the model providers' rates with no markup; Opper charges a 3% fee when you buy credits.",
1055
+ },
938
1056
  {
939
1057
  id: "opencode-free",
940
1058
  label: "OpenCode Free",
@@ -315,6 +315,13 @@ export const DEEPSEEK_VISION_PREVIEW_MODEL = "deepseek-v4-flash-vision-exp";
315
315
  */
316
316
  export const COMMAND_CODE_IMAGE_MODELS = [
317
317
  `deepseek/${DEEPSEEK_VISION_PREVIEW_MODEL}`,
318
+ // Probed 2026-09-18 through a running 2.58.0 proxy: a 3x3 random-color grid
319
+ // (180x180 PNG, six candidate colors) came back 9/9 correct both as a user
320
+ // message and as a tool_result, and the request logs show the route served
321
+ // the image natively — no vision-sidecar call in either window. #4505 asked
322
+ // for exactly this upstream probe before promoting the id. The sibling
323
+ // deepseek/deepseek-v4-flash route remains verified-negative above.
324
+ "deepseek/deepseek-v4.1-flash",
318
325
  "gpt-5.6-luna",
319
326
  "gpt-5.6-sol",
320
327
  "MiniMaxAI/MiniMax-M3",
@@ -334,20 +341,19 @@ export const COMMAND_CODE_IMAGE_MODELS = [
334
341
  /**
335
342
  * Native image stays sourced from COMMAND_CODE_IMAGE_MODELS. Text-only routes
336
343
  * sit beside that list so the catalog can still advertise sidecar coverage
337
- * without claiming the gateway itself accepts a picture.
338
- *
339
- * The gateway-prefixed DeepSeek V4.1 Flash route has no verified native image
340
- * support, so declaring it image-capable would hand it a picture it drops. A
341
- * positive text-only declaration makes it a vision-sidecar consumer
344
+ * without claiming the gateway itself accepts a picture. A positive text-only
345
+ * declaration makes the route a vision-sidecar consumer
342
346
  * (src/vision/eligibility.ts), so the catalog advertises image input on its
343
- * behalf and the four-target combo in #4505 intersects to ["text","image"]
344
- * instead of ["text"] — without claiming native vision. modelInputModalities
345
- * is per-key filled, so this reaches an existing install even when
346
- * noVisionModels was persisted before the id joined that list.
347
+ * behalf — without claiming native vision — and modelInputModalities is
348
+ * per-key filled, so that reaches an existing install even when noVisionModels
349
+ * was persisted before the id joined a list.
350
+ *
351
+ * Empty as of 2026-09-18. Its only entry, deepseek/deepseek-v4.1-flash, moved
352
+ * to COMMAND_CODE_IMAGE_MODELS once the #4505-requested probe passed on both
353
+ * the user-message and tool-result paths (see the note at that entry). The
354
+ * mechanism stays for the next route that measures text-only.
347
355
  */
348
- export const COMMAND_CODE_TEXT_ONLY_MODELS = [
349
- "deepseek/deepseek-v4.1-flash",
350
- ] as const;
356
+ export const COMMAND_CODE_TEXT_ONLY_MODELS = [] as const;
351
357
  export const COMMAND_CODE_MODEL_INPUT_MODALITIES: Record<string, ["text"] | ["text", "image"]> = {
352
358
  ...Object.fromEntries(COMMAND_CODE_IMAGE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
353
359
  ...Object.fromEntries(COMMAND_CODE_TEXT_ONLY_MODELS.map(id => [id, ["text"] as ["text"]])),
@@ -437,36 +443,68 @@ export const deepseekReasoningMapFor = (modelId: string): Record<string, string>
437
443
  // Coding Plan: the products use different exact allowlists and different base URLs.
438
444
  // Evidence: https://help.aliyun.com/en/model-studio/token-plan-personal-overview
439
445
  // https://help.aliyun.com/en/model-studio/token-plan-quickstart
446
+ // 260909 refresh, re-probed against the live gateway (both regions, both tiers):
447
+ // https://github.com/oliver-mee/alibaba-token-plan-wiki (machine-readable catalog).
448
+ // 260918: glm-5.3 returns. The 260909 removal was correct at the time (the id
449
+ // 404'd on every plan key), but the gateway started serving glm-5.3 on 260917:
450
+ // it now appears on /models for global Team, global Personal, and CN Team, and
451
+ // answers a completion on a Personal key (probed 260918). Contract on the plan
452
+ // gateway: effort low/high/max (default max), thinking always-on (the gateway
453
+ // rejects enable_thinking:false with 400), 1M context, 131,072 max output,
454
+ // text-only input, strict json_schema accepted. glm-5.3-flash REMAINS OUT:
455
+ // still never served by the Token Plan gateway (docs.z.ai VLM id, not plan
456
+ // entitlement).
457
+ // The Beijing preset keeps the Personal Edition subset; non-chat ids (audio/image/
458
+ // video families) stay out: they answer only on async endpoints openai-chat cannot
459
+ // reach. deepseek-v4-pro-0813 is callable but NOT listed by /models, which is the
460
+ // reason liveModels must stay false for this provider. deepseek-v4.1-flash is the
461
+ // 260910 DeepSeek rename row: listed on /models on both tiers and regions from 260915,
462
+ // hybrid thinking, vision via user message and tool result, json_object but not
463
+ // json_schema (see noJsonSchemaModels on the entries).
464
+ // Beijing serves the Personal Edition, so this is the Personal-tier roster probed
465
+ // 260909 (a strict subset of Team). deepseek-v4-pro-0813 stays out of the Beijing
466
+ // entry: its callability is only proven on Team keys, and no Personal key has been
467
+ // shown to reach it. The Beijing entry also shares the intl maps, so it carries a
468
+ // few orphan keys (kimi/glm-5/MiniMax rows); harmless, and one map beats two
469
+ // drifting ones.
440
470
  export const ALIBABA_TOKEN_PLAN_MODELS = [
441
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
442
- "glm-5.3", "glm-5.3-flash", "glm-5.2",
471
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
472
+ "deepseek-v4-pro", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "glm-5.2", "glm-5.3",
443
473
  ];
444
474
  export const ALIBABA_TOKEN_PLAN_QWEN_MODELS = [
445
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
475
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
446
476
  ];
447
477
  export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
448
478
  "qwen3.8-max": ["text", "image"],
449
- "qwen3.7-max": ["text", "image"],
479
+ "qwen3.8-flash": ["text", "image"],
480
+ "qwen3.7-max": ["text"],
450
481
  "qwen3.7-plus": ["text", "image"],
451
482
  "qwen3.6-flash": ["text", "image"],
452
- "glm-5.3": ["text"],
453
- "glm-5.3-flash": ["text", "image"],
483
+ "deepseek-v4-pro": ["text"],
484
+ "deepseek-v4-pro-0813": ["text"],
485
+ "deepseek-v4-flash-0731": ["text"],
486
+ // Vision probed on the plan gateway 260915 (user message and tool result, both 200).
487
+ "deepseek-v4.1-flash": ["text", "image"],
454
488
  "glm-5.2": ["text"],
489
+ "glm-5.3": ["text"],
455
490
  };
456
491
 
457
492
  // 260721 Alibaba Token Plan International (ap-southeast-1 / Singapore, hardened 260721).
458
493
  // Multi-vendor lineup distinct from Beijing — includes DeepSeek V4 flash, Kimi K2.7, MiniMax.
459
494
  // Evidence: https://www.alibabacloud.com/help/en/model-studio/token-plan-overview
460
495
  // https://qwencloud.com/pricing/token-plan (qwen3.8 metadata)
496
+ // The Team Edition roster (Singapore), verified identical to the CN Team set on 260909.
497
+ // deepseek-v4-pro is restored: it remains callable on the plan gateway (probed 260909,
498
+ // listed on /models on both regions) after being dropped as "retired" upstream.
461
499
  export const ALIBABA_INTL_TOKEN_PLAN_MODELS = [
462
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
463
- "deepseek-v4-flash", "deepseek-v3.2",
500
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
501
+ "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "deepseek-v3.2",
464
502
  "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5",
465
- "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5",
503
+ "glm-5.2", "glm-5.3", "glm-5.1", "glm-5",
466
504
  "MiniMax-M2.5",
467
505
  ];
468
506
  export const ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS = [
469
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
507
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
470
508
  ];
471
509
 
472
510
  // 260722 Tencent Cloud Coding Plan. The plan's model set is explicitly dynamic; these are the
@@ -543,24 +581,49 @@ export const VOLCENGINE_PLAN_TEXT_ONLY_MODELS = [
543
581
  "doubao-seed-2.0-pro",
544
582
  ];
545
583
  export const ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
546
- "qwen3.8-max": ["text", "image"],
547
- "qwen3.7-max": ["text", "image"],
548
- "qwen3.7-plus": ["text", "image"],
584
+ ...ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
549
585
  "qwen3.6-plus": ["text", "image"],
550
- "qwen3.6-flash": ["text", "image"],
551
586
  "deepseek-v4-flash": ["text"],
552
587
  "deepseek-v3.2": ["text"],
553
588
  "kimi-k2.7-code": ["text", "image"],
554
589
  "kimi-k2.6": ["text", "image"],
555
590
  "kimi-k2.5": ["text", "image"],
556
- "glm-5.3": ["text"],
557
- "glm-5.3-flash": ["text", "image"],
558
- "glm-5.2": ["text"],
559
591
  "glm-5.1": ["text"],
560
592
  "glm-5": ["text"],
561
593
  "MiniMax-M2.5": ["text"],
562
594
  };
563
595
 
596
+ // Shared Token Plan metadata (260909 gateway probes; output ceilings are max_tokens
597
+ // boundary probes: accept at N, reject at N+1).
598
+ export const QWEN38_FAMILY = ["qwen3.8-max", "qwen3.8-flash"];
599
+ export const ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS: Record<string, number> = {
600
+ "qwen3.8-max": 1_000_000, "qwen3.8-flash": 1_000_000, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
601
+ "qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
602
+ "deepseek-v4-pro": 1_000_000, "deepseek-v4-pro-0813": 1_000_000, "deepseek-v4-flash": 1_000_000,
603
+ "deepseek-v4-flash-0731": 1_000_000, "deepseek-v4.1-flash": 1_000_000, "deepseek-v3.2": 131_072,
604
+ "kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
605
+ "glm-5.2": 1_000_000, "glm-5.3": 1_000_000, "glm-5.1": 202_752, "glm-5": 202_752,
606
+ "MiniMax-M2.5": 196_608,
607
+ };
608
+ export const ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS: Record<string, number> = {
609
+ "qwen3.8-max": 131_072, "qwen3.8-flash": 131_072, "qwen3.7-max": 131_072, "qwen3.7-plus": 131_072,
610
+ "qwen3.6-plus": 65_536, "qwen3.6-flash": 65_536,
611
+ "deepseek-v4-pro": 393_216, "deepseek-v4-pro-0813": 393_216, "deepseek-v4-flash": 393_216,
612
+ "deepseek-v4-flash-0731": 393_216, "deepseek-v4.1-flash": 393_216, "deepseek-v3.2": 65_536,
613
+ "kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 98_304,
614
+ "glm-5.2": 131_072, "glm-5.3": 131_072, "glm-5.1": 128_000, "glm-5": 16_384,
615
+ "MiniMax-M2.5": 32_768,
616
+ };
617
+ export const ALIBABA_TOKEN_PLAN_NO_VISION = [
618
+ "qwen3.7-max", "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash",
619
+ "deepseek-v4-flash-0731", "deepseek-v3.2", "glm-5.2", "glm-5.3", "glm-5.1", "glm-5", "MiniMax-M2.5",
620
+ ];
621
+ export const ALIBABA_TOKEN_PLAN_PRESERVE_REASONING = [
622
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
623
+ "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731",
624
+ "deepseek-v4.1-flash", "glm-5.2", "glm-5.3",
625
+ ];
626
+
564
627
  // 260717 Kimi K3: the subscription endpoint uses one upstream id (`k3`) for both
565
628
  // entitlement tiers. Bare `k3` advertises the Moderato 256K ceiling; the local `[1m]`
566
629
  // alias advertises Allegretto's 1M ceiling and is stripped before the upstream request.
@@ -910,3 +973,47 @@ export const CLINE_PASS_TEXT_ONLY_MODELS = CLINE_PASS_MODALITY_KNOWN_MODELS.filt
910
973
  export const CLINE_PASS_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
911
974
  CLINE_PASS_MODALITY_KNOWN_MODELS.map(id => [id, CLINE_PASS_IMAGE_MODELS.has(id) ? ["text", "image"] : ["text"]]),
912
975
  );
976
+
977
+ // Opper seed: bare *pool* names. A pool is every provider Opper serves that model through; Opper
978
+ // picks the route per request. Each name is the `.model` of a `pooled: true` entry in the public
979
+ // catalogue snapshot supplied by the original provider author
980
+ // (https://api.opper.ai/v3/models?limit=2000, captured 2026-09-14); `vendor/model` ids
981
+ // (anthropic/claude-sonnet-4-6) pin one route and stay valid, they are just not seeded.
982
+ export const OPPER_MODELS = [
983
+ "claude-sonnet-4-6",
984
+ "claude-opus-5",
985
+ "gpt-5.5",
986
+ "gpt-5.4-mini",
987
+ "gemini-3.8-flash",
988
+ "deepseek-v4-pro",
989
+ "kimi-k3",
990
+ "mistral-large-2512",
991
+ ];
992
+ // Smallest value across each pool's members in that snapshot, capped at the lab model's own limit
993
+ // (kimi-k3 output); live discovery owns which models exist.
994
+ export const OPPER_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
995
+ "claude-sonnet-4-6": 1_000_000,
996
+ "claude-opus-5": 1_000_000,
997
+ "gpt-5.5": 1_050_000,
998
+ "gpt-5.4-mini": 400_000,
999
+ "gemini-3.8-flash": 1_048_576,
1000
+ "deepseek-v4-pro": 1_000_000,
1001
+ "kimi-k3": 1_048_576,
1002
+ "mistral-large-2512": 256_000,
1003
+ };
1004
+ export const OPPER_MODEL_MAX_OUTPUT_TOKENS: Record<string, number> = {
1005
+ "claude-sonnet-4-6": 64_000,
1006
+ "claude-opus-5": 128_000,
1007
+ "gpt-5.5": 128_000,
1008
+ "gpt-5.4-mini": 128_000,
1009
+ "gemini-3.8-flash": 65_536,
1010
+ "deepseek-v4-pro": 65_536,
1011
+ "kimi-k3": 131_072,
1012
+ "mistral-large-2512": 8_192,
1013
+ };
1014
+ // Pools whose members do not all accept image input (deepseek-v4-pro: no member does; kimi-k3: the
1015
+ // sference route is text-only), so the shared modality set is text.
1016
+ export const OPPER_TEXT_ONLY_MODELS = ["deepseek-v4-pro", "kimi-k3"];
1017
+ export const OPPER_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
1018
+ OPPER_MODELS.map(id => [id, OPPER_TEXT_ONLY_MODELS.includes(id) ? ["text"] : ["text", "image"]]),
1019
+ );
@@ -64,6 +64,10 @@ export interface ProviderModelDiscoveryFilter {
64
64
  interface ProviderModelDiscoverySharedSpec {
65
65
  /** Query parameters applied to the resolved discovery URL. */
66
66
  query?: Readonly<Record<string, string>>;
67
+ /** Top-level response key containing model rows; defaults to `data`. */
68
+ envelopeKey?: string;
69
+ /** Model-row field containing the provider-native identifier; defaults to `id`. */
70
+ idField?: string;
67
71
  /** Declarative eligibility rules evaluated against each untrusted model row. */
68
72
  filter?: ProviderModelDiscoveryFilter;
69
73
  /** Optional lower byte ceiling; the process-wide hard ceiling still wins. */
@@ -217,6 +221,11 @@ export interface ProviderRegistryEntry {
217
221
  * to stay contiguous. This is seeded/backfilled like other fixed wire capabilities.
218
222
  */
219
223
  requiresAdjacentResponsesToolResults?: boolean;
224
+ /**
225
+ * Responses upstream that also rejects a tool call with no matching output anywhere in the
226
+ * replayed input. Seeded/backfilled like other fixed wire capabilities.
227
+ */
228
+ requiresPairedResponsesToolResults?: boolean;
220
229
  /**
221
230
  * When enabled, tool results that are present but empty are annotated on the wire.
222
231
  * Seeded/backfilled like other fixed wire capabilities.
@@ -2,9 +2,9 @@
2
2
  //
3
3
  // Some routed models decorate the first and last lines as
4
4
  // `*** Begin Patch ***` / `*** End Patch ***`. Codex rejects those otherwise
5
- // valid custom-tool payloads. Repair is deliberately limited to a complete,
6
- // structurally recognizable top-level patch: arbitrary `exec` JavaScript is
7
- // caller-authored executable input and must remain byte-identical.
5
+ // valid custom-tool payloads. Repair of executable bodies is deliberately limited to
6
+ // unambiguous wrapper mistakes: one recognized alternate field or one complete outer
7
+ // Markdown fence. Ordinary `exec` JavaScript remains byte-identical.
8
8
  //
9
9
  // This is the same intent boundary as `src/lib/tool-argument-integers.ts`:
10
10
  // repair the one faithful reading, leave genuine patch content alone.
@@ -22,20 +22,52 @@ const PATCH_BEGIN = "*** Begin Patch";
22
22
  const PATCH_END = "*** End Patch";
23
23
  const TOP_LEVEL_PATCH_ENVELOPE = /^(\*\*\* Begin Patch(?: \*\*\*)?)(\r?\n)([\s\S]*)(\r?\n)(\*\*\* End Patch(?: \*\*\*)?)(\r?\n)?$/;
24
24
  const PATCH_OPERATION_LINE = /^\*\*\* (?:Add|Update|Delete) File: .+$/m;
25
+ const OUTER_MARKDOWN_CODE_FENCE = /^```[^\r\n]*\r?\n([\s\S]*?)\r?\n```$/;
26
+ const FREEFORM_FALLBACK_KEYS: Readonly<Record<string, readonly string[]>> = {
27
+ exec: ["code", "script", "js", "javascript", "command", "cmd", "content"],
28
+ apply_patch: ["patch", "content"],
29
+ };
30
+
31
+ function stripMarkdownCodeFence(text: string, toolName: string): string {
32
+ if (toolName !== "exec" && toolName !== "apply_patch") return text;
33
+ const match = OUTER_MARKDOWN_CODE_FENCE.exec(text.trim());
34
+ return match ? match[1] : text;
35
+ }
36
+
37
+ /**
38
+ * The single-field wrappers `unwrapFreeformToolInput` accepts for one tool name, besides the
39
+ * canonical `input`.
40
+ *
41
+ * Exported so the streaming side can hold a buffer that is still turning into one of these.
42
+ * A second list of key names beside this one is how the streamed bytes and the completed item
43
+ * come to disagree, which is the defect it exists to prevent (#5047).
44
+ */
45
+ export function freeformFallbackKeys(toolName: string): readonly string[] {
46
+ return FREEFORM_FALLBACK_KEYS[toolName] ?? [];
47
+ }
25
48
 
26
49
  /** Unwrap the `{input:string}` function-call wrapper used for freeform tools. */
27
- export function unwrapFreeformToolInput(argumentsText: unknown): string {
50
+ export function unwrapFreeformToolInput(argumentsText: unknown, toolName = ""): string {
28
51
  if (typeof argumentsText !== "string") return "";
29
52
  try {
30
53
  const parsed: unknown = JSON.parse(argumentsText);
31
54
  if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
32
- const input = (parsed as { input?: unknown }).input;
33
- if (typeof input === "string") return input;
55
+ const record = parsed as Record<string, unknown>;
56
+ if (Object.prototype.hasOwnProperty.call(record, "input")) {
57
+ return typeof record.input === "string"
58
+ ? stripMarkdownCodeFence(record.input, toolName)
59
+ : argumentsText;
60
+ }
61
+ const fallbackKeys = FREEFORM_FALLBACK_KEYS[toolName] ?? [];
62
+ const candidates = fallbackKeys.filter(key => typeof record[key] === "string");
63
+ if (candidates.length === 1) {
64
+ return stripMarkdownCodeFence(record[candidates[0]] as string, toolName);
65
+ }
34
66
  }
35
67
  } catch {
36
68
  // The string is the freeform body, not nested JSON.
37
69
  }
38
- return argumentsText;
70
+ return stripMarkdownCodeFence(argumentsText, toolName);
39
71
  }
40
72
 
41
73
  /**
@@ -92,17 +124,18 @@ export function mayBecomePatchEnvelope(text: string): boolean {
92
124
  /**
93
125
  * Repair freeform input before Codex sees it.
94
126
  *
95
- * Only a bare or reserved-`functions` `apply_patch` payload may receive delimiter
96
- * repair. Remote namespaces own their grammar; those bodies and every other
97
- * freeform input are unwrapped and left byte-exact.
127
+ * Only a bare or reserved-`functions` tool may receive fallback-field or outer-fence
128
+ * repair, and only `apply_patch` may receive delimiter repair. Remote namespaces own
129
+ * their grammar; those bodies and every other freeform input are unwrapped and left
130
+ * byte-exact.
98
131
  */
99
132
  export function repairFreeformToolInput(
100
133
  argumentsText: unknown,
101
134
  toolName = "",
102
135
  namespace?: string,
103
136
  ): string {
104
- const unwrapped = unwrapFreeformToolInput(argumentsText);
105
137
  const ownsApplyPatchGrammar = namespace === undefined || namespace === "functions";
138
+ const unwrapped = unwrapFreeformToolInput(argumentsText, ownsApplyPatchGrammar ? toolName : "");
106
139
  return ownsApplyPatchGrammar && toolName === "apply_patch"
107
140
  ? normalizeApplyPatchDelimiters(unwrapped)
108
141
  : unwrapped;