@bitkyc08/opencodex 2.57.0 → 2.59.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -10
- package/gui/dist/assets/index-C5IebErG.js +136 -0
- package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
- package/gui/dist/index.html +2 -2
- package/gui/dist/provider-icons/crusoe.svg +1 -0
- package/gui/dist/provider-icons/opper.svg +3 -0
- package/package.json +2 -2
- package/src/adapters/base.ts +11 -1
- package/src/adapters/codebuddy/scaffold-guard.ts +5 -4
- package/src/adapters/command-code.ts +13 -4
- package/src/adapters/cursor/catalog.ts +11 -0
- package/src/adapters/cursor/cursor-errors.ts +15 -0
- package/src/adapters/cursor/discovery.ts +65 -1
- package/src/adapters/cursor/effort-map.ts +16 -2
- package/src/adapters/cursor/envelope-echo.ts +55 -2
- package/src/adapters/cursor/live-transport.ts +5 -1
- package/src/adapters/cursor/message-mapper.ts +3 -2
- package/src/adapters/cursor/protobuf-events.ts +110 -11
- package/src/adapters/cursor/protobuf-request.ts +27 -6
- package/src/adapters/cursor/request-builder.ts +14 -3
- package/src/adapters/cursor/text-toolcall.ts +230 -0
- package/src/adapters/cursor/thread-continuity.ts +141 -0
- package/src/adapters/cursor/tool-guidance.ts +5 -4
- package/src/adapters/cursor/types.ts +5 -0
- package/src/adapters/cursor.ts +97 -6
- package/src/adapters/devin/cloud-direct/chat.ts +11 -2
- package/src/adapters/devin/cloud-direct/index.ts +7 -0
- package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
- package/src/adapters/devin.ts +75 -13
- package/src/adapters/google-antigravity-wire.ts +29 -2
- package/src/adapters/google-http.ts +45 -13
- package/src/adapters/google.ts +23 -4
- package/src/adapters/mimo-free.ts +32 -17
- package/src/adapters/ollama-native.ts +42 -8
- package/src/adapters/openai-chat/response-events.ts +61 -0
- package/src/adapters/openai-chat.ts +5 -10
- package/src/adapters/openai-responses/passthrough.ts +40 -5
- package/src/adapters/openai-responses/request-strips.ts +43 -0
- package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
- package/src/adapters/openai-responses/tool-schema.ts +19 -7
- package/src/adapters/physical-send.ts +50 -0
- package/src/adapters/responses-tool-schema.ts +76 -46
- package/src/adapters/run-turn-queue.ts +17 -4
- package/src/bridge/response-json.ts +2 -2
- package/src/bridge/sse.ts +166 -25
- package/src/claude/context-windows.ts +22 -0
- package/src/claude/outbound.ts +46 -5
- package/src/cli/account-api.ts +4 -3
- package/src/cli/account-extended.ts +22 -2
- package/src/cli/account-orca-import.ts +63 -0
- package/src/cli/account.ts +32 -4
- package/src/cli/capabilities.ts +40 -0
- package/src/cli/claude.ts +29 -1
- package/src/cli/codex-cli-update.ts +97 -2
- package/src/cli/config-command.ts +35 -18
- package/src/cli/dispatch.ts +71 -4
- package/src/cli/doctor.ts +197 -2
- package/src/cli/help.ts +4 -1
- package/src/cli/index.ts +132 -22
- package/src/cli/models-runtime.ts +33 -4
- package/src/cli/registry.ts +11 -1
- package/src/cli/runtime-api.ts +44 -0
- package/src/cli/start-args.ts +94 -0
- package/src/cli/system-command.ts +72 -1
- package/src/cli/uninstall-client-state.ts +12 -0
- package/src/client/machine-api.ts +4 -3
- package/src/client/machine-listener.ts +14 -1
- package/src/clients/config-export/constants.ts +2 -3
- package/src/clients/config-export.ts +5 -5
- package/src/codex/account-store.ts +81 -5
- package/src/codex/auth-api/pool-quota-probe.ts +14 -3
- package/src/codex/auth-api/routes.ts +17 -2
- package/src/codex/auth-context.ts +58 -20
- package/src/codex/catalog/build-entries.ts +25 -4
- package/src/codex/catalog/derive-entry.ts +8 -1
- package/src/codex/catalog/effort.ts +10 -6
- package/src/codex/catalog/gather-capture.ts +1 -0
- package/src/codex/catalog/model-hints.ts +37 -5
- package/src/codex/catalog/parsing.ts +83 -5
- package/src/codex/catalog/reserve-warn.ts +96 -0
- package/src/codex/catalog/retained-sync.ts +19 -0
- package/src/codex/catalog/routed-gather.ts +42 -3
- package/src/codex/cli-installation-identity.ts +210 -0
- package/src/codex/cli-installation-targets.ts +158 -0
- package/src/codex/convergence.ts +5 -0
- package/src/codex/desktop-switches.ts +145 -0
- package/src/codex/history-job.ts +5 -1
- package/src/codex/history-provider.ts +37 -5
- package/src/codex/history-state-open.ts +105 -0
- package/src/codex/history-worker.ts +14 -1
- package/src/codex/inject/config-toml.ts +44 -2
- package/src/codex/inject/remove.ts +145 -7
- package/src/codex/inject/restore.ts +204 -32
- package/src/codex/inject.ts +6 -9
- package/src/codex/lineage.ts +83 -32
- package/src/codex/loopback-target.ts +40 -0
- package/src/codex/main-account-hard-lock.ts +2 -1
- package/src/codex/main-account.ts +10 -3
- package/src/codex/main-device-reauth.ts +17 -9
- package/src/codex/model-entitlements.ts +60 -1
- package/src/codex/native-profile-startup.ts +64 -20
- package/src/codex/observed-model-denials.ts +137 -0
- package/src/codex/orca-auth-source.ts +94 -0
- package/src/codex/orca-import.ts +219 -0
- package/src/codex/prompt-text-probe.ts +282 -12
- package/src/codex/quota-401-recovery.ts +12 -0
- package/src/codex/quota-types.ts +65 -0
- package/src/codex/quota.ts +24 -19
- package/src/codex/routing/cooldown-math.ts +8 -47
- package/src/codex/routing/pin-drain.ts +57 -0
- package/src/codex/routing.ts +13 -15
- package/src/codex/subagent-model-fallback.ts +94 -0
- package/src/codex/windows-installation-files.ts +224 -0
- package/src/combos/failover.ts +122 -5
- package/src/config/atomic-write.ts +83 -8
- package/src/config/diagnostics.ts +21 -0
- package/src/config/load-degrade.ts +15 -0
- package/src/config/pending-teardown.ts +8 -0
- package/src/config/process-state.ts +36 -3
- package/src/config/provider-relative-send-path.ts +16 -0
- package/src/config/proxy-env.ts +23 -5
- package/src/config/schema/config-schema.ts +23 -0
- package/src/config/schema/leaf-validators.ts +65 -17
- package/src/generated/compatibility-version.json +337 -201
- package/src/generated/model-metadata.ts +1 -1
- package/src/lib/bounded-body.ts +4 -2
- package/src/lib/bounded-subprocess.ts +62 -10
- package/src/lib/destination-policy.ts +48 -6
- package/src/lib/errors.ts +3 -15
- package/src/lib/local-destinations.ts +32 -5
- package/src/lib/provider-outbound.ts +3 -3
- package/src/lib/proxy-env.ts +70 -3
- package/src/lib/request-execution-budget.ts +11 -3
- package/src/lib/response-body-inactivity.ts +193 -0
- package/src/lib/retry-delay.ts +69 -0
- package/src/lib/socks5-fetch.ts +631 -0
- package/src/lib/spend-reservation-ledger.ts +115 -9
- package/src/lib/windows-secret-acl.ts +151 -15
- package/src/lib/windows-user-principal.ts +5 -1
- package/src/lib/workflow-budget.ts +145 -8
- package/src/oauth/account-quota-rank.ts +72 -15
- package/src/oauth/generic-account-failover.ts +40 -27
- package/src/oauth/orcarouter.ts +15 -2
- package/src/oauth/store.ts +8 -0
- package/src/providers/codex-capacity.ts +9 -0
- package/src/providers/derive.ts +6 -0
- package/src/providers/devin-provider-merge-migration.ts +33 -12
- package/src/providers/free-directory.ts +20 -2
- package/src/providers/key-failover.ts +261 -7
- package/src/providers/model-discovery.ts +19 -7
- package/src/providers/model-rename-migration.ts +1 -0
- package/src/providers/openai-sidecar.ts +4 -0
- package/src/providers/opencode-go-transport.ts +14 -5
- package/src/providers/quota/report-cache.ts +3 -0
- package/src/providers/registry/entries-core.ts +11 -0
- package/src/providers/registry/entries-extended.ts +146 -28
- package/src/providers/registry/model-seeds.ts +136 -29
- package/src/providers/registry/types.ts +9 -0
- package/src/responses/apply-patch-envelope.ts +44 -11
- package/src/responses/bridge-search-replay-cache.ts +152 -0
- package/src/responses/code-mode-helper-compat.ts +26 -16
- package/src/responses/custom-tool-compat.ts +1 -1
- package/src/responses/hosted-tool-policy.ts +85 -2
- package/src/responses/schema.ts +9 -2
- package/src/responses/spill-store.ts +17 -0
- package/src/responses/state/body-policy.ts +25 -0
- package/src/responses/state/spill-queue.ts +8 -6
- package/src/responses/state.ts +3 -22
- package/src/router.ts +4 -0
- package/src/server/auth-cors.ts +27 -0
- package/src/server/chat-completions.ts +9 -4
- package/src/server/chat-native-sse.ts +26 -9
- package/src/server/chat-native.ts +10 -4
- package/src/server/claude-messages.ts +24 -2
- package/src/server/gui-static.ts +36 -2
- package/src/server/inbound-body-admission.ts +187 -0
- package/src/server/index/websocket-handler.ts +48 -1
- package/src/server/index.ts +15 -19
- package/src/server/management/api-access.ts +3 -4
- package/src/server/management/config-routes.ts +57 -10
- package/src/server/management/provider-capability-config.ts +35 -7
- package/src/server/management/provider-routes.ts +70 -18
- package/src/server/models-capabilities.ts +24 -3
- package/src/server/proxy-liveness.ts +97 -2
- package/src/server/relay.ts +17 -24
- package/src/server/request-log.ts +25 -1
- package/src/server/responses/adapter-continuation.ts +71 -27
- package/src/server/responses/adapter-delivery.ts +39 -8
- package/src/server/responses/adapter-dispatch.ts +52 -24
- package/src/server/responses/codex-ws-exchange.ts +65 -4
- package/src/server/responses/combo-stream-preflight.ts +68 -5
- package/src/server/responses/compact.ts +60 -11
- package/src/server/responses/core-codex-account.ts +83 -22
- package/src/server/responses/core-combo.ts +26 -0
- package/src/server/responses/core-normalize.ts +12 -5
- package/src/server/responses/core-options.ts +3 -0
- package/src/server/responses/fetch-helpers.ts +72 -3
- package/src/server/responses/native-injection-protocol.ts +42 -0
- package/src/server/responses/native-injection-replay.ts +105 -0
- package/src/server/responses/native-injection.ts +242 -0
- package/src/server/responses/native-response-control.ts +56 -0
- package/src/server/responses/native-response-json.ts +14 -0
- package/src/server/responses/native-response-output.ts +37 -0
- package/src/server/responses/native-steering-log.ts +44 -0
- package/src/server/responses/native-steering-policy.ts +49 -0
- package/src/server/responses/native-steering-replay.ts +126 -0
- package/src/server/responses/native-steering-settings.ts +76 -0
- package/src/server/responses/native-steering.ts +400 -0
- package/src/server/responses/native-tool-results.ts +130 -0
- package/src/server/responses/passthrough-delivery.ts +21 -1
- package/src/server/responses/passthrough-dispatch.ts +146 -49
- package/src/server/responses/passthrough-execution.ts +11 -1
- package/src/server/responses/request-prepare.ts +70 -0
- package/src/server/responses/request-send-budget.ts +84 -7
- package/src/server/responses/request-sidecar-auth.ts +16 -8
- package/src/server/responses/request-spend.ts +38 -9
- package/src/server/responses/request-transport.ts +13 -10
- package/src/server/responses/run-turn-execution.ts +20 -5
- package/src/server/responses/sidecar-execution.ts +2 -0
- package/src/server/responses/ws-upstream.ts +23 -2
- package/src/server/responses-custom-tool-repair.ts +2 -2
- package/src/server/sse-frame-buffer.ts +12 -10
- package/src/server/sse-payload-rewrite.ts +36 -9
- package/src/server/stop-teardown.ts +8 -1
- package/src/server/system-env-shell.ts +5 -1
- package/src/server/system-env.ts +7 -1
- package/src/server/workflow-refusal.ts +56 -2
- package/src/server/ws-bridge.ts +16 -1
- package/src/service/cli.ts +29 -7
- package/src/service/guards.ts +10 -0
- package/src/service/health.ts +43 -0
- package/src/service/state.ts +7 -2
- package/src/types/accounts.ts +4 -0
- package/src/types/config.ts +104 -3
- package/src/types/provider.ts +32 -0
- package/src/types/request.ts +7 -1
- package/src/types/wire.ts +9 -1
- package/src/usage/expected-prices.ts +28 -0
- package/src/usage/log.ts +87 -4
- package/src/web-search/passthrough-bridge.ts +39 -5
- package/gui/dist/assets/index-Cz7CLdif.js +0 -128
|
@@ -282,6 +282,17 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
|
282
282
|
forwardCallerServiceTier: false,
|
|
283
283
|
},
|
|
284
284
|
},
|
|
285
|
+
// Grok 4.6/4.5 OAuth Responses replays Codex tool history. After a mid-stream 502/reset,
|
|
286
|
+
// the client can resend a function_call without a matching output, or with hook-injected
|
|
287
|
+
// developer context between the pair. Google already synthesizes a missing tool_result
|
|
288
|
+
// (#2199). xAI's Responses parser does not, so the next turns 400 and the thread snowballs.
|
|
289
|
+
// Reuse the existing adjacency capability (Kimi #4726, DeepSeek #1292). Do not set
|
|
290
|
+
// statelessResponses: xAI stores responses for 30 days and documents previous_response_id.
|
|
291
|
+
// https://docs.x.ai/developers/model-capabilities/text/comparison
|
|
292
|
+
requiresAdjacentResponsesToolResults: true,
|
|
293
|
+
// The dangling half of the same failure: a call whose output never arrived. Kimi accepts that
|
|
294
|
+
// shape, so this is a second capability rather than a widening of the one above.
|
|
295
|
+
requiresPairedResponsesToolResults: true,
|
|
285
296
|
// Vision lineup per docs.x.ai model-capabilities/images/understanding: the grok-4.x chat
|
|
286
297
|
// models accept image input (JPEG/PNG, URL or base64). Without this the catalog leaves
|
|
287
298
|
// inputModalities undefined, and deriveComboCatalogModel defaults an undefined member to
|
|
@@ -56,6 +56,11 @@ import {
|
|
|
56
56
|
ALIBABA_TOKEN_PLAN_MODELS,
|
|
57
57
|
ALIBABA_TOKEN_PLAN_QWEN_MODELS,
|
|
58
58
|
ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
59
|
+
ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
|
|
60
|
+
ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
|
|
61
|
+
ALIBABA_TOKEN_PLAN_NO_VISION,
|
|
62
|
+
ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
|
|
63
|
+
QWEN38_FAMILY,
|
|
59
64
|
ALIBABA_INTL_TOKEN_PLAN_MODELS,
|
|
60
65
|
ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS,
|
|
61
66
|
TENCENT_CODING_PLAN_MODELS,
|
|
@@ -93,6 +98,10 @@ import {
|
|
|
93
98
|
DIGITALOCEAN_CHAT_COMPLETION_MODELS,
|
|
94
99
|
SCALEWAY_SERVERLESS_CHAT_MODELS,
|
|
95
100
|
SCALEWAY_MODEL_INPUT_MODALITIES,
|
|
101
|
+
OPPER_MODELS,
|
|
102
|
+
OPPER_MODEL_CONTEXT_WINDOWS,
|
|
103
|
+
OPPER_MODEL_MAX_OUTPUT_TOKENS,
|
|
104
|
+
OPPER_MODEL_INPUT_MODALITIES,
|
|
96
105
|
} from "./model-seeds";
|
|
97
106
|
|
|
98
107
|
export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
@@ -214,6 +223,72 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
214
223
|
},
|
|
215
224
|
note: "Shared Token Factory text-output inference only; live discovery excludes embedding and image-generation rows.",
|
|
216
225
|
},
|
|
226
|
+
{
|
|
227
|
+
// Primary sources checked 2026-09-11:
|
|
228
|
+
// - https://docs.crusoecloud.com/quickstart/getting-started-with-serverless-inference documents
|
|
229
|
+
// the fixed OpenAI-compatible host https://api.inference.crusoecloud.com/v1, Bearer API keys
|
|
230
|
+
// created in the Cloud console (Intelligence Foundry > Inference > Create API Key), and an
|
|
231
|
+
// OpenAI SDK chat.completions example against meta-llama/Llama-3.3-70B-Instruct.
|
|
232
|
+
// - https://docs.crusoecloud.com/serverless-inference/available-models lists the served models
|
|
233
|
+
// with slash-delimited ids; https://docs.crusoecloud.com/serverless-inference/rate-limits
|
|
234
|
+
// documents per-project, per-model TPM/RPM limits (429 when exceeded, 503 under shared load).
|
|
235
|
+
// - GET /v1/models rejects unauthenticated requests with 401 {"errors":["Authentication failed"]},
|
|
236
|
+
// so a successful authenticated list response is evidence that the supplied key is valid.
|
|
237
|
+
// An authenticated capture on 2026-09-12 returned 18 rows shaped like OpenRouter's catalog
|
|
238
|
+
// (`is_public`, `type`, `context_length`, `architecture.modality` of "text" or "multimodal",
|
|
239
|
+
// `tags`, `pricing`, `supported_parameters`); 17 were public serverless models and one was an
|
|
240
|
+
// account-private dedicated deployment with empty `type`/`modality`. `type` is blank on one
|
|
241
|
+
// public model, so the filter keys on `is_public` plus `architecture.modality` instead.
|
|
242
|
+
// - https://legal.crusoe.ai/ hosts the Crusoe Cloud Platform Terms of Service v1.10 (effective
|
|
243
|
+
// 2026-08-10), which name Crusoe Technologies LLC as the contracting entity, and the Service
|
|
244
|
+
// Specific Terms v5.0 (effective 2026-07-14), whose Crusoe Intelligence Foundry Terms cover the
|
|
245
|
+
// Managed Inference Service reached through the Crusoe API.
|
|
246
|
+
// - https://models.dev/api.json (provider "crusoe") records openai/gpt-oss-120b as the one served
|
|
247
|
+
// model with a low/medium/high reasoning_effort ladder; the other reasoning models expose an
|
|
248
|
+
// on/off toggle only.
|
|
249
|
+
// Maintainer: @acheamponge, who works at Crusoe (affiliation disclosed) and also maintains the
|
|
250
|
+
// models.dev crusoe entry.
|
|
251
|
+
id: "crusoe",
|
|
252
|
+
label: "Crusoe",
|
|
253
|
+
baseUrl: "https://api.inference.crusoecloud.com/v1",
|
|
254
|
+
adapter: "openai-chat",
|
|
255
|
+
authKind: "key",
|
|
256
|
+
dashboardUrl: "https://console.crusoecloud.com",
|
|
257
|
+
liveModels: true,
|
|
258
|
+
preserveCustomDestination: true,
|
|
259
|
+
// The getting-started guide documents tools through the OpenAI SDK but no provider-wide
|
|
260
|
+
// parallel tool-call contract.
|
|
261
|
+
parallelToolCalls: false,
|
|
262
|
+
// Only gpt-oss-120b has a real effort ladder; toggle-style reasoning models must not be promoted
|
|
263
|
+
// to Codex's full fallback ladder.
|
|
264
|
+
reasoningEfforts: [],
|
|
265
|
+
modelReasoningEfforts: { "openai/gpt-oss-120b": ["low", "medium", "high"] },
|
|
266
|
+
directReasoningEffortModels: ["openai/gpt-oss-120b"],
|
|
267
|
+
// The catalog reports `architecture.modality: "multimodal"` without an input list. Four rows
|
|
268
|
+
// also carry the explicit "image text to text" tag; yutori/n2 instead reports multimodal
|
|
269
|
+
// type/modality plus browser/computer-use tags. Those five captured rows are classified here.
|
|
270
|
+
modelInputModalities: {
|
|
271
|
+
"google/gemma-4-31b-it": ["text", "image"],
|
|
272
|
+
"moonshotai/Kimi-K2.6": ["text", "image"],
|
|
273
|
+
"nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B": ["text", "image"],
|
|
274
|
+
"yutori/n2": ["text", "image"],
|
|
275
|
+
"zai-org/GLM-5.3-Flash": ["text", "image"],
|
|
276
|
+
},
|
|
277
|
+
modelDiscovery: {
|
|
278
|
+
path: "models",
|
|
279
|
+
maxResponseBytes: 256 * 1024,
|
|
280
|
+
maxModels: 256,
|
|
281
|
+
filter: {
|
|
282
|
+
// Keep public serverless rows whose architecture produces text; account-private
|
|
283
|
+
// deployments (blank modality) and any embedding or media rows fail closed.
|
|
284
|
+
allOf: [
|
|
285
|
+
{ path: ["is_public"], equalsAny: [true] },
|
|
286
|
+
{ path: ["architecture", "modality"], equalsAny: ["text", "multimodal"] },
|
|
287
|
+
],
|
|
288
|
+
},
|
|
289
|
+
},
|
|
290
|
+
note: "Public Serverless Inference chat models on the shared OpenAI-compatible host; account-private and self-serve dedicated deployments are excluded from discovery and out of scope.",
|
|
291
|
+
},
|
|
217
292
|
{
|
|
218
293
|
id: "digitalocean",
|
|
219
294
|
label: "DigitalOcean Serverless Inference",
|
|
@@ -427,6 +502,7 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
427
502
|
// model_access_denied, which is why the Chat path cannot simply hang off the new base.
|
|
428
503
|
responsesPath: "/api/v1/responses",
|
|
429
504
|
chatCompletionsPath: "/api/coding/paas/v4/chat/completions",
|
|
505
|
+
modelDiscovery: { path: "/api/v1/models", envelopeKey: "models", idField: "slug" },
|
|
430
506
|
// The address this row occupied before the move. A saved custom provider still pointing
|
|
431
507
|
// at the Chat endpoint keeps receiving this row's metadata (#1100).
|
|
432
508
|
destinationAliases: [{ baseUrl: "https://api.z.ai/api/coding/paas/v4", adapter: "openai-chat" }],
|
|
@@ -724,22 +800,36 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
724
800
|
liveModels: false,
|
|
725
801
|
note: "Token Plan Personal Edition · China (Beijing)",
|
|
726
802
|
modelInputModalities: ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
727
|
-
modelContextWindows:
|
|
728
|
-
|
|
729
|
-
"qwen3.6-flash": 1_000_000, "glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000,
|
|
730
|
-
},
|
|
803
|
+
modelContextWindows: ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
|
|
804
|
+
modelMaxOutputTokens: ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
|
|
731
805
|
modelReasoningEfforts: {
|
|
732
806
|
...Object.fromEntries(ALIBABA_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
|
|
733
|
-
|
|
734
|
-
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
735
|
-
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
807
|
+
...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
|
|
736
808
|
"glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
|
|
809
|
+
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
810
|
+
"deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
|
|
811
|
+
"deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
|
|
812
|
+
"deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
|
|
813
|
+
"deepseek-v4.1-flash": deepseekThinkingEffortsFor("deepseek-v4.1-flash"),
|
|
814
|
+
},
|
|
815
|
+
modelReasoningEffortMap: {
|
|
816
|
+
"deepseek-v4-pro": deepseekReasoningMapFor("deepseek-v4-pro"),
|
|
817
|
+
"deepseek-v4-pro-0813": deepseekReasoningMapFor("deepseek-v4-pro-0813"),
|
|
818
|
+
"deepseek-v4-flash-0731": deepseekReasoningMapFor("deepseek-v4-flash-0731"),
|
|
819
|
+
"deepseek-v4.1-flash": deepseekReasoningMapFor("deepseek-v4.1-flash"),
|
|
737
820
|
},
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
821
|
+
// Probed 260915 on the plan gateway: json_object returns valid JSON, strict
|
|
822
|
+
// json_schema is rejected 400 ("This response_format type is unavailable now")
|
|
823
|
+
// in both thinking modes, so requests downgrade to json_object rather than
|
|
824
|
+
// sending a schema the gateway refuses.
|
|
825
|
+
noJsonSchemaModels: ["deepseek-v4.1-flash"],
|
|
826
|
+
modelDefaultReasoningEfforts: Object.fromEntries(QWEN38_FAMILY.map(id => [id, "xhigh"])),
|
|
827
|
+
directReasoningEffortModels: QWEN38_FAMILY,
|
|
828
|
+
thinkingBudgetModels: ALIBABA_TOKEN_PLAN_QWEN_MODELS.filter(id => !QWEN38_FAMILY.includes(id)),
|
|
829
|
+
preserveReasoningContentModels: ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
|
|
830
|
+
noVisionModels: ALIBABA_TOKEN_PLAN_NO_VISION,
|
|
831
|
+
// The gateway accepts prompt_cache_key on every Token Plan chat model (probed 260902).
|
|
832
|
+
promptCacheKey: true,
|
|
743
833
|
},
|
|
744
834
|
{
|
|
745
835
|
id: "alibaba-token-plan-intl",
|
|
@@ -756,31 +846,35 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
756
846
|
note: "Token Plan Team Edition · Singapore (ap-southeast-1)",
|
|
757
847
|
metadataModelIdNormalize: "case-insensitive",
|
|
758
848
|
modelInputModalities: ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES,
|
|
759
|
-
modelContextWindows:
|
|
760
|
-
|
|
761
|
-
"qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000, "qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
|
|
762
|
-
"deepseek-v4-flash": 1_000_000, "deepseek-v3.2": 131_072,
|
|
763
|
-
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
|
|
764
|
-
"glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000, "glm-5.1": 1_000_000, "glm-5": 1_000_000,
|
|
765
|
-
"MiniMax-M2.5": 204_800,
|
|
766
|
-
},
|
|
849
|
+
modelContextWindows: ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
|
|
850
|
+
modelMaxOutputTokens: ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
|
|
767
851
|
modelReasoningEfforts: {
|
|
768
852
|
...Object.fromEntries(ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
|
|
769
|
-
|
|
770
|
-
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
771
|
-
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
853
|
+
...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
|
|
772
854
|
"glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
|
|
855
|
+
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
856
|
+
"deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
|
|
857
|
+
"deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
|
|
773
858
|
"deepseek-v4-flash": deepseekThinkingEffortsFor("deepseek-v4-flash"),
|
|
859
|
+
"deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
|
|
860
|
+
"deepseek-v4.1-flash": deepseekThinkingEffortsFor("deepseek-v4.1-flash"),
|
|
774
861
|
},
|
|
775
862
|
modelReasoningEffortMap: {
|
|
863
|
+
"deepseek-v4-pro": deepseekReasoningMapFor("deepseek-v4-pro"),
|
|
864
|
+
"deepseek-v4-pro-0813": deepseekReasoningMapFor("deepseek-v4-pro-0813"),
|
|
776
865
|
"deepseek-v4-flash": deepseekReasoningMapFor("deepseek-v4-flash"),
|
|
866
|
+
"deepseek-v4-flash-0731": deepseekReasoningMapFor("deepseek-v4-flash-0731"),
|
|
867
|
+
"deepseek-v4.1-flash": deepseekReasoningMapFor("deepseek-v4.1-flash"),
|
|
777
868
|
},
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
869
|
+
// Same 260915 json_schema rejection probe as the Beijing entry.
|
|
870
|
+
noJsonSchemaModels: ["deepseek-v4.1-flash"],
|
|
871
|
+
directReasoningEffortModels: QWEN38_FAMILY,
|
|
872
|
+
thinkingBudgetModels: ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.filter(id => !QWEN38_FAMILY.includes(id)),
|
|
873
|
+
preserveReasoningContentModels: ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
|
|
874
|
+
noVisionModels: ALIBABA_TOKEN_PLAN_NO_VISION,
|
|
782
875
|
noReasoningModels: ["kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "deepseek-v3.2", "glm-5.1", "glm-5", "MiniMax-M2.5"],
|
|
783
|
-
modelDefaultReasoningEfforts:
|
|
876
|
+
modelDefaultReasoningEfforts: Object.fromEntries(QWEN38_FAMILY.map(id => [id, "xhigh"])),
|
|
877
|
+
promptCacheKey: true,
|
|
784
878
|
},
|
|
785
879
|
// NEEDS_HUMAN 2026-07-10: kept for config compatibility, but this is a dashboard URL,
|
|
786
880
|
// no /models endpoint is documented, and tools are silently ignored upstream per docs.parallel.ai.
|
|
@@ -935,6 +1029,30 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
935
1029
|
noJsonSchemaModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS, ...OPENCODE_FREE_DEEPSEEK_MODELS],
|
|
936
1030
|
},
|
|
937
1031
|
{ id: "vercel-ai-gateway", label: "Vercel AI Gateway", baseUrl: "https://ai-gateway.vercel.sh/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://vercel.com/dashboard" },
|
|
1032
|
+
{
|
|
1033
|
+
// Opper: EU-hosted AI gateway (Opper AI AB, Stockholm). One OpenAI-compatible endpoint and one
|
|
1034
|
+
// key in front of 30+ upstream providers. Seeded ids are Opper *pools* (bare names such as
|
|
1035
|
+
// `claude-sonnet-4-6`): the gateway chooses the provider/region per request, and a
|
|
1036
|
+
// `vendor/model` id (`anthropic/claude-sonnet-4-6`, `aws/claude-sonnet-4-6-eu`) pins one route.
|
|
1037
|
+
// The original provider author reported on 2026-09-08 that GET /v3/compat/models answers 401
|
|
1038
|
+
// without a key, so the default discovery URL doubles as key validation. Windows, output caps
|
|
1039
|
+
// and modalities live in model-seeds.ts (smallest value / shared modality across each pool's
|
|
1040
|
+
// members); live discovery owns which models exist.
|
|
1041
|
+
id: "opper",
|
|
1042
|
+
label: "Opper",
|
|
1043
|
+
adapter: "openai-chat",
|
|
1044
|
+
baseUrl: "https://api.opper.ai/v3/compat",
|
|
1045
|
+
authKind: "key",
|
|
1046
|
+
dashboardUrl: "https://platform.opper.ai",
|
|
1047
|
+
liveModels: true,
|
|
1048
|
+
preserveCustomDestination: true,
|
|
1049
|
+
defaultModel: "claude-sonnet-4-6",
|
|
1050
|
+
models: OPPER_MODELS,
|
|
1051
|
+
modelContextWindows: OPPER_MODEL_CONTEXT_WINDOWS,
|
|
1052
|
+
modelMaxOutputTokens: OPPER_MODEL_MAX_OUTPUT_TOKENS,
|
|
1053
|
+
modelInputModalities: OPPER_MODEL_INPUT_MODALITIES,
|
|
1054
|
+
note: "EU-hosted AI gateway: one OpenAI-compatible endpoint and one key in front of 30+ providers. Bare model ids are pools (claude-sonnet-4-6, gpt-5.5) and Opper picks the route per request; vendor/model ids (anthropic/claude-sonnet-4-6) pin one provider. The catalogue is discovered live from /v3/compat/models with your key; the public list is at opper.ai/models. Token rates are the model providers' rates with no markup; Opper charges a 3% fee when you buy credits.",
|
|
1055
|
+
},
|
|
938
1056
|
{
|
|
939
1057
|
id: "opencode-free",
|
|
940
1058
|
label: "OpenCode Free",
|
|
@@ -315,6 +315,13 @@ export const DEEPSEEK_VISION_PREVIEW_MODEL = "deepseek-v4-flash-vision-exp";
|
|
|
315
315
|
*/
|
|
316
316
|
export const COMMAND_CODE_IMAGE_MODELS = [
|
|
317
317
|
`deepseek/${DEEPSEEK_VISION_PREVIEW_MODEL}`,
|
|
318
|
+
// Probed 2026-09-18 through a running 2.58.0 proxy: a 3x3 random-color grid
|
|
319
|
+
// (180x180 PNG, six candidate colors) came back 9/9 correct both as a user
|
|
320
|
+
// message and as a tool_result, and the request logs show the route served
|
|
321
|
+
// the image natively — no vision-sidecar call in either window. #4505 asked
|
|
322
|
+
// for exactly this upstream probe before promoting the id. The sibling
|
|
323
|
+
// deepseek/deepseek-v4-flash route remains verified-negative above.
|
|
324
|
+
"deepseek/deepseek-v4.1-flash",
|
|
318
325
|
"gpt-5.6-luna",
|
|
319
326
|
"gpt-5.6-sol",
|
|
320
327
|
"MiniMaxAI/MiniMax-M3",
|
|
@@ -334,20 +341,19 @@ export const COMMAND_CODE_IMAGE_MODELS = [
|
|
|
334
341
|
/**
|
|
335
342
|
* Native image stays sourced from COMMAND_CODE_IMAGE_MODELS. Text-only routes
|
|
336
343
|
* sit beside that list so the catalog can still advertise sidecar coverage
|
|
337
|
-
* without claiming the gateway itself accepts a picture.
|
|
338
|
-
*
|
|
339
|
-
* The gateway-prefixed DeepSeek V4.1 Flash route has no verified native image
|
|
340
|
-
* support, so declaring it image-capable would hand it a picture it drops. A
|
|
341
|
-
* positive text-only declaration makes it a vision-sidecar consumer
|
|
344
|
+
* without claiming the gateway itself accepts a picture. A positive text-only
|
|
345
|
+
* declaration makes the route a vision-sidecar consumer
|
|
342
346
|
* (src/vision/eligibility.ts), so the catalog advertises image input on its
|
|
343
|
-
* behalf
|
|
344
|
-
*
|
|
345
|
-
*
|
|
346
|
-
*
|
|
347
|
+
* behalf — without claiming native vision — and modelInputModalities is
|
|
348
|
+
* per-key filled, so that reaches an existing install even when noVisionModels
|
|
349
|
+
* was persisted before the id joined a list.
|
|
350
|
+
*
|
|
351
|
+
* Empty as of 2026-09-18. Its only entry, deepseek/deepseek-v4.1-flash, moved
|
|
352
|
+
* to COMMAND_CODE_IMAGE_MODELS once the #4505-requested probe passed on both
|
|
353
|
+
* the user-message and tool-result paths (see the note at that entry). The
|
|
354
|
+
* mechanism stays for the next route that measures text-only.
|
|
347
355
|
*/
|
|
348
|
-
export const COMMAND_CODE_TEXT_ONLY_MODELS = [
|
|
349
|
-
"deepseek/deepseek-v4.1-flash",
|
|
350
|
-
] as const;
|
|
356
|
+
export const COMMAND_CODE_TEXT_ONLY_MODELS = [] as const;
|
|
351
357
|
export const COMMAND_CODE_MODEL_INPUT_MODALITIES: Record<string, ["text"] | ["text", "image"]> = {
|
|
352
358
|
...Object.fromEntries(COMMAND_CODE_IMAGE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
|
|
353
359
|
...Object.fromEntries(COMMAND_CODE_TEXT_ONLY_MODELS.map(id => [id, ["text"] as ["text"]])),
|
|
@@ -437,36 +443,68 @@ export const deepseekReasoningMapFor = (modelId: string): Record<string, string>
|
|
|
437
443
|
// Coding Plan: the products use different exact allowlists and different base URLs.
|
|
438
444
|
// Evidence: https://help.aliyun.com/en/model-studio/token-plan-personal-overview
|
|
439
445
|
// https://help.aliyun.com/en/model-studio/token-plan-quickstart
|
|
446
|
+
// 260909 refresh, re-probed against the live gateway (both regions, both tiers):
|
|
447
|
+
// https://github.com/oliver-mee/alibaba-token-plan-wiki (machine-readable catalog).
|
|
448
|
+
// 260918: glm-5.3 returns. The 260909 removal was correct at the time (the id
|
|
449
|
+
// 404'd on every plan key), but the gateway started serving glm-5.3 on 260917:
|
|
450
|
+
// it now appears on /models for global Team, global Personal, and CN Team, and
|
|
451
|
+
// answers a completion on a Personal key (probed 260918). Contract on the plan
|
|
452
|
+
// gateway: effort low/high/max (default max), thinking always-on (the gateway
|
|
453
|
+
// rejects enable_thinking:false with 400), 1M context, 131,072 max output,
|
|
454
|
+
// text-only input, strict json_schema accepted. glm-5.3-flash REMAINS OUT:
|
|
455
|
+
// still never served by the Token Plan gateway (docs.z.ai VLM id, not plan
|
|
456
|
+
// entitlement).
|
|
457
|
+
// The Beijing preset keeps the Personal Edition subset; non-chat ids (audio/image/
|
|
458
|
+
// video families) stay out: they answer only on async endpoints openai-chat cannot
|
|
459
|
+
// reach. deepseek-v4-pro-0813 is callable but NOT listed by /models, which is the
|
|
460
|
+
// reason liveModels must stay false for this provider. deepseek-v4.1-flash is the
|
|
461
|
+
// 260910 DeepSeek rename row: listed on /models on both tiers and regions from 260915,
|
|
462
|
+
// hybrid thinking, vision via user message and tool result, json_object but not
|
|
463
|
+
// json_schema (see noJsonSchemaModels on the entries).
|
|
464
|
+
// Beijing serves the Personal Edition, so this is the Personal-tier roster probed
|
|
465
|
+
// 260909 (a strict subset of Team). deepseek-v4-pro-0813 stays out of the Beijing
|
|
466
|
+
// entry: its callability is only proven on Team keys, and no Personal key has been
|
|
467
|
+
// shown to reach it. The Beijing entry also shares the intl maps, so it carries a
|
|
468
|
+
// few orphan keys (kimi/glm-5/MiniMax rows); harmless, and one map beats two
|
|
469
|
+
// drifting ones.
|
|
440
470
|
export const ALIBABA_TOKEN_PLAN_MODELS = [
|
|
441
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
442
|
-
"
|
|
471
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
472
|
+
"deepseek-v4-pro", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "glm-5.2", "glm-5.3",
|
|
443
473
|
];
|
|
444
474
|
export const ALIBABA_TOKEN_PLAN_QWEN_MODELS = [
|
|
445
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
475
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
446
476
|
];
|
|
447
477
|
export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
448
478
|
"qwen3.8-max": ["text", "image"],
|
|
449
|
-
"qwen3.
|
|
479
|
+
"qwen3.8-flash": ["text", "image"],
|
|
480
|
+
"qwen3.7-max": ["text"],
|
|
450
481
|
"qwen3.7-plus": ["text", "image"],
|
|
451
482
|
"qwen3.6-flash": ["text", "image"],
|
|
452
|
-
"
|
|
453
|
-
"
|
|
483
|
+
"deepseek-v4-pro": ["text"],
|
|
484
|
+
"deepseek-v4-pro-0813": ["text"],
|
|
485
|
+
"deepseek-v4-flash-0731": ["text"],
|
|
486
|
+
// Vision probed on the plan gateway 260915 (user message and tool result, both 200).
|
|
487
|
+
"deepseek-v4.1-flash": ["text", "image"],
|
|
454
488
|
"glm-5.2": ["text"],
|
|
489
|
+
"glm-5.3": ["text"],
|
|
455
490
|
};
|
|
456
491
|
|
|
457
492
|
// 260721 Alibaba Token Plan International (ap-southeast-1 / Singapore, hardened 260721).
|
|
458
493
|
// Multi-vendor lineup distinct from Beijing — includes DeepSeek V4 flash, Kimi K2.7, MiniMax.
|
|
459
494
|
// Evidence: https://www.alibabacloud.com/help/en/model-studio/token-plan-overview
|
|
460
495
|
// https://qwencloud.com/pricing/token-plan (qwen3.8 metadata)
|
|
496
|
+
// The Team Edition roster (Singapore), verified identical to the CN Team set on 260909.
|
|
497
|
+
// deepseek-v4-pro is restored: it remains callable on the plan gateway (probed 260909,
|
|
498
|
+
// listed on /models on both regions) after being dropped as "retired" upstream.
|
|
461
499
|
export const ALIBABA_INTL_TOKEN_PLAN_MODELS = [
|
|
462
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
463
|
-
"deepseek-v4-flash", "deepseek-v3.2",
|
|
500
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
501
|
+
"deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "deepseek-v3.2",
|
|
464
502
|
"kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5",
|
|
465
|
-
"glm-5.
|
|
503
|
+
"glm-5.2", "glm-5.3", "glm-5.1", "glm-5",
|
|
466
504
|
"MiniMax-M2.5",
|
|
467
505
|
];
|
|
468
506
|
export const ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS = [
|
|
469
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
507
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
470
508
|
];
|
|
471
509
|
|
|
472
510
|
// 260722 Tencent Cloud Coding Plan. The plan's model set is explicitly dynamic; these are the
|
|
@@ -543,24 +581,49 @@ export const VOLCENGINE_PLAN_TEXT_ONLY_MODELS = [
|
|
|
543
581
|
"doubao-seed-2.0-pro",
|
|
544
582
|
];
|
|
545
583
|
export const ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
546
|
-
|
|
547
|
-
"qwen3.7-max": ["text", "image"],
|
|
548
|
-
"qwen3.7-plus": ["text", "image"],
|
|
584
|
+
...ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
549
585
|
"qwen3.6-plus": ["text", "image"],
|
|
550
|
-
"qwen3.6-flash": ["text", "image"],
|
|
551
586
|
"deepseek-v4-flash": ["text"],
|
|
552
587
|
"deepseek-v3.2": ["text"],
|
|
553
588
|
"kimi-k2.7-code": ["text", "image"],
|
|
554
589
|
"kimi-k2.6": ["text", "image"],
|
|
555
590
|
"kimi-k2.5": ["text", "image"],
|
|
556
|
-
"glm-5.3": ["text"],
|
|
557
|
-
"glm-5.3-flash": ["text", "image"],
|
|
558
|
-
"glm-5.2": ["text"],
|
|
559
591
|
"glm-5.1": ["text"],
|
|
560
592
|
"glm-5": ["text"],
|
|
561
593
|
"MiniMax-M2.5": ["text"],
|
|
562
594
|
};
|
|
563
595
|
|
|
596
|
+
// Shared Token Plan metadata (260909 gateway probes; output ceilings are max_tokens
|
|
597
|
+
// boundary probes: accept at N, reject at N+1).
|
|
598
|
+
export const QWEN38_FAMILY = ["qwen3.8-max", "qwen3.8-flash"];
|
|
599
|
+
export const ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS: Record<string, number> = {
|
|
600
|
+
"qwen3.8-max": 1_000_000, "qwen3.8-flash": 1_000_000, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
|
|
601
|
+
"qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
|
|
602
|
+
"deepseek-v4-pro": 1_000_000, "deepseek-v4-pro-0813": 1_000_000, "deepseek-v4-flash": 1_000_000,
|
|
603
|
+
"deepseek-v4-flash-0731": 1_000_000, "deepseek-v4.1-flash": 1_000_000, "deepseek-v3.2": 131_072,
|
|
604
|
+
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
|
|
605
|
+
"glm-5.2": 1_000_000, "glm-5.3": 1_000_000, "glm-5.1": 202_752, "glm-5": 202_752,
|
|
606
|
+
"MiniMax-M2.5": 196_608,
|
|
607
|
+
};
|
|
608
|
+
export const ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS: Record<string, number> = {
|
|
609
|
+
"qwen3.8-max": 131_072, "qwen3.8-flash": 131_072, "qwen3.7-max": 131_072, "qwen3.7-plus": 131_072,
|
|
610
|
+
"qwen3.6-plus": 65_536, "qwen3.6-flash": 65_536,
|
|
611
|
+
"deepseek-v4-pro": 393_216, "deepseek-v4-pro-0813": 393_216, "deepseek-v4-flash": 393_216,
|
|
612
|
+
"deepseek-v4-flash-0731": 393_216, "deepseek-v4.1-flash": 393_216, "deepseek-v3.2": 65_536,
|
|
613
|
+
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 98_304,
|
|
614
|
+
"glm-5.2": 131_072, "glm-5.3": 131_072, "glm-5.1": 128_000, "glm-5": 16_384,
|
|
615
|
+
"MiniMax-M2.5": 32_768,
|
|
616
|
+
};
|
|
617
|
+
export const ALIBABA_TOKEN_PLAN_NO_VISION = [
|
|
618
|
+
"qwen3.7-max", "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash",
|
|
619
|
+
"deepseek-v4-flash-0731", "deepseek-v3.2", "glm-5.2", "glm-5.3", "glm-5.1", "glm-5", "MiniMax-M2.5",
|
|
620
|
+
];
|
|
621
|
+
export const ALIBABA_TOKEN_PLAN_PRESERVE_REASONING = [
|
|
622
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
623
|
+
"deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731",
|
|
624
|
+
"deepseek-v4.1-flash", "glm-5.2", "glm-5.3",
|
|
625
|
+
];
|
|
626
|
+
|
|
564
627
|
// 260717 Kimi K3: the subscription endpoint uses one upstream id (`k3`) for both
|
|
565
628
|
// entitlement tiers. Bare `k3` advertises the Moderato 256K ceiling; the local `[1m]`
|
|
566
629
|
// alias advertises Allegretto's 1M ceiling and is stripped before the upstream request.
|
|
@@ -910,3 +973,47 @@ export const CLINE_PASS_TEXT_ONLY_MODELS = CLINE_PASS_MODALITY_KNOWN_MODELS.filt
|
|
|
910
973
|
export const CLINE_PASS_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
|
|
911
974
|
CLINE_PASS_MODALITY_KNOWN_MODELS.map(id => [id, CLINE_PASS_IMAGE_MODELS.has(id) ? ["text", "image"] : ["text"]]),
|
|
912
975
|
);
|
|
976
|
+
|
|
977
|
+
// Opper seed: bare *pool* names. A pool is every provider Opper serves that model through; Opper
|
|
978
|
+
// picks the route per request. Each name is the `.model` of a `pooled: true` entry in the public
|
|
979
|
+
// catalogue snapshot supplied by the original provider author
|
|
980
|
+
// (https://api.opper.ai/v3/models?limit=2000, captured 2026-09-14); `vendor/model` ids
|
|
981
|
+
// (anthropic/claude-sonnet-4-6) pin one route and stay valid, they are just not seeded.
|
|
982
|
+
export const OPPER_MODELS = [
|
|
983
|
+
"claude-sonnet-4-6",
|
|
984
|
+
"claude-opus-5",
|
|
985
|
+
"gpt-5.5",
|
|
986
|
+
"gpt-5.4-mini",
|
|
987
|
+
"gemini-3.8-flash",
|
|
988
|
+
"deepseek-v4-pro",
|
|
989
|
+
"kimi-k3",
|
|
990
|
+
"mistral-large-2512",
|
|
991
|
+
];
|
|
992
|
+
// Smallest value across each pool's members in that snapshot, capped at the lab model's own limit
|
|
993
|
+
// (kimi-k3 output); live discovery owns which models exist.
|
|
994
|
+
export const OPPER_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
|
|
995
|
+
"claude-sonnet-4-6": 1_000_000,
|
|
996
|
+
"claude-opus-5": 1_000_000,
|
|
997
|
+
"gpt-5.5": 1_050_000,
|
|
998
|
+
"gpt-5.4-mini": 400_000,
|
|
999
|
+
"gemini-3.8-flash": 1_048_576,
|
|
1000
|
+
"deepseek-v4-pro": 1_000_000,
|
|
1001
|
+
"kimi-k3": 1_048_576,
|
|
1002
|
+
"mistral-large-2512": 256_000,
|
|
1003
|
+
};
|
|
1004
|
+
export const OPPER_MODEL_MAX_OUTPUT_TOKENS: Record<string, number> = {
|
|
1005
|
+
"claude-sonnet-4-6": 64_000,
|
|
1006
|
+
"claude-opus-5": 128_000,
|
|
1007
|
+
"gpt-5.5": 128_000,
|
|
1008
|
+
"gpt-5.4-mini": 128_000,
|
|
1009
|
+
"gemini-3.8-flash": 65_536,
|
|
1010
|
+
"deepseek-v4-pro": 65_536,
|
|
1011
|
+
"kimi-k3": 131_072,
|
|
1012
|
+
"mistral-large-2512": 8_192,
|
|
1013
|
+
};
|
|
1014
|
+
// Pools whose members do not all accept image input (deepseek-v4-pro: no member does; kimi-k3: the
|
|
1015
|
+
// sference route is text-only), so the shared modality set is text.
|
|
1016
|
+
export const OPPER_TEXT_ONLY_MODELS = ["deepseek-v4-pro", "kimi-k3"];
|
|
1017
|
+
export const OPPER_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
|
|
1018
|
+
OPPER_MODELS.map(id => [id, OPPER_TEXT_ONLY_MODELS.includes(id) ? ["text"] : ["text", "image"]]),
|
|
1019
|
+
);
|
|
@@ -64,6 +64,10 @@ export interface ProviderModelDiscoveryFilter {
|
|
|
64
64
|
interface ProviderModelDiscoverySharedSpec {
|
|
65
65
|
/** Query parameters applied to the resolved discovery URL. */
|
|
66
66
|
query?: Readonly<Record<string, string>>;
|
|
67
|
+
/** Top-level response key containing model rows; defaults to `data`. */
|
|
68
|
+
envelopeKey?: string;
|
|
69
|
+
/** Model-row field containing the provider-native identifier; defaults to `id`. */
|
|
70
|
+
idField?: string;
|
|
67
71
|
/** Declarative eligibility rules evaluated against each untrusted model row. */
|
|
68
72
|
filter?: ProviderModelDiscoveryFilter;
|
|
69
73
|
/** Optional lower byte ceiling; the process-wide hard ceiling still wins. */
|
|
@@ -217,6 +221,11 @@ export interface ProviderRegistryEntry {
|
|
|
217
221
|
* to stay contiguous. This is seeded/backfilled like other fixed wire capabilities.
|
|
218
222
|
*/
|
|
219
223
|
requiresAdjacentResponsesToolResults?: boolean;
|
|
224
|
+
/**
|
|
225
|
+
* Responses upstream that also rejects a tool call with no matching output anywhere in the
|
|
226
|
+
* replayed input. Seeded/backfilled like other fixed wire capabilities.
|
|
227
|
+
*/
|
|
228
|
+
requiresPairedResponsesToolResults?: boolean;
|
|
220
229
|
/**
|
|
221
230
|
* When enabled, tool results that are present but empty are annotated on the wire.
|
|
222
231
|
* Seeded/backfilled like other fixed wire capabilities.
|
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
//
|
|
3
3
|
// Some routed models decorate the first and last lines as
|
|
4
4
|
// `*** Begin Patch ***` / `*** End Patch ***`. Codex rejects those otherwise
|
|
5
|
-
// valid custom-tool payloads. Repair is deliberately limited to
|
|
6
|
-
//
|
|
7
|
-
//
|
|
5
|
+
// valid custom-tool payloads. Repair of executable bodies is deliberately limited to
|
|
6
|
+
// unambiguous wrapper mistakes: one recognized alternate field or one complete outer
|
|
7
|
+
// Markdown fence. Ordinary `exec` JavaScript remains byte-identical.
|
|
8
8
|
//
|
|
9
9
|
// This is the same intent boundary as `src/lib/tool-argument-integers.ts`:
|
|
10
10
|
// repair the one faithful reading, leave genuine patch content alone.
|
|
@@ -22,20 +22,52 @@ const PATCH_BEGIN = "*** Begin Patch";
|
|
|
22
22
|
const PATCH_END = "*** End Patch";
|
|
23
23
|
const TOP_LEVEL_PATCH_ENVELOPE = /^(\*\*\* Begin Patch(?: \*\*\*)?)(\r?\n)([\s\S]*)(\r?\n)(\*\*\* End Patch(?: \*\*\*)?)(\r?\n)?$/;
|
|
24
24
|
const PATCH_OPERATION_LINE = /^\*\*\* (?:Add|Update|Delete) File: .+$/m;
|
|
25
|
+
const OUTER_MARKDOWN_CODE_FENCE = /^```[^\r\n]*\r?\n([\s\S]*?)\r?\n```$/;
|
|
26
|
+
const FREEFORM_FALLBACK_KEYS: Readonly<Record<string, readonly string[]>> = {
|
|
27
|
+
exec: ["code", "script", "js", "javascript", "command", "cmd", "content"],
|
|
28
|
+
apply_patch: ["patch", "content"],
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
function stripMarkdownCodeFence(text: string, toolName: string): string {
|
|
32
|
+
if (toolName !== "exec" && toolName !== "apply_patch") return text;
|
|
33
|
+
const match = OUTER_MARKDOWN_CODE_FENCE.exec(text.trim());
|
|
34
|
+
return match ? match[1] : text;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* The single-field wrappers `unwrapFreeformToolInput` accepts for one tool name, besides the
|
|
39
|
+
* canonical `input`.
|
|
40
|
+
*
|
|
41
|
+
* Exported so the streaming side can hold a buffer that is still turning into one of these.
|
|
42
|
+
* A second list of key names beside this one is how the streamed bytes and the completed item
|
|
43
|
+
* come to disagree, which is the defect it exists to prevent (#5047).
|
|
44
|
+
*/
|
|
45
|
+
export function freeformFallbackKeys(toolName: string): readonly string[] {
|
|
46
|
+
return FREEFORM_FALLBACK_KEYS[toolName] ?? [];
|
|
47
|
+
}
|
|
25
48
|
|
|
26
49
|
/** Unwrap the `{input:string}` function-call wrapper used for freeform tools. */
|
|
27
|
-
export function unwrapFreeformToolInput(argumentsText: unknown): string {
|
|
50
|
+
export function unwrapFreeformToolInput(argumentsText: unknown, toolName = ""): string {
|
|
28
51
|
if (typeof argumentsText !== "string") return "";
|
|
29
52
|
try {
|
|
30
53
|
const parsed: unknown = JSON.parse(argumentsText);
|
|
31
54
|
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
|
|
32
|
-
const
|
|
33
|
-
if (
|
|
55
|
+
const record = parsed as Record<string, unknown>;
|
|
56
|
+
if (Object.prototype.hasOwnProperty.call(record, "input")) {
|
|
57
|
+
return typeof record.input === "string"
|
|
58
|
+
? stripMarkdownCodeFence(record.input, toolName)
|
|
59
|
+
: argumentsText;
|
|
60
|
+
}
|
|
61
|
+
const fallbackKeys = FREEFORM_FALLBACK_KEYS[toolName] ?? [];
|
|
62
|
+
const candidates = fallbackKeys.filter(key => typeof record[key] === "string");
|
|
63
|
+
if (candidates.length === 1) {
|
|
64
|
+
return stripMarkdownCodeFence(record[candidates[0]] as string, toolName);
|
|
65
|
+
}
|
|
34
66
|
}
|
|
35
67
|
} catch {
|
|
36
68
|
// The string is the freeform body, not nested JSON.
|
|
37
69
|
}
|
|
38
|
-
return argumentsText;
|
|
70
|
+
return stripMarkdownCodeFence(argumentsText, toolName);
|
|
39
71
|
}
|
|
40
72
|
|
|
41
73
|
/**
|
|
@@ -92,17 +124,18 @@ export function mayBecomePatchEnvelope(text: string): boolean {
|
|
|
92
124
|
/**
|
|
93
125
|
* Repair freeform input before Codex sees it.
|
|
94
126
|
*
|
|
95
|
-
* Only a bare or reserved-`functions`
|
|
96
|
-
* repair
|
|
97
|
-
* freeform input are unwrapped and left
|
|
127
|
+
* Only a bare or reserved-`functions` tool may receive fallback-field or outer-fence
|
|
128
|
+
* repair, and only `apply_patch` may receive delimiter repair. Remote namespaces own
|
|
129
|
+
* their grammar; those bodies and every other freeform input are unwrapped and left
|
|
130
|
+
* byte-exact.
|
|
98
131
|
*/
|
|
99
132
|
export function repairFreeformToolInput(
|
|
100
133
|
argumentsText: unknown,
|
|
101
134
|
toolName = "",
|
|
102
135
|
namespace?: string,
|
|
103
136
|
): string {
|
|
104
|
-
const unwrapped = unwrapFreeformToolInput(argumentsText);
|
|
105
137
|
const ownsApplyPatchGrammar = namespace === undefined || namespace === "functions";
|
|
138
|
+
const unwrapped = unwrapFreeformToolInput(argumentsText, ownsApplyPatchGrammar ? toolName : "");
|
|
106
139
|
return ownsApplyPatchGrammar && toolName === "apply_patch"
|
|
107
140
|
? normalizeApplyPatchDelimiters(unwrapped)
|
|
108
141
|
: unwrapped;
|