@bitkyc08/opencodex 2.56.0 → 2.58.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/ocx.mjs +10 -0
- package/gui/dist/assets/{index-D4zuyIxQ.js → index-BbrHOIY0.js} +21 -21
- package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
- package/gui/dist/index.html +2 -2
- package/package.json +4 -4
- package/src/adapters/codebuddy/adapter.ts +2 -1
- package/src/adapters/codebuddy/scaffold-guard.ts +249 -0
- package/src/adapters/command-code.ts +12 -3
- package/src/adapters/cursor/cursor-errors.ts +15 -0
- package/src/adapters/cursor/discovery.ts +65 -1
- package/src/adapters/cursor/envelope-echo.ts +8 -2
- package/src/adapters/cursor/live-transport.ts +5 -1
- package/src/adapters/cursor/protobuf-events.ts +110 -11
- package/src/adapters/cursor/protobuf-request.ts +19 -1
- package/src/adapters/cursor/text-toolcall.ts +230 -0
- package/src/adapters/cursor/thread-continuity.ts +67 -0
- package/src/adapters/cursor/types.ts +5 -0
- package/src/adapters/cursor.ts +55 -5
- package/src/adapters/google-http.ts +38 -13
- package/src/adapters/google.ts +7 -7
- package/src/adapters/kiro/payload.ts +17 -3
- package/src/adapters/kiro/reasoning.ts +70 -7
- package/src/adapters/kiro/stream.ts +8 -2
- package/src/adapters/kiro/wire.ts +2 -1
- package/src/adapters/kiro-events.ts +21 -13
- package/src/adapters/mimo-free.ts +32 -17
- package/src/adapters/ollama-native.ts +42 -8
- package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
- package/src/adapters/openai-chat/tool-schema.ts +25 -7
- package/src/adapters/openai-chat.ts +8 -8
- package/src/adapters/openai-responses/passthrough.ts +62 -5
- package/src/adapters/openai-responses/request-strips.ts +43 -0
- package/src/adapters/physical-send.ts +50 -0
- package/src/bridge/errors.ts +26 -2
- package/src/bridge/response-json.ts +8 -2
- package/src/bridge/sse.ts +20 -2
- package/src/claude/desktop-profile.ts +66 -9
- package/src/claude/outbound.ts +32 -4
- package/src/cli/account-main.ts +1 -1
- package/src/cli/capabilities.ts +2 -2
- package/src/cli/combo.ts +10 -1
- package/src/cli/config-command.ts +35 -18
- package/src/cli/dispatch.ts +17 -4
- package/src/cli/index.ts +92 -7
- package/src/cli/registry.ts +2 -1
- package/src/cli/system-command.ts +74 -5
- package/src/cli/uninstall-client-state.ts +12 -0
- package/src/clients/config-export.ts +7 -3
- package/src/codex/account-label.ts +14 -3
- package/src/codex/account-store.ts +113 -26
- package/src/codex/account-usability.ts +21 -0
- package/src/codex/auth-api/login-flow.ts +14 -2
- package/src/codex/auth-api/reset-credit-service.ts +11 -2
- package/src/codex/auth-context.ts +199 -15
- package/src/codex/catalog/aggregation.ts +80 -1
- package/src/codex/catalog/model-visibility.ts +1 -0
- package/src/codex/catalog/remote.ts +30 -0
- package/src/codex/catalog/retained-sync.ts +9 -1
- package/src/codex/catalog/routed-gather.ts +38 -1
- package/src/codex/cli-install-provenance.ts +7 -1
- package/src/codex/convergence.ts +7 -2
- package/src/codex/desktop-app/types.ts +11 -2
- package/src/codex/desktop-app/windows.ts +5 -5
- package/src/codex/desktop-switches.ts +145 -0
- package/src/codex/history-job.ts +5 -1
- package/src/codex/history-provider.ts +33 -4
- package/src/codex/history-worker.ts +14 -1
- package/src/codex/inject/remove.ts +145 -7
- package/src/codex/inject/restore.ts +231 -32
- package/src/codex/inject.ts +12 -16
- package/src/codex/loopback-target.ts +9 -0
- package/src/codex/model-entitlements.ts +152 -15
- package/src/codex/native-profile-startup.ts +64 -20
- package/src/codex/pool-refresh-backoff.ts +12 -3
- package/src/codex/quota-rejection.ts +104 -15
- package/src/codex/routing/cache-affinity.ts +70 -0
- package/src/codex/routing/cooldown-math.ts +10 -0
- package/src/codex/routing/selection.ts +79 -2
- package/src/codex/routing/thread-affinity.ts +50 -2
- package/src/codex/routing/transient-hold-dispatch.ts +141 -0
- package/src/codex/routing.ts +29 -49
- package/src/codex/warmup.ts +1 -1
- package/src/combos/failover.ts +85 -0
- package/src/combos/request.ts +17 -10
- package/src/combos/types.ts +23 -2
- package/src/config/atomic-write.ts +83 -8
- package/src/config/pending-teardown.ts +31 -0
- package/src/config/schema/config-schema.ts +2 -0
- package/src/config/schema/leaf-validators.ts +1 -0
- package/src/generated/compatibility-version.json +272 -180
- package/src/images/loop.ts +1 -1
- package/src/lib/bounded-subprocess.ts +62 -10
- package/src/lib/errors.ts +17 -0
- package/src/lib/request-execution-budget.ts +147 -21
- package/src/lib/spend-reservation-ledger.ts +18 -0
- package/src/lib/state-store-registrations.ts +6 -2
- package/src/lib/test-home-guard.ts +85 -1
- package/src/lib/upstream-retry.ts +77 -10
- package/src/lib/windows-elevation.ts +76 -14
- package/src/lib/windows-secret-acl.ts +151 -15
- package/src/lib/windows-user-principal.ts +5 -1
- package/src/oauth/index.ts +2 -2
- package/src/oauth/key-providers.ts +2 -2
- package/src/providers/derive.ts +6 -0
- package/src/providers/kiro-models.ts +4 -3
- package/src/providers/label.ts +19 -1
- package/src/providers/model-discovery.ts +35 -7
- package/src/providers/registry/entries-core.ts +18 -0
- package/src/providers/registry/entries-extended.ts +59 -28
- package/src/providers/registry/model-seeds.ts +71 -17
- package/src/providers/registry/types.ts +9 -0
- package/src/responses/reasoning-envelope.ts +6 -3
- package/src/responses/spill-store.ts +17 -0
- package/src/responses/state/body-policy.ts +25 -0
- package/src/responses/state/spill-queue.ts +8 -6
- package/src/responses/state.ts +3 -22
- package/src/router.ts +4 -0
- package/src/routing/identity-domains.ts +21 -14
- package/src/routing/probe-lease.ts +103 -1
- package/src/server/auth-cors.ts +1 -0
- package/src/server/chat-completions.ts +3 -1
- package/src/server/chat-native.ts +37 -9
- package/src/server/index/live-sideband.ts +37 -1
- package/src/server/index/websocket-handler.ts +54 -3
- package/src/server/index.ts +5 -5
- package/src/server/inspection-tee.ts +107 -0
- package/src/server/live.ts +46 -1
- package/src/server/management/combo-routes.ts +10 -1
- package/src/server/management/config-routes.ts +27 -5
- package/src/server/models-capabilities.ts +24 -3
- package/src/server/relay-eager.ts +2 -0
- package/src/server/relay.ts +14 -19
- package/src/server/request-log.ts +127 -3
- package/src/server/response-log-body.ts +153 -0
- package/src/server/responses/account-change-state.ts +74 -0
- package/src/server/responses/adapter-continuation.ts +33 -7
- package/src/server/responses/adapter-delivery.ts +5 -11
- package/src/server/responses/adapter-dispatch.ts +84 -13
- package/src/server/responses/codex-ws-exchange.ts +65 -4
- package/src/server/responses/codex-ws-wire.ts +5 -0
- package/src/server/responses/collaboration.ts +74 -4
- package/src/server/responses/combo-session-recall.ts +68 -8
- package/src/server/responses/combo-stream-preflight.ts +68 -5
- package/src/server/responses/compact.ts +54 -13
- package/src/server/responses/core-auth.ts +2 -0
- package/src/server/responses/core-codex-account.ts +51 -3
- package/src/server/responses/core-combo.ts +129 -23
- package/src/server/responses/core-errors.ts +18 -0
- package/src/server/responses/core-options.ts +3 -0
- package/src/server/responses/core-replay.ts +105 -32
- package/src/server/responses/core.ts +3 -3
- package/src/server/responses/encrypted-payload.ts +0 -1
- package/src/server/responses/fetch-helpers.ts +4 -1
- package/src/server/responses/input-admission.ts +126 -6
- package/src/server/responses/native-injection-protocol.ts +42 -0
- package/src/server/responses/native-injection-replay.ts +105 -0
- package/src/server/responses/native-injection.ts +242 -0
- package/src/server/responses/native-response-control.ts +56 -0
- package/src/server/responses/native-response-json.ts +14 -0
- package/src/server/responses/native-response-output.ts +37 -0
- package/src/server/responses/native-steering-log.ts +44 -0
- package/src/server/responses/native-steering-policy.ts +49 -0
- package/src/server/responses/native-steering-replay.ts +126 -0
- package/src/server/responses/native-steering-settings.ts +76 -0
- package/src/server/responses/native-steering.ts +400 -0
- package/src/server/responses/native-tool-results.ts +130 -0
- package/src/server/responses/passthrough-delivery.ts +30 -6
- package/src/server/responses/passthrough-dispatch.ts +61 -11
- package/src/server/responses/passthrough-error.ts +38 -2
- package/src/server/responses/request-prepare.ts +173 -22
- package/src/server/responses/request-send-budget.ts +97 -2
- package/src/server/responses/request-spend.ts +147 -0
- package/src/server/responses/request-transport.ts +62 -3
- package/src/server/responses/run-turn-execution.ts +59 -31
- package/src/server/responses/sidecar-execution.ts +7 -13
- package/src/server/responses/terminal-guard.ts +65 -4
- package/src/server/responses/ws-upstream.ts +21 -1
- package/src/server/responses-undeclared-tool-guard.ts +9 -5
- package/src/server/stop-teardown.ts +8 -1
- package/src/server/ws-bridge.ts +16 -1
- package/src/service/cli.ts +13 -1
- package/src/service/windows-ops.ts +210 -16
- package/src/service/windows-scheduler.ts +28 -21
- package/src/service.ts +1 -1
- package/src/types/config.ts +8 -1
- package/src/types/provider.ts +13 -0
- package/src/types/request.ts +8 -5
- package/src/types/tools.ts +24 -0
- package/src/types.ts +2 -0
- package/src/update/index.ts +10 -0
- package/src/update/stop-contract.d.mts +1 -0
- package/src/update/stop-contract.mjs +19 -0
- package/src/update/stop-decision.d.mts +1 -1
- package/src/update/stop-decision.mjs +12 -3
- package/src/usage/log.ts +1 -1
- package/src/vision/anthropic-describe.ts +1 -1
- package/src/vision/describe.ts +5 -5
- package/src/web-search/anthropic-executor.ts +1 -1
- package/src/web-search/exa-executor.ts +1 -1
- package/src/web-search/executor.ts +1 -1
- package/src/web-search/gemini-executor.ts +1 -1
- package/src/web-search/loop.ts +1 -1
- package/src/web-search/ollama-executor.ts +1 -1
- package/src/web-search/parse.ts +67 -14
- package/src/web-search/passthrough-bridge.ts +64 -31
- package/src/web-search/xai-executor.ts +1 -1
|
@@ -23,6 +23,8 @@ import {
|
|
|
23
23
|
|
|
24
24
|
const MODEL_DISCOVERY_MAX_FILTER_VALUES = 256;
|
|
25
25
|
const MODEL_DISCOVERY_MAX_FILTER_STRING_LENGTH = 1_024;
|
|
26
|
+
const TRAILING_SLASHES = /\/+$/;
|
|
27
|
+
const TRAILING_MODELS = /\/models$/;
|
|
26
28
|
|
|
27
29
|
export interface ResolvedProviderModelDiscovery {
|
|
28
30
|
spec?: ProviderModelDiscoverySpec;
|
|
@@ -50,6 +52,20 @@ export type ModelEnvelopeRowsResult =
|
|
|
50
52
|
| { ok: true; rows: unknown[] }
|
|
51
53
|
| { ok: false; reason: "invalid_shape" | "too_many_models" };
|
|
52
54
|
|
|
55
|
+
/**
|
|
56
|
+
* Build the default OpenAI-compatible model-discovery URL from a configured baseUrl.
|
|
57
|
+
*
|
|
58
|
+
* `baseUrl` is required on both `OcxProviderConfig` and the persisted-config schema, so a row
|
|
59
|
+
* without one is not a state configuration loading can produce. It is deliberately not tolerated
|
|
60
|
+
* here: the old template-literal join silently produced `"undefined/models"`, which is not a usable
|
|
61
|
+
* fallback either — it only ever survived because a static row returns before the URL is parsed.
|
|
62
|
+
*/
|
|
63
|
+
export function providerModelsUrl(baseUrl: string): string {
|
|
64
|
+
const trimmed = baseUrl.trim().replace(TRAILING_SLASHES, "");
|
|
65
|
+
const withoutEndpoint = trimmed.replace(TRAILING_MODELS, "");
|
|
66
|
+
return `${withoutEndpoint}/models`;
|
|
67
|
+
}
|
|
68
|
+
|
|
53
69
|
function positiveIntegerAtMost(value: number | undefined, hardLimit: number): number {
|
|
54
70
|
if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return hardLimit;
|
|
55
71
|
return Math.min(Math.floor(value), hardLimit);
|
|
@@ -113,6 +129,14 @@ export function providerModelDiscoverySpecError(spec: ProviderModelDiscoverySpec
|
|
|
113
129
|
if (queryEntries.some(([key, value]) => !key.trim() || key.length > 128 || typeof value !== "string" || value.length > 512)) {
|
|
114
130
|
return "discovery query keys/values exceed their bounds";
|
|
115
131
|
}
|
|
132
|
+
for (const [field, value] of [
|
|
133
|
+
["envelopeKey", spec.envelopeKey],
|
|
134
|
+
["idField", spec.idField],
|
|
135
|
+
] as const) {
|
|
136
|
+
if (value !== undefined && (
|
|
137
|
+
typeof value !== "string" || !value || value !== value.trim() || value.length > 128
|
|
138
|
+
)) return `${field} must be a nonblank field name up to 128 characters`;
|
|
139
|
+
}
|
|
116
140
|
for (const [field, value, hardLimit] of [
|
|
117
141
|
["maxResponseBytes", spec.maxResponseBytes, MODEL_DISCOVERY_MAX_RESPONSE_BYTES],
|
|
118
142
|
["maxModels", spec.maxModels, MODEL_DISCOVERY_MAX_MODELS],
|
|
@@ -406,7 +430,7 @@ export function extractModelEnvelopeRows(
|
|
|
406
430
|
return { ok: true, rows };
|
|
407
431
|
}
|
|
408
432
|
|
|
409
|
-
/** Validate, bound, deduplicate, and
|
|
433
|
+
/** Validate, bound, deduplicate, and filter the declared envelope or a top-level array (Together `#617`). */
|
|
410
434
|
/**
|
|
411
435
|
* Metadata a sibling `models[]` array may contribute to an ALREADY-ADMITTED
|
|
412
436
|
* `data[]` row (#1797).
|
|
@@ -485,24 +509,26 @@ export function extractProviderModelItems(
|
|
|
485
509
|
let data: unknown[];
|
|
486
510
|
let siblings: SiblingIndex | null = null;
|
|
487
511
|
if (Array.isArray(value)) {
|
|
488
|
-
// Together-style top-level /models arrays.
|
|
489
|
-
// `models` key on openai-chat responses as valid
|
|
512
|
+
// Together-style top-level /models arrays. The default contract must not treat a stray
|
|
513
|
+
// `models` key on openai-chat responses as valid; only a provider spec may opt into it.
|
|
490
514
|
if (value.length > limit) return { ok: false, reason: "too_many_models" };
|
|
491
515
|
data = value;
|
|
492
516
|
} else {
|
|
493
|
-
const
|
|
517
|
+
const envelopeKey = discovery.spec?.envelopeKey ?? "data";
|
|
518
|
+
const envelope = extractModelEnvelopeRows(value, discovery.maxModels, [envelopeKey]);
|
|
494
519
|
if (!envelope.ok) return envelope;
|
|
495
520
|
data = envelope.rows;
|
|
496
|
-
siblings = buildSiblingIndex(value, limit);
|
|
521
|
+
siblings = envelopeKey === "data" ? buildSiblingIndex(value, limit) : null;
|
|
497
522
|
}
|
|
498
523
|
|
|
499
524
|
const items: ProviderModelsApiItem[] = [];
|
|
500
525
|
const seen = new Set<string>();
|
|
526
|
+
const idField = discovery.spec?.idField ?? "id";
|
|
501
527
|
for (const raw of data) {
|
|
502
528
|
if (raw === null || typeof raw !== "object" || Array.isArray(raw)) {
|
|
503
529
|
return { ok: false, reason: "invalid_shape" };
|
|
504
530
|
}
|
|
505
|
-
const id = (raw as
|
|
531
|
+
const id = (raw as Record<string, unknown>)[idField];
|
|
506
532
|
if (!isValidModelDiscoveryModelId(id)) return { ok: false, reason: "invalid_shape" };
|
|
507
533
|
const prefix = discovery.spec?.stripIdPrefix;
|
|
508
534
|
let finalId = id;
|
|
@@ -510,7 +536,9 @@ export function extractProviderModelItems(
|
|
|
510
536
|
finalId = finalId.slice(prefix.length);
|
|
511
537
|
if (!isValidModelDiscoveryModelId(finalId)) continue;
|
|
512
538
|
}
|
|
513
|
-
const item = finalId === id
|
|
539
|
+
const item = finalId === id && idField === "id"
|
|
540
|
+
? raw as ProviderModelsApiItem
|
|
541
|
+
: { ...(raw as Record<string, unknown>), id: finalId };
|
|
514
542
|
// Admission is decided on the ORIGINAL `data[]` row, before any sibling
|
|
515
543
|
// enrichment. Merging first let a `models[]` entry supply the very field a
|
|
516
544
|
// provider filter requires — reproduced against the real Chutes policy,
|
|
@@ -17,6 +17,7 @@ import type { ProviderRegistryEntry } from "./types";
|
|
|
17
17
|
import {
|
|
18
18
|
ANTHROPIC_MODELS,
|
|
19
19
|
ANTHROPIC_MODEL_CONTEXT_WINDOWS,
|
|
20
|
+
ANTHROPIC_MODEL_INPUT_MODALITIES,
|
|
20
21
|
ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
|
|
21
22
|
ANTHROPIC_MODEL_REASONING_EFFORTS,
|
|
22
23
|
ZAI_GLM_52_REASONING_EFFORTS,
|
|
@@ -281,6 +282,17 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
|
281
282
|
forwardCallerServiceTier: false,
|
|
282
283
|
},
|
|
283
284
|
},
|
|
285
|
+
// Grok 4.6/4.5 OAuth Responses replays Codex tool history. After a mid-stream 502/reset,
|
|
286
|
+
// the client can resend a function_call without a matching output, or with hook-injected
|
|
287
|
+
// developer context between the pair. Google already synthesizes a missing tool_result
|
|
288
|
+
// (#2199). xAI's Responses parser does not, so the next turns 400 and the thread snowballs.
|
|
289
|
+
// Reuse the existing adjacency capability (Kimi #4726, DeepSeek #1292). Do not set
|
|
290
|
+
// statelessResponses: xAI stores responses for 30 days and documents previous_response_id.
|
|
291
|
+
// https://docs.x.ai/developers/model-capabilities/text/comparison
|
|
292
|
+
requiresAdjacentResponsesToolResults: true,
|
|
293
|
+
// The dangling half of the same failure: a call whose output never arrived. Kimi accepts that
|
|
294
|
+
// shape, so this is a second capability rather than a widening of the one above.
|
|
295
|
+
requiresPairedResponsesToolResults: true,
|
|
284
296
|
// Vision lineup per docs.x.ai model-capabilities/images/understanding: the grok-4.x chat
|
|
285
297
|
// models accept image input (JPEG/PNG, URL or base64). Without this the catalog leaves
|
|
286
298
|
// inputModalities undefined, and deriveComboCatalogModel defaults an undefined member to
|
|
@@ -383,6 +395,7 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
|
383
395
|
note: "Log in with your Claude account",
|
|
384
396
|
models: [...ANTHROPIC_MODELS],
|
|
385
397
|
modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
|
|
398
|
+
modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
|
|
386
399
|
modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
|
|
387
400
|
// Codex omits max_output_tokens; without a provider budget the Anthropic adapter
|
|
388
401
|
// falls back to 8192, which truncates long answers with stop_reason=max_tokens.
|
|
@@ -403,6 +416,7 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
|
403
416
|
models: [...ANTHROPIC_MODELS],
|
|
404
417
|
liveModels: true,
|
|
405
418
|
modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
|
|
419
|
+
modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
|
|
406
420
|
modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
|
|
407
421
|
defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
|
|
408
422
|
defaultModel: "claude-sonnet-5",
|
|
@@ -420,6 +434,10 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
|
420
434
|
// or the one the Claude /v1/messages inbound derives); the adapter itself never invents one.
|
|
421
435
|
// Evidence: https://platform.kimi.com/docs/api/chat
|
|
422
436
|
promptCacheKey: true,
|
|
437
|
+
// Kimi's Responses endpoint rejects hook-provided context between a tool call and
|
|
438
|
+
// its matching result (#4726), the same strict shape DeepSeek exposed in #1292.
|
|
439
|
+
// The flag is inert while this preset uses the Chat wire.
|
|
440
|
+
requiresAdjacentResponsesToolResults: true,
|
|
423
441
|
featured: true,
|
|
424
442
|
oauthId: "kimi",
|
|
425
443
|
jawcodeBundle: "moonshot",
|
|
@@ -56,6 +56,11 @@ import {
|
|
|
56
56
|
ALIBABA_TOKEN_PLAN_MODELS,
|
|
57
57
|
ALIBABA_TOKEN_PLAN_QWEN_MODELS,
|
|
58
58
|
ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
59
|
+
ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
|
|
60
|
+
ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
|
|
61
|
+
ALIBABA_TOKEN_PLAN_NO_VISION,
|
|
62
|
+
ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
|
|
63
|
+
QWEN38_FAMILY,
|
|
59
64
|
ALIBABA_INTL_TOKEN_PLAN_MODELS,
|
|
60
65
|
ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS,
|
|
61
66
|
TENCENT_CODING_PLAN_MODELS,
|
|
@@ -110,6 +115,13 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
110
115
|
// Baseten says models outside its reasoning table do not support reasoning. Keep
|
|
111
116
|
// unknown/new live slugs conservative until an official-docs registry refresh proves it.
|
|
112
117
|
reasoningEfforts: [],
|
|
118
|
+
// `text.verbosity` is an OpenAI Responses parameter. Baseten documents its Model
|
|
119
|
+
// APIs as Chat Completions compatible, so there is nothing on that wire for it to
|
|
120
|
+
// become, and a routed row must not inherit the Codex template's verbosity picker
|
|
121
|
+
// (#4630: Codex sent `text: { verbosity: "low" }` and the turn 400'd before any
|
|
122
|
+
// model output). Provider-wide rather than per-model because this catalog is live-
|
|
123
|
+
// discovered: a slug that arrives tomorrow supports it no more than the seeded ones.
|
|
124
|
+
supportsVerbosity: false,
|
|
113
125
|
modelReasoningEfforts: BASETEN_MODEL_REASONING_EFFORTS,
|
|
114
126
|
modelReasoningEffortMap: BASETEN_MODEL_REASONING_EFFORT_MAP,
|
|
115
127
|
modelDefaultReasoningEfforts: BASETEN_MODEL_DEFAULT_REASONING_EFFORTS,
|
|
@@ -420,6 +432,7 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
420
432
|
// model_access_denied, which is why the Chat path cannot simply hang off the new base.
|
|
421
433
|
responsesPath: "/api/v1/responses",
|
|
422
434
|
chatCompletionsPath: "/api/coding/paas/v4/chat/completions",
|
|
435
|
+
modelDiscovery: { path: "/api/v1/models", envelopeKey: "models", idField: "slug" },
|
|
423
436
|
// The address this row occupied before the move. A saved custom provider still pointing
|
|
424
437
|
// at the Chat endpoint keeps receiving this row's metadata (#1100).
|
|
425
438
|
destinationAliases: [{ baseUrl: "https://api.z.ai/api/coding/paas/v4", adapter: "openai-chat" }],
|
|
@@ -717,22 +730,35 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
717
730
|
liveModels: false,
|
|
718
731
|
note: "Token Plan Personal Edition · China (Beijing)",
|
|
719
732
|
modelInputModalities: ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
720
|
-
modelContextWindows:
|
|
721
|
-
|
|
722
|
-
"qwen3.6-flash": 1_000_000, "glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000,
|
|
723
|
-
},
|
|
733
|
+
modelContextWindows: ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
|
|
734
|
+
modelMaxOutputTokens: ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
|
|
724
735
|
modelReasoningEfforts: {
|
|
725
736
|
...Object.fromEntries(ALIBABA_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
|
|
726
|
-
|
|
727
|
-
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
728
|
-
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
737
|
+
...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
|
|
729
738
|
"glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
|
|
739
|
+
"deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
|
|
740
|
+
"deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
|
|
741
|
+
"deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
|
|
742
|
+
"deepseek-v4.1-flash": deepseekThinkingEffortsFor("deepseek-v4.1-flash"),
|
|
743
|
+
},
|
|
744
|
+
modelReasoningEffortMap: {
|
|
745
|
+
"deepseek-v4-pro": deepseekReasoningMapFor("deepseek-v4-pro"),
|
|
746
|
+
"deepseek-v4-pro-0813": deepseekReasoningMapFor("deepseek-v4-pro-0813"),
|
|
747
|
+
"deepseek-v4-flash-0731": deepseekReasoningMapFor("deepseek-v4-flash-0731"),
|
|
748
|
+
"deepseek-v4.1-flash": deepseekReasoningMapFor("deepseek-v4.1-flash"),
|
|
730
749
|
},
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
750
|
+
// Probed 260915 on the plan gateway: json_object returns valid JSON, strict
|
|
751
|
+
// json_schema is rejected 400 ("This response_format type is unavailable now")
|
|
752
|
+
// in both thinking modes, so requests downgrade to json_object rather than
|
|
753
|
+
// sending a schema the gateway refuses.
|
|
754
|
+
noJsonSchemaModels: ["deepseek-v4.1-flash"],
|
|
755
|
+
modelDefaultReasoningEfforts: Object.fromEntries(QWEN38_FAMILY.map(id => [id, "xhigh"])),
|
|
756
|
+
directReasoningEffortModels: QWEN38_FAMILY,
|
|
757
|
+
thinkingBudgetModels: ALIBABA_TOKEN_PLAN_QWEN_MODELS.filter(id => !QWEN38_FAMILY.includes(id)),
|
|
758
|
+
preserveReasoningContentModels: ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
|
|
759
|
+
noVisionModels: ALIBABA_TOKEN_PLAN_NO_VISION,
|
|
760
|
+
// The gateway accepts prompt_cache_key on every Token Plan chat model (probed 260902).
|
|
761
|
+
promptCacheKey: true,
|
|
736
762
|
},
|
|
737
763
|
{
|
|
738
764
|
id: "alibaba-token-plan-intl",
|
|
@@ -749,31 +775,34 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
749
775
|
note: "Token Plan Team Edition · Singapore (ap-southeast-1)",
|
|
750
776
|
metadataModelIdNormalize: "case-insensitive",
|
|
751
777
|
modelInputModalities: ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES,
|
|
752
|
-
modelContextWindows:
|
|
753
|
-
|
|
754
|
-
"qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000, "qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
|
|
755
|
-
"deepseek-v4-flash": 1_000_000, "deepseek-v3.2": 131_072,
|
|
756
|
-
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
|
|
757
|
-
"glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000, "glm-5.1": 1_000_000, "glm-5": 1_000_000,
|
|
758
|
-
"MiniMax-M2.5": 204_800,
|
|
759
|
-
},
|
|
778
|
+
modelContextWindows: ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
|
|
779
|
+
modelMaxOutputTokens: ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
|
|
760
780
|
modelReasoningEfforts: {
|
|
761
781
|
...Object.fromEntries(ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
|
|
762
|
-
|
|
763
|
-
"glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
|
|
764
|
-
"glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
|
|
782
|
+
...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
|
|
765
783
|
"glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
|
|
784
|
+
"deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
|
|
785
|
+
"deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
|
|
766
786
|
"deepseek-v4-flash": deepseekThinkingEffortsFor("deepseek-v4-flash"),
|
|
787
|
+
"deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
|
|
788
|
+
"deepseek-v4.1-flash": deepseekThinkingEffortsFor("deepseek-v4.1-flash"),
|
|
767
789
|
},
|
|
768
790
|
modelReasoningEffortMap: {
|
|
791
|
+
"deepseek-v4-pro": deepseekReasoningMapFor("deepseek-v4-pro"),
|
|
792
|
+
"deepseek-v4-pro-0813": deepseekReasoningMapFor("deepseek-v4-pro-0813"),
|
|
769
793
|
"deepseek-v4-flash": deepseekReasoningMapFor("deepseek-v4-flash"),
|
|
794
|
+
"deepseek-v4-flash-0731": deepseekReasoningMapFor("deepseek-v4-flash-0731"),
|
|
795
|
+
"deepseek-v4.1-flash": deepseekReasoningMapFor("deepseek-v4.1-flash"),
|
|
770
796
|
},
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
797
|
+
// Same 260915 json_schema rejection probe as the Beijing entry.
|
|
798
|
+
noJsonSchemaModels: ["deepseek-v4.1-flash"],
|
|
799
|
+
directReasoningEffortModels: QWEN38_FAMILY,
|
|
800
|
+
thinkingBudgetModels: ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.filter(id => !QWEN38_FAMILY.includes(id)),
|
|
801
|
+
preserveReasoningContentModels: ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
|
|
802
|
+
noVisionModels: ALIBABA_TOKEN_PLAN_NO_VISION,
|
|
775
803
|
noReasoningModels: ["kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "deepseek-v3.2", "glm-5.1", "glm-5", "MiniMax-M2.5"],
|
|
776
|
-
modelDefaultReasoningEfforts:
|
|
804
|
+
modelDefaultReasoningEfforts: Object.fromEntries(QWEN38_FAMILY.map(id => [id, "xhigh"])),
|
|
805
|
+
promptCacheKey: true,
|
|
777
806
|
},
|
|
778
807
|
// NEEDS_HUMAN 2026-07-10: kept for config compatibility, but this is a dashboard URL,
|
|
779
808
|
// no /models endpoint is documented, and tools are silently ignored upstream per docs.parallel.ai.
|
|
@@ -883,6 +912,8 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
883
912
|
modelSuffixBracketStrip: true,
|
|
884
913
|
// API-key form of the same Kimi Code Plan transport; keep cache affinity identical to OAuth.
|
|
885
914
|
promptCacheKey: true,
|
|
915
|
+
// Keep Responses tool-result adjacency aligned with the OAuth preset (#4726).
|
|
916
|
+
requiresAdjacentResponsesToolResults: true,
|
|
886
917
|
models: KIMI_CODING_MODELS,
|
|
887
918
|
modelContextWindows: KIMI_CODING_MODEL_CONTEXT_WINDOWS,
|
|
888
919
|
modelInputModalities: KIMI_CODING_MODEL_INPUT_MODALITIES,
|
|
@@ -8,6 +8,10 @@ import type { ProviderModelDiscoverySpec } from "./types";
|
|
|
8
8
|
// always on, per the official models overview and pricing page (platform.claude.com).
|
|
9
9
|
export const ANTHROPIC_MODELS = ["claude-fable-5-1", "claude-fable-5", "claude-sonnet-5", "claude-opus-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-sonnet-4-6", "claude-haiku-4-5"];
|
|
10
10
|
export const ANTHROPIC_MODEL_CONTEXT_WINDOWS: Record<string, number> = { "claude-fable-5-1": 1_000_000, "claude-sonnet-5": 1_000_000, "claude-fable-5": 1_000_000, "claude-opus-5": 1_000_000, "claude-opus-4-8": 1_000_000, "claude-opus-4-7": 1_000_000, "claude-opus-4-6": 1_000_000, "claude-sonnet-4-6": 1_000_000, "claude-haiku-4-5": 200_000 };
|
|
11
|
+
// All seeded Claude models support vision: https://platform.claude.com/docs/en/models/overview
|
|
12
|
+
export const ANTHROPIC_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
|
|
13
|
+
ANTHROPIC_MODELS.map(id => [id, ["text", "image"]]),
|
|
14
|
+
);
|
|
11
15
|
// Every current Claude family accepts at least 64k output tokens (Haiku 4.5 / Sonnet 4.x
|
|
12
16
|
// through Opus 5 and Fable 5). Anthropic caps max_tokens per model server-side, so a
|
|
13
17
|
// larger request never over-allocates; it only stops the 8192 truncation.
|
|
@@ -433,20 +437,42 @@ export const deepseekReasoningMapFor = (modelId: string): Record<string, string>
|
|
|
433
437
|
// Coding Plan: the products use different exact allowlists and different base URLs.
|
|
434
438
|
// Evidence: https://help.aliyun.com/en/model-studio/token-plan-personal-overview
|
|
435
439
|
// https://help.aliyun.com/en/model-studio/token-plan-quickstart
|
|
440
|
+
// 260909 refresh, re-probed against the live gateway (both regions, both tiers):
|
|
441
|
+
// https://github.com/oliver-mee/alibaba-token-plan-wiki (machine-readable catalog).
|
|
442
|
+
// glm-5.3 / glm-5.3-flash removed from both Token Plan catalogs: they exist on Z.AI
|
|
443
|
+
// endpoints but the Token Plan gateway has never served either id (the 260826 seed
|
|
444
|
+
// propagated them across every GLM-carrying catalog; a selected row 404s).
|
|
445
|
+
// The Beijing preset keeps the Personal Edition subset; non-chat ids (audio/image/
|
|
446
|
+
// video families) stay out: they answer only on async endpoints openai-chat cannot
|
|
447
|
+
// reach. deepseek-v4-pro-0813 is callable but NOT listed by /models, which is the
|
|
448
|
+
// reason liveModels must stay false for this provider. deepseek-v4.1-flash is the
|
|
449
|
+
// 260910 DeepSeek rename row: listed on /models on both tiers and regions from 260915,
|
|
450
|
+
// hybrid thinking, vision via user message and tool result, json_object but not
|
|
451
|
+
// json_schema (see noJsonSchemaModels on the entries).
|
|
452
|
+
// Beijing serves the Personal Edition, so this is the Personal-tier roster probed
|
|
453
|
+
// 260909 (a strict subset of Team). deepseek-v4-pro-0813 stays out of the Beijing
|
|
454
|
+
// entry: its callability is only proven on Team keys, and no Personal key has been
|
|
455
|
+
// shown to reach it. The Beijing entry also shares the intl maps, so it carries a
|
|
456
|
+
// few orphan keys (kimi/glm-5/MiniMax rows); harmless, and one map beats two
|
|
457
|
+
// drifting ones.
|
|
436
458
|
export const ALIBABA_TOKEN_PLAN_MODELS = [
|
|
437
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
438
|
-
"
|
|
459
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
460
|
+
"deepseek-v4-pro", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "glm-5.2",
|
|
439
461
|
];
|
|
440
462
|
export const ALIBABA_TOKEN_PLAN_QWEN_MODELS = [
|
|
441
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
463
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
442
464
|
];
|
|
443
465
|
export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
444
466
|
"qwen3.8-max": ["text", "image"],
|
|
445
|
-
"qwen3.
|
|
467
|
+
"qwen3.8-flash": ["text", "image"],
|
|
468
|
+
"qwen3.7-max": ["text"],
|
|
446
469
|
"qwen3.7-plus": ["text", "image"],
|
|
447
470
|
"qwen3.6-flash": ["text", "image"],
|
|
448
|
-
"
|
|
449
|
-
"
|
|
471
|
+
"deepseek-v4-pro": ["text"],
|
|
472
|
+
"deepseek-v4-pro-0813": ["text"],
|
|
473
|
+
"deepseek-v4-flash-0731": ["text"],
|
|
474
|
+
// Vision probed on the plan gateway 260915 (user message and tool result, both 200).
|
|
475
|
+
"deepseek-v4.1-flash": ["text", "image"],
|
|
450
476
|
"glm-5.2": ["text"],
|
|
451
477
|
};
|
|
452
478
|
|
|
@@ -454,15 +480,18 @@ export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
|
454
480
|
// Multi-vendor lineup distinct from Beijing — includes DeepSeek V4 flash, Kimi K2.7, MiniMax.
|
|
455
481
|
// Evidence: https://www.alibabacloud.com/help/en/model-studio/token-plan-overview
|
|
456
482
|
// https://qwencloud.com/pricing/token-plan (qwen3.8 metadata)
|
|
483
|
+
// The Team Edition roster (Singapore), verified identical to the CN Team set on 260909.
|
|
484
|
+
// deepseek-v4-pro is restored: it remains callable on the plan gateway (probed 260909,
|
|
485
|
+
// listed on /models on both regions) after being dropped as "retired" upstream.
|
|
457
486
|
export const ALIBABA_INTL_TOKEN_PLAN_MODELS = [
|
|
458
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
459
|
-
"deepseek-v4-flash", "deepseek-v3.2",
|
|
487
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
488
|
+
"deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "deepseek-v3.2",
|
|
460
489
|
"kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5",
|
|
461
|
-
"glm-5.
|
|
490
|
+
"glm-5.2", "glm-5.1", "glm-5",
|
|
462
491
|
"MiniMax-M2.5",
|
|
463
492
|
];
|
|
464
493
|
export const ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS = [
|
|
465
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
494
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
466
495
|
];
|
|
467
496
|
|
|
468
497
|
// 260722 Tencent Cloud Coding Plan. The plan's model set is explicitly dynamic; these are the
|
|
@@ -539,24 +568,49 @@ export const VOLCENGINE_PLAN_TEXT_ONLY_MODELS = [
|
|
|
539
568
|
"doubao-seed-2.0-pro",
|
|
540
569
|
];
|
|
541
570
|
export const ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
542
|
-
|
|
543
|
-
"qwen3.7-max": ["text", "image"],
|
|
544
|
-
"qwen3.7-plus": ["text", "image"],
|
|
571
|
+
...ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
545
572
|
"qwen3.6-plus": ["text", "image"],
|
|
546
|
-
"qwen3.6-flash": ["text", "image"],
|
|
547
573
|
"deepseek-v4-flash": ["text"],
|
|
548
574
|
"deepseek-v3.2": ["text"],
|
|
549
575
|
"kimi-k2.7-code": ["text", "image"],
|
|
550
576
|
"kimi-k2.6": ["text", "image"],
|
|
551
577
|
"kimi-k2.5": ["text", "image"],
|
|
552
|
-
"glm-5.3": ["text"],
|
|
553
|
-
"glm-5.3-flash": ["text", "image"],
|
|
554
|
-
"glm-5.2": ["text"],
|
|
555
578
|
"glm-5.1": ["text"],
|
|
556
579
|
"glm-5": ["text"],
|
|
557
580
|
"MiniMax-M2.5": ["text"],
|
|
558
581
|
};
|
|
559
582
|
|
|
583
|
+
// Shared Token Plan metadata (260909 gateway probes; output ceilings are max_tokens
|
|
584
|
+
// boundary probes: accept at N, reject at N+1).
|
|
585
|
+
export const QWEN38_FAMILY = ["qwen3.8-max", "qwen3.8-flash"];
|
|
586
|
+
export const ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS: Record<string, number> = {
|
|
587
|
+
"qwen3.8-max": 1_000_000, "qwen3.8-flash": 1_000_000, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
|
|
588
|
+
"qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
|
|
589
|
+
"deepseek-v4-pro": 1_000_000, "deepseek-v4-pro-0813": 1_000_000, "deepseek-v4-flash": 1_000_000,
|
|
590
|
+
"deepseek-v4-flash-0731": 1_000_000, "deepseek-v4.1-flash": 1_000_000, "deepseek-v3.2": 131_072,
|
|
591
|
+
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
|
|
592
|
+
"glm-5.2": 1_000_000, "glm-5.1": 202_752, "glm-5": 202_752,
|
|
593
|
+
"MiniMax-M2.5": 196_608,
|
|
594
|
+
};
|
|
595
|
+
export const ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS: Record<string, number> = {
|
|
596
|
+
"qwen3.8-max": 131_072, "qwen3.8-flash": 131_072, "qwen3.7-max": 131_072, "qwen3.7-plus": 131_072,
|
|
597
|
+
"qwen3.6-plus": 65_536, "qwen3.6-flash": 65_536,
|
|
598
|
+
"deepseek-v4-pro": 393_216, "deepseek-v4-pro-0813": 393_216, "deepseek-v4-flash": 393_216,
|
|
599
|
+
"deepseek-v4-flash-0731": 393_216, "deepseek-v4.1-flash": 393_216, "deepseek-v3.2": 65_536,
|
|
600
|
+
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 98_304,
|
|
601
|
+
"glm-5.2": 131_072, "glm-5.1": 128_000, "glm-5": 16_384,
|
|
602
|
+
"MiniMax-M2.5": 32_768,
|
|
603
|
+
};
|
|
604
|
+
export const ALIBABA_TOKEN_PLAN_NO_VISION = [
|
|
605
|
+
"qwen3.7-max", "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash",
|
|
606
|
+
"deepseek-v4-flash-0731", "deepseek-v3.2", "glm-5.2", "glm-5.1", "glm-5", "MiniMax-M2.5",
|
|
607
|
+
];
|
|
608
|
+
export const ALIBABA_TOKEN_PLAN_PRESERVE_REASONING = [
|
|
609
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
610
|
+
"deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731",
|
|
611
|
+
"deepseek-v4.1-flash", "glm-5.2",
|
|
612
|
+
];
|
|
613
|
+
|
|
560
614
|
// 260717 Kimi K3: the subscription endpoint uses one upstream id (`k3`) for both
|
|
561
615
|
// entitlement tiers. Bare `k3` advertises the Moderato 256K ceiling; the local `[1m]`
|
|
562
616
|
// alias advertises Allegretto's 1M ceiling and is stripped before the upstream request.
|
|
@@ -64,6 +64,10 @@ export interface ProviderModelDiscoveryFilter {
|
|
|
64
64
|
interface ProviderModelDiscoverySharedSpec {
|
|
65
65
|
/** Query parameters applied to the resolved discovery URL. */
|
|
66
66
|
query?: Readonly<Record<string, string>>;
|
|
67
|
+
/** Top-level response key containing model rows; defaults to `data`. */
|
|
68
|
+
envelopeKey?: string;
|
|
69
|
+
/** Model-row field containing the provider-native identifier; defaults to `id`. */
|
|
70
|
+
idField?: string;
|
|
67
71
|
/** Declarative eligibility rules evaluated against each untrusted model row. */
|
|
68
72
|
filter?: ProviderModelDiscoveryFilter;
|
|
69
73
|
/** Optional lower byte ceiling; the process-wide hard ceiling still wins. */
|
|
@@ -217,6 +221,11 @@ export interface ProviderRegistryEntry {
|
|
|
217
221
|
* to stay contiguous. This is seeded/backfilled like other fixed wire capabilities.
|
|
218
222
|
*/
|
|
219
223
|
requiresAdjacentResponsesToolResults?: boolean;
|
|
224
|
+
/**
|
|
225
|
+
* Responses upstream that also rejects a tool call with no matching output anywhere in the
|
|
226
|
+
* replayed input. Seeded/backfilled like other fixed wire capabilities.
|
|
227
|
+
*/
|
|
228
|
+
requiresPairedResponsesToolResults?: boolean;
|
|
220
229
|
/**
|
|
221
230
|
* When enabled, tool results that are present but empty are annotated on the wire.
|
|
222
231
|
* Seeded/backfilled like other fixed wire capabilities.
|
|
@@ -28,9 +28,12 @@ export interface ReasoningEnvelope {
|
|
|
28
28
|
*/
|
|
29
29
|
txt?: string;
|
|
30
30
|
/**
|
|
31
|
-
* Kiro `reasoningContentEvent
|
|
32
|
-
* the
|
|
33
|
-
*
|
|
31
|
+
* Kiro's reasoning blob from `reasoningContentEvent`: a KMS-encrypted value that is opaque to the
|
|
32
|
+
* proxy (the GPT-5.6 family sends it as `signature`, other models as the base64
|
|
33
|
+
* `redactedContent`, and the value carries a tag naming which one — see
|
|
34
|
+
* src/adapters/kiro/reasoning.ts). Kiro's own CLI replays it on the matching
|
|
35
|
+
* `assistantResponseMessage` to preserve model reasoning across turns, so it round-trips here the
|
|
36
|
+
* same way a signature does.
|
|
34
37
|
*/
|
|
35
38
|
krc?: string;
|
|
36
39
|
}
|
|
@@ -159,6 +159,23 @@ function spillNow(): number {
|
|
|
159
159
|
return spillNowOverride?.() ?? Date.now();
|
|
160
160
|
}
|
|
161
161
|
|
|
162
|
+
/**
|
|
163
|
+
* The spill deadline clock, shared with the shutdown drain in `state/spill-queue.ts`.
|
|
164
|
+
*
|
|
165
|
+
* Every deadline the shutdown path enforces has to read the same clock the work it
|
|
166
|
+
* budgets reads. When the drain measured its reserve on `Date.now()` while the ACL
|
|
167
|
+
* harden it was budgeting ran on this injected clock, a test could freeze the clock,
|
|
168
|
+
* believe it had removed wall time from the case, and still lose an 80 ms reserve to
|
|
169
|
+
* real elapsed time on a loaded runner — which is what turned
|
|
170
|
+
* `shutdown fallback prices the job-owned superseded generation before publishing`
|
|
171
|
+
* red on macOS 2/2 in run 35137850114 while the assertion it was written for never ran.
|
|
172
|
+
*
|
|
173
|
+
* Production is unchanged: with no override installed this is `Date.now()`.
|
|
174
|
+
*/
|
|
175
|
+
export function responseSpillNow(): number {
|
|
176
|
+
return spillNow();
|
|
177
|
+
}
|
|
178
|
+
|
|
162
179
|
function record(event: "write" | "fsync" | "close" | "harden" | "publish" | "dir-fsync" | "stub-swap"): void {
|
|
163
180
|
spillIoForTest?.record?.(event);
|
|
164
181
|
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Request bodies that must never enter the continuation cache.
|
|
3
|
+
*
|
|
4
|
+
* The cache is persisted to `responses-state.json`, so anything recorded here reaches disk.
|
|
5
|
+
* Encrypted-agent-task recovery decrypts task text into the request body and promises
|
|
6
|
+
* in-memory, TTL-bounded retention; recording that body would put the plaintext on disk with
|
|
7
|
+
* no TTL and break the promise.
|
|
8
|
+
*
|
|
9
|
+
* A WeakSet rather than a body field on purpose: `_rawBody` is serialized verbatim by the
|
|
10
|
+
* native passthrough, so any marker written into the body itself would be sent upstream.
|
|
11
|
+
* Marking is enforced once here rather than at each call site, because every recording path
|
|
12
|
+
* (streaming, non-streaming, passthrough, forced) funnels through `rememberResponseState` —
|
|
13
|
+
* a new call site cannot reintroduce the leak by forgetting a guard.
|
|
14
|
+
*/
|
|
15
|
+
const nonPersistableBodies = new WeakSet<object>();
|
|
16
|
+
|
|
17
|
+
/** Bar this exact request body from the continuation cache, and therefore from disk. */
|
|
18
|
+
export function markBodyNonPersistable(body: unknown): void {
|
|
19
|
+
if (body && typeof body === "object") nonPersistableBodies.add(body as object);
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** Test the body's in-memory persistence restriction without adding a wire marker. */
|
|
23
|
+
export function isBodyNonPersistable(body: unknown): boolean {
|
|
24
|
+
return !!body && typeof body === "object" && nonPersistableBodies.has(body);
|
|
25
|
+
}
|
|
@@ -7,6 +7,7 @@ import {
|
|
|
7
7
|
MAX_RESPONSE_SPILL_PAYLOAD_BYTES,
|
|
8
8
|
prospectiveResponseSpillBytes,
|
|
9
9
|
responseSpillPayloadCap,
|
|
10
|
+
responseSpillNow,
|
|
10
11
|
type ResponseSpillPublicationControl,
|
|
11
12
|
type ResponseSpillRef,
|
|
12
13
|
writeResponseSpillDurably,
|
|
@@ -377,7 +378,7 @@ function responseSpillShutdownBudget(): { totalMs: number; fallbackReserveMs: nu
|
|
|
377
378
|
}
|
|
378
379
|
|
|
379
380
|
function awaitResponseSpillTailUntil(observed: Promise<void>, deadline: number): Promise<boolean> {
|
|
380
|
-
const remaining = deadline -
|
|
381
|
+
const remaining = deadline - responseSpillNow();
|
|
381
382
|
if (remaining <= 0) return Promise.resolve(false);
|
|
382
383
|
return new Promise(resolve => {
|
|
383
384
|
let finished = false;
|
|
@@ -554,12 +555,13 @@ function terminalizeExhaustedShutdownFallback(
|
|
|
554
555
|
}
|
|
555
556
|
|
|
556
557
|
function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
557
|
-
|
|
558
|
+
// Same clock as the harden work this reserve is budgeting — see `responseSpillNow`.
|
|
559
|
+
const deadline = responseSpillNow() + reserveMs;
|
|
558
560
|
const failures: Error[] = [];
|
|
559
561
|
for (;;) {
|
|
560
562
|
const pending = pendingShutdownFallbackCandidates();
|
|
561
563
|
if (pending.length === 0) return failures;
|
|
562
|
-
if (
|
|
564
|
+
if (responseSpillNow() >= deadline) {
|
|
563
565
|
terminalizeExhaustedShutdownFallback(pending, failures);
|
|
564
566
|
return failures;
|
|
565
567
|
}
|
|
@@ -569,7 +571,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
|
569
571
|
for (let index = 0; index < pending.length; index += 1) {
|
|
570
572
|
const { job, candidate } = pending[index]!;
|
|
571
573
|
if (requireStore().currentEntry(job.id) !== candidate) continue;
|
|
572
|
-
const remaining = deadline -
|
|
574
|
+
const remaining = deadline - responseSpillNow();
|
|
573
575
|
if (remaining <= 0) {
|
|
574
576
|
reserveExhausted = true;
|
|
575
577
|
for (const exhausted of pending.slice(index)) {
|
|
@@ -587,7 +589,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
|
587
589
|
requireStore().recomputeOldestResident();
|
|
588
590
|
requireStore().pruneResponses();
|
|
589
591
|
enforceAppOwnedMemoryBudget();
|
|
590
|
-
if (reserveExhausted ||
|
|
592
|
+
if (reserveExhausted || responseSpillNow() >= deadline) {
|
|
591
593
|
terminalizeExhaustedShutdownFallback(pendingShutdownFallbackCandidates(), failures);
|
|
592
594
|
return failures;
|
|
593
595
|
}
|
|
@@ -597,7 +599,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
|
597
599
|
export async function drainResponseSpillPublications(): Promise<void> {
|
|
598
600
|
const budget = responseSpillShutdownBudget();
|
|
599
601
|
const fallbackReserveMs = Math.min(budget.totalMs, Math.max(1, budget.fallbackReserveMs));
|
|
600
|
-
const drainDeadline =
|
|
602
|
+
const drainDeadline = responseSpillNow() + Math.max(0, budget.totalMs - fallbackReserveMs);
|
|
601
603
|
|
|
602
604
|
for (;;) {
|
|
603
605
|
if (pendingResponseSpills.size === 0) return;
|
package/src/responses/state.ts
CHANGED
|
@@ -23,6 +23,8 @@ import type { ResponseSpillWriteFailureCode, ResponseSpillWriteStatus, ResponseS
|
|
|
23
23
|
export { responseAdmissionCountersForTests } from "./state/spill-failure";
|
|
24
24
|
import { admissionCounters, noteSpillWriteFailure, noteSpillWriteSuccess, spillCounters, spillWriteHealth } from "./state/spill-failure";
|
|
25
25
|
import { loadSnapshotEntry } from "./state/snapshot-codec";
|
|
26
|
+
import { isBodyNonPersistable } from "./state/body-policy";
|
|
27
|
+
export { isBodyNonPersistable, markBodyNonPersistable } from "./state/body-policy";
|
|
26
28
|
export { flushPendingResponseSpillsForTests, awaitResponseSpillPublicationTailForTests, pendingResponseSpillMetricsForTests, setResponseSpillShutdownBudgetForTests, setResponseSpillAsyncAclAttemptBudgetForTests, setResponseSpillShutdownTerminalizationPassLimitForTests } from "./state/spill-queue";
|
|
27
29
|
import {
|
|
28
30
|
bindSpillQueueStore,
|
|
@@ -1235,27 +1237,6 @@ export function responseStateMetrics(): ResponseStateMetrics {
|
|
|
1235
1237
|
* Cache completed output and max_output_tokens partial output for previous_response_id replay.
|
|
1236
1238
|
* Content-filtered incomplete and failed output are not authoritative replay history.
|
|
1237
1239
|
*/
|
|
1238
|
-
/**
|
|
1239
|
-
* Request bodies that must never enter the continuation cache.
|
|
1240
|
-
*
|
|
1241
|
-
* The cache is persisted to `responses-state.json`, so anything recorded here reaches disk.
|
|
1242
|
-
* Encrypted-agent-task recovery decrypts task text into the request body and promises
|
|
1243
|
-
* in-memory, TTL-bounded retention; recording that body would put the plaintext on disk with
|
|
1244
|
-
* no TTL and break the promise.
|
|
1245
|
-
*
|
|
1246
|
-
* A WeakSet rather than a body field on purpose: `_rawBody` is serialized verbatim by the
|
|
1247
|
-
* native passthrough, so any marker written into the body itself would be sent upstream.
|
|
1248
|
-
* Marking is enforced once here rather than at each call site, because every recording path
|
|
1249
|
-
* (streaming, non-streaming, passthrough, forced) funnels through `rememberResponseState` —
|
|
1250
|
-
* a new call site cannot reintroduce the leak by forgetting a guard.
|
|
1251
|
-
*/
|
|
1252
|
-
const nonPersistableBodies = new WeakSet<object>();
|
|
1253
|
-
|
|
1254
|
-
/** Bar this exact request body from the continuation cache, and therefore from disk. */
|
|
1255
|
-
export function markBodyNonPersistable(body: unknown): void {
|
|
1256
|
-
if (body && typeof body === "object") nonPersistableBodies.add(body as object);
|
|
1257
|
-
}
|
|
1258
|
-
|
|
1259
1240
|
export function rememberResponseState(
|
|
1260
1241
|
requestBody: unknown,
|
|
1261
1242
|
response: { id?: unknown; output?: unknown; status?: unknown; incomplete_details?: unknown },
|
|
@@ -1264,7 +1245,7 @@ export function rememberResponseState(
|
|
|
1264
1245
|
): void {
|
|
1265
1246
|
if (!requestBody || typeof requestBody !== "object" || Array.isArray(requestBody)) return;
|
|
1266
1247
|
const request = requestBody as Record<string, unknown>;
|
|
1267
|
-
if (
|
|
1248
|
+
if (isBodyNonPersistable(request)) return;
|
|
1268
1249
|
// `force` bypasses only the store:false skip: Codex sends `store:false` on every non-Azure
|
|
1269
1250
|
// HTTP request (and WS inherits it), yet its WS turns still chain with previous_response_id.
|
|
1270
1251
|
// The passthrough branch records with force so those chains can be expanded locally; the
|