@bitkyc08/opencodex 2.56.0 → 2.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (206) hide show
  1. package/bin/ocx.mjs +10 -0
  2. package/gui/dist/assets/{index-D4zuyIxQ.js → index-BbrHOIY0.js} +21 -21
  3. package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
  4. package/gui/dist/index.html +2 -2
  5. package/package.json +4 -4
  6. package/src/adapters/codebuddy/adapter.ts +2 -1
  7. package/src/adapters/codebuddy/scaffold-guard.ts +249 -0
  8. package/src/adapters/command-code.ts +12 -3
  9. package/src/adapters/cursor/cursor-errors.ts +15 -0
  10. package/src/adapters/cursor/discovery.ts +65 -1
  11. package/src/adapters/cursor/envelope-echo.ts +8 -2
  12. package/src/adapters/cursor/live-transport.ts +5 -1
  13. package/src/adapters/cursor/protobuf-events.ts +110 -11
  14. package/src/adapters/cursor/protobuf-request.ts +19 -1
  15. package/src/adapters/cursor/text-toolcall.ts +230 -0
  16. package/src/adapters/cursor/thread-continuity.ts +67 -0
  17. package/src/adapters/cursor/types.ts +5 -0
  18. package/src/adapters/cursor.ts +55 -5
  19. package/src/adapters/google-http.ts +38 -13
  20. package/src/adapters/google.ts +7 -7
  21. package/src/adapters/kiro/payload.ts +17 -3
  22. package/src/adapters/kiro/reasoning.ts +70 -7
  23. package/src/adapters/kiro/stream.ts +8 -2
  24. package/src/adapters/kiro/wire.ts +2 -1
  25. package/src/adapters/kiro-events.ts +21 -13
  26. package/src/adapters/mimo-free.ts +32 -17
  27. package/src/adapters/ollama-native.ts +42 -8
  28. package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
  29. package/src/adapters/openai-chat/tool-schema.ts +25 -7
  30. package/src/adapters/openai-chat.ts +8 -8
  31. package/src/adapters/openai-responses/passthrough.ts +62 -5
  32. package/src/adapters/openai-responses/request-strips.ts +43 -0
  33. package/src/adapters/physical-send.ts +50 -0
  34. package/src/bridge/errors.ts +26 -2
  35. package/src/bridge/response-json.ts +8 -2
  36. package/src/bridge/sse.ts +20 -2
  37. package/src/claude/desktop-profile.ts +66 -9
  38. package/src/claude/outbound.ts +32 -4
  39. package/src/cli/account-main.ts +1 -1
  40. package/src/cli/capabilities.ts +2 -2
  41. package/src/cli/combo.ts +10 -1
  42. package/src/cli/config-command.ts +35 -18
  43. package/src/cli/dispatch.ts +17 -4
  44. package/src/cli/index.ts +92 -7
  45. package/src/cli/registry.ts +2 -1
  46. package/src/cli/system-command.ts +74 -5
  47. package/src/cli/uninstall-client-state.ts +12 -0
  48. package/src/clients/config-export.ts +7 -3
  49. package/src/codex/account-label.ts +14 -3
  50. package/src/codex/account-store.ts +113 -26
  51. package/src/codex/account-usability.ts +21 -0
  52. package/src/codex/auth-api/login-flow.ts +14 -2
  53. package/src/codex/auth-api/reset-credit-service.ts +11 -2
  54. package/src/codex/auth-context.ts +199 -15
  55. package/src/codex/catalog/aggregation.ts +80 -1
  56. package/src/codex/catalog/model-visibility.ts +1 -0
  57. package/src/codex/catalog/remote.ts +30 -0
  58. package/src/codex/catalog/retained-sync.ts +9 -1
  59. package/src/codex/catalog/routed-gather.ts +38 -1
  60. package/src/codex/cli-install-provenance.ts +7 -1
  61. package/src/codex/convergence.ts +7 -2
  62. package/src/codex/desktop-app/types.ts +11 -2
  63. package/src/codex/desktop-app/windows.ts +5 -5
  64. package/src/codex/desktop-switches.ts +145 -0
  65. package/src/codex/history-job.ts +5 -1
  66. package/src/codex/history-provider.ts +33 -4
  67. package/src/codex/history-worker.ts +14 -1
  68. package/src/codex/inject/remove.ts +145 -7
  69. package/src/codex/inject/restore.ts +231 -32
  70. package/src/codex/inject.ts +12 -16
  71. package/src/codex/loopback-target.ts +9 -0
  72. package/src/codex/model-entitlements.ts +152 -15
  73. package/src/codex/native-profile-startup.ts +64 -20
  74. package/src/codex/pool-refresh-backoff.ts +12 -3
  75. package/src/codex/quota-rejection.ts +104 -15
  76. package/src/codex/routing/cache-affinity.ts +70 -0
  77. package/src/codex/routing/cooldown-math.ts +10 -0
  78. package/src/codex/routing/selection.ts +79 -2
  79. package/src/codex/routing/thread-affinity.ts +50 -2
  80. package/src/codex/routing/transient-hold-dispatch.ts +141 -0
  81. package/src/codex/routing.ts +29 -49
  82. package/src/codex/warmup.ts +1 -1
  83. package/src/combos/failover.ts +85 -0
  84. package/src/combos/request.ts +17 -10
  85. package/src/combos/types.ts +23 -2
  86. package/src/config/atomic-write.ts +83 -8
  87. package/src/config/pending-teardown.ts +31 -0
  88. package/src/config/schema/config-schema.ts +2 -0
  89. package/src/config/schema/leaf-validators.ts +1 -0
  90. package/src/generated/compatibility-version.json +272 -180
  91. package/src/images/loop.ts +1 -1
  92. package/src/lib/bounded-subprocess.ts +62 -10
  93. package/src/lib/errors.ts +17 -0
  94. package/src/lib/request-execution-budget.ts +147 -21
  95. package/src/lib/spend-reservation-ledger.ts +18 -0
  96. package/src/lib/state-store-registrations.ts +6 -2
  97. package/src/lib/test-home-guard.ts +85 -1
  98. package/src/lib/upstream-retry.ts +77 -10
  99. package/src/lib/windows-elevation.ts +76 -14
  100. package/src/lib/windows-secret-acl.ts +151 -15
  101. package/src/lib/windows-user-principal.ts +5 -1
  102. package/src/oauth/index.ts +2 -2
  103. package/src/oauth/key-providers.ts +2 -2
  104. package/src/providers/derive.ts +6 -0
  105. package/src/providers/kiro-models.ts +4 -3
  106. package/src/providers/label.ts +19 -1
  107. package/src/providers/model-discovery.ts +35 -7
  108. package/src/providers/registry/entries-core.ts +18 -0
  109. package/src/providers/registry/entries-extended.ts +59 -28
  110. package/src/providers/registry/model-seeds.ts +71 -17
  111. package/src/providers/registry/types.ts +9 -0
  112. package/src/responses/reasoning-envelope.ts +6 -3
  113. package/src/responses/spill-store.ts +17 -0
  114. package/src/responses/state/body-policy.ts +25 -0
  115. package/src/responses/state/spill-queue.ts +8 -6
  116. package/src/responses/state.ts +3 -22
  117. package/src/router.ts +4 -0
  118. package/src/routing/identity-domains.ts +21 -14
  119. package/src/routing/probe-lease.ts +103 -1
  120. package/src/server/auth-cors.ts +1 -0
  121. package/src/server/chat-completions.ts +3 -1
  122. package/src/server/chat-native.ts +37 -9
  123. package/src/server/index/live-sideband.ts +37 -1
  124. package/src/server/index/websocket-handler.ts +54 -3
  125. package/src/server/index.ts +5 -5
  126. package/src/server/inspection-tee.ts +107 -0
  127. package/src/server/live.ts +46 -1
  128. package/src/server/management/combo-routes.ts +10 -1
  129. package/src/server/management/config-routes.ts +27 -5
  130. package/src/server/models-capabilities.ts +24 -3
  131. package/src/server/relay-eager.ts +2 -0
  132. package/src/server/relay.ts +14 -19
  133. package/src/server/request-log.ts +127 -3
  134. package/src/server/response-log-body.ts +153 -0
  135. package/src/server/responses/account-change-state.ts +74 -0
  136. package/src/server/responses/adapter-continuation.ts +33 -7
  137. package/src/server/responses/adapter-delivery.ts +5 -11
  138. package/src/server/responses/adapter-dispatch.ts +84 -13
  139. package/src/server/responses/codex-ws-exchange.ts +65 -4
  140. package/src/server/responses/codex-ws-wire.ts +5 -0
  141. package/src/server/responses/collaboration.ts +74 -4
  142. package/src/server/responses/combo-session-recall.ts +68 -8
  143. package/src/server/responses/combo-stream-preflight.ts +68 -5
  144. package/src/server/responses/compact.ts +54 -13
  145. package/src/server/responses/core-auth.ts +2 -0
  146. package/src/server/responses/core-codex-account.ts +51 -3
  147. package/src/server/responses/core-combo.ts +129 -23
  148. package/src/server/responses/core-errors.ts +18 -0
  149. package/src/server/responses/core-options.ts +3 -0
  150. package/src/server/responses/core-replay.ts +105 -32
  151. package/src/server/responses/core.ts +3 -3
  152. package/src/server/responses/encrypted-payload.ts +0 -1
  153. package/src/server/responses/fetch-helpers.ts +4 -1
  154. package/src/server/responses/input-admission.ts +126 -6
  155. package/src/server/responses/native-injection-protocol.ts +42 -0
  156. package/src/server/responses/native-injection-replay.ts +105 -0
  157. package/src/server/responses/native-injection.ts +242 -0
  158. package/src/server/responses/native-response-control.ts +56 -0
  159. package/src/server/responses/native-response-json.ts +14 -0
  160. package/src/server/responses/native-response-output.ts +37 -0
  161. package/src/server/responses/native-steering-log.ts +44 -0
  162. package/src/server/responses/native-steering-policy.ts +49 -0
  163. package/src/server/responses/native-steering-replay.ts +126 -0
  164. package/src/server/responses/native-steering-settings.ts +76 -0
  165. package/src/server/responses/native-steering.ts +400 -0
  166. package/src/server/responses/native-tool-results.ts +130 -0
  167. package/src/server/responses/passthrough-delivery.ts +30 -6
  168. package/src/server/responses/passthrough-dispatch.ts +61 -11
  169. package/src/server/responses/passthrough-error.ts +38 -2
  170. package/src/server/responses/request-prepare.ts +173 -22
  171. package/src/server/responses/request-send-budget.ts +97 -2
  172. package/src/server/responses/request-spend.ts +147 -0
  173. package/src/server/responses/request-transport.ts +62 -3
  174. package/src/server/responses/run-turn-execution.ts +59 -31
  175. package/src/server/responses/sidecar-execution.ts +7 -13
  176. package/src/server/responses/terminal-guard.ts +65 -4
  177. package/src/server/responses/ws-upstream.ts +21 -1
  178. package/src/server/responses-undeclared-tool-guard.ts +9 -5
  179. package/src/server/stop-teardown.ts +8 -1
  180. package/src/server/ws-bridge.ts +16 -1
  181. package/src/service/cli.ts +13 -1
  182. package/src/service/windows-ops.ts +210 -16
  183. package/src/service/windows-scheduler.ts +28 -21
  184. package/src/service.ts +1 -1
  185. package/src/types/config.ts +8 -1
  186. package/src/types/provider.ts +13 -0
  187. package/src/types/request.ts +8 -5
  188. package/src/types/tools.ts +24 -0
  189. package/src/types.ts +2 -0
  190. package/src/update/index.ts +10 -0
  191. package/src/update/stop-contract.d.mts +1 -0
  192. package/src/update/stop-contract.mjs +19 -0
  193. package/src/update/stop-decision.d.mts +1 -1
  194. package/src/update/stop-decision.mjs +12 -3
  195. package/src/usage/log.ts +1 -1
  196. package/src/vision/anthropic-describe.ts +1 -1
  197. package/src/vision/describe.ts +5 -5
  198. package/src/web-search/anthropic-executor.ts +1 -1
  199. package/src/web-search/exa-executor.ts +1 -1
  200. package/src/web-search/executor.ts +1 -1
  201. package/src/web-search/gemini-executor.ts +1 -1
  202. package/src/web-search/loop.ts +1 -1
  203. package/src/web-search/ollama-executor.ts +1 -1
  204. package/src/web-search/parse.ts +67 -14
  205. package/src/web-search/passthrough-bridge.ts +64 -31
  206. package/src/web-search/xai-executor.ts +1 -1
@@ -23,6 +23,8 @@ import {
23
23
 
24
24
  const MODEL_DISCOVERY_MAX_FILTER_VALUES = 256;
25
25
  const MODEL_DISCOVERY_MAX_FILTER_STRING_LENGTH = 1_024;
26
+ const TRAILING_SLASHES = /\/+$/;
27
+ const TRAILING_MODELS = /\/models$/;
26
28
 
27
29
  export interface ResolvedProviderModelDiscovery {
28
30
  spec?: ProviderModelDiscoverySpec;
@@ -50,6 +52,20 @@ export type ModelEnvelopeRowsResult =
50
52
  | { ok: true; rows: unknown[] }
51
53
  | { ok: false; reason: "invalid_shape" | "too_many_models" };
52
54
 
55
+ /**
56
+ * Build the default OpenAI-compatible model-discovery URL from a configured baseUrl.
57
+ *
58
+ * `baseUrl` is required on both `OcxProviderConfig` and the persisted-config schema, so a row
59
+ * without one is not a state configuration loading can produce. It is deliberately not tolerated
60
+ * here: the old template-literal join silently produced `"undefined/models"`, which is not a usable
61
+ * fallback either — it only ever survived because a static row returns before the URL is parsed.
62
+ */
63
+ export function providerModelsUrl(baseUrl: string): string {
64
+ const trimmed = baseUrl.trim().replace(TRAILING_SLASHES, "");
65
+ const withoutEndpoint = trimmed.replace(TRAILING_MODELS, "");
66
+ return `${withoutEndpoint}/models`;
67
+ }
68
+
53
69
  function positiveIntegerAtMost(value: number | undefined, hardLimit: number): number {
54
70
  if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return hardLimit;
55
71
  return Math.min(Math.floor(value), hardLimit);
@@ -113,6 +129,14 @@ export function providerModelDiscoverySpecError(spec: ProviderModelDiscoverySpec
113
129
  if (queryEntries.some(([key, value]) => !key.trim() || key.length > 128 || typeof value !== "string" || value.length > 512)) {
114
130
  return "discovery query keys/values exceed their bounds";
115
131
  }
132
+ for (const [field, value] of [
133
+ ["envelopeKey", spec.envelopeKey],
134
+ ["idField", spec.idField],
135
+ ] as const) {
136
+ if (value !== undefined && (
137
+ typeof value !== "string" || !value || value !== value.trim() || value.length > 128
138
+ )) return `${field} must be a nonblank field name up to 128 characters`;
139
+ }
116
140
  for (const [field, value, hardLimit] of [
117
141
  ["maxResponseBytes", spec.maxResponseBytes, MODEL_DISCOVERY_MAX_RESPONSE_BYTES],
118
142
  ["maxModels", spec.maxModels, MODEL_DISCOVERY_MAX_MODELS],
@@ -406,7 +430,7 @@ export function extractModelEnvelopeRows(
406
430
  return { ok: true, rows };
407
431
  }
408
432
 
409
- /** Validate, bound, deduplicate, and declaratively filter OpenAI `{data:[...]}` or top-level arrays (Together `#617`). */
433
+ /** Validate, bound, deduplicate, and filter the declared envelope or a top-level array (Together `#617`). */
410
434
  /**
411
435
  * Metadata a sibling `models[]` array may contribute to an ALREADY-ADMITTED
412
436
  * `data[]` row (#1797).
@@ -485,24 +509,26 @@ export function extractProviderModelItems(
485
509
  let data: unknown[];
486
510
  let siblings: SiblingIndex | null = null;
487
511
  if (Array.isArray(value)) {
488
- // Together-style top-level /models arrays. Catalog discovery must not treat a stray
489
- // `models` key on openai-chat responses as valid — only `data` envelopes or top-level arrays.
512
+ // Together-style top-level /models arrays. The default contract must not treat a stray
513
+ // `models` key on openai-chat responses as valid; only a provider spec may opt into it.
490
514
  if (value.length > limit) return { ok: false, reason: "too_many_models" };
491
515
  data = value;
492
516
  } else {
493
- const envelope = extractModelEnvelopeRows(value, discovery.maxModels, ["data"]);
517
+ const envelopeKey = discovery.spec?.envelopeKey ?? "data";
518
+ const envelope = extractModelEnvelopeRows(value, discovery.maxModels, [envelopeKey]);
494
519
  if (!envelope.ok) return envelope;
495
520
  data = envelope.rows;
496
- siblings = buildSiblingIndex(value, limit);
521
+ siblings = envelopeKey === "data" ? buildSiblingIndex(value, limit) : null;
497
522
  }
498
523
 
499
524
  const items: ProviderModelsApiItem[] = [];
500
525
  const seen = new Set<string>();
526
+ const idField = discovery.spec?.idField ?? "id";
501
527
  for (const raw of data) {
502
528
  if (raw === null || typeof raw !== "object" || Array.isArray(raw)) {
503
529
  return { ok: false, reason: "invalid_shape" };
504
530
  }
505
- const id = (raw as { id?: unknown }).id;
531
+ const id = (raw as Record<string, unknown>)[idField];
506
532
  if (!isValidModelDiscoveryModelId(id)) return { ok: false, reason: "invalid_shape" };
507
533
  const prefix = discovery.spec?.stripIdPrefix;
508
534
  let finalId = id;
@@ -510,7 +536,9 @@ export function extractProviderModelItems(
510
536
  finalId = finalId.slice(prefix.length);
511
537
  if (!isValidModelDiscoveryModelId(finalId)) continue;
512
538
  }
513
- const item = finalId === id ? raw as ProviderModelsApiItem : { ...(raw as ProviderModelsApiItem), id: finalId };
539
+ const item = finalId === id && idField === "id"
540
+ ? raw as ProviderModelsApiItem
541
+ : { ...(raw as Record<string, unknown>), id: finalId };
514
542
  // Admission is decided on the ORIGINAL `data[]` row, before any sibling
515
543
  // enrichment. Merging first let a `models[]` entry supply the very field a
516
544
  // provider filter requires — reproduced against the real Chutes policy,
@@ -17,6 +17,7 @@ import type { ProviderRegistryEntry } from "./types";
17
17
  import {
18
18
  ANTHROPIC_MODELS,
19
19
  ANTHROPIC_MODEL_CONTEXT_WINDOWS,
20
+ ANTHROPIC_MODEL_INPUT_MODALITIES,
20
21
  ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
21
22
  ANTHROPIC_MODEL_REASONING_EFFORTS,
22
23
  ZAI_GLM_52_REASONING_EFFORTS,
@@ -281,6 +282,17 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
281
282
  forwardCallerServiceTier: false,
282
283
  },
283
284
  },
285
+ // Grok 4.6/4.5 OAuth Responses replays Codex tool history. After a mid-stream 502/reset,
286
+ // the client can resend a function_call without a matching output, or with hook-injected
287
+ // developer context between the pair. Google already synthesizes a missing tool_result
288
+ // (#2199). xAI's Responses parser does not, so the next turns 400 and the thread snowballs.
289
+ // Reuse the existing adjacency capability (Kimi #4726, DeepSeek #1292). Do not set
290
+ // statelessResponses: xAI stores responses for 30 days and documents previous_response_id.
291
+ // https://docs.x.ai/developers/model-capabilities/text/comparison
292
+ requiresAdjacentResponsesToolResults: true,
293
+ // The dangling half of the same failure: a call whose output never arrived. Kimi accepts that
294
+ // shape, so this is a second capability rather than a widening of the one above.
295
+ requiresPairedResponsesToolResults: true,
284
296
  // Vision lineup per docs.x.ai model-capabilities/images/understanding: the grok-4.x chat
285
297
  // models accept image input (JPEG/PNG, URL or base64). Without this the catalog leaves
286
298
  // inputModalities undefined, and deriveComboCatalogModel defaults an undefined member to
@@ -383,6 +395,7 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
383
395
  note: "Log in with your Claude account",
384
396
  models: [...ANTHROPIC_MODELS],
385
397
  modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
398
+ modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
386
399
  modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
387
400
  // Codex omits max_output_tokens; without a provider budget the Anthropic adapter
388
401
  // falls back to 8192, which truncates long answers with stop_reason=max_tokens.
@@ -403,6 +416,7 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
403
416
  models: [...ANTHROPIC_MODELS],
404
417
  liveModels: true,
405
418
  modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
419
+ modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
406
420
  modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
407
421
  defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
408
422
  defaultModel: "claude-sonnet-5",
@@ -420,6 +434,10 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
420
434
  // or the one the Claude /v1/messages inbound derives); the adapter itself never invents one.
421
435
  // Evidence: https://platform.kimi.com/docs/api/chat
422
436
  promptCacheKey: true,
437
+ // Kimi's Responses endpoint rejects hook-provided context between a tool call and
438
+ // its matching result (#4726), the same strict shape DeepSeek exposed in #1292.
439
+ // The flag is inert while this preset uses the Chat wire.
440
+ requiresAdjacentResponsesToolResults: true,
423
441
  featured: true,
424
442
  oauthId: "kimi",
425
443
  jawcodeBundle: "moonshot",
@@ -56,6 +56,11 @@ import {
56
56
  ALIBABA_TOKEN_PLAN_MODELS,
57
57
  ALIBABA_TOKEN_PLAN_QWEN_MODELS,
58
58
  ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
59
+ ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
60
+ ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
61
+ ALIBABA_TOKEN_PLAN_NO_VISION,
62
+ ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
63
+ QWEN38_FAMILY,
59
64
  ALIBABA_INTL_TOKEN_PLAN_MODELS,
60
65
  ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS,
61
66
  TENCENT_CODING_PLAN_MODELS,
@@ -110,6 +115,13 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
110
115
  // Baseten says models outside its reasoning table do not support reasoning. Keep
111
116
  // unknown/new live slugs conservative until an official-docs registry refresh proves it.
112
117
  reasoningEfforts: [],
118
+ // `text.verbosity` is an OpenAI Responses parameter. Baseten documents its Model
119
+ // APIs as Chat Completions compatible, so there is nothing on that wire for it to
120
+ // become, and a routed row must not inherit the Codex template's verbosity picker
121
+ // (#4630: Codex sent `text: { verbosity: "low" }` and the turn 400'd before any
122
+ // model output). Provider-wide rather than per-model because this catalog is live-
123
+ // discovered: a slug that arrives tomorrow supports it no more than the seeded ones.
124
+ supportsVerbosity: false,
113
125
  modelReasoningEfforts: BASETEN_MODEL_REASONING_EFFORTS,
114
126
  modelReasoningEffortMap: BASETEN_MODEL_REASONING_EFFORT_MAP,
115
127
  modelDefaultReasoningEfforts: BASETEN_MODEL_DEFAULT_REASONING_EFFORTS,
@@ -420,6 +432,7 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
420
432
  // model_access_denied, which is why the Chat path cannot simply hang off the new base.
421
433
  responsesPath: "/api/v1/responses",
422
434
  chatCompletionsPath: "/api/coding/paas/v4/chat/completions",
435
+ modelDiscovery: { path: "/api/v1/models", envelopeKey: "models", idField: "slug" },
423
436
  // The address this row occupied before the move. A saved custom provider still pointing
424
437
  // at the Chat endpoint keeps receiving this row's metadata (#1100).
425
438
  destinationAliases: [{ baseUrl: "https://api.z.ai/api/coding/paas/v4", adapter: "openai-chat" }],
@@ -717,22 +730,35 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
717
730
  liveModels: false,
718
731
  note: "Token Plan Personal Edition · China (Beijing)",
719
732
  modelInputModalities: ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
720
- modelContextWindows: {
721
- "qwen3.8-max": 983_616, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
722
- "qwen3.6-flash": 1_000_000, "glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000,
723
- },
733
+ modelContextWindows: ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
734
+ modelMaxOutputTokens: ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
724
735
  modelReasoningEfforts: {
725
736
  ...Object.fromEntries(ALIBABA_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
726
- "qwen3.8-max": QWEN38_REASONING_EFFORTS,
727
- "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
728
- "glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
737
+ ...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
729
738
  "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
739
+ "deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
740
+ "deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
741
+ "deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
742
+ "deepseek-v4.1-flash": deepseekThinkingEffortsFor("deepseek-v4.1-flash"),
743
+ },
744
+ modelReasoningEffortMap: {
745
+ "deepseek-v4-pro": deepseekReasoningMapFor("deepseek-v4-pro"),
746
+ "deepseek-v4-pro-0813": deepseekReasoningMapFor("deepseek-v4-pro-0813"),
747
+ "deepseek-v4-flash-0731": deepseekReasoningMapFor("deepseek-v4-flash-0731"),
748
+ "deepseek-v4.1-flash": deepseekReasoningMapFor("deepseek-v4.1-flash"),
730
749
  },
731
- modelDefaultReasoningEfforts: { "qwen3.8-max": "xhigh" },
732
- directReasoningEffortModels: ["qwen3.8-max"],
733
- thinkingBudgetModels: ALIBABA_TOKEN_PLAN_QWEN_MODELS.filter(id => id !== "qwen3.8-max"),
734
- preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash"],
735
- noVisionModels: ["glm-5.3", "glm-5.2"],
750
+ // Probed 260915 on the plan gateway: json_object returns valid JSON, strict
751
+ // json_schema is rejected 400 ("This response_format type is unavailable now")
752
+ // in both thinking modes, so requests downgrade to json_object rather than
753
+ // sending a schema the gateway refuses.
754
+ noJsonSchemaModels: ["deepseek-v4.1-flash"],
755
+ modelDefaultReasoningEfforts: Object.fromEntries(QWEN38_FAMILY.map(id => [id, "xhigh"])),
756
+ directReasoningEffortModels: QWEN38_FAMILY,
757
+ thinkingBudgetModels: ALIBABA_TOKEN_PLAN_QWEN_MODELS.filter(id => !QWEN38_FAMILY.includes(id)),
758
+ preserveReasoningContentModels: ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
759
+ noVisionModels: ALIBABA_TOKEN_PLAN_NO_VISION,
760
+ // The gateway accepts prompt_cache_key on every Token Plan chat model (probed 260902).
761
+ promptCacheKey: true,
736
762
  },
737
763
  {
738
764
  id: "alibaba-token-plan-intl",
@@ -749,31 +775,34 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
749
775
  note: "Token Plan Team Edition · Singapore (ap-southeast-1)",
750
776
  metadataModelIdNormalize: "case-insensitive",
751
777
  modelInputModalities: ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES,
752
- modelContextWindows: {
753
- "qwen3.8-max": 983_616,
754
- "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000, "qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
755
- "deepseek-v4-flash": 1_000_000, "deepseek-v3.2": 131_072,
756
- "kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
757
- "glm-5.3": 1_000_000, "glm-5.3-flash": 1_000_000, "glm-5.2": 1_000_000, "glm-5.1": 1_000_000, "glm-5": 1_000_000,
758
- "MiniMax-M2.5": 204_800,
759
- },
778
+ modelContextWindows: ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS,
779
+ modelMaxOutputTokens: ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS,
760
780
  modelReasoningEfforts: {
761
781
  ...Object.fromEntries(ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
762
- "qwen3.8-max": QWEN38_REASONING_EFFORTS,
763
- "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
764
- "glm-5.3-flash": ZAI_GLM_53_REASONING_EFFORTS,
782
+ ...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
765
783
  "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
784
+ "deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
785
+ "deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
766
786
  "deepseek-v4-flash": deepseekThinkingEffortsFor("deepseek-v4-flash"),
787
+ "deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
788
+ "deepseek-v4.1-flash": deepseekThinkingEffortsFor("deepseek-v4.1-flash"),
767
789
  },
768
790
  modelReasoningEffortMap: {
791
+ "deepseek-v4-pro": deepseekReasoningMapFor("deepseek-v4-pro"),
792
+ "deepseek-v4-pro-0813": deepseekReasoningMapFor("deepseek-v4-pro-0813"),
769
793
  "deepseek-v4-flash": deepseekReasoningMapFor("deepseek-v4-flash"),
794
+ "deepseek-v4-flash-0731": deepseekReasoningMapFor("deepseek-v4-flash-0731"),
795
+ "deepseek-v4.1-flash": deepseekReasoningMapFor("deepseek-v4.1-flash"),
770
796
  },
771
- directReasoningEffortModels: ["qwen3.8-max"],
772
- thinkingBudgetModels: ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.filter(id => id !== "qwen3.8-max"),
773
- preserveReasoningContentModels: ["glm-5.3", "glm-5.3-flash", "glm-5.2", "deepseek-v4-flash", "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash"],
774
- noVisionModels: ["deepseek-v4-flash", "deepseek-v3.2", "glm-5.3", "glm-5.2", "glm-5.1", "glm-5", "MiniMax-M2.5"],
797
+ // Same 260915 json_schema rejection probe as the Beijing entry.
798
+ noJsonSchemaModels: ["deepseek-v4.1-flash"],
799
+ directReasoningEffortModels: QWEN38_FAMILY,
800
+ thinkingBudgetModels: ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.filter(id => !QWEN38_FAMILY.includes(id)),
801
+ preserveReasoningContentModels: ALIBABA_TOKEN_PLAN_PRESERVE_REASONING,
802
+ noVisionModels: ALIBABA_TOKEN_PLAN_NO_VISION,
775
803
  noReasoningModels: ["kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "deepseek-v3.2", "glm-5.1", "glm-5", "MiniMax-M2.5"],
776
- modelDefaultReasoningEfforts: { "qwen3.8-max": "xhigh" },
804
+ modelDefaultReasoningEfforts: Object.fromEntries(QWEN38_FAMILY.map(id => [id, "xhigh"])),
805
+ promptCacheKey: true,
777
806
  },
778
807
  // NEEDS_HUMAN 2026-07-10: kept for config compatibility, but this is a dashboard URL,
779
808
  // no /models endpoint is documented, and tools are silently ignored upstream per docs.parallel.ai.
@@ -883,6 +912,8 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
883
912
  modelSuffixBracketStrip: true,
884
913
  // API-key form of the same Kimi Code Plan transport; keep cache affinity identical to OAuth.
885
914
  promptCacheKey: true,
915
+ // Keep Responses tool-result adjacency aligned with the OAuth preset (#4726).
916
+ requiresAdjacentResponsesToolResults: true,
886
917
  models: KIMI_CODING_MODELS,
887
918
  modelContextWindows: KIMI_CODING_MODEL_CONTEXT_WINDOWS,
888
919
  modelInputModalities: KIMI_CODING_MODEL_INPUT_MODALITIES,
@@ -8,6 +8,10 @@ import type { ProviderModelDiscoverySpec } from "./types";
8
8
  // always on, per the official models overview and pricing page (platform.claude.com).
9
9
  export const ANTHROPIC_MODELS = ["claude-fable-5-1", "claude-fable-5", "claude-sonnet-5", "claude-opus-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-sonnet-4-6", "claude-haiku-4-5"];
10
10
  export const ANTHROPIC_MODEL_CONTEXT_WINDOWS: Record<string, number> = { "claude-fable-5-1": 1_000_000, "claude-sonnet-5": 1_000_000, "claude-fable-5": 1_000_000, "claude-opus-5": 1_000_000, "claude-opus-4-8": 1_000_000, "claude-opus-4-7": 1_000_000, "claude-opus-4-6": 1_000_000, "claude-sonnet-4-6": 1_000_000, "claude-haiku-4-5": 200_000 };
11
+ // All seeded Claude models support vision: https://platform.claude.com/docs/en/models/overview
12
+ export const ANTHROPIC_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
13
+ ANTHROPIC_MODELS.map(id => [id, ["text", "image"]]),
14
+ );
11
15
  // Every current Claude family accepts at least 64k output tokens (Haiku 4.5 / Sonnet 4.x
12
16
  // through Opus 5 and Fable 5). Anthropic caps max_tokens per model server-side, so a
13
17
  // larger request never over-allocates; it only stops the 8192 truncation.
@@ -433,20 +437,42 @@ export const deepseekReasoningMapFor = (modelId: string): Record<string, string>
433
437
  // Coding Plan: the products use different exact allowlists and different base URLs.
434
438
  // Evidence: https://help.aliyun.com/en/model-studio/token-plan-personal-overview
435
439
  // https://help.aliyun.com/en/model-studio/token-plan-quickstart
440
+ // 260909 refresh, re-probed against the live gateway (both regions, both tiers):
441
+ // https://github.com/oliver-mee/alibaba-token-plan-wiki (machine-readable catalog).
442
+ // glm-5.3 / glm-5.3-flash removed from both Token Plan catalogs: they exist on Z.AI
443
+ // endpoints but the Token Plan gateway has never served either id (the 260826 seed
444
+ // propagated them across every GLM-carrying catalog; a selected row 404s).
445
+ // The Beijing preset keeps the Personal Edition subset; non-chat ids (audio/image/
446
+ // video families) stay out: they answer only on async endpoints openai-chat cannot
447
+ // reach. deepseek-v4-pro-0813 is callable but NOT listed by /models, which is the
448
+ // reason liveModels must stay false for this provider. deepseek-v4.1-flash is the
449
+ // 260910 DeepSeek rename row: listed on /models on both tiers and regions from 260915,
450
+ // hybrid thinking, vision via user message and tool result, json_object but not
451
+ // json_schema (see noJsonSchemaModels on the entries).
452
+ // Beijing serves the Personal Edition, so this is the Personal-tier roster probed
453
+ // 260909 (a strict subset of Team). deepseek-v4-pro-0813 stays out of the Beijing
454
+ // entry: its callability is only proven on Team keys, and no Personal key has been
455
+ // shown to reach it. The Beijing entry also shares the intl maps, so it carries a
456
+ // few orphan keys (kimi/glm-5/MiniMax rows); harmless, and one map beats two
457
+ // drifting ones.
436
458
  export const ALIBABA_TOKEN_PLAN_MODELS = [
437
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
438
- "glm-5.3", "glm-5.3-flash", "glm-5.2",
459
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
460
+ "deepseek-v4-pro", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "glm-5.2",
439
461
  ];
440
462
  export const ALIBABA_TOKEN_PLAN_QWEN_MODELS = [
441
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
463
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
442
464
  ];
443
465
  export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
444
466
  "qwen3.8-max": ["text", "image"],
445
- "qwen3.7-max": ["text", "image"],
467
+ "qwen3.8-flash": ["text", "image"],
468
+ "qwen3.7-max": ["text"],
446
469
  "qwen3.7-plus": ["text", "image"],
447
470
  "qwen3.6-flash": ["text", "image"],
448
- "glm-5.3": ["text"],
449
- "glm-5.3-flash": ["text", "image"],
471
+ "deepseek-v4-pro": ["text"],
472
+ "deepseek-v4-pro-0813": ["text"],
473
+ "deepseek-v4-flash-0731": ["text"],
474
+ // Vision probed on the plan gateway 260915 (user message and tool result, both 200).
475
+ "deepseek-v4.1-flash": ["text", "image"],
450
476
  "glm-5.2": ["text"],
451
477
  };
452
478
 
@@ -454,15 +480,18 @@ export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
454
480
  // Multi-vendor lineup distinct from Beijing — includes DeepSeek V4 flash, Kimi K2.7, MiniMax.
455
481
  // Evidence: https://www.alibabacloud.com/help/en/model-studio/token-plan-overview
456
482
  // https://qwencloud.com/pricing/token-plan (qwen3.8 metadata)
483
+ // The Team Edition roster (Singapore), verified identical to the CN Team set on 260909.
484
+ // deepseek-v4-pro is restored: it remains callable on the plan gateway (probed 260909,
485
+ // listed on /models on both regions) after being dropped as "retired" upstream.
457
486
  export const ALIBABA_INTL_TOKEN_PLAN_MODELS = [
458
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
459
- "deepseek-v4-flash", "deepseek-v3.2",
487
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
488
+ "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "deepseek-v3.2",
460
489
  "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5",
461
- "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5",
490
+ "glm-5.2", "glm-5.1", "glm-5",
462
491
  "MiniMax-M2.5",
463
492
  ];
464
493
  export const ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS = [
465
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
494
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
466
495
  ];
467
496
 
468
497
  // 260722 Tencent Cloud Coding Plan. The plan's model set is explicitly dynamic; these are the
@@ -539,24 +568,49 @@ export const VOLCENGINE_PLAN_TEXT_ONLY_MODELS = [
539
568
  "doubao-seed-2.0-pro",
540
569
  ];
541
570
  export const ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
542
- "qwen3.8-max": ["text", "image"],
543
- "qwen3.7-max": ["text", "image"],
544
- "qwen3.7-plus": ["text", "image"],
571
+ ...ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
545
572
  "qwen3.6-plus": ["text", "image"],
546
- "qwen3.6-flash": ["text", "image"],
547
573
  "deepseek-v4-flash": ["text"],
548
574
  "deepseek-v3.2": ["text"],
549
575
  "kimi-k2.7-code": ["text", "image"],
550
576
  "kimi-k2.6": ["text", "image"],
551
577
  "kimi-k2.5": ["text", "image"],
552
- "glm-5.3": ["text"],
553
- "glm-5.3-flash": ["text", "image"],
554
- "glm-5.2": ["text"],
555
578
  "glm-5.1": ["text"],
556
579
  "glm-5": ["text"],
557
580
  "MiniMax-M2.5": ["text"],
558
581
  };
559
582
 
583
+ // Shared Token Plan metadata (260909 gateway probes; output ceilings are max_tokens
584
+ // boundary probes: accept at N, reject at N+1).
585
+ export const QWEN38_FAMILY = ["qwen3.8-max", "qwen3.8-flash"];
586
+ export const ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS: Record<string, number> = {
587
+ "qwen3.8-max": 1_000_000, "qwen3.8-flash": 1_000_000, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
588
+ "qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
589
+ "deepseek-v4-pro": 1_000_000, "deepseek-v4-pro-0813": 1_000_000, "deepseek-v4-flash": 1_000_000,
590
+ "deepseek-v4-flash-0731": 1_000_000, "deepseek-v4.1-flash": 1_000_000, "deepseek-v3.2": 131_072,
591
+ "kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
592
+ "glm-5.2": 1_000_000, "glm-5.1": 202_752, "glm-5": 202_752,
593
+ "MiniMax-M2.5": 196_608,
594
+ };
595
+ export const ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS: Record<string, number> = {
596
+ "qwen3.8-max": 131_072, "qwen3.8-flash": 131_072, "qwen3.7-max": 131_072, "qwen3.7-plus": 131_072,
597
+ "qwen3.6-plus": 65_536, "qwen3.6-flash": 65_536,
598
+ "deepseek-v4-pro": 393_216, "deepseek-v4-pro-0813": 393_216, "deepseek-v4-flash": 393_216,
599
+ "deepseek-v4-flash-0731": 393_216, "deepseek-v4.1-flash": 393_216, "deepseek-v3.2": 65_536,
600
+ "kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 98_304,
601
+ "glm-5.2": 131_072, "glm-5.1": 128_000, "glm-5": 16_384,
602
+ "MiniMax-M2.5": 32_768,
603
+ };
604
+ export const ALIBABA_TOKEN_PLAN_NO_VISION = [
605
+ "qwen3.7-max", "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash",
606
+ "deepseek-v4-flash-0731", "deepseek-v3.2", "glm-5.2", "glm-5.1", "glm-5", "MiniMax-M2.5",
607
+ ];
608
+ export const ALIBABA_TOKEN_PLAN_PRESERVE_REASONING = [
609
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
610
+ "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731",
611
+ "deepseek-v4.1-flash", "glm-5.2",
612
+ ];
613
+
560
614
  // 260717 Kimi K3: the subscription endpoint uses one upstream id (`k3`) for both
561
615
  // entitlement tiers. Bare `k3` advertises the Moderato 256K ceiling; the local `[1m]`
562
616
  // alias advertises Allegretto's 1M ceiling and is stripped before the upstream request.
@@ -64,6 +64,10 @@ export interface ProviderModelDiscoveryFilter {
64
64
  interface ProviderModelDiscoverySharedSpec {
65
65
  /** Query parameters applied to the resolved discovery URL. */
66
66
  query?: Readonly<Record<string, string>>;
67
+ /** Top-level response key containing model rows; defaults to `data`. */
68
+ envelopeKey?: string;
69
+ /** Model-row field containing the provider-native identifier; defaults to `id`. */
70
+ idField?: string;
67
71
  /** Declarative eligibility rules evaluated against each untrusted model row. */
68
72
  filter?: ProviderModelDiscoveryFilter;
69
73
  /** Optional lower byte ceiling; the process-wide hard ceiling still wins. */
@@ -217,6 +221,11 @@ export interface ProviderRegistryEntry {
217
221
  * to stay contiguous. This is seeded/backfilled like other fixed wire capabilities.
218
222
  */
219
223
  requiresAdjacentResponsesToolResults?: boolean;
224
+ /**
225
+ * Responses upstream that also rejects a tool call with no matching output anywhere in the
226
+ * replayed input. Seeded/backfilled like other fixed wire capabilities.
227
+ */
228
+ requiresPairedResponsesToolResults?: boolean;
220
229
  /**
221
230
  * When enabled, tool results that are present but empty are annotated on the wire.
222
231
  * Seeded/backfilled like other fixed wire capabilities.
@@ -28,9 +28,12 @@ export interface ReasoningEnvelope {
28
28
  */
29
29
  txt?: string;
30
30
  /**
31
- * Kiro `reasoningContentEvent.redactedContent`: a KMS-encrypted reasoning blob that is opaque to
32
- * the proxy. Kiro's own CLI replays it on the matching `assistantResponseMessage` to preserve
33
- * model reasoning across turns, so it round-trips here the same way a signature does.
31
+ * Kiro's reasoning blob from `reasoningContentEvent`: a KMS-encrypted value that is opaque to the
32
+ * proxy (the GPT-5.6 family sends it as `signature`, other models as the base64
33
+ * `redactedContent`, and the value carries a tag naming which one — see
34
+ * src/adapters/kiro/reasoning.ts). Kiro's own CLI replays it on the matching
35
+ * `assistantResponseMessage` to preserve model reasoning across turns, so it round-trips here the
36
+ * same way a signature does.
34
37
  */
35
38
  krc?: string;
36
39
  }
@@ -159,6 +159,23 @@ function spillNow(): number {
159
159
  return spillNowOverride?.() ?? Date.now();
160
160
  }
161
161
 
162
+ /**
163
+ * The spill deadline clock, shared with the shutdown drain in `state/spill-queue.ts`.
164
+ *
165
+ * Every deadline the shutdown path enforces has to read the same clock the work it
166
+ * budgets reads. When the drain measured its reserve on `Date.now()` while the ACL
167
+ * harden it was budgeting ran on this injected clock, a test could freeze the clock,
168
+ * believe it had removed wall time from the case, and still lose an 80 ms reserve to
169
+ * real elapsed time on a loaded runner — which is what turned
170
+ * `shutdown fallback prices the job-owned superseded generation before publishing`
171
+ * red on macOS 2/2 in run 35137850114 while the assertion it was written for never ran.
172
+ *
173
+ * Production is unchanged: with no override installed this is `Date.now()`.
174
+ */
175
+ export function responseSpillNow(): number {
176
+ return spillNow();
177
+ }
178
+
162
179
  function record(event: "write" | "fsync" | "close" | "harden" | "publish" | "dir-fsync" | "stub-swap"): void {
163
180
  spillIoForTest?.record?.(event);
164
181
  }
@@ -0,0 +1,25 @@
1
+ /**
2
+ * Request bodies that must never enter the continuation cache.
3
+ *
4
+ * The cache is persisted to `responses-state.json`, so anything recorded here reaches disk.
5
+ * Encrypted-agent-task recovery decrypts task text into the request body and promises
6
+ * in-memory, TTL-bounded retention; recording that body would put the plaintext on disk with
7
+ * no TTL and break the promise.
8
+ *
9
+ * A WeakSet rather than a body field on purpose: `_rawBody` is serialized verbatim by the
10
+ * native passthrough, so any marker written into the body itself would be sent upstream.
11
+ * Marking is enforced once here rather than at each call site, because every recording path
12
+ * (streaming, non-streaming, passthrough, forced) funnels through `rememberResponseState` —
13
+ * a new call site cannot reintroduce the leak by forgetting a guard.
14
+ */
15
+ const nonPersistableBodies = new WeakSet<object>();
16
+
17
+ /** Bar this exact request body from the continuation cache, and therefore from disk. */
18
+ export function markBodyNonPersistable(body: unknown): void {
19
+ if (body && typeof body === "object") nonPersistableBodies.add(body as object);
20
+ }
21
+
22
+ /** Test the body's in-memory persistence restriction without adding a wire marker. */
23
+ export function isBodyNonPersistable(body: unknown): boolean {
24
+ return !!body && typeof body === "object" && nonPersistableBodies.has(body);
25
+ }
@@ -7,6 +7,7 @@ import {
7
7
  MAX_RESPONSE_SPILL_PAYLOAD_BYTES,
8
8
  prospectiveResponseSpillBytes,
9
9
  responseSpillPayloadCap,
10
+ responseSpillNow,
10
11
  type ResponseSpillPublicationControl,
11
12
  type ResponseSpillRef,
12
13
  writeResponseSpillDurably,
@@ -377,7 +378,7 @@ function responseSpillShutdownBudget(): { totalMs: number; fallbackReserveMs: nu
377
378
  }
378
379
 
379
380
  function awaitResponseSpillTailUntil(observed: Promise<void>, deadline: number): Promise<boolean> {
380
- const remaining = deadline - Date.now();
381
+ const remaining = deadline - responseSpillNow();
381
382
  if (remaining <= 0) return Promise.resolve(false);
382
383
  return new Promise(resolve => {
383
384
  let finished = false;
@@ -554,12 +555,13 @@ function terminalizeExhaustedShutdownFallback(
554
555
  }
555
556
 
556
557
  function fallbackPendingResponseSpills(reserveMs: number): Error[] {
557
- const deadline = Date.now() + reserveMs;
558
+ // Same clock as the harden work this reserve is budgeting — see `responseSpillNow`.
559
+ const deadline = responseSpillNow() + reserveMs;
558
560
  const failures: Error[] = [];
559
561
  for (;;) {
560
562
  const pending = pendingShutdownFallbackCandidates();
561
563
  if (pending.length === 0) return failures;
562
- if (Date.now() >= deadline) {
564
+ if (responseSpillNow() >= deadline) {
563
565
  terminalizeExhaustedShutdownFallback(pending, failures);
564
566
  return failures;
565
567
  }
@@ -569,7 +571,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
569
571
  for (let index = 0; index < pending.length; index += 1) {
570
572
  const { job, candidate } = pending[index]!;
571
573
  if (requireStore().currentEntry(job.id) !== candidate) continue;
572
- const remaining = deadline - Date.now();
574
+ const remaining = deadline - responseSpillNow();
573
575
  if (remaining <= 0) {
574
576
  reserveExhausted = true;
575
577
  for (const exhausted of pending.slice(index)) {
@@ -587,7 +589,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
587
589
  requireStore().recomputeOldestResident();
588
590
  requireStore().pruneResponses();
589
591
  enforceAppOwnedMemoryBudget();
590
- if (reserveExhausted || Date.now() >= deadline) {
592
+ if (reserveExhausted || responseSpillNow() >= deadline) {
591
593
  terminalizeExhaustedShutdownFallback(pendingShutdownFallbackCandidates(), failures);
592
594
  return failures;
593
595
  }
@@ -597,7 +599,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
597
599
  export async function drainResponseSpillPublications(): Promise<void> {
598
600
  const budget = responseSpillShutdownBudget();
599
601
  const fallbackReserveMs = Math.min(budget.totalMs, Math.max(1, budget.fallbackReserveMs));
600
- const drainDeadline = Date.now() + Math.max(0, budget.totalMs - fallbackReserveMs);
602
+ const drainDeadline = responseSpillNow() + Math.max(0, budget.totalMs - fallbackReserveMs);
601
603
 
602
604
  for (;;) {
603
605
  if (pendingResponseSpills.size === 0) return;
@@ -23,6 +23,8 @@ import type { ResponseSpillWriteFailureCode, ResponseSpillWriteStatus, ResponseS
23
23
  export { responseAdmissionCountersForTests } from "./state/spill-failure";
24
24
  import { admissionCounters, noteSpillWriteFailure, noteSpillWriteSuccess, spillCounters, spillWriteHealth } from "./state/spill-failure";
25
25
  import { loadSnapshotEntry } from "./state/snapshot-codec";
26
+ import { isBodyNonPersistable } from "./state/body-policy";
27
+ export { isBodyNonPersistable, markBodyNonPersistable } from "./state/body-policy";
26
28
  export { flushPendingResponseSpillsForTests, awaitResponseSpillPublicationTailForTests, pendingResponseSpillMetricsForTests, setResponseSpillShutdownBudgetForTests, setResponseSpillAsyncAclAttemptBudgetForTests, setResponseSpillShutdownTerminalizationPassLimitForTests } from "./state/spill-queue";
27
29
  import {
28
30
  bindSpillQueueStore,
@@ -1235,27 +1237,6 @@ export function responseStateMetrics(): ResponseStateMetrics {
1235
1237
  * Cache completed output and max_output_tokens partial output for previous_response_id replay.
1236
1238
  * Content-filtered incomplete and failed output are not authoritative replay history.
1237
1239
  */
1238
- /**
1239
- * Request bodies that must never enter the continuation cache.
1240
- *
1241
- * The cache is persisted to `responses-state.json`, so anything recorded here reaches disk.
1242
- * Encrypted-agent-task recovery decrypts task text into the request body and promises
1243
- * in-memory, TTL-bounded retention; recording that body would put the plaintext on disk with
1244
- * no TTL and break the promise.
1245
- *
1246
- * A WeakSet rather than a body field on purpose: `_rawBody` is serialized verbatim by the
1247
- * native passthrough, so any marker written into the body itself would be sent upstream.
1248
- * Marking is enforced once here rather than at each call site, because every recording path
1249
- * (streaming, non-streaming, passthrough, forced) funnels through `rememberResponseState` —
1250
- * a new call site cannot reintroduce the leak by forgetting a guard.
1251
- */
1252
- const nonPersistableBodies = new WeakSet<object>();
1253
-
1254
- /** Bar this exact request body from the continuation cache, and therefore from disk. */
1255
- export function markBodyNonPersistable(body: unknown): void {
1256
- if (body && typeof body === "object") nonPersistableBodies.add(body as object);
1257
- }
1258
-
1259
1240
  export function rememberResponseState(
1260
1241
  requestBody: unknown,
1261
1242
  response: { id?: unknown; output?: unknown; status?: unknown; incomplete_details?: unknown },
@@ -1264,7 +1245,7 @@ export function rememberResponseState(
1264
1245
  ): void {
1265
1246
  if (!requestBody || typeof requestBody !== "object" || Array.isArray(requestBody)) return;
1266
1247
  const request = requestBody as Record<string, unknown>;
1267
- if (nonPersistableBodies.has(request)) return;
1248
+ if (isBodyNonPersistable(request)) return;
1268
1249
  // `force` bypasses only the store:false skip: Codex sends `store:false` on every non-Azure
1269
1250
  // HTTP request (and WS inherits it), yet its WS turns still chain with previous_response_id.
1270
1251
  // The passthrough branch records with force so those chains can be expanded locally; the