@bitkyc08/opencodex 2.57.0 → 2.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/gui/dist/assets/{index-Cz7CLdif.js → index-BbrHOIY0.js} +2 -2
  2. package/gui/dist/index.html +1 -1
  3. package/package.json +2 -2
  4. package/src/adapters/codebuddy/scaffold-guard.ts +5 -4
  5. package/src/adapters/command-code.ts +13 -4
  6. package/src/adapters/cursor/cursor-errors.ts +15 -0
  7. package/src/adapters/cursor/discovery.ts +65 -1
  8. package/src/adapters/cursor/live-transport.ts +5 -1
  9. package/src/adapters/cursor/protobuf-events.ts +110 -11
  10. package/src/adapters/cursor/protobuf-request.ts +19 -1
  11. package/src/adapters/cursor/text-toolcall.ts +230 -0
  12. package/src/adapters/cursor/thread-continuity.ts +67 -0
  13. package/src/adapters/cursor/types.ts +5 -0
  14. package/src/adapters/cursor.ts +55 -5
  15. package/src/adapters/google-http.ts +38 -13
  16. package/src/adapters/mimo-free.ts +32 -17
  17. package/src/adapters/ollama-native.ts +42 -8
  18. package/src/adapters/openai-responses/passthrough.ts +30 -4
  19. package/src/adapters/openai-responses/request-strips.ts +43 -0
  20. package/src/adapters/physical-send.ts +50 -0
  21. package/src/bridge/response-json.ts +1 -1
  22. package/src/bridge/sse.ts +1 -1
  23. package/src/claude/outbound.ts +14 -4
  24. package/src/cli/config-command.ts +35 -18
  25. package/src/cli/dispatch.ts +17 -4
  26. package/src/cli/index.ts +44 -2
  27. package/src/cli/system-command.ts +70 -1
  28. package/src/cli/uninstall-client-state.ts +12 -0
  29. package/src/codex/auth-context.ts +42 -8
  30. package/src/codex/desktop-switches.ts +145 -0
  31. package/src/codex/history-job.ts +5 -1
  32. package/src/codex/history-provider.ts +33 -4
  33. package/src/codex/history-worker.ts +14 -1
  34. package/src/codex/inject/remove.ts +145 -7
  35. package/src/codex/inject/restore.ts +204 -32
  36. package/src/codex/inject.ts +3 -7
  37. package/src/codex/loopback-target.ts +9 -0
  38. package/src/codex/native-profile-startup.ts +64 -20
  39. package/src/config/atomic-write.ts +83 -8
  40. package/src/config/schema/config-schema.ts +2 -0
  41. package/src/config/schema/leaf-validators.ts +1 -0
  42. package/src/generated/compatibility-version.json +132 -68
  43. package/src/lib/bounded-subprocess.ts +62 -10
  44. package/src/lib/windows-secret-acl.ts +151 -15
  45. package/src/lib/windows-user-principal.ts +5 -1
  46. package/src/providers/derive.ts +6 -0
  47. package/src/providers/model-discovery.ts +19 -7
  48. package/src/providers/registry/entries-core.ts +11 -0
  49. package/src/providers/registry/entries-extended.ts +50 -28
  50. package/src/providers/registry/model-seeds.ts +67 -17
  51. package/src/providers/registry/types.ts +9 -0
  52. package/src/responses/spill-store.ts +17 -0
  53. package/src/responses/state/body-policy.ts +25 -0
  54. package/src/responses/state/spill-queue.ts +8 -6
  55. package/src/responses/state.ts +3 -22
  56. package/src/router.ts +4 -0
  57. package/src/server/auth-cors.ts +1 -0
  58. package/src/server/index/websocket-handler.ts +48 -1
  59. package/src/server/management/config-routes.ts +27 -5
  60. package/src/server/models-capabilities.ts +24 -3
  61. package/src/server/responses/codex-ws-exchange.ts +65 -4
  62. package/src/server/responses/combo-stream-preflight.ts +68 -5
  63. package/src/server/responses/core-combo.ts +26 -0
  64. package/src/server/responses/core-options.ts +3 -0
  65. package/src/server/responses/fetch-helpers.ts +4 -1
  66. package/src/server/responses/native-injection-protocol.ts +42 -0
  67. package/src/server/responses/native-injection-replay.ts +105 -0
  68. package/src/server/responses/native-injection.ts +242 -0
  69. package/src/server/responses/native-response-control.ts +56 -0
  70. package/src/server/responses/native-response-json.ts +14 -0
  71. package/src/server/responses/native-response-output.ts +37 -0
  72. package/src/server/responses/native-steering-log.ts +44 -0
  73. package/src/server/responses/native-steering-policy.ts +49 -0
  74. package/src/server/responses/native-steering-replay.ts +126 -0
  75. package/src/server/responses/native-steering-settings.ts +76 -0
  76. package/src/server/responses/native-steering.ts +400 -0
  77. package/src/server/responses/native-tool-results.ts +130 -0
  78. package/src/server/responses/passthrough-delivery.ts +11 -0
  79. package/src/server/responses/passthrough-dispatch.ts +33 -1
  80. package/src/server/responses/request-prepare.ts +41 -0
  81. package/src/server/responses/ws-upstream.ts +21 -1
  82. package/src/server/stop-teardown.ts +8 -1
  83. package/src/server/ws-bridge.ts +16 -1
  84. package/src/service/cli.ts +13 -1
  85. package/src/types/config.ts +4 -0
  86. package/src/types/provider.ts +13 -0
@@ -437,20 +437,42 @@ export const deepseekReasoningMapFor = (modelId: string): Record<string, string>
437
437
  // Coding Plan: the products use different exact allowlists and different base URLs.
438
438
  // Evidence: https://help.aliyun.com/en/model-studio/token-plan-personal-overview
439
439
  // https://help.aliyun.com/en/model-studio/token-plan-quickstart
440
+ // 260909 refresh, re-probed against the live gateway (both regions, both tiers):
441
+ // https://github.com/oliver-mee/alibaba-token-plan-wiki (machine-readable catalog).
442
+ // glm-5.3 / glm-5.3-flash removed from both Token Plan catalogs: they exist on Z.AI
443
+ // endpoints but the Token Plan gateway has never served either id (the 260826 seed
444
+ // propagated them across every GLM-carrying catalog; a selected row 404s).
445
+ // The Beijing preset keeps the Personal Edition subset; non-chat ids (audio/image/
446
+ // video families) stay out: they answer only on async endpoints openai-chat cannot
447
+ // reach. deepseek-v4-pro-0813 is callable but NOT listed by /models, which is the
448
+ // reason liveModels must stay false for this provider. deepseek-v4.1-flash is the
449
+ // 260910 DeepSeek rename row: listed on /models on both tiers and regions from 260915,
450
+ // hybrid thinking, vision via user message and tool result, json_object but not
451
+ // json_schema (see noJsonSchemaModels on the entries).
452
+ // Beijing serves the Personal Edition, so this is the Personal-tier roster probed
453
+ // 260909 (a strict subset of Team). deepseek-v4-pro-0813 stays out of the Beijing
454
+ // entry: its callability is only proven on Team keys, and no Personal key has been
455
+ // shown to reach it. The Beijing entry also shares the intl maps, so it carries a
456
+ // few orphan keys (kimi/glm-5/MiniMax rows); harmless, and one map beats two
457
+ // drifting ones.
440
458
  export const ALIBABA_TOKEN_PLAN_MODELS = [
441
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
442
- "glm-5.3", "glm-5.3-flash", "glm-5.2",
459
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
460
+ "deepseek-v4-pro", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "glm-5.2",
443
461
  ];
444
462
  export const ALIBABA_TOKEN_PLAN_QWEN_MODELS = [
445
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
463
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
446
464
  ];
447
465
  export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
448
466
  "qwen3.8-max": ["text", "image"],
449
- "qwen3.7-max": ["text", "image"],
467
+ "qwen3.8-flash": ["text", "image"],
468
+ "qwen3.7-max": ["text"],
450
469
  "qwen3.7-plus": ["text", "image"],
451
470
  "qwen3.6-flash": ["text", "image"],
452
- "glm-5.3": ["text"],
453
- "glm-5.3-flash": ["text", "image"],
471
+ "deepseek-v4-pro": ["text"],
472
+ "deepseek-v4-pro-0813": ["text"],
473
+ "deepseek-v4-flash-0731": ["text"],
474
+ // Vision probed on the plan gateway 260915 (user message and tool result, both 200).
475
+ "deepseek-v4.1-flash": ["text", "image"],
454
476
  "glm-5.2": ["text"],
455
477
  };
456
478
 
@@ -458,15 +480,18 @@ export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
458
480
  // Multi-vendor lineup distinct from Beijing — includes DeepSeek V4 flash, Kimi K2.7, MiniMax.
459
481
  // Evidence: https://www.alibabacloud.com/help/en/model-studio/token-plan-overview
460
482
  // https://qwencloud.com/pricing/token-plan (qwen3.8 metadata)
483
+ // The Team Edition roster (Singapore), verified identical to the CN Team set on 260909.
484
+ // deepseek-v4-pro is restored: it remains callable on the plan gateway (probed 260909,
485
+ // listed on /models on both regions) after being dropped as "retired" upstream.
461
486
  export const ALIBABA_INTL_TOKEN_PLAN_MODELS = [
462
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
463
- "deepseek-v4-flash", "deepseek-v3.2",
487
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
488
+ "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "deepseek-v3.2",
464
489
  "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5",
465
- "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5",
490
+ "glm-5.2", "glm-5.1", "glm-5",
466
491
  "MiniMax-M2.5",
467
492
  ];
468
493
  export const ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS = [
469
- "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
494
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
470
495
  ];
471
496
 
472
497
  // 260722 Tencent Cloud Coding Plan. The plan's model set is explicitly dynamic; these are the
@@ -543,24 +568,49 @@ export const VOLCENGINE_PLAN_TEXT_ONLY_MODELS = [
543
568
  "doubao-seed-2.0-pro",
544
569
  ];
545
570
  export const ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
546
- "qwen3.8-max": ["text", "image"],
547
- "qwen3.7-max": ["text", "image"],
548
- "qwen3.7-plus": ["text", "image"],
571
+ ...ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
549
572
  "qwen3.6-plus": ["text", "image"],
550
- "qwen3.6-flash": ["text", "image"],
551
573
  "deepseek-v4-flash": ["text"],
552
574
  "deepseek-v3.2": ["text"],
553
575
  "kimi-k2.7-code": ["text", "image"],
554
576
  "kimi-k2.6": ["text", "image"],
555
577
  "kimi-k2.5": ["text", "image"],
556
- "glm-5.3": ["text"],
557
- "glm-5.3-flash": ["text", "image"],
558
- "glm-5.2": ["text"],
559
578
  "glm-5.1": ["text"],
560
579
  "glm-5": ["text"],
561
580
  "MiniMax-M2.5": ["text"],
562
581
  };
563
582
 
583
+ // Shared Token Plan metadata (260909 gateway probes; output ceilings are max_tokens
584
+ // boundary probes: accept at N, reject at N+1).
585
+ export const QWEN38_FAMILY = ["qwen3.8-max", "qwen3.8-flash"];
586
+ export const ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS: Record<string, number> = {
587
+ "qwen3.8-max": 1_000_000, "qwen3.8-flash": 1_000_000, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
588
+ "qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
589
+ "deepseek-v4-pro": 1_000_000, "deepseek-v4-pro-0813": 1_000_000, "deepseek-v4-flash": 1_000_000,
590
+ "deepseek-v4-flash-0731": 1_000_000, "deepseek-v4.1-flash": 1_000_000, "deepseek-v3.2": 131_072,
591
+ "kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
592
+ "glm-5.2": 1_000_000, "glm-5.1": 202_752, "glm-5": 202_752,
593
+ "MiniMax-M2.5": 196_608,
594
+ };
595
+ export const ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS: Record<string, number> = {
596
+ "qwen3.8-max": 131_072, "qwen3.8-flash": 131_072, "qwen3.7-max": 131_072, "qwen3.7-plus": 131_072,
597
+ "qwen3.6-plus": 65_536, "qwen3.6-flash": 65_536,
598
+ "deepseek-v4-pro": 393_216, "deepseek-v4-pro-0813": 393_216, "deepseek-v4-flash": 393_216,
599
+ "deepseek-v4-flash-0731": 393_216, "deepseek-v4.1-flash": 393_216, "deepseek-v3.2": 65_536,
600
+ "kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 98_304,
601
+ "glm-5.2": 131_072, "glm-5.1": 128_000, "glm-5": 16_384,
602
+ "MiniMax-M2.5": 32_768,
603
+ };
604
+ export const ALIBABA_TOKEN_PLAN_NO_VISION = [
605
+ "qwen3.7-max", "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash",
606
+ "deepseek-v4-flash-0731", "deepseek-v3.2", "glm-5.2", "glm-5.1", "glm-5", "MiniMax-M2.5",
607
+ ];
608
+ export const ALIBABA_TOKEN_PLAN_PRESERVE_REASONING = [
609
+ "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
610
+ "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731",
611
+ "deepseek-v4.1-flash", "glm-5.2",
612
+ ];
613
+
564
614
  // 260717 Kimi K3: the subscription endpoint uses one upstream id (`k3`) for both
565
615
  // entitlement tiers. Bare `k3` advertises the Moderato 256K ceiling; the local `[1m]`
566
616
  // alias advertises Allegretto's 1M ceiling and is stripped before the upstream request.
@@ -64,6 +64,10 @@ export interface ProviderModelDiscoveryFilter {
64
64
  interface ProviderModelDiscoverySharedSpec {
65
65
  /** Query parameters applied to the resolved discovery URL. */
66
66
  query?: Readonly<Record<string, string>>;
67
+ /** Top-level response key containing model rows; defaults to `data`. */
68
+ envelopeKey?: string;
69
+ /** Model-row field containing the provider-native identifier; defaults to `id`. */
70
+ idField?: string;
67
71
  /** Declarative eligibility rules evaluated against each untrusted model row. */
68
72
  filter?: ProviderModelDiscoveryFilter;
69
73
  /** Optional lower byte ceiling; the process-wide hard ceiling still wins. */
@@ -217,6 +221,11 @@ export interface ProviderRegistryEntry {
217
221
  * to stay contiguous. This is seeded/backfilled like other fixed wire capabilities.
218
222
  */
219
223
  requiresAdjacentResponsesToolResults?: boolean;
224
+ /**
225
+ * Responses upstream that also rejects a tool call with no matching output anywhere in the
226
+ * replayed input. Seeded/backfilled like other fixed wire capabilities.
227
+ */
228
+ requiresPairedResponsesToolResults?: boolean;
220
229
  /**
221
230
  * When enabled, tool results that are present but empty are annotated on the wire.
222
231
  * Seeded/backfilled like other fixed wire capabilities.
@@ -159,6 +159,23 @@ function spillNow(): number {
159
159
  return spillNowOverride?.() ?? Date.now();
160
160
  }
161
161
 
162
+ /**
163
+ * The spill deadline clock, shared with the shutdown drain in `state/spill-queue.ts`.
164
+ *
165
+ * Every deadline the shutdown path enforces has to read the same clock the work it
166
+ * budgets reads. When the drain measured its reserve on `Date.now()` while the ACL
167
+ * harden it was budgeting ran on this injected clock, a test could freeze the clock,
168
+ * believe it had removed wall time from the case, and still lose an 80 ms reserve to
169
+ * real elapsed time on a loaded runner — which is what turned
170
+ * `shutdown fallback prices the job-owned superseded generation before publishing`
171
+ * red on macOS 2/2 in run 35137850114 while the assertion it was written for never ran.
172
+ *
173
+ * Production is unchanged: with no override installed this is `Date.now()`.
174
+ */
175
+ export function responseSpillNow(): number {
176
+ return spillNow();
177
+ }
178
+
162
179
  function record(event: "write" | "fsync" | "close" | "harden" | "publish" | "dir-fsync" | "stub-swap"): void {
163
180
  spillIoForTest?.record?.(event);
164
181
  }
@@ -0,0 +1,25 @@
1
+ /**
2
+ * Request bodies that must never enter the continuation cache.
3
+ *
4
+ * The cache is persisted to `responses-state.json`, so anything recorded here reaches disk.
5
+ * Encrypted-agent-task recovery decrypts task text into the request body and promises
6
+ * in-memory, TTL-bounded retention; recording that body would put the plaintext on disk with
7
+ * no TTL and break the promise.
8
+ *
9
+ * A WeakSet rather than a body field on purpose: `_rawBody` is serialized verbatim by the
10
+ * native passthrough, so any marker written into the body itself would be sent upstream.
11
+ * Marking is enforced once here rather than at each call site, because every recording path
12
+ * (streaming, non-streaming, passthrough, forced) funnels through `rememberResponseState` —
13
+ * a new call site cannot reintroduce the leak by forgetting a guard.
14
+ */
15
+ const nonPersistableBodies = new WeakSet<object>();
16
+
17
+ /** Bar this exact request body from the continuation cache, and therefore from disk. */
18
+ export function markBodyNonPersistable(body: unknown): void {
19
+ if (body && typeof body === "object") nonPersistableBodies.add(body as object);
20
+ }
21
+
22
+ /** Test the body's in-memory persistence restriction without adding a wire marker. */
23
+ export function isBodyNonPersistable(body: unknown): boolean {
24
+ return !!body && typeof body === "object" && nonPersistableBodies.has(body);
25
+ }
@@ -7,6 +7,7 @@ import {
7
7
  MAX_RESPONSE_SPILL_PAYLOAD_BYTES,
8
8
  prospectiveResponseSpillBytes,
9
9
  responseSpillPayloadCap,
10
+ responseSpillNow,
10
11
  type ResponseSpillPublicationControl,
11
12
  type ResponseSpillRef,
12
13
  writeResponseSpillDurably,
@@ -377,7 +378,7 @@ function responseSpillShutdownBudget(): { totalMs: number; fallbackReserveMs: nu
377
378
  }
378
379
 
379
380
  function awaitResponseSpillTailUntil(observed: Promise<void>, deadline: number): Promise<boolean> {
380
- const remaining = deadline - Date.now();
381
+ const remaining = deadline - responseSpillNow();
381
382
  if (remaining <= 0) return Promise.resolve(false);
382
383
  return new Promise(resolve => {
383
384
  let finished = false;
@@ -554,12 +555,13 @@ function terminalizeExhaustedShutdownFallback(
554
555
  }
555
556
 
556
557
  function fallbackPendingResponseSpills(reserveMs: number): Error[] {
557
- const deadline = Date.now() + reserveMs;
558
+ // Same clock as the harden work this reserve is budgeting — see `responseSpillNow`.
559
+ const deadline = responseSpillNow() + reserveMs;
558
560
  const failures: Error[] = [];
559
561
  for (;;) {
560
562
  const pending = pendingShutdownFallbackCandidates();
561
563
  if (pending.length === 0) return failures;
562
- if (Date.now() >= deadline) {
564
+ if (responseSpillNow() >= deadline) {
563
565
  terminalizeExhaustedShutdownFallback(pending, failures);
564
566
  return failures;
565
567
  }
@@ -569,7 +571,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
569
571
  for (let index = 0; index < pending.length; index += 1) {
570
572
  const { job, candidate } = pending[index]!;
571
573
  if (requireStore().currentEntry(job.id) !== candidate) continue;
572
- const remaining = deadline - Date.now();
574
+ const remaining = deadline - responseSpillNow();
573
575
  if (remaining <= 0) {
574
576
  reserveExhausted = true;
575
577
  for (const exhausted of pending.slice(index)) {
@@ -587,7 +589,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
587
589
  requireStore().recomputeOldestResident();
588
590
  requireStore().pruneResponses();
589
591
  enforceAppOwnedMemoryBudget();
590
- if (reserveExhausted || Date.now() >= deadline) {
592
+ if (reserveExhausted || responseSpillNow() >= deadline) {
591
593
  terminalizeExhaustedShutdownFallback(pendingShutdownFallbackCandidates(), failures);
592
594
  return failures;
593
595
  }
@@ -597,7 +599,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
597
599
  export async function drainResponseSpillPublications(): Promise<void> {
598
600
  const budget = responseSpillShutdownBudget();
599
601
  const fallbackReserveMs = Math.min(budget.totalMs, Math.max(1, budget.fallbackReserveMs));
600
- const drainDeadline = Date.now() + Math.max(0, budget.totalMs - fallbackReserveMs);
602
+ const drainDeadline = responseSpillNow() + Math.max(0, budget.totalMs - fallbackReserveMs);
601
603
 
602
604
  for (;;) {
603
605
  if (pendingResponseSpills.size === 0) return;
@@ -23,6 +23,8 @@ import type { ResponseSpillWriteFailureCode, ResponseSpillWriteStatus, ResponseS
23
23
  export { responseAdmissionCountersForTests } from "./state/spill-failure";
24
24
  import { admissionCounters, noteSpillWriteFailure, noteSpillWriteSuccess, spillCounters, spillWriteHealth } from "./state/spill-failure";
25
25
  import { loadSnapshotEntry } from "./state/snapshot-codec";
26
+ import { isBodyNonPersistable } from "./state/body-policy";
27
+ export { isBodyNonPersistable, markBodyNonPersistable } from "./state/body-policy";
26
28
  export { flushPendingResponseSpillsForTests, awaitResponseSpillPublicationTailForTests, pendingResponseSpillMetricsForTests, setResponseSpillShutdownBudgetForTests, setResponseSpillAsyncAclAttemptBudgetForTests, setResponseSpillShutdownTerminalizationPassLimitForTests } from "./state/spill-queue";
27
29
  import {
28
30
  bindSpillQueueStore,
@@ -1235,27 +1237,6 @@ export function responseStateMetrics(): ResponseStateMetrics {
1235
1237
  * Cache completed output and max_output_tokens partial output for previous_response_id replay.
1236
1238
  * Content-filtered incomplete and failed output are not authoritative replay history.
1237
1239
  */
1238
- /**
1239
- * Request bodies that must never enter the continuation cache.
1240
- *
1241
- * The cache is persisted to `responses-state.json`, so anything recorded here reaches disk.
1242
- * Encrypted-agent-task recovery decrypts task text into the request body and promises
1243
- * in-memory, TTL-bounded retention; recording that body would put the plaintext on disk with
1244
- * no TTL and break the promise.
1245
- *
1246
- * A WeakSet rather than a body field on purpose: `_rawBody` is serialized verbatim by the
1247
- * native passthrough, so any marker written into the body itself would be sent upstream.
1248
- * Marking is enforced once here rather than at each call site, because every recording path
1249
- * (streaming, non-streaming, passthrough, forced) funnels through `rememberResponseState` —
1250
- * a new call site cannot reintroduce the leak by forgetting a guard.
1251
- */
1252
- const nonPersistableBodies = new WeakSet<object>();
1253
-
1254
- /** Bar this exact request body from the continuation cache, and therefore from disk. */
1255
- export function markBodyNonPersistable(body: unknown): void {
1256
- if (body && typeof body === "object") nonPersistableBodies.add(body as object);
1257
- }
1258
-
1259
1240
  export function rememberResponseState(
1260
1241
  requestBody: unknown,
1261
1242
  response: { id?: unknown; output?: unknown; status?: unknown; incomplete_details?: unknown },
@@ -1264,7 +1245,7 @@ export function rememberResponseState(
1264
1245
  ): void {
1265
1246
  if (!requestBody || typeof requestBody !== "object" || Array.isArray(requestBody)) return;
1266
1247
  const request = requestBody as Record<string, unknown>;
1267
- if (nonPersistableBodies.has(request)) return;
1248
+ if (isBodyNonPersistable(request)) return;
1268
1249
  // `force` bypasses only the store:false skip: Codex sends `store:false` on every non-Azure
1269
1250
  // HTTP request (and WS inherits it), yet its WS turns still chain with previous_response_id.
1270
1251
  // The passthrough branch records with force so those chains can be expanded locally; the
package/src/router.ts CHANGED
@@ -384,6 +384,10 @@ export function routedProviderConfig(providerName: string, provider: OcxProvider
384
384
  && registryEntry.requiresAdjacentResponsesToolResults !== undefined
385
385
  ? { requiresAdjacentResponsesToolResults: registryEntry.requiresAdjacentResponsesToolResults }
386
386
  : {}),
387
+ ...(provider.requiresPairedResponsesToolResults === undefined
388
+ && registryEntry.requiresPairedResponsesToolResults !== undefined
389
+ ? { requiresPairedResponsesToolResults: registryEntry.requiresPairedResponsesToolResults }
390
+ : {}),
387
391
  ...(provider.annotateEmptyToolOutputs === undefined
388
392
  && registryEntry.annotateEmptyToolOutputs !== undefined
389
393
  ? { annotateEmptyToolOutputs: registryEntry.annotateEmptyToolOutputs }
@@ -913,6 +913,7 @@ const PROVIDER_CONFIG_FIELD_POLICY = {
913
913
  commandCodeVersion: "editor",
914
914
  statelessResponses: "editor",
915
915
  requiresAdjacentResponsesToolResults: "editor",
916
+ requiresPairedResponsesToolResults: "editor",
916
917
  annotateEmptyToolOutputs: "editor",
917
918
  supportsServiceTier: "editor",
918
919
  modelSupportsServiceTier: "editor",
@@ -1,3 +1,7 @@
1
+ import { nativeSteeringUnavailableReason, nativeResponseControlMode, type NativeResponseControl } from "../responses/native-response-control";
2
+ import { NativeInjectionChannel } from "../responses/native-injection";
3
+ import { NativeSteeringChannel, NativeSteeringError } from "../responses/native-steering";
4
+ import { createNativeSteeringLogObserver } from "../responses/native-steering-log";
1
5
  import type { Server, ServerWebSocket } from "bun";
2
6
  import {
3
7
  LIVE_SIDEBAND_UPSTREAM_OPEN_TIMEOUT_MS,
@@ -188,11 +192,47 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
188
192
  } catch {
189
193
  return; // text-only contract; ignore unparseable frames
190
194
  }
195
+ if (frame.type === "response.inject" || frame.type === "response.steer" || (frame.type === "response.create" && ws.data.nativeControl)) {
196
+ try {
197
+ if (frame.type === "response.inject") {
198
+ if (!ws.data.nativeControl?.inject) throw new NativeSteeringError("injection_not_supported", "Native injection is disabled or unavailable on this route.");
199
+ ws.data.nativeControl.inject(frame);
200
+ return;
201
+ }
202
+ if (frame.type === "response.steer") {
203
+ if (!ws.data.nativeControl) throw new NativeSteeringError("steering_not_supported", ws.data.nativeSteeringUnavailable ?? "Native steering transport is unavailable; the route may be unsupported or using HTTP fallback.");
204
+ ws.data.nativeControl.steer(frame);
205
+ return;
206
+ }
207
+ if (ws.data.nativeControl?.continue(frame)) return;
208
+ } catch (error) {
209
+ sendJsonFrame(ws, buildWsErrorFrame(400, {
210
+ type: "invalid_request_error",
211
+ code: error instanceof NativeSteeringError ? error.code : "native_steering_error",
212
+ message: error instanceof NativeSteeringError ? error.message : "Native steering transport failed; delivery may be unknown. Do not automatically replay input.",
213
+ }));
214
+ return;
215
+ }
216
+ }
191
217
  if (frame.type === "response.processed") return; // ack — no-op
192
218
  if (frame.type !== "response.create") return;
193
219
  markActivity("ws response.create");
194
220
 
221
+ let nativeControl: NativeResponseControl | undefined;
222
+ try {
223
+ const idleMs = typeof config.stallTimeoutSec === "number" && Number.isFinite(config.stallTimeoutSec)
224
+ ? Math.max(1, config.stallTimeoutSec) * 1000 : 300_000;
225
+ const mode = nativeResponseControlMode(frame, config);
226
+ nativeControl = mode === "injection" ? new NativeInjectionChannel(frame, idleMs)
227
+ : mode === "steering" ? new NativeSteeringChannel(frame, idleMs) : undefined;
228
+ } catch {
229
+ sendJsonFrame(ws, buildWsErrorFrame(400, { type: "invalid_request_error", message: "Invalid native steering request settings" }));
230
+ return;
231
+ }
195
232
  ws.data.cancel?.();
233
+ // A superseded turn must not keep ownership during warmup or refusal.
234
+ ws.data.nativeControl = undefined;
235
+ ws.data.nativeSteeringUnavailable = nativeSteeringUnavailableReason(frame, config.codexNativeSteering);
196
236
  const turnId = (ws.data.turnId ?? 0) + 1;
197
237
  ws.data.turnId = turnId;
198
238
  const isCurrent = () => ws.data.turnId === turnId;
@@ -227,6 +267,8 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
227
267
  return;
228
268
  }
229
269
 
270
+ // Only a genuinely admitted turn may receive steering or continuations.
271
+ ws.data.nativeControl = nativeControl;
230
272
  const payload: Record<string, unknown> = { ...frame };
231
273
  delete payload.type;
232
274
  turnAdmissionLease.bindAbortController(turnAbort);
@@ -267,6 +309,7 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
267
309
  ...(wsAdmission ? { admission: wsAdmission } : {}),
268
310
  forceEmptyResponseId: true,
269
311
  inboundTransport: "websocket",
312
+ nativeControl,
270
313
  abortSignal: turnAbort.signal,
271
314
  turnAdmissionLease,
272
315
  onFirstOutput: () => recordFirstOutput(logCtx, start),
@@ -277,7 +320,10 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
277
320
  },
278
321
  });
279
322
  await sendResponseToWebSocket(ws, response, isCurrent, {
280
- onSsePayload: payload => inspectResponseLogSsePayload(logCtx, payload),
323
+ untilEof: nativeControl?.relayActive === true,
324
+ onSsePayload: nativeControl?.relayActive
325
+ ? createNativeSteeringLogObserver(logCtx, () => recordFirstOutput(logCtx, start))
326
+ : payload => inspectResponseLogSsePayload(logCtx, payload),
281
327
  onTerminal: status => {
282
328
  terminalRecorder?.(status, logCtx.terminalHttpStatus);
283
329
  finalizeLog(httpStatusForRequestLogTerminal(status, logCtx), {
@@ -313,6 +359,7 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
313
359
  }
314
360
  } finally {
315
361
  turnAdmissionLease.release();
362
+ if (ws.data.nativeControl === nativeControl) ws.data.nativeControl = undefined;
316
363
  if (!logged && turnAbort.signal.aborted) finalizeLog(499);
317
364
  if (ws.data.cancel === cancelTurn) ws.data.cancel = undefined;
318
365
  }
@@ -3,6 +3,11 @@ import { randomUUID } from "node:crypto";
3
3
  import { readFileSync } from "node:fs";
4
4
  import type { CatalogModel } from "../../codex/catalog";
5
5
  import { catalogModelSlug, invalidateCodexModelsCache, nativeContextLimits, nativeModelRows, uniqueCatalogModelsForPublicList } from "../../codex/catalog";
6
+ import {
7
+ applyCodexDesktopSwitches,
8
+ describeCodexDesktopSwitches,
9
+ type CodexDesktopSwitchApply,
10
+ } from "../../codex/desktop-switches";
6
11
  import {
7
12
  DEFAULT_SUBAGENT_MODELS,
8
13
  codexAutoStartEnabled,
@@ -328,6 +333,11 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise<Respon
328
333
  codexDesktopAuthless: config.codexDesktopAuthless === true,
329
334
  // Absent keeps Design B remote compaction; true selects the dedicated provider identity.
330
335
  codexClientCompaction: config.codexClientCompaction === true,
336
+ codexDesktopSwitches: describeCodexDesktopSwitches(config, {
337
+ applied: false,
338
+ reason: "not_requested",
339
+ retryable: false,
340
+ }),
331
341
  startupHealth: await readStartupHealth(config),
332
342
  codexRuntime: {
333
343
  path: displayCodexRuntimePath(resolved.runtime.command),
@@ -597,15 +607,26 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise<Respon
597
607
  configureAppOwnedMemoryBudget(resolveAppOwnedMemoryBudgetBytes(body.appOwnedMemoryBudgetMb));
598
608
  enforceAppOwnedMemoryBudget();
599
609
  }
600
- // Both Desktop compatibility switches change the injected config.toml shape, so converge now
601
- // rather than waiting for the next start; the injector re-reads config and rewrites the form.
602
610
  const authlessIsEnabled = config.codexDesktopAuthless === true;
603
611
  const clientCompactionIsEnabled = config.codexClientCompaction === true;
604
- const catalogRefresh = pickerWasEnabled !== pickerIsEnabled
605
- || authlessWasEnabled !== authlessIsEnabled
606
- || clientCompactionWasEnabled !== clientCompactionIsEnabled
612
+ const desktopSwitchesChanged = authlessWasEnabled !== authlessIsEnabled
613
+ || clientCompactionWasEnabled !== clientCompactionIsEnabled;
614
+ // Catalog convergence is not config injection, and the comment that used to sit here said
615
+ // it was. `convergeCodexCatalog` rejects any scope but `catalog` and never reaches the
616
+ // injector, which is why flipping either switch left `config.toml` in its old shape until
617
+ // a separate `ocx sync` (#4809). Both halves are needed when a Desktop switch changes; a
618
+ // picker-only update still refreshes just the catalog.
619
+ const catalogRefresh = pickerWasEnabled !== pickerIsEnabled || desktopSwitchesChanged
607
620
  ? await convergeCodexCatalog()
608
621
  : undefined;
622
+ // Injection second, matching `syncModelsToCodex`: the injected `model_catalog_json` should
623
+ // point at a catalog that has already settled. And it runs here rather than inside the save
624
+ // because coordinated Codex writes acquire the Codex write lock N before the config mutation
625
+ // lock C — awaiting N while still holding C would invert that order.
626
+ const desktopSwitchApply: CodexDesktopSwitchApply = desktopSwitchesChanged
627
+ ? await applyCodexDesktopSwitches(config)
628
+ : { applied: false, reason: "not_requested", retryable: false };
629
+ const codexDesktopSwitches = describeCodexDesktopSwitches(config, desktopSwitchApply);
609
630
  const catalogRefreshPending = catalogRefresh
610
631
  ? catalogRefreshIsPending(catalogRefresh)
611
632
  : false;
@@ -622,6 +643,7 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise<Respon
622
643
  catalogRefreshPending,
623
644
  codexDesktopAuthless: authlessIsEnabled,
624
645
  codexClientCompaction: clientCompactionIsEnabled,
646
+ codexDesktopSwitches,
625
647
  codexMainAccountHardLock: config.codexMainAccountHardLock === true,
626
648
  mainAccountHardLock: getMainAccountHardLockStatus(config),
627
649
  startupHealth: await readStartupHealth(config),
@@ -10,6 +10,11 @@ import type { CursorEffortTable } from "../integrations/cursor-effort-table";
10
10
  * Completions, Responses and Anthropic Messages, streams, and accepts tool calls, so those are
11
11
  * constants; context length and vision come from catalog data when known and are omitted
12
12
  * otherwise, matching Cursor's optional-field schema.
13
+ *
14
+ * Top-level capacity metrics (`context_window`, `context_length`, `max_output_tokens`) are
15
+ * mirrored directly on each model row for external client discovery (e.g. pi-ai, DSH,
16
+ * LibreChat) that inspects flat properties rather than Cursor's nested `capabilities.*` shape.
17
+ * A row that gains a nested capacity value must gain the top-level mirror in the same change.
13
18
  */
14
19
 
15
20
  /**
@@ -132,6 +137,19 @@ export interface ModelCapabilityFields {
132
137
  supports_vision?: boolean;
133
138
  reasoning_effort?: string[];
134
139
  };
140
+ /**
141
+ * Mirrored top-level context window for external/legacy client discovery (e.g. pi-ai, DSH)
142
+ * that reads top-level context_window / context_length instead of nested capabilities.
143
+ */
144
+ context_window?: number;
145
+ /**
146
+ * Top-level context length alias matching capabilities.context_length for clients expecting context_length.
147
+ */
148
+ context_length?: number;
149
+ /**
150
+ * Mirrored top-level max output token limit for external/legacy client discovery.
151
+ */
152
+ max_output_tokens?: number;
135
153
  /**
136
154
  * Cursor reads the long-context threshold from `pricing.overrides[].min_prompt_tokens`. That
137
155
  * key sits outside its validated capability schema, so it is the one place a threshold can
@@ -153,6 +171,7 @@ export function modelCapabilityFields(input: ModelCapabilityInput): ModelCapabil
153
171
  const longContextLength = positiveInt(input.longContextWindow);
154
172
  const maxOutputTokens = positiveInt(input.maxOutputTokens);
155
173
  const hasLongTier = contextLength !== undefined && longContextLength !== undefined && longContextLength > contextLength;
174
+ const effectiveContextLength = hasLongTier ? longContextLength : contextLength;
156
175
  const modalities = Array.isArray(input.inputModalities)
157
176
  ? input.inputModalities.filter(modality => typeof modality === "string" && modality.length > 0)
158
177
  : undefined;
@@ -160,9 +179,7 @@ export function modelCapabilityFields(input: ModelCapabilityInput): ModelCapabil
160
179
  return {
161
180
  api_types: [...OPENCODEX_MODEL_API_TYPES],
162
181
  capabilities: {
163
- ...(hasLongTier
164
- ? { context_length: longContextLength }
165
- : contextLength !== undefined ? { context_length: contextLength } : {}),
182
+ ...(effectiveContextLength !== undefined ? { context_length: effectiveContextLength } : {}),
166
183
  ...(maxOutputTokens !== undefined ? { max_output_tokens: maxOutputTokens } : {}),
167
184
  // Once a gateway advertises api_types, Cursor keeps only rows whose output_modalities
168
185
  // include "text"; omitting the key drops the row from the extended catalog.
@@ -174,6 +191,10 @@ export function modelCapabilityFields(input: ModelCapabilityInput): ModelCapabil
174
191
  ...(supportsVision !== undefined ? { supports_vision: supportsVision } : {}),
175
192
  ...(efforts.length > 0 ? { reasoning_effort: [...efforts] } : {}),
176
193
  },
194
+ ...(effectiveContextLength !== undefined
195
+ ? { context_window: effectiveContextLength, context_length: effectiveContextLength }
196
+ : {}),
197
+ ...(maxOutputTokens !== undefined ? { max_output_tokens: maxOutputTokens } : {}),
177
198
  ...(hasLongTier ? { pricing: { overrides: [{ min_prompt_tokens: contextLength }] } } : {}),
178
199
  };
179
200
  }