@bitkyc08/opencodex 2.56.0 → 2.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (206) hide show
  1. package/bin/ocx.mjs +10 -0
  2. package/gui/dist/assets/{index-D4zuyIxQ.js → index-BbrHOIY0.js} +21 -21
  3. package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
  4. package/gui/dist/index.html +2 -2
  5. package/package.json +4 -4
  6. package/src/adapters/codebuddy/adapter.ts +2 -1
  7. package/src/adapters/codebuddy/scaffold-guard.ts +249 -0
  8. package/src/adapters/command-code.ts +12 -3
  9. package/src/adapters/cursor/cursor-errors.ts +15 -0
  10. package/src/adapters/cursor/discovery.ts +65 -1
  11. package/src/adapters/cursor/envelope-echo.ts +8 -2
  12. package/src/adapters/cursor/live-transport.ts +5 -1
  13. package/src/adapters/cursor/protobuf-events.ts +110 -11
  14. package/src/adapters/cursor/protobuf-request.ts +19 -1
  15. package/src/adapters/cursor/text-toolcall.ts +230 -0
  16. package/src/adapters/cursor/thread-continuity.ts +67 -0
  17. package/src/adapters/cursor/types.ts +5 -0
  18. package/src/adapters/cursor.ts +55 -5
  19. package/src/adapters/google-http.ts +38 -13
  20. package/src/adapters/google.ts +7 -7
  21. package/src/adapters/kiro/payload.ts +17 -3
  22. package/src/adapters/kiro/reasoning.ts +70 -7
  23. package/src/adapters/kiro/stream.ts +8 -2
  24. package/src/adapters/kiro/wire.ts +2 -1
  25. package/src/adapters/kiro-events.ts +21 -13
  26. package/src/adapters/mimo-free.ts +32 -17
  27. package/src/adapters/ollama-native.ts +42 -8
  28. package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
  29. package/src/adapters/openai-chat/tool-schema.ts +25 -7
  30. package/src/adapters/openai-chat.ts +8 -8
  31. package/src/adapters/openai-responses/passthrough.ts +62 -5
  32. package/src/adapters/openai-responses/request-strips.ts +43 -0
  33. package/src/adapters/physical-send.ts +50 -0
  34. package/src/bridge/errors.ts +26 -2
  35. package/src/bridge/response-json.ts +8 -2
  36. package/src/bridge/sse.ts +20 -2
  37. package/src/claude/desktop-profile.ts +66 -9
  38. package/src/claude/outbound.ts +32 -4
  39. package/src/cli/account-main.ts +1 -1
  40. package/src/cli/capabilities.ts +2 -2
  41. package/src/cli/combo.ts +10 -1
  42. package/src/cli/config-command.ts +35 -18
  43. package/src/cli/dispatch.ts +17 -4
  44. package/src/cli/index.ts +92 -7
  45. package/src/cli/registry.ts +2 -1
  46. package/src/cli/system-command.ts +74 -5
  47. package/src/cli/uninstall-client-state.ts +12 -0
  48. package/src/clients/config-export.ts +7 -3
  49. package/src/codex/account-label.ts +14 -3
  50. package/src/codex/account-store.ts +113 -26
  51. package/src/codex/account-usability.ts +21 -0
  52. package/src/codex/auth-api/login-flow.ts +14 -2
  53. package/src/codex/auth-api/reset-credit-service.ts +11 -2
  54. package/src/codex/auth-context.ts +199 -15
  55. package/src/codex/catalog/aggregation.ts +80 -1
  56. package/src/codex/catalog/model-visibility.ts +1 -0
  57. package/src/codex/catalog/remote.ts +30 -0
  58. package/src/codex/catalog/retained-sync.ts +9 -1
  59. package/src/codex/catalog/routed-gather.ts +38 -1
  60. package/src/codex/cli-install-provenance.ts +7 -1
  61. package/src/codex/convergence.ts +7 -2
  62. package/src/codex/desktop-app/types.ts +11 -2
  63. package/src/codex/desktop-app/windows.ts +5 -5
  64. package/src/codex/desktop-switches.ts +145 -0
  65. package/src/codex/history-job.ts +5 -1
  66. package/src/codex/history-provider.ts +33 -4
  67. package/src/codex/history-worker.ts +14 -1
  68. package/src/codex/inject/remove.ts +145 -7
  69. package/src/codex/inject/restore.ts +231 -32
  70. package/src/codex/inject.ts +12 -16
  71. package/src/codex/loopback-target.ts +9 -0
  72. package/src/codex/model-entitlements.ts +152 -15
  73. package/src/codex/native-profile-startup.ts +64 -20
  74. package/src/codex/pool-refresh-backoff.ts +12 -3
  75. package/src/codex/quota-rejection.ts +104 -15
  76. package/src/codex/routing/cache-affinity.ts +70 -0
  77. package/src/codex/routing/cooldown-math.ts +10 -0
  78. package/src/codex/routing/selection.ts +79 -2
  79. package/src/codex/routing/thread-affinity.ts +50 -2
  80. package/src/codex/routing/transient-hold-dispatch.ts +141 -0
  81. package/src/codex/routing.ts +29 -49
  82. package/src/codex/warmup.ts +1 -1
  83. package/src/combos/failover.ts +85 -0
  84. package/src/combos/request.ts +17 -10
  85. package/src/combos/types.ts +23 -2
  86. package/src/config/atomic-write.ts +83 -8
  87. package/src/config/pending-teardown.ts +31 -0
  88. package/src/config/schema/config-schema.ts +2 -0
  89. package/src/config/schema/leaf-validators.ts +1 -0
  90. package/src/generated/compatibility-version.json +272 -180
  91. package/src/images/loop.ts +1 -1
  92. package/src/lib/bounded-subprocess.ts +62 -10
  93. package/src/lib/errors.ts +17 -0
  94. package/src/lib/request-execution-budget.ts +147 -21
  95. package/src/lib/spend-reservation-ledger.ts +18 -0
  96. package/src/lib/state-store-registrations.ts +6 -2
  97. package/src/lib/test-home-guard.ts +85 -1
  98. package/src/lib/upstream-retry.ts +77 -10
  99. package/src/lib/windows-elevation.ts +76 -14
  100. package/src/lib/windows-secret-acl.ts +151 -15
  101. package/src/lib/windows-user-principal.ts +5 -1
  102. package/src/oauth/index.ts +2 -2
  103. package/src/oauth/key-providers.ts +2 -2
  104. package/src/providers/derive.ts +6 -0
  105. package/src/providers/kiro-models.ts +4 -3
  106. package/src/providers/label.ts +19 -1
  107. package/src/providers/model-discovery.ts +35 -7
  108. package/src/providers/registry/entries-core.ts +18 -0
  109. package/src/providers/registry/entries-extended.ts +59 -28
  110. package/src/providers/registry/model-seeds.ts +71 -17
  111. package/src/providers/registry/types.ts +9 -0
  112. package/src/responses/reasoning-envelope.ts +6 -3
  113. package/src/responses/spill-store.ts +17 -0
  114. package/src/responses/state/body-policy.ts +25 -0
  115. package/src/responses/state/spill-queue.ts +8 -6
  116. package/src/responses/state.ts +3 -22
  117. package/src/router.ts +4 -0
  118. package/src/routing/identity-domains.ts +21 -14
  119. package/src/routing/probe-lease.ts +103 -1
  120. package/src/server/auth-cors.ts +1 -0
  121. package/src/server/chat-completions.ts +3 -1
  122. package/src/server/chat-native.ts +37 -9
  123. package/src/server/index/live-sideband.ts +37 -1
  124. package/src/server/index/websocket-handler.ts +54 -3
  125. package/src/server/index.ts +5 -5
  126. package/src/server/inspection-tee.ts +107 -0
  127. package/src/server/live.ts +46 -1
  128. package/src/server/management/combo-routes.ts +10 -1
  129. package/src/server/management/config-routes.ts +27 -5
  130. package/src/server/models-capabilities.ts +24 -3
  131. package/src/server/relay-eager.ts +2 -0
  132. package/src/server/relay.ts +14 -19
  133. package/src/server/request-log.ts +127 -3
  134. package/src/server/response-log-body.ts +153 -0
  135. package/src/server/responses/account-change-state.ts +74 -0
  136. package/src/server/responses/adapter-continuation.ts +33 -7
  137. package/src/server/responses/adapter-delivery.ts +5 -11
  138. package/src/server/responses/adapter-dispatch.ts +84 -13
  139. package/src/server/responses/codex-ws-exchange.ts +65 -4
  140. package/src/server/responses/codex-ws-wire.ts +5 -0
  141. package/src/server/responses/collaboration.ts +74 -4
  142. package/src/server/responses/combo-session-recall.ts +68 -8
  143. package/src/server/responses/combo-stream-preflight.ts +68 -5
  144. package/src/server/responses/compact.ts +54 -13
  145. package/src/server/responses/core-auth.ts +2 -0
  146. package/src/server/responses/core-codex-account.ts +51 -3
  147. package/src/server/responses/core-combo.ts +129 -23
  148. package/src/server/responses/core-errors.ts +18 -0
  149. package/src/server/responses/core-options.ts +3 -0
  150. package/src/server/responses/core-replay.ts +105 -32
  151. package/src/server/responses/core.ts +3 -3
  152. package/src/server/responses/encrypted-payload.ts +0 -1
  153. package/src/server/responses/fetch-helpers.ts +4 -1
  154. package/src/server/responses/input-admission.ts +126 -6
  155. package/src/server/responses/native-injection-protocol.ts +42 -0
  156. package/src/server/responses/native-injection-replay.ts +105 -0
  157. package/src/server/responses/native-injection.ts +242 -0
  158. package/src/server/responses/native-response-control.ts +56 -0
  159. package/src/server/responses/native-response-json.ts +14 -0
  160. package/src/server/responses/native-response-output.ts +37 -0
  161. package/src/server/responses/native-steering-log.ts +44 -0
  162. package/src/server/responses/native-steering-policy.ts +49 -0
  163. package/src/server/responses/native-steering-replay.ts +126 -0
  164. package/src/server/responses/native-steering-settings.ts +76 -0
  165. package/src/server/responses/native-steering.ts +400 -0
  166. package/src/server/responses/native-tool-results.ts +130 -0
  167. package/src/server/responses/passthrough-delivery.ts +30 -6
  168. package/src/server/responses/passthrough-dispatch.ts +61 -11
  169. package/src/server/responses/passthrough-error.ts +38 -2
  170. package/src/server/responses/request-prepare.ts +173 -22
  171. package/src/server/responses/request-send-budget.ts +97 -2
  172. package/src/server/responses/request-spend.ts +147 -0
  173. package/src/server/responses/request-transport.ts +62 -3
  174. package/src/server/responses/run-turn-execution.ts +59 -31
  175. package/src/server/responses/sidecar-execution.ts +7 -13
  176. package/src/server/responses/terminal-guard.ts +65 -4
  177. package/src/server/responses/ws-upstream.ts +21 -1
  178. package/src/server/responses-undeclared-tool-guard.ts +9 -5
  179. package/src/server/stop-teardown.ts +8 -1
  180. package/src/server/ws-bridge.ts +16 -1
  181. package/src/service/cli.ts +13 -1
  182. package/src/service/windows-ops.ts +210 -16
  183. package/src/service/windows-scheduler.ts +28 -21
  184. package/src/service.ts +1 -1
  185. package/src/types/config.ts +8 -1
  186. package/src/types/provider.ts +13 -0
  187. package/src/types/request.ts +8 -5
  188. package/src/types/tools.ts +24 -0
  189. package/src/types.ts +2 -0
  190. package/src/update/index.ts +10 -0
  191. package/src/update/stop-contract.d.mts +1 -0
  192. package/src/update/stop-contract.mjs +19 -0
  193. package/src/update/stop-decision.d.mts +1 -1
  194. package/src/update/stop-decision.mjs +12 -3
  195. package/src/usage/log.ts +1 -1
  196. package/src/vision/anthropic-describe.ts +1 -1
  197. package/src/vision/describe.ts +5 -5
  198. package/src/web-search/anthropic-executor.ts +1 -1
  199. package/src/web-search/exa-executor.ts +1 -1
  200. package/src/web-search/executor.ts +1 -1
  201. package/src/web-search/gemini-executor.ts +1 -1
  202. package/src/web-search/loop.ts +1 -1
  203. package/src/web-search/ollama-executor.ts +1 -1
  204. package/src/web-search/parse.ts +67 -14
  205. package/src/web-search/passthrough-bridge.ts +64 -31
  206. package/src/web-search/xai-executor.ts +1 -1
package/src/router.ts CHANGED
@@ -384,6 +384,10 @@ export function routedProviderConfig(providerName: string, provider: OcxProvider
384
384
  && registryEntry.requiresAdjacentResponsesToolResults !== undefined
385
385
  ? { requiresAdjacentResponsesToolResults: registryEntry.requiresAdjacentResponsesToolResults }
386
386
  : {}),
387
+ ...(provider.requiresPairedResponsesToolResults === undefined
388
+ && registryEntry.requiresPairedResponsesToolResults !== undefined
389
+ ? { requiresPairedResponsesToolResults: registryEntry.requiresPairedResponsesToolResults }
390
+ : {}),
387
391
  ...(provider.annotateEmptyToolOutputs === undefined
388
392
  && registryEntry.annotateEmptyToolOutputs !== undefined
389
393
  ? { annotateEmptyToolOutputs: registryEntry.annotateEmptyToolOutputs }
@@ -37,13 +37,16 @@ export type IdentityDomainProvenance = "operator-declared" | "provider-documente
37
37
  * What a domain key is evidence FOR, which is two facts rather than one.
38
38
  *
39
39
  * Proven SEPARATION and proven SHARING are different claims, and a provider routinely
40
- * gives the first without the second. OpenAI documents that prompt caches are not shared
41
- * across organizations or processing regions, and in the same breath documents that
42
- * changing keys inside one organization does not guarantee a hit. So a different
43
- * org-or-region key proves two domains, while an identical one proves nothing: a
44
- * positive cache inference needs the provider to actually promise the hit, and here the
45
- * provider declines to. Inferring "shared" from an equal key would be the same guess
46
- * this module exists to refuse, only pointed the other way.
40
+ * gives the first without the second. OpenAI's prompt-caching guide states the separating
41
+ * half outright -- "Caches are not shared across organizations and cannot be reused across
42
+ * regional processing boundaries" -- and never states a sharing half at all. The page does
43
+ * not discuss two API keys inside one organization, and what it does say about keys is that
44
+ * they "influence routing; they do not pin requests to a machine or guarantee a cache hit."
45
+ * So a different org-or-region key proves two domains, while an identical one proves
46
+ * nothing. A positive cache inference needs the provider to promise the hit, and no such
47
+ * promise exists here -- the silence is the evidence, not a documented denial. Inferring
48
+ * "shared" from an equal key would be the same guess this module exists to refuse, only
49
+ * pointed the other way.
47
50
  *
48
51
  * "separates" therefore means two different keys are two different domains while two
49
52
  * identical keys stay "unknown". "separates-and-shares" means the same source also
@@ -130,7 +133,9 @@ export interface DeclaredCredentialGroup {
130
133
  * then classifies "unknown" rather than extrapolating.
131
134
  *
132
135
  * - OpenAI: rate limits are per organization and project, with model groups sharing a
133
- * limit; prompt caches are not shared across organizations or processing regions.
136
+ * limit ("Rate limits are defined at the organization level and at the project level, not
137
+ * user level", plus the documented shared limit across a model family); prompt caches are
138
+ * not shared across organizations or regional processing boundaries.
134
139
  * - Anthropic: prompt cache is isolated per workspace even inside one organization.
135
140
  * (Cache-read tokens are also excluded from input TPM there, which is quota
136
141
  * accounting, not domain shape, so it does not appear here.)
@@ -158,12 +163,14 @@ const PROVIDER_DOCUMENTED_DOMAINS: Record<string, {
158
163
  evidence: "separates-and-shares",
159
164
  },
160
165
  cache: {
161
- // Separation only. The documentation says caches are not shared across
162
- // organizations or processing regions, and says in the same place that changing
163
- // keys inside one organization does not guarantee a hit. So a different org or
164
- // region is proven distinct, while same org and region is "unknown" -- claiming
165
- // "shared" there would assert a warm prefix the provider explicitly refuses to
166
- // promise, and the caller would pay for it by replaying a long prompt that misses.
166
+ // Separation only, and the asymmetry is in the source. The prompt-caching guide says
167
+ // "Caches are not shared across organizations and cannot be reused across regional
168
+ // processing boundaries", which settles a DIFFERENT org or region as distinct. It
169
+ // states no counterpart for an identical one: the guide never discusses two keys in
170
+ // one organization, and a key is documented to "influence routing" without
171
+ // guaranteeing a hit. Same org and region therefore stays "unknown" -- claiming
172
+ // "shared" would assert a warm prefix the provider never promised, and the caller
173
+ // would pay for the guess by replaying a long prompt that misses.
167
174
  key: (ref) => ref.organizationId !== undefined && ref.region !== undefined
168
175
  ? `openai:org:${ref.organizationId}:region:${ref.region}`
169
176
  : undefined,
@@ -349,7 +349,14 @@ export function resolveHeldAccountDispatch(input: {
349
349
  kind: "withheld",
350
350
  boundAccountId: input.boundAccountId,
351
351
  ...(input.detourAccountId !== undefined ? { detourAccountId: input.detourAccountId } : {}),
352
- retryAt: nextProbeAt(input.boundAccountId, now, input.minProbeIntervalMs),
352
+ // Both bounds, not just the probe pacing. A request refused by the RATIO has no probe state
353
+ // of its own yet, so `nextProbeAt` answered `now` and the refusal told the caller to try
354
+ // again immediately -- a withheld dispatch that busy-loops is the same load as the dispatch
355
+ // it refused. The limiter is the only thing that knows when its window moves.
356
+ retryAt: Math.max(
357
+ nextProbeAt(input.boundAccountId, now, input.minProbeIntervalMs),
358
+ limiter.nextRecoveryAt(now),
359
+ ),
353
360
  };
354
361
  }
355
362
 
@@ -408,6 +415,16 @@ export interface PoolBackpressureLimiter {
408
415
  tryPermitRetryDispatch(now?: number): boolean;
409
416
  /** Admit one probe dispatch under the same shared recovery budget. */
410
417
  tryPermitProbeDispatch(now?: number): boolean;
418
+ /**
419
+ * Earliest moment this limiter could admit another recovery dispatch.
420
+ *
421
+ * A refusal has to hand back a time, or the caller has nothing to wait on and busy-loops
422
+ * against a pool that is already failing -- which is the load this limiter exists to remove.
423
+ * `now` when the allowance is not spent; otherwise the moment the oldest bucket still inside
424
+ * the window falls out of it, which is strictly in the future and is a real change point
425
+ * rather than a guess.
426
+ */
427
+ nextRecoveryAt(now?: number): number;
411
428
  state(now?: number): PoolBackpressureState;
412
429
  }
413
430
 
@@ -461,6 +478,19 @@ export function createPoolBackpressureLimiter(
461
478
  return true;
462
479
  }
463
480
 
481
+ function nextRecoveryAt(now: number): number {
482
+ const { initials, recoveries } = totals(now);
483
+ if (recoveries + 1 <= allowanceFor(initials)) return now;
484
+ // The window has to move before another recovery fits. The earliest that can happen is the
485
+ // moment the oldest bucket still inside it leaves, and every such bucket started after
486
+ // `now - windowMs`, so the answer is always strictly in the future.
487
+ for (const bucket of buckets) {
488
+ if (bucket.start <= now - policy.windowMs) continue;
489
+ return bucket.start + policy.windowMs;
490
+ }
491
+ return now + policy.windowMs;
492
+ }
493
+
464
494
  return {
465
495
  recordInitialSend(now = Date.now()): void {
466
496
  bucketFor(now).initials += 1;
@@ -471,6 +501,9 @@ export function createPoolBackpressureLimiter(
471
501
  tryPermitProbeDispatch(now = Date.now()): boolean {
472
502
  return tryPermit(now);
473
503
  },
504
+ nextRecoveryAt(now = Date.now()): number {
505
+ return nextRecoveryAt(now);
506
+ },
474
507
  state(now = Date.now()): PoolBackpressureState {
475
508
  const { initials, recoveries } = totals(now);
476
509
  return {
@@ -509,3 +542,72 @@ export function configureSharedPoolBackpressure(policy: PoolBackpressurePolicy):
509
542
  export function resetSharedPoolBackpressureForTests(): void {
510
543
  sharedLimiter = undefined;
511
544
  }
545
+
546
+ /**
547
+ * Forget every account's probe pacing AND the shared recovery window.
548
+ *
549
+ * Called when the pool's routing state is reset wholesale -- a roster change, a config reload,
550
+ * an account removal. Both halves describe a pool that no longer exists: pacing is keyed on
551
+ * account ids that may be gone, and the window's buckets count sends made by a roster that
552
+ * changed underneath them. Keeping either across such a reset lets one context's recovery
553
+ * decisions govern the next one, which is also how it leaks between test files.
554
+ *
555
+ * This is the production reset. The two `ForTests` seams above stay separate because a test
556
+ * frequently wants exactly one half of it.
557
+ */
558
+ export function clearPoolRecoveryState(): void {
559
+ probeStates.clear();
560
+ sharedLimiter = undefined;
561
+ }
562
+
563
+ /**
564
+ * What one physical send IS, as far as the recovery window is concerned.
565
+ *
566
+ * The window measures recovery traffic against observed demand, so it needs the distinction
567
+ * made where the send happens -- and the transport wrapper cannot make it. That layer sees a
568
+ * URL and an init; whether this is a conversation's first attempt, its third retry, or the one
569
+ * trial admitted against a held account is knowledge only the caller has. So the caller names
570
+ * it, and the classification lives here with the window rather than in the transport, which
571
+ * owns no routing policy and has an enforced import boundary saying so.
572
+ *
573
+ * - `initial`: a new request's first send. Recorded, never refused -- it is the denominator,
574
+ * and refusing it would make this a throughput cap rather than a recovery bound.
575
+ * - `retry`: a re-send of a request that already reached upstream once. Admitted only while
576
+ * recovery traffic stays under its ratio of observed demand.
577
+ * - `probe`: the half-open trial against a held account. It ALREADY paid at selection, inside
578
+ * {@link resolveHeldAccountDispatch}; charging it again would bill one send twice and shrink
579
+ * the very budget it was admitted from.
580
+ */
581
+ export type PoolRecoveryDispatchClass = "initial" | "retry" | "probe";
582
+
583
+ export interface PoolRecoveryDispatchDecision {
584
+ readonly admitted: boolean;
585
+ /**
586
+ * Earliest moment another recovery dispatch could be admitted. `now` when the send was
587
+ * admitted; otherwise a real change point strictly in the future, so a refused caller has
588
+ * something to wait on instead of busy-looping against a pool that is already failing.
589
+ */
590
+ readonly retryAt: number;
591
+ }
592
+
593
+ /**
594
+ * Admit one physical send against the process-wide recovery window.
595
+ *
596
+ * Per-request send budgets cannot see a storm: thousands of requests each staying inside their
597
+ * own allowance still compose into an unbounded rate against one failing upstream. This is the
598
+ * layer above them, and it is shared by construction.
599
+ */
600
+ export function classifyPoolRecoveryDispatch(
601
+ dispatchClass: PoolRecoveryDispatchClass,
602
+ now = Date.now(),
603
+ limiter: PoolBackpressureLimiter = sharedPoolBackpressure(),
604
+ ): PoolRecoveryDispatchDecision {
605
+ if (dispatchClass === "initial") {
606
+ limiter.recordInitialSend(now);
607
+ return { admitted: true, retryAt: now };
608
+ }
609
+ if (dispatchClass === "probe") return { admitted: true, retryAt: now };
610
+ return limiter.tryPermitRetryDispatch(now)
611
+ ? { admitted: true, retryAt: now }
612
+ : { admitted: false, retryAt: limiter.nextRecoveryAt(now) };
613
+ }
@@ -913,6 +913,7 @@ const PROVIDER_CONFIG_FIELD_POLICY = {
913
913
  commandCodeVersion: "editor",
914
914
  statelessResponses: "editor",
915
915
  requiresAdjacentResponsesToolResults: "editor",
916
+ requiresPairedResponsesToolResults: "editor",
916
917
  annotateEmptyToolOutputs: "editor",
917
918
  supportsServiceTier: "editor",
918
919
  modelSupportsServiceTier: "editor",
@@ -170,7 +170,9 @@ async function handleChatCompletionsWithBudget(
170
170
  if (chatBody.tools !== undefined) parts.push(JSON.stringify(chatBody.tools));
171
171
  logCtx.usageLogInputTokens = Math.max(1, estimateTokens(parts.join("\n"), requestedModel));
172
172
  }
173
- if (!effortRow && isNativeChatRouteEligible(route, chatBody, config)) chatNativeRoute = route;
173
+ // Combos must enter the Responses routing path so child selection, forced default
174
+ // effort, failover, and per-attempt telemetry run before any native Chat send.
175
+ if (!route.combo && !effortRow && isNativeChatRouteEligible(route, chatBody, config)) chatNativeRoute = route;
174
176
  } catch (err) {
175
177
  if (err instanceof UnknownRoutingPolicyError) {
176
178
  logCtx.requestedModel = requestedModel;
@@ -25,8 +25,12 @@ import {
25
25
  applyUpstreamRecoveryInit,
26
26
  fetchWithResetRetry,
27
27
  fetchWithTransientRetry,
28
+ isNonReplayableResponse,
29
+ isReplayRefusalCode,
30
+ isReplayRefusalResponse,
28
31
  prepareSameTarget429Wait,
29
32
  type UpstreamSendRecovery,
33
+ UPSTREAM_RESET_REPLAY_REFUSED_CODE,
30
34
  } from "../lib/upstream-retry";
31
35
  import {
32
36
  isTranslatorBudgetExceededError,
@@ -51,7 +55,9 @@ import { linkAbortSignal } from "./responses";
51
55
  import {
52
56
  addFinalRequestLog,
53
57
  beginRequestAttempt,
54
- noteAttemptSend,
58
+ noteProviderAttemptSend,
59
+ recordKeyAttemptFailure,
60
+ recordKeyWireAttemptUsage,
55
61
  recordFirstOutput,
56
62
  recordAttemptCredentialSource,
57
63
  sealRequestAttemptIdentity,
@@ -344,10 +350,12 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
344
350
  const encoding = new Headers(init.headers).get("accept-encoding");
345
351
  if (!headers.has("accept-encoding") && encoding) headers.set("accept-encoding", encoding);
346
352
  if (init.signal?.aborted) throw init.signal.reason;
347
- noteAttemptSend(attempt, logCtx.usageLogInputTokens, transportRecovery ?? recovery);
348
- return ((activeProvider as OcxProviderTransport).fetch ?? execute)(request.url, applyUpstreamRecoveryInit({
353
+ noteProviderAttemptSend(logCtx, route.providerName, activeProvider, logCtx.usageLogInputTokens, transportRecovery ?? recovery);
354
+ const dispatched = await ((activeProvider as OcxProviderTransport).fetch ?? execute)(request.url, applyUpstreamRecoveryInit({
349
355
  ...init, method: request.method, headers, body: request.body,
350
356
  }, transportRecovery));
357
+ if (!dispatched.ok) await recordKeyAttemptFailure(logCtx, dispatched, init.signal ?? upstream.signal);
358
+ return dispatched;
351
359
  },
352
360
  }),
353
361
  );
@@ -375,6 +383,11 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
375
383
  let retries = 0;
376
384
  while (
377
385
  response.status === 429
386
+ // A 429 this proxy synthesized for a refused reset replay is not a provider rate
387
+ // limit: waiting and re-sending here is exactly the duplicate inference the refusal
388
+ // exists to stop. It kept the same shape under the old 502 only because 502 never
389
+ // matched this branch.
390
+ && !isNonReplayableResponse(response)
378
391
  && retryPolicy
379
392
  && retries < retryPolicy.attempts
380
393
  && transientSendAvailable()
@@ -388,7 +401,9 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
388
401
  if (upstream.signal.aborted) throw upstream.signal.reason;
389
402
  response = await send(activeRequest, "rate-limit-429");
390
403
  }
391
- while (response.status === 429 && hasKeyPoolFailover(activeProvider)) {
404
+ // Same reason as above, plus a second one: rotating here would write a cooldown against
405
+ // a key that rate-limited nothing, and that false signal outlives the request.
406
+ while (response.status === 429 && !isNonReplayableResponse(response) && hasKeyPoolFailover(activeProvider)) {
392
407
  const rotated = rotateProviderTransportOn429(config, route.providerName, activeProvider, {
393
408
  retryAfter: response.headers.get("retry-after"),
394
409
  now: Date.now(),
@@ -474,6 +489,12 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
474
489
  if (isCyberPolicyCode(upstreamCode) || classified.code === CYBER_POLICY_ERROR_CODE) {
475
490
  classified.code = CYBER_POLICY_ERROR_CODE;
476
491
  classified.type = cyberPolicyErrorType(upstreamType);
492
+ } else if (isReplayRefusalResponse(response) || isReplayRefusalCode(upstreamCode)) {
493
+ // 429 classifies as a rate limit and a rate limit already carries a code, so the branch
494
+ // below -- which only fills an EMPTY code -- could never restore this one. Without it the
495
+ // client is told the provider throttled the turn, when what happened is that this proxy
496
+ // declined to send it a second time.
497
+ classified.code = UPSTREAM_RESET_REPLAY_REFUSED_CODE;
477
498
  } else if (upstreamCode === "model_not_found") {
478
499
  classified.code = "model_not_found";
479
500
  classified.type = "invalid_request_error";
@@ -481,7 +502,10 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
481
502
  classified.code = upstreamCode;
482
503
  }
483
504
  const status = isCyberPolicyCode(classified.code) ? 400 : response.status;
484
- const retryAfter = isCyberPolicyCode(classified.code)
505
+ // A refusal this proxy made has no wait to report. Synthesizing one here would hand the
506
+ // client the default two-second retry for a rate limit that never happened, which is the
507
+ // duplicate send the refusal exists to prevent.
508
+ const retryAfter = isCyberPolicyCode(classified.code) || isReplayRefusalCode(classified.code)
485
509
  ? undefined
486
510
  : resolveClientRetryAfter({
487
511
  status: response.status,
@@ -509,8 +533,10 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
509
533
  stallTimeoutSec: config.stallTimeoutSec,
510
534
  onFirstOutput: logIds ? () => recordFirstOutput(logCtx, logIds.start) : undefined,
511
535
  onUsage: usage => {
512
- logCtx.usage = usage;
513
- attempt.usage = usage;
536
+ if (!recordKeyWireAttemptUsage(logCtx, usage)) {
537
+ logCtx.usage = usage;
538
+ attempt.usage = usage;
539
+ }
514
540
  },
515
541
  onTerminal: (status: number, message?: string) => {
516
542
  terminalStatus = status;
@@ -600,8 +626,10 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
600
626
  if (!completion) return fail(502, "upstream response contained no choices", "upstream_error");
601
627
  const usage = usageFromChat(completion.usage);
602
628
  if (usage) {
603
- logCtx.usage = usage;
604
- attempt.usage = usage;
629
+ if (!recordKeyWireAttemptUsage(logCtx, usage)) {
630
+ logCtx.usage = usage;
631
+ attempt.usage = usage;
632
+ }
605
633
  }
606
634
  if (logIds) recordFirstOutput(logCtx, logIds.start);
607
635
  try {
@@ -11,7 +11,13 @@ import {
11
11
  type WsData,
12
12
  } from "../ws-bridge";
13
13
  import type { Server, ServerWebSocket } from "bun";
14
- import { handleLive, logLiveSidebandFrame, parseLiveSidebandTarget, resolveLiveSidebandUpgrade } from "../live";
14
+ import {
15
+ handleLive,
16
+ logLiveSidebandFrame,
17
+ logLiveSidebandStage,
18
+ parseLiveSidebandTarget,
19
+ resolveLiveSidebandUpgrade,
20
+ } from "../live";
15
21
  import { RESPONSE_TTL_MS } from "../../responses/state";
16
22
 
17
23
  export const MAX_WS_FRAME_BYTES = 50 * 1024 * 1024;
@@ -125,8 +131,33 @@ export function sendUpstreamFrame(upstream: WebSocket, frame: string | Buffer):
125
131
  upstream.send(Uint8Array.from(frame));
126
132
  }
127
133
 
134
+ /**
135
+ * Translate the close a downstream client sent into one the upstream socket can carry.
136
+ *
137
+ * The client's code is the only evidence of WHY the call ended, and the upstream needs it:
138
+ * a user hanging up (1001) and a protocol fault (1002/1011) are different events on the
139
+ * account, and collapsing both into a bare 1000 erases that at the proxy. Two bounds make
140
+ * the relay safe anyway. Codes a WebSocket endpoint may never send — 1005 and 1006 are
141
+ * status codes the local runtime synthesizes for "no status" and "abnormal", 1015 is
142
+ * TLS-reserved, and anything outside the registered and private ranges is undefined — become
143
+ * 1000, because `upstream.close` throws on them and a throw here would strand the upstream
144
+ * socket. The reason is truncated to the 123-byte control-frame payload limit by bytes, not
145
+ * characters, so a multibyte reason cannot overrun the frame.
146
+ */
147
+ export function clientCloseForUpstream(code: number, reason?: string): { code: number; reason: string } {
148
+ const sendable = code === 1000
149
+ || code === 1001
150
+ || code === 1003
151
+ || (code >= 1007 && code <= 1011)
152
+ || (code >= 3000 && code <= 4999);
153
+ let text = reason ?? "";
154
+ while (Buffer.byteLength(text) > 123) text = text.slice(0, -1);
155
+ return { code: sendable ? code : 1000, reason: text };
156
+ }
157
+
128
158
  function finalizeLiveSideband(ws: ServerWebSocket<WsData>, upstream?: WebSocket): void {
129
159
  if (upstream && ws.data.liveUpstream !== upstream) return;
160
+ logLiveSidebandStage("relay-closed");
130
161
  if (ws.data.liveCloseFallback !== undefined) {
131
162
  clearTimeout(ws.data.liveCloseFallback);
132
163
  ws.data.liveCloseFallback = undefined;
@@ -281,6 +312,7 @@ export function openLiveSidebandUpstream(
281
312
  try {
282
313
  socket = createWebSocket(url, headers);
283
314
  } catch {
315
+ logLiveSidebandStage("upstream-failed", { status: 502, code: "upstream_error" });
284
316
  resolve({ ok: false, status: 502, code: "upstream_error", message: "voice upstream connect failed" });
285
317
  return;
286
318
  }
@@ -297,6 +329,8 @@ export function openLiveSidebandUpstream(
297
329
  settled = true;
298
330
  clearTimeout(timer);
299
331
  removeAbortListener();
332
+ if (result.ok) logLiveSidebandStage("upstream-open");
333
+ else logLiveSidebandStage("upstream-failed", { status: result.status, code: result.code });
300
334
  resolve(result);
301
335
  };
302
336
  const timer = setTimeout(() => {
@@ -505,6 +539,7 @@ export function attachLiveSidebandUpstream(
505
539
  // session, not the connect phase.
506
540
  if (ws.data.liveConnectTimer !== undefined) clearTimeout(ws.data.liveConnectTimer);
507
541
  ws.data.liveConnectTimer = undefined;
542
+ logLiveSidebandStage("relay-attached");
508
543
  for (const frame of takeover.frames) {
509
544
  try {
510
545
  // Mirror the live message listener exactly: same ceiling, same diagnostic
@@ -527,6 +562,7 @@ export function attachLiveSidebandUpstream(
527
562
  ws.data.liveOpened = true;
528
563
  if (ws.data.liveConnectTimer !== undefined) clearTimeout(ws.data.liveConnectTimer);
529
564
  ws.data.liveConnectTimer = undefined;
565
+ logLiveSidebandStage("relay-attached");
530
566
  // An accepted transport alone does not prove inference/quota recovery.
531
567
  // Keep healthy closes neutral; explicit transport failures are recorded below.
532
568
  const pending = ws.data.livePending ?? [];
@@ -1,9 +1,14 @@
1
+ import { nativeSteeringUnavailableReason, nativeResponseControlMode, type NativeResponseControl } from "../responses/native-response-control";
2
+ import { NativeInjectionChannel } from "../responses/native-injection";
3
+ import { NativeSteeringChannel, NativeSteeringError } from "../responses/native-steering";
4
+ import { createNativeSteeringLogObserver } from "../responses/native-steering-log";
1
5
  import type { Server, ServerWebSocket } from "bun";
2
6
  import {
3
7
  LIVE_SIDEBAND_UPSTREAM_OPEN_TIMEOUT_MS,
4
8
  MAX_WS_FRAME_BYTES,
5
9
  WEBSOCKET_IDLE_TIMEOUT_SECONDS,
6
10
  attachLiveSidebandUpstream,
11
+ clientCloseForUpstream,
7
12
  closeLiveSideband,
8
13
  closeLiveSidebandBeforeUpgrade,
9
14
  enqueueLiveSidebandPendingFrame,
@@ -187,11 +192,47 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
187
192
  } catch {
188
193
  return; // text-only contract; ignore unparseable frames
189
194
  }
195
+ if (frame.type === "response.inject" || frame.type === "response.steer" || (frame.type === "response.create" && ws.data.nativeControl)) {
196
+ try {
197
+ if (frame.type === "response.inject") {
198
+ if (!ws.data.nativeControl?.inject) throw new NativeSteeringError("injection_not_supported", "Native injection is disabled or unavailable on this route.");
199
+ ws.data.nativeControl.inject(frame);
200
+ return;
201
+ }
202
+ if (frame.type === "response.steer") {
203
+ if (!ws.data.nativeControl) throw new NativeSteeringError("steering_not_supported", ws.data.nativeSteeringUnavailable ?? "Native steering transport is unavailable; the route may be unsupported or using HTTP fallback.");
204
+ ws.data.nativeControl.steer(frame);
205
+ return;
206
+ }
207
+ if (ws.data.nativeControl?.continue(frame)) return;
208
+ } catch (error) {
209
+ sendJsonFrame(ws, buildWsErrorFrame(400, {
210
+ type: "invalid_request_error",
211
+ code: error instanceof NativeSteeringError ? error.code : "native_steering_error",
212
+ message: error instanceof NativeSteeringError ? error.message : "Native steering transport failed; delivery may be unknown. Do not automatically replay input.",
213
+ }));
214
+ return;
215
+ }
216
+ }
190
217
  if (frame.type === "response.processed") return; // ack — no-op
191
218
  if (frame.type !== "response.create") return;
192
219
  markActivity("ws response.create");
193
220
 
221
+ let nativeControl: NativeResponseControl | undefined;
222
+ try {
223
+ const idleMs = typeof config.stallTimeoutSec === "number" && Number.isFinite(config.stallTimeoutSec)
224
+ ? Math.max(1, config.stallTimeoutSec) * 1000 : 300_000;
225
+ const mode = nativeResponseControlMode(frame, config);
226
+ nativeControl = mode === "injection" ? new NativeInjectionChannel(frame, idleMs)
227
+ : mode === "steering" ? new NativeSteeringChannel(frame, idleMs) : undefined;
228
+ } catch {
229
+ sendJsonFrame(ws, buildWsErrorFrame(400, { type: "invalid_request_error", message: "Invalid native steering request settings" }));
230
+ return;
231
+ }
194
232
  ws.data.cancel?.();
233
+ // A superseded turn must not keep ownership during warmup or refusal.
234
+ ws.data.nativeControl = undefined;
235
+ ws.data.nativeSteeringUnavailable = nativeSteeringUnavailableReason(frame, config.codexNativeSteering);
195
236
  const turnId = (ws.data.turnId ?? 0) + 1;
196
237
  ws.data.turnId = turnId;
197
238
  const isCurrent = () => ws.data.turnId === turnId;
@@ -226,6 +267,8 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
226
267
  return;
227
268
  }
228
269
 
270
+ // Only a genuinely admitted turn may receive steering or continuations.
271
+ ws.data.nativeControl = nativeControl;
229
272
  const payload: Record<string, unknown> = { ...frame };
230
273
  delete payload.type;
231
274
  turnAdmissionLease.bindAbortController(turnAbort);
@@ -266,6 +309,7 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
266
309
  ...(wsAdmission ? { admission: wsAdmission } : {}),
267
310
  forceEmptyResponseId: true,
268
311
  inboundTransport: "websocket",
312
+ nativeControl,
269
313
  abortSignal: turnAbort.signal,
270
314
  turnAdmissionLease,
271
315
  onFirstOutput: () => recordFirstOutput(logCtx, start),
@@ -276,7 +320,10 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
276
320
  },
277
321
  });
278
322
  await sendResponseToWebSocket(ws, response, isCurrent, {
279
- onSsePayload: payload => inspectResponseLogSsePayload(logCtx, payload),
323
+ untilEof: nativeControl?.relayActive === true,
324
+ onSsePayload: nativeControl?.relayActive
325
+ ? createNativeSteeringLogObserver(logCtx, () => recordFirstOutput(logCtx, start))
326
+ : payload => inspectResponseLogSsePayload(logCtx, payload),
280
327
  onTerminal: status => {
281
328
  terminalRecorder?.(status, logCtx.terminalHttpStatus);
282
329
  finalizeLog(httpStatusForRequestLogTerminal(status, logCtx), {
@@ -312,18 +359,22 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
312
359
  }
313
360
  } finally {
314
361
  turnAdmissionLease.release();
362
+ if (ws.data.nativeControl === nativeControl) ws.data.nativeControl = undefined;
315
363
  if (!logged && turnAbort.signal.aborted) finalizeLog(499);
316
364
  if (ws.data.cancel === cancelTurn) ws.data.cancel = undefined;
317
365
  }
318
366
  })();
319
367
  },
320
- close(ws: ServerWebSocket<WsData>) {
368
+ close(ws: ServerWebSocket<WsData>, code: number, reason: string) {
321
369
  if (ws.data.kind === "remote-workspace-agent") {
322
370
  ws.data.remoteWorkspaceClose?.();
323
371
  return;
324
372
  }
325
373
  if (ws.data.kind === "live-sideband") {
326
- closeLiveSideband(ws);
374
+ // Carry the client's own close through to the upstream instead of reporting every
375
+ // hang-up as a plain 1000.
376
+ const forwarded = clientCloseForUpstream(code, reason);
377
+ closeLiveSideband(ws, forwarded.code, forwarded.reason);
327
378
  return;
328
379
  }
329
380
  unregisterCodexWebSocket(ws);
@@ -268,11 +268,11 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
268
268
  startupOwnershipStatePaths,
269
269
  startupWindowsTaskListingCache,
270
270
  );
271
- // Startup cache invalidation is best-effort and must never block the server from
272
- // serving. It now takes K so it cannot race a convergence commit. Use the home
273
- // paired with the ownership inspection; re-reading ambient CODEX_HOME here could
274
- // invalidate a different installation after an environment or mount change.
275
- if (startupCacheOwnership.ownership === "owned" && startupOwnershipHomes !== null) {
271
+ // Startup cache invalidation is best-effort and applies only when startup sync is enabled.
272
+ // Check OFF before taking K: on Windows K resolves SID + LocalAppData through PowerShell,
273
+ // and run 35093667426 exceeded healthy controls by 33.8 s against that 30 s child budget.
274
+ // The permit callback still re-reads intent under K to close a concurrent disable race.
275
+ if (shouldSyncCodexOnStart(config) && startupCacheOwnership.ownership === "owned" && startupOwnershipHomes !== null) {
276
276
  try {
277
277
  const startupCodexHome = startupOwnershipHomes.codexHome;
278
278
  // #1046: record whether this actually rewrote the cache. `handleStart` ORs this