@bitkyc08/opencodex 2.56.0 → 2.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (145) hide show
  1. package/bin/ocx.mjs +10 -0
  2. package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
  3. package/gui/dist/assets/{index-D4zuyIxQ.js → index-Cz7CLdif.js} +21 -21
  4. package/gui/dist/index.html +2 -2
  5. package/package.json +3 -3
  6. package/src/adapters/codebuddy/adapter.ts +2 -1
  7. package/src/adapters/codebuddy/scaffold-guard.ts +248 -0
  8. package/src/adapters/command-code.ts +1 -1
  9. package/src/adapters/cursor/envelope-echo.ts +8 -2
  10. package/src/adapters/google.ts +7 -7
  11. package/src/adapters/kiro/payload.ts +17 -3
  12. package/src/adapters/kiro/reasoning.ts +70 -7
  13. package/src/adapters/kiro/stream.ts +8 -2
  14. package/src/adapters/kiro/wire.ts +2 -1
  15. package/src/adapters/kiro-events.ts +21 -13
  16. package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
  17. package/src/adapters/openai-chat/tool-schema.ts +25 -7
  18. package/src/adapters/openai-chat.ts +8 -8
  19. package/src/adapters/openai-responses/passthrough.ts +32 -1
  20. package/src/bridge/errors.ts +26 -2
  21. package/src/bridge/response-json.ts +7 -1
  22. package/src/bridge/sse.ts +19 -1
  23. package/src/claude/desktop-profile.ts +66 -9
  24. package/src/claude/outbound.ts +18 -0
  25. package/src/cli/account-main.ts +1 -1
  26. package/src/cli/capabilities.ts +2 -2
  27. package/src/cli/combo.ts +10 -1
  28. package/src/cli/index.ts +48 -5
  29. package/src/cli/registry.ts +2 -1
  30. package/src/cli/system-command.ts +4 -4
  31. package/src/clients/config-export.ts +7 -3
  32. package/src/codex/account-label.ts +14 -3
  33. package/src/codex/account-store.ts +113 -26
  34. package/src/codex/account-usability.ts +21 -0
  35. package/src/codex/auth-api/login-flow.ts +14 -2
  36. package/src/codex/auth-api/reset-credit-service.ts +11 -2
  37. package/src/codex/auth-context.ts +157 -7
  38. package/src/codex/catalog/aggregation.ts +80 -1
  39. package/src/codex/catalog/model-visibility.ts +1 -0
  40. package/src/codex/catalog/remote.ts +30 -0
  41. package/src/codex/catalog/retained-sync.ts +9 -1
  42. package/src/codex/catalog/routed-gather.ts +38 -1
  43. package/src/codex/cli-install-provenance.ts +7 -1
  44. package/src/codex/convergence.ts +7 -2
  45. package/src/codex/desktop-app/types.ts +11 -2
  46. package/src/codex/desktop-app/windows.ts +5 -5
  47. package/src/codex/inject/restore.ts +29 -2
  48. package/src/codex/inject.ts +9 -9
  49. package/src/codex/model-entitlements.ts +152 -15
  50. package/src/codex/pool-refresh-backoff.ts +12 -3
  51. package/src/codex/quota-rejection.ts +104 -15
  52. package/src/codex/routing/cache-affinity.ts +70 -0
  53. package/src/codex/routing/cooldown-math.ts +10 -0
  54. package/src/codex/routing/selection.ts +79 -2
  55. package/src/codex/routing/thread-affinity.ts +50 -2
  56. package/src/codex/routing/transient-hold-dispatch.ts +141 -0
  57. package/src/codex/routing.ts +29 -49
  58. package/src/codex/warmup.ts +1 -1
  59. package/src/combos/failover.ts +85 -0
  60. package/src/combos/request.ts +17 -10
  61. package/src/combos/types.ts +23 -2
  62. package/src/config/pending-teardown.ts +31 -0
  63. package/src/generated/compatibility-version.json +163 -135
  64. package/src/images/loop.ts +1 -1
  65. package/src/lib/errors.ts +17 -0
  66. package/src/lib/request-execution-budget.ts +147 -21
  67. package/src/lib/spend-reservation-ledger.ts +18 -0
  68. package/src/lib/state-store-registrations.ts +6 -2
  69. package/src/lib/test-home-guard.ts +85 -1
  70. package/src/lib/upstream-retry.ts +77 -10
  71. package/src/lib/windows-elevation.ts +76 -14
  72. package/src/oauth/index.ts +2 -2
  73. package/src/oauth/key-providers.ts +2 -2
  74. package/src/providers/kiro-models.ts +4 -3
  75. package/src/providers/label.ts +19 -1
  76. package/src/providers/model-discovery.ts +16 -0
  77. package/src/providers/registry/entries-core.ts +7 -0
  78. package/src/providers/registry/entries-extended.ts +9 -0
  79. package/src/providers/registry/model-seeds.ts +4 -0
  80. package/src/responses/reasoning-envelope.ts +6 -3
  81. package/src/routing/identity-domains.ts +21 -14
  82. package/src/routing/probe-lease.ts +103 -1
  83. package/src/server/chat-completions.ts +3 -1
  84. package/src/server/chat-native.ts +37 -9
  85. package/src/server/index/live-sideband.ts +37 -1
  86. package/src/server/index/websocket-handler.ts +6 -2
  87. package/src/server/index.ts +5 -5
  88. package/src/server/inspection-tee.ts +107 -0
  89. package/src/server/live.ts +46 -1
  90. package/src/server/management/combo-routes.ts +10 -1
  91. package/src/server/relay-eager.ts +2 -0
  92. package/src/server/relay.ts +14 -19
  93. package/src/server/request-log.ts +127 -3
  94. package/src/server/response-log-body.ts +153 -0
  95. package/src/server/responses/account-change-state.ts +74 -0
  96. package/src/server/responses/adapter-continuation.ts +33 -7
  97. package/src/server/responses/adapter-delivery.ts +5 -11
  98. package/src/server/responses/adapter-dispatch.ts +84 -13
  99. package/src/server/responses/codex-ws-wire.ts +5 -0
  100. package/src/server/responses/collaboration.ts +74 -4
  101. package/src/server/responses/combo-session-recall.ts +68 -8
  102. package/src/server/responses/compact.ts +54 -13
  103. package/src/server/responses/core-auth.ts +2 -0
  104. package/src/server/responses/core-codex-account.ts +51 -3
  105. package/src/server/responses/core-combo.ts +103 -23
  106. package/src/server/responses/core-errors.ts +18 -0
  107. package/src/server/responses/core-replay.ts +105 -32
  108. package/src/server/responses/core.ts +3 -3
  109. package/src/server/responses/encrypted-payload.ts +0 -1
  110. package/src/server/responses/input-admission.ts +126 -6
  111. package/src/server/responses/passthrough-delivery.ts +19 -6
  112. package/src/server/responses/passthrough-dispatch.ts +28 -10
  113. package/src/server/responses/passthrough-error.ts +38 -2
  114. package/src/server/responses/request-prepare.ts +132 -22
  115. package/src/server/responses/request-send-budget.ts +97 -2
  116. package/src/server/responses/request-spend.ts +147 -0
  117. package/src/server/responses/request-transport.ts +62 -3
  118. package/src/server/responses/run-turn-execution.ts +59 -31
  119. package/src/server/responses/sidecar-execution.ts +7 -13
  120. package/src/server/responses/terminal-guard.ts +65 -4
  121. package/src/server/responses-undeclared-tool-guard.ts +9 -5
  122. package/src/service/windows-ops.ts +210 -16
  123. package/src/service/windows-scheduler.ts +28 -21
  124. package/src/service.ts +1 -1
  125. package/src/types/config.ts +4 -1
  126. package/src/types/request.ts +8 -5
  127. package/src/types/tools.ts +24 -0
  128. package/src/types.ts +2 -0
  129. package/src/update/index.ts +10 -0
  130. package/src/update/stop-contract.d.mts +1 -0
  131. package/src/update/stop-contract.mjs +19 -0
  132. package/src/update/stop-decision.d.mts +1 -1
  133. package/src/update/stop-decision.mjs +12 -3
  134. package/src/usage/log.ts +1 -1
  135. package/src/vision/anthropic-describe.ts +1 -1
  136. package/src/vision/describe.ts +5 -5
  137. package/src/web-search/anthropic-executor.ts +1 -1
  138. package/src/web-search/exa-executor.ts +1 -1
  139. package/src/web-search/executor.ts +1 -1
  140. package/src/web-search/gemini-executor.ts +1 -1
  141. package/src/web-search/loop.ts +1 -1
  142. package/src/web-search/ollama-executor.ts +1 -1
  143. package/src/web-search/parse.ts +67 -14
  144. package/src/web-search/passthrough-bridge.ts +64 -31
  145. package/src/web-search/xai-executor.ts +1 -1
@@ -10,7 +10,13 @@
10
10
  * catches the pathological case and stays out of the way otherwise. Every uncertainty
11
11
  * resolves toward admitting.
12
12
  */
13
- import { nativeOpenAiContextWindow, nativeOpenAiMaxInputTokens, type NativeContextLimitsInput } from "../../codex/catalog/metadata";
13
+ import {
14
+ nativeOpenAiContextWindow,
15
+ nativeOpenAiMaxInputTokens,
16
+ nativeOpenAiMaxOutputTokens,
17
+ type NativeContextLimitsInput,
18
+ } from "../../codex/catalog/metadata";
19
+ import { getModelMetadata } from "../../generated/model-metadata";
14
20
  import { estimateTokens } from "../../lib/token-estimate";
15
21
  import { isCanonicalOpenAiForwardProvider, OPENAI_CODEX_PROVIDER_ID } from "../../providers/openai-tiers";
16
22
  import { modelRecordValue } from "../../reasoning-effort";
@@ -54,6 +60,8 @@ export interface InputAdmissionResult {
54
60
  estimatedTokens: number;
55
61
  /** Resolved ceiling, or null when nothing could be resolved (=> always admitted). */
56
62
  ceiling: number | null;
63
+ /** Output space reserved by the combo preflight; absent on the loose direct gate. */
64
+ requiredOutputHeadroom?: number;
57
65
  }
58
66
 
59
67
  function positive(value: unknown): number | null {
@@ -135,14 +143,19 @@ export function estimateInputTokens(parsed: OcxParsedRequest, modelId: string):
135
143
  * reject a user-defined provider that merely shares a built-in name using limits that
136
144
  * belong to a different service.
137
145
  */
138
- export function resolveInputCeiling(
146
+ interface ResolvedContextLimits {
147
+ /** The target's total context window: input and output share it. */
148
+ window: number | null;
149
+ /** Largest admissible input, which input-only caps may tighten below the window. */
150
+ ceiling: number | null;
151
+ }
152
+
153
+ function resolveContextLimits(
139
154
  provider: OcxProviderConfig,
140
155
  providerName: string,
141
156
  modelId: string,
142
- // Operator cap for the canonical native provider. Passed in rather than read from a
143
- // config here so this stays pure: no filesystem, no catalog, no registry scan.
144
157
  nativeContextCap?: NativeContextLimitsInput,
145
- ): number | null {
158
+ ): ResolvedContextLimits {
146
159
  // `modelRecordValue`, not a bare lookup: the catalog resolves these same two maps that
147
160
  // way, so a `gpt-oss` entry covers `gpt-oss:120b`. Reading raw here made the gate fall
148
161
  // back to the provider-wide window and refuse turns the model can plainly hold.
@@ -168,13 +181,120 @@ export function resolveInputCeiling(
168
181
  : null;
169
182
  const nativeMaxInput = canonicalNativeBare ? positive(nativeOpenAiMaxInputTokens(modelId, nativeLimits)) : null;
170
183
 
171
- const window = canonicalNativeBare ? native : configured;
184
+ const window = canonicalNativeBare ? (native ?? generatedNativeWindow(modelId, configured, nativeContextCap)) : configured;
172
185
  // modelMaxInputTokens is an input-only cap, so it can only tighten the window.
173
186
  const configuredMaxInput = positive(modelRecordValue(provider.modelMaxInputTokens, modelId));
174
187
  const limits = [window, configuredMaxInput, nativeMaxInput].filter((v): v is number => v !== null);
188
+ return { window, ceiling: limits.length === 0 ? null : Math.min(...limits) };
189
+ }
190
+
191
+ /**
192
+ * Generated-catalog keys, not routing provider names. `OPENAI_CODEX_PROVIDER_ID` is the string
193
+ * `"openai"` -- the canonical Codex forward route -- so using it to index the generated bundle
194
+ * would silently skip the native Codex rows and read the public API rows instead.
195
+ */
196
+ const NATIVE_METADATA_CATALOGS = ["openai-codex", "openai"] as const;
197
+
198
+ /**
199
+ * Static in-tree metadata for a canonical native slug the narrower override and pinned-native
200
+ * tables do not carry. Falling through to null made input admission completely blind for
201
+ * exactly those models, which is how a 128k target accepted a turn it could not finish.
202
+ *
203
+ * This deliberately covers slugs that are no longer offered in the picker: a retired slug is
204
+ * still dispatchable when an operator names it explicitly in a combo target, and that is the
205
+ * configuration where the gate was inert. This is a generated bundle compiled into the binary,
206
+ * not a live catalog read, so it adds no I/O. Explicit provider and operator caps may only
207
+ * narrow the result, never widen it.
208
+ */
209
+ function generatedNativeWindow(
210
+ modelId: string,
211
+ configured: number | null,
212
+ nativeContextCap: NativeContextLimitsInput | undefined,
213
+ ): number | null {
214
+ let generated: number | null = null;
215
+ for (const catalog of NATIVE_METADATA_CATALOGS) {
216
+ generated = positive(getModelMetadata(catalog, modelId)?.contextWindow);
217
+ if (generated !== null) break;
218
+ }
219
+ if (generated === null) return null;
220
+ const cap = typeof nativeContextCap === "number"
221
+ ? positive(nativeContextCap)
222
+ : positive(nativeContextCap?.cap);
223
+ return Math.min(generated, configured ?? generated, cap ?? generated);
224
+ }
225
+
226
+ export function resolveInputCeiling(
227
+ provider: OcxProviderConfig,
228
+ providerName: string,
229
+ modelId: string,
230
+ // Operator cap for the canonical native provider. Passed in rather than read from a
231
+ // config here so this stays pure: no filesystem, no catalog, no registry scan.
232
+ nativeContextCap?: NativeContextLimitsInput,
233
+ ): number | null {
234
+ return resolveContextLimits(provider, providerName, modelId, nativeContextCap).ceiling;
235
+ }
236
+
237
+ /**
238
+ * Largest output the concrete target can emit. Used only to avoid reserving MORE than the
239
+ * target could ever produce when a client asks for a bigger allowance than the model has.
240
+ * Unknown stays unknown rather than inventing a capability.
241
+ */
242
+ export function resolveOutputCeiling(
243
+ provider: OcxProviderConfig,
244
+ providerName: string,
245
+ modelId: string,
246
+ ): number | null {
247
+ const configured = positive(modelRecordValue(provider.modelMaxOutputTokens, modelId))
248
+ ?? positive(provider.defaultMaxOutputTokens);
249
+ const canonicalNativeBare = providerName === OPENAI_CODEX_PROVIDER_ID
250
+ && isCanonicalOpenAiForwardProvider(provider)
251
+ && !modelId.includes("/");
252
+ const native = canonicalNativeBare ? positive(nativeOpenAiMaxOutputTokens(modelId)) : null;
253
+ const limits = [configured, native].filter((v): v is number => v !== null);
175
254
  return limits.length === 0 ? null : Math.min(...limits);
176
255
  }
177
256
 
257
+ /**
258
+ * Combo-only admission. A fallback must be able to satisfy the caller's declared output
259
+ * allowance inside its OWN context window. Otherwise it returns 200, emits a few hundred
260
+ * tokens, and terminates on `finish_reason: length` — which the Anthropic surface renders as
261
+ * "response exceeded the output token maximum" even though the real cause was the total
262
+ * window. By then the next target cannot be tried, because output has already committed.
263
+ *
264
+ * Two budgets are checked separately so the reserve is counted exactly once. `ceiling` is an
265
+ * input-only budget once `modelMaxInputTokens` tightens it below the window, so the output
266
+ * reserve belongs against `window`, not against `ceiling`.
267
+ *
268
+ * Direct and single-target requests keep the deliberately loose 2.5x pathological-input gate.
269
+ * This stricter rule applies only to synthetic combo children, where skipping one known-small
270
+ * target is safe and the ladder continues before any upstream bytes are sent. Unknown context
271
+ * stays fail-open, and a caller that declared no output allowance is unaffected.
272
+ */
273
+ export function checkComboTargetInputAdmission(
274
+ parsed: OcxParsedRequest,
275
+ provider: OcxProviderConfig,
276
+ providerName: string,
277
+ modelId: string,
278
+ nativeContextCap?: NativeContextLimitsInput,
279
+ ): InputAdmissionResult {
280
+ const { window, ceiling } = resolveContextLimits(provider, providerName, modelId, nativeContextCap);
281
+ const requestedOutput = positive(parsed.options.maxOutputTokens);
282
+ if (window === null || ceiling === null || requestedOutput === null) {
283
+ return checkInputAdmission(parsed, provider, providerName, modelId, nativeContextCap);
284
+ }
285
+ const targetOutput = resolveOutputCeiling(provider, providerName, modelId);
286
+ const requiredOutputHeadroom = targetOutput === null
287
+ ? requestedOutput
288
+ : Math.min(requestedOutput, targetOutput);
289
+ const estimatedTokens = estimateInputTokens(parsed, modelId);
290
+ return {
291
+ admitted: estimatedTokens <= ceiling && estimatedTokens + requiredOutputHeadroom <= window,
292
+ estimatedTokens,
293
+ ceiling,
294
+ requiredOutputHeadroom,
295
+ };
296
+ }
297
+
178
298
  /**
179
299
  * Fail-open when no ceiling is known; refuse only past `ceiling * ADMISSION_TOLERANCE`.
180
300
  *
@@ -15,6 +15,8 @@ import {
15
15
  relayWithAbort,
16
16
  } from "../relay";
17
17
  import { isUsageDebugEnabled } from "../../usage/debug";
18
+ import { isReplayRefusalResponse } from "../../lib/upstream-retry";
19
+ import { teeWithBoundedInspection } from "../inspection-tee";
18
20
  import {
19
21
  codexForwardTerminalOutcomeRecorder,
20
22
  usesCodexForwardPoolAuth,
@@ -27,7 +29,7 @@ import type { ResponsesTerminalStatus } from "../../bridge";
27
29
  import { isCodexWsQuotaObservedResponse, isCodexWsUpstreamResponse } from "./ws-upstream";
28
30
  import { recordSubagentQuotaFailureForThreadSpawn } from "../../codex/subagent-model-fallback";
29
31
  import { recordCodexUpstreamOutcome } from "../../codex/routing";
30
- import { codexProbeLeaseId, codexProbeQuotaScope } from "../../codex/auth-context";
32
+ import { codexProbeLeaseId, codexProbeQuotaScope, codexTransientProbeGrant } from "../../codex/auth-context";
31
33
  import { consumeComboFailure } from "./core-combo-failure";
32
34
  import { readDisplaySafeErrorText } from "./core-errors";
33
35
  import { streamingContextOverflowResponse, jsonContextOverflowResponse } from "./context-overflow";
@@ -243,7 +245,12 @@ export async function deliverPassthroughResponse(
243
245
  } else if (!shouldDeferCodexResetDerivedCooldown(
244
246
  upstreamResponse,
245
247
  options.deferCodexResetDerivedCooldown,
246
- )) {
248
+ ) && !isReplayRefusalResponse(upstreamResponse)) {
249
+ // A refusal this proxy made is not evidence about the account. Recording it would
250
+ // classify the synthetic 429 as quota exhaustion and write a default cooldown against
251
+ // a credential the request may never have reached, and that false signal outlives the
252
+ // request. The sibling recorders on this path already decline: the terminal recorder
253
+ // needs an ok streaming body, and the quota-header snapshot finds no quota headers.
247
254
  recordCodexUpstreamOutcome(config, admissionState.authCtx.accountId, upstreamResponse.status, {
248
255
  ...quotaMeta,
249
256
  threadId: admissionState.authCtx.affinityKey,
@@ -251,6 +258,7 @@ export async function deliverPassthroughResponse(
251
258
  modelId: route.modelId,
252
259
  probeLeaseId: codexProbeLeaseId(admissionState.authCtx),
253
260
  probeQuotaScope: codexProbeQuotaScope(admissionState.authCtx),
261
+ transientProbe: codexTransientProbeGrant(admissionState.authCtx),
254
262
  writerGeneration: admissionState.authCtx.writerGeneration,
255
263
  // Includes a replay's second 401, which is the case that actually retires the
256
264
  // account — fence it on the credential the request was holding.
@@ -298,6 +306,9 @@ export async function deliverPassthroughResponse(
298
306
  return formatPassthroughUpstreamError(upstreamResponse.status, errorText, {
299
307
  statusText: upstreamResponse.statusText,
300
308
  headers,
309
+ // Provenance, not inference: `errorText` is empty when the bounded read finds nothing
310
+ // display-safe, and an empty body is exactly what the retryable-429 default fires on.
311
+ replayRefusal: isReplayRefusalResponse(upstreamResponse),
301
312
  });
302
313
  }
303
314
 
@@ -593,16 +604,18 @@ export async function deliverPassthroughResponse(
593
604
  })),
594
605
  );
595
606
  }
596
- const [nativeBody, inspectBody] = passthroughSseBody.tee();
597
607
  const turnAc = new AbortController();
598
608
  const clientGone = new AbortController();
609
+ const clientGoneSignal = options.abortSignal
610
+ ? AbortSignal.any([clientGone.signal, options.abortSignal])
611
+ : clientGone.signal;
612
+ // Pace against raw bytes before rewrites, without detaching terminal ownership.
613
+ const [nativeBody, inspectBody] = teeWithBoundedInspection(passthroughSseBody, { clientGoneSignal });
599
614
  linkAbortSignal(upstream, turnAc.signal);
600
615
  registerTurn(turnAc, options.turnAdmissionLease);
601
616
  const inspectionConsumerOptions = {
602
617
  // Request abort can reject the fetch body before the response cancel hook runs.
603
- clientGoneSignal: options.abortSignal
604
- ? AbortSignal.any([clientGone.signal, options.abortSignal])
605
- : clientGone.signal,
618
+ clientGoneSignal,
606
619
  drainBounds: { ms: 15_000, bytes: 32 * 1024 * 1024 },
607
620
  upstream,
608
621
  pinCompletedResponseIdToFirstSeen: githubCopilotRepairEnabled,
@@ -29,6 +29,7 @@ import {
29
29
  unwrapUpstreamRetryEvidenceError,
30
30
  codexProbeLeaseId,
31
31
  codexProbeQuotaScope,
32
+ codexTransientProbeGrant,
32
33
  createCodexReserveDispatchGuard,
33
34
  } from "../../codex/auth-context";
34
35
  import {
@@ -60,7 +61,6 @@ import { restorePlaintextV2AgentMessageCalls } from "../../responses/plaintext-v
60
61
  import {
61
62
  recordAdapterReasoning,
62
63
  recordAdapterTier,
63
- noteAttemptSend,
64
64
  sealRequestAttemptIdentity,
65
65
  recordAttemptCredentialSource,
66
66
  } from "../request-log";
@@ -80,6 +80,7 @@ import {
80
80
  safeHostLabel,
81
81
  storedPoolReplayDispatchNotifier,
82
82
  } from "./fetch-helpers";
83
+ import { classifyPoolRecoveryDispatch } from "../../routing/probe-lease";
83
84
  import { clientCancelledResponse } from "./core-errors";
84
85
  import {
85
86
  upstreamHostCircuitOpenResponse,
@@ -101,6 +102,7 @@ import {
101
102
  fetchWithTransientRetry,
102
103
  applyUpstreamRecoveryInit,
103
104
  TRANSIENT_RETRY_MAX_ATTEMPTS,
105
+ isNonReplayableResponse,
104
106
  prepareSameTarget429Wait,
105
107
  sleepWithAbort,
106
108
  } from "../../lib/upstream-retry";
@@ -171,6 +173,7 @@ export async function preparePassthroughExchange(
171
173
  | "replayOAuthCredentialSnapshot"
172
174
  | "genericFailovers"
173
175
  | "applyFailoverSnapshot"
176
+ | "noteRoutedAttemptSend"
174
177
  >,
175
178
  responseEffects: Pick<
176
179
  ResponsesEffects,
@@ -346,10 +349,13 @@ export async function preparePassthroughExchange(
346
349
  const declaredWireToolNames = new Set<string>();
347
350
  const declaredBareWireToolNames = new Set<string>();
348
351
  const declaredNamelessClientCallTypes = new Set<string>();
349
- // `buildToolBridgeMaps` creates a bare alias only when the caller selected exactly one
350
- // namespaced tool through a bare tool_choice. Restore that request-bounded identity before
351
- // authorization checks instead of admitting the bare name into the declared set: for `exec`,
352
- // the latter would also authorize the unrelated code-mode helper names.
352
+ // `buildToolBridgeMaps` adds each eligible bare alias to `declaredToolNames` and `toolNsMap`
353
+ // (one authorized identity claims the bare name). `refreshUndeclaredToolGuard` normally copies
354
+ // those entries into `declaredWireToolNames`, but passthrough restoration runs before the
355
+ // undeclared-tool guard, so restore that request-bounded identity here, before authorization
356
+ // checks. `exec` uses separate handling: its bridge alias is copied into the declared set only
357
+ // when the client itself declared bare `exec`, because otherwise code-mode normalization could
358
+ // authorize the unrelated code-mode helper names.
353
359
  const authorizedBareNamespaceToolAliases: RoutedNamespaceToolAliases = new Map(
354
360
  [...toolBridgeMaps.toolNsMap].flatMap(([alias, identity]) =>
355
361
  alias === identity.name
@@ -736,6 +742,7 @@ export async function preparePassthroughExchange(
736
742
  modelId: route.modelId,
737
743
  probeLeaseId: codexProbeLeaseId(admissionState.authCtx),
738
744
  probeQuotaScope: codexProbeQuotaScope(admissionState.authCtx),
745
+ transientProbe: codexTransientProbeGrant(admissionState.authCtx),
739
746
  writerGeneration: admissionState.authCtx.writerGeneration,
740
747
  });
741
748
  }
@@ -752,7 +759,13 @@ export async function preparePassthroughExchange(
752
759
  // Body is a replayable string; nothing has streamed to the client yet.
753
760
  upstreamResponse = await fetchWithTransientRetry(
754
761
  recovery => {
755
- noteAttemptSend(logCtx.activeAttempt, passthroughEstimate, recovery);
762
+ // The pool-wide recovery window measures recovery traffic against observed demand,
763
+ // and this is where demand is observed: `recovery === undefined` is a new request's
764
+ // first send, everything after it is the same request trying again. Without this the
765
+ // ratio has no denominator and the window collapses to its quiet-pool floor, which
766
+ // would throttle recovery on a busy proxy exactly as hard as on an idle one (#4701).
767
+ if (recovery === undefined) classifyPoolRecoveryDispatch("initial");
768
+ transportState.noteRoutedAttemptSend(passthroughEstimate, recovery);
756
769
  return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
757
770
  method: request.method,
758
771
  headers: request.headers,
@@ -848,7 +861,7 @@ export async function preparePassthroughExchange(
848
861
  if (allowance.permit && !allowance.permit.use()) {
849
862
  throw new SendBudgetExhaustedError(safeHostLabel(request.url));
850
863
  }
851
- noteAttemptSend(logCtx.activeAttempt, passthroughEstimate, innerRecovery ?? recovery);
864
+ transportState.noteRoutedAttemptSend(passthroughEstimate, innerRecovery ?? recovery);
852
865
  return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
853
866
  method: request.method,
854
867
  headers: request.headers,
@@ -946,7 +959,7 @@ export async function preparePassthroughExchange(
946
959
  // every other build site; a replay is exactly when a grown payload reappears.
947
960
  const replayBodyRefusal = refuseOversizedOutboundBody(request);
948
961
  if (replayBodyRefusal) return replayBodyRefusal;
949
- noteAttemptSend(logCtx.activeAttempt, passthroughEstimate, "oauth-401");
962
+ transportState.noteRoutedAttemptSend(passthroughEstimate, "oauth-401");
950
963
  upstreamResponse = await fetchWithHeaderTimeout(
951
964
  request.url,
952
965
  { method: request.method, headers: request.headers, body: request.body },
@@ -1075,7 +1088,7 @@ export async function preparePassthroughExchange(
1075
1088
  try {
1076
1089
  upstreamResponse = await fetchWithTransientRetry(
1077
1090
  recovery => {
1078
- noteAttemptSend(logCtx.activeAttempt, passthroughEstimate, recovery ?? "oauth-401");
1091
+ transportState.noteRoutedAttemptSend(passthroughEstimate, recovery ?? "oauth-401");
1079
1092
  return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
1080
1093
  method: request.method,
1081
1094
  headers: request.headers,
@@ -1105,6 +1118,10 @@ export async function preparePassthroughExchange(
1105
1118
  // the same quorum, cooldown and request budget here, before any client bytes flow.
1106
1119
  if (
1107
1120
  upstreamResponse.status === 429
1121
+ // Not a provider rate limit when this proxy synthesized it for a refused reset
1122
+ // replay; rotating accounts on it would re-send an inference that may already
1123
+ // have run and would cool down an account that refused nothing.
1124
+ && !isNonReplayableResponse(upstreamResponse)
1108
1125
  && transportState.genericFailoverAccountId
1109
1126
  && transportState.genericFailovers < GENERIC_OAUTH_MAX_FAILOVERS_PER_REQUEST
1110
1127
  && isGenericOAuthFailoverEnabled(config, route.providerName)
@@ -1159,6 +1176,7 @@ export async function preparePassthroughExchange(
1159
1176
  // keep their pool logic below (rateLimitRetryPolicyFor returns null for them).
1160
1177
  while (
1161
1178
  upstreamResponse.status === 429
1179
+ && !isNonReplayableResponse(upstreamResponse)
1162
1180
  && rateLimitPolicy !== null
1163
1181
  && rateLimitRetries < rateLimitPolicy.attempts
1164
1182
  // Checked here rather than inside the helper: prepareSameTarget429Wait releases the 429
@@ -1192,7 +1210,7 @@ export async function preparePassthroughExchange(
1192
1210
  recovery => {
1193
1211
  // The first send of every replay is itself a rate-limit retry; inner transient-5xx
1194
1212
  // recoveries keep their own label (recovery is provided for those).
1195
- noteAttemptSend(logCtx.activeAttempt, passthroughEstimate, recovery ?? "rate-limit-429");
1213
+ transportState.noteRoutedAttemptSend(passthroughEstimate, recovery ?? "rate-limit-429");
1196
1214
  return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
1197
1215
  method: request.method,
1198
1216
  headers: request.headers,
@@ -1,5 +1,6 @@
1
1
  import { formatErrorResponse } from "../../bridge";
2
2
  import { isCyberPolicyCode, isCyberPolicyMessage } from "../../lib/errors";
3
+ import { isReplayRefusalCode, UPSTREAM_RESET_REPLAY_REFUSED_CODE } from "../../lib/upstream-retry";
3
4
  import {
4
5
  resolveClientRetryAfter,
5
6
  validateClientRetryAfterHeader,
@@ -24,6 +25,26 @@ function isCyberPolicyBody(body: string): boolean {
24
25
  return false;
25
26
  }
26
27
 
28
+ /**
29
+ * True for a body this proxy wrote to refuse replaying an ambiguous pre-header reset.
30
+ *
31
+ * It is read off the body rather than a marker because this formatter is handed bytes, not
32
+ * the response they came from, and the refusal reaches it after the original body was read.
33
+ * The code is this proxy's own, so an upstream echoing it is not a case worth widening for.
34
+ */
35
+ function isReplayRefusalBody(body: string): boolean {
36
+ if (!body.includes(UPSTREAM_RESET_REPLAY_REFUSED_CODE)) return false;
37
+ try {
38
+ const parsed = JSON.parse(body) as Record<string, unknown>;
39
+ const error = parsed.error && typeof parsed.error === "object" && !Array.isArray(parsed.error)
40
+ ? parsed.error as Record<string, unknown>
41
+ : undefined;
42
+ return isReplayRefusalCode(error?.code) || isReplayRefusalCode(parsed.code);
43
+ } catch {
44
+ return false;
45
+ }
46
+ }
47
+
27
48
  /**
28
49
  * Passthrough adapters historically relayed upstream non-2xx bodies verbatim.
29
50
  * Codex maps an *empty* body to the literal client string "Unknown error"
@@ -38,6 +59,9 @@ function isCyberPolicyBody(body: string): boolean {
38
59
  * - missing/malformed values are replaced when resolveClientRetryAfter yields a value
39
60
  * - malformed/expired values are removed when the resolver returns undefined
40
61
  * (e.g. quota-exhausted 429s must not keep junk headers or get the synthetic "2")
62
+ * - a replay refusal this proxy wrote gets none and keeps none: the whole point of the
63
+ * refusal is that the turn may already be running, and the synthetic default for a
64
+ * retryable 429 is a direct instruction to the client to send it a second time
41
65
  */
42
66
  export function formatPassthroughUpstreamError(
43
67
  status: number,
@@ -46,6 +70,13 @@ export function formatPassthroughUpstreamError(
46
70
  statusText?: string;
47
71
  headers?: Headers;
48
72
  now?: number;
73
+ /**
74
+ * Provenance from the caller that still holds the response: this body is a refusal this
75
+ * proxy synthesized. The body check below is the fallback for a re-wrapped body, and it
76
+ * cannot answer at all when the bounded read returned nothing display-safe -- which is
77
+ * precisely when the empty-body branch would invent the retryable-429 default.
78
+ */
79
+ replayRefusal?: boolean;
49
80
  },
50
81
  ): Response {
51
82
  const trimmed = bodyText.trim();
@@ -53,7 +84,12 @@ export function formatPassthroughUpstreamError(
53
84
  const upstreamRetryAfter = options?.headers?.get("retry-after")?.trim() || undefined;
54
85
  const originalValid = validateClientRetryAfterHeader(upstreamRetryAfter, now);
55
86
  const cyberPolicyFailure = isCyberPolicyBody(trimmed);
56
- const resolved = cyberPolicyFailure
87
+ // Two different reasons to answer with no wait at all, handled the same way: a hard policy
88
+ // block will not become servable, and a refusal we made was never a rate limit.
89
+ const suppressRetryAfter = cyberPolicyFailure
90
+ || options?.replayRefusal === true
91
+ || isReplayRefusalBody(trimmed);
92
+ const resolved = suppressRetryAfter
57
93
  ? undefined
58
94
  : resolveClientRetryAfter({
59
95
  status,
@@ -64,7 +100,7 @@ export function formatPassthroughUpstreamError(
64
100
 
65
101
  if (trimmed) {
66
102
  const needsSet = resolved !== undefined && upstreamRetryAfter !== resolved;
67
- const needsDelete = (cyberPolicyFailure && upstreamRetryAfter !== undefined)
103
+ const needsDelete = (suppressRetryAfter && upstreamRetryAfter !== undefined)
68
104
  || (resolved === undefined
69
105
  && upstreamRetryAfter !== undefined
70
106
  && originalValid === undefined);