@bitkyc08/opencodex 2.54.0 → 2.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/gui/dist/assets/{index-CkvITofZ.js → index-VuoiWj9J.js} +10 -10
  2. package/gui/dist/index.html +1 -1
  3. package/package.json +1 -1
  4. package/src/adapters/anthropic-image-codec.ts +57 -0
  5. package/src/adapters/anthropic-image-normalize.ts +28 -1
  6. package/src/adapters/anthropic.ts +68 -6
  7. package/src/adapters/base.ts +8 -0
  8. package/src/adapters/coding-agent/protocol.ts +41 -16
  9. package/src/adapters/cursor/cursor-errors.ts +1 -1
  10. package/src/adapters/cursor/live-transport.ts +5 -1
  11. package/src/adapters/cursor/native-exec-fs.ts +10 -10
  12. package/src/adapters/cursor/native-exec-network.ts +2 -2
  13. package/src/adapters/cursor/native-exec-shell.ts +13 -12
  14. package/src/adapters/cursor/native-exec.ts +51 -10
  15. package/src/adapters/cursor/policy-error.ts +75 -0
  16. package/src/adapters/cursor/protobuf-request.ts +105 -1
  17. package/src/adapters/devin/cloud-direct/catalog.ts +34 -2
  18. package/src/adapters/devin/live-models.ts +33 -2
  19. package/src/adapters/google-wire-compiler.ts +8 -0
  20. package/src/adapters/google.ts +46 -0
  21. package/src/adapters/input-media-guard.ts +45 -0
  22. package/src/adapters/kiro/adapter.ts +8 -0
  23. package/src/adapters/kiro/payload.ts +28 -6
  24. package/src/adapters/kiro-events.ts +25 -6
  25. package/src/adapters/kiro-images.ts +30 -0
  26. package/src/adapters/kiro-retry.ts +8 -0
  27. package/src/adapters/openai-chat.ts +33 -4
  28. package/src/adapters/openai-responses.ts +26 -0
  29. package/src/adapters/registry.ts +4 -0
  30. package/src/bridge.ts +163 -116
  31. package/src/chat/image-parts.ts +151 -0
  32. package/src/chat/inbound.ts +70 -33
  33. package/src/cli/connect.ts +30 -9
  34. package/src/cli/dispatch.ts +7 -3
  35. package/src/cli/index.ts +3 -0
  36. package/src/cli/runtime-api.ts +25 -0
  37. package/src/cli/status.ts +21 -19
  38. package/src/cli/system-restart-client.ts +25 -0
  39. package/src/clients/config-export.ts +14 -4
  40. package/src/codex/app-server-processes.ts +25 -0
  41. package/src/codex/auth-context.ts +8 -0
  42. package/src/codex/autostart-health.ts +36 -2
  43. package/src/codex/catalog/provider-fetch.ts +41 -0
  44. package/src/codex/catalog-auto-refresh.ts +182 -0
  45. package/src/codex/catalog-refresh-status.ts +93 -0
  46. package/src/codex/history-provider.ts +55 -0
  47. package/src/codex/model-entitlements.ts +78 -0
  48. package/src/codex/native-profile-processes.ts +114 -15
  49. package/src/codex/prompt-text-probe.ts +274 -41
  50. package/src/codex/routing-adoption.ts +189 -0
  51. package/src/codex/routing.ts +520 -48
  52. package/src/codex/runtime.ts +249 -7
  53. package/src/combos/failover.ts +45 -0
  54. package/src/config.ts +124 -4
  55. package/src/generated/compatibility-version.json +110 -74
  56. package/src/generated/model-metadata.ts +1 -0
  57. package/src/lib/request-execution-budget.ts +202 -0
  58. package/src/lib/upstream-retry.ts +95 -8
  59. package/src/lib/workflow-budget.ts +172 -0
  60. package/src/oauth/devin.ts +57 -12
  61. package/src/providers/quota.ts +37 -6
  62. package/src/providers/registry.ts +53 -6
  63. package/src/responses/input-media.ts +65 -0
  64. package/src/responses/parser-content.ts +42 -0
  65. package/src/responses/schema.ts +12 -2
  66. package/src/server/audio-live.ts +1 -2
  67. package/src/server/audio-transcriptions.ts +1 -2
  68. package/src/server/auth-cors.ts +1 -1
  69. package/src/server/background-lifecycle.ts +18 -0
  70. package/src/server/chat-completions.ts +23 -8
  71. package/src/server/chat-native.ts +17 -17
  72. package/src/server/index.ts +24 -0
  73. package/src/server/management/request-history-routes.ts +5 -0
  74. package/src/server/request-log.ts +8 -2
  75. package/src/server/responses/compact.ts +51 -3
  76. package/src/server/responses/core.ts +238 -28
  77. package/src/server/search.ts +7 -9
  78. package/src/types/config.ts +51 -6
  79. package/src/usage/log.ts +37 -0
  80. package/src/vision/eligibility.ts +37 -4
  81. package/src/vision/index.ts +1 -0
  82. package/src/vision/plan.ts +45 -10
  83. package/src/web-search/alpha-search.ts +324 -0
  84. package/src/web-search/index.ts +13 -22
  85. package/src/web-search/passthrough-bridge.ts +195 -22
  86. package/src/web-search/sidecar-providers.ts +22 -0
@@ -165,7 +165,7 @@ import {
165
165
  shouldResolveOpenAiPassthroughWebSearchBridge,
166
166
  } from "../../web-search/passthrough-bridge";
167
167
  import { buildImageTool, buildVideoTool, planImageBridge, planVideoBridge, runWithImageBridge, clampImageMaxRounds, IMAGE_GEN_TOOL_NAME, VIDEO_GEN_TOOL_NAME } from "../../images";
168
- import { describeImagesInPlace, isModelTextOnly, planVisionSidecar, resolveOpenAiVisionModel, shouldResolveOpenAiVisionSidecar, stripImagesInPlace } from "../../vision";
168
+ import { describeImagesInPlace, planVisionSidecar, requiresVisionPreprocessing, resolveOpenAiVisionModel, shouldResolveOpenAiVisionSidecar, stripImagesInPlace } from "../../vision";
169
169
  import { createAdapterEventQueue, preflightAdapterEvents, type AdapterEventQueue } from "../../adapters/run-turn-queue";
170
170
  import {
171
171
  applyCodexAuthContextToProvider,
@@ -223,7 +223,20 @@ import {
223
223
  isTransientUpstreamStatus,
224
224
  prepareSameTarget429Wait,
225
225
  sleepWithAbort,
226
+ TRANSIENT_RETRY_MAX_ATTEMPTS,
227
+ SendBudgetExhaustedError,
228
+ type TransientSendBudget,
226
229
  } from "../../lib/upstream-retry";
230
+ import {
231
+ createRequestExecutionBudget,
232
+ isRequestExecutionBudget,
233
+ type SendClass,
234
+ type SingleUseDispatchPermit,
235
+ } from "../../lib/request-execution-budget";
236
+ import {
237
+ chargeWorkflowSends,
238
+ workflowSendCeilingReached,
239
+ } from "../../lib/workflow-budget";
227
240
  import {
228
241
  ForwardAdmissionCredentialError,
229
242
  hasForwardableCodexBearer,
@@ -781,6 +794,18 @@ function isEncryptedFunctionOutputRejection(bodyText: string): boolean {
781
794
  }
782
795
  }
783
796
 
797
+ /**
798
+ * #4469: reasoning encrypted_content is minted per caller identity, so replaying it under a
799
+ * different caller is rejected with "reasoning `encrypted_content` was not issued to this
800
+ * caller". Substring checks tolerate the optional backticks and a leading or trailing
801
+ * sentence, while the "was not issued to this caller" anchor plus an encrypted-content or
802
+ * reasoning subject keep unrelated invalid_request_error prose from gaining a hidden resend.
803
+ */
804
+ function isReasoningBlobCallerMismatchMessage(message: string): boolean {
805
+ if (!message.includes("was not issued to this caller")) return false;
806
+ return message.includes("encrypted_content") || message.includes("reasoning");
807
+ }
808
+
784
809
  function isSelfIdentifiedOpaqueBlobRejection(bodyText: string): boolean {
785
810
  if (isEncryptedFunctionOutputRejection(bodyText)) return true;
786
811
  try {
@@ -793,7 +818,7 @@ function isSelfIdentifiedOpaqueBlobRejection(bodyText: string): boolean {
793
818
  try {
794
819
  const payload = JSON.parse(bodyText) as unknown;
795
820
  if (!payload || typeof payload !== "object" || Array.isArray(payload)) return false;
796
- const record = payload as { code?: unknown; error?: unknown };
821
+ const record = payload as { code?: unknown; type?: unknown; message?: unknown; error?: unknown };
797
822
 
798
823
  if (record.error && typeof record.error === "object" && !Array.isArray(record.error)) {
799
824
  const error = record.error as { type?: unknown; code?: unknown; message?: unknown };
@@ -807,9 +832,23 @@ function isSelfIdentifiedOpaqueBlobRejection(bodyText: string): boolean {
807
832
  " could not be verified. Reason: Encrypted content could not be decrypted or parsed.",
808
833
  )
809
834
  ) return true;
835
+ // #4469: the caller-mismatch wording arrives without a dedicated code, so the
836
+ // message itself is the identity. It is not gated on code being null — the upstream
837
+ // may attach a generic code — because the anchored phrase is already specific.
838
+ if (typeof error.message === "string" && isReasoningBlobCallerMismatchMessage(error.message)) {
839
+ return true;
840
+ }
810
841
  }
811
842
  }
812
843
 
844
+ // The flat stream-error envelope carries type/message at the top level rather than under
845
+ // an error object; the same anchored identity applies there.
846
+ if (
847
+ record.type === "invalid_request_error"
848
+ && typeof record.message === "string"
849
+ && isReasoningBlobCallerMismatchMessage(record.message)
850
+ ) return true;
851
+
813
852
  if (record.code !== "invalid-argument" || typeof record.error !== "string") return false;
814
853
  return record.error.startsWith("Could not decode the compaction blob")
815
854
  || record.error.startsWith("Could not decrypt the provided encrypted_content");
@@ -824,8 +863,9 @@ function isSelfIdentifiedOpaqueBlobRejection(bodyText: string): boolean {
824
863
  * The outbound-body check is intentional: the inbound transcript may contain a proxy envelope or
825
864
  * compaction blob that the adapter already lowered, in which case a replay would be byte-identical.
826
865
  * OpenAI usually exposes a dedicated nested code; ChatGPT also emits one exact code-less
827
- * unverifiable-ciphertext message. xAI's code is generic, so its two concrete decoder error
828
- * identities are also required. Unrelated error prose must never gain a hidden resend.
866
+ * unverifiable-ciphertext message, and #4469 added the anchored caller-mismatch wording for
867
+ * reasoning blobs minted under a different caller. xAI's code is generic, so its two concrete
868
+ * decoder error identities are also required. Unrelated error prose must never gain a hidden resend.
829
869
  */
830
870
  export function shouldAttemptOpaqueBlobRecovery(args: {
831
871
  status: number;
@@ -1259,6 +1299,10 @@ interface CodexPoolAccountRetryArgs {
1259
1299
  translatorBudget: TranslatorBudget;
1260
1300
  turnAdmissionLease?: AdmissionLease;
1261
1301
  resolveCodexModelEntitlements?: typeof resolveCodexModelEntitlements;
1302
+ /** The logical request's execution budget: the account move is its fourth send. */
1303
+ sendBudget?: TransientSendBudget;
1304
+ /** Root workflow this turn belongs to, so the move is charged there as well. */
1305
+ workflowRootId?: string;
1262
1306
  };
1263
1307
  firstAuthCtx: Extract<CodexAuthContext, { kind: "pool" | "main-pool" }>;
1264
1308
  firstResponse: Response;
@@ -1453,6 +1497,27 @@ async function retryCodexPoolOnAlternateAccount(
1453
1497
  recordUnmovedTransientOutcome();
1454
1498
  return { kind: "no-alternate" };
1455
1499
  }
1500
+ // An account move is the guarded profile's fourth send and draws the single shared
1501
+ // final-recovery reserve. Nothing bounded it per request before: `excludeAccountId` excludes
1502
+ // only the account that just failed, and the caller's recovery loop can return here after the
1503
+ // alternate fails too, so one request could walk the pool an account at a time. The permit is
1504
+ // consumed immediately before the physical send, so a resolution that finds no alternate
1505
+ // costs nothing.
1506
+ const executionBudget = isRequestExecutionBudget(args.options.sendBudget)
1507
+ ? args.options.sendBudget
1508
+ : undefined;
1509
+ let accountMovePermit: SingleUseDispatchPermit | undefined;
1510
+ if (!retryAuthCtx && executionBudget) {
1511
+ const decision = executionBudget.reserveDispatch({
1512
+ sendClass: "account-failover",
1513
+ targetKey: `${route.providerName}|${route.modelId}|alternate-account`,
1514
+ });
1515
+ if (!decision.allowed) {
1516
+ recordUnmovedTransientOutcome();
1517
+ return { kind: "no-alternate" };
1518
+ }
1519
+ accountMovePermit = decision.permit;
1520
+ }
1456
1521
  try {
1457
1522
  retryAuthCtx ??= await resolveCodexAuthContext(
1458
1523
  callerAuthHeaders,
@@ -1591,6 +1656,18 @@ async function retryCodexPoolOnAlternateAccount(
1591
1656
  let upstreamResponse: Response;
1592
1657
  try {
1593
1658
  while (true) {
1659
+ // The same-account gated-model 400 ladder below keeps its own `maxRetrySends` bound and
1660
+ // does not take the reserve again; only the move itself does.
1661
+ if (accountMovePermit) {
1662
+ const charged = accountMovePermit.use();
1663
+ accountMovePermit = undefined;
1664
+ if (!charged) {
1665
+ recordUnmovedTransientOutcome();
1666
+ return { kind: "no-alternate" };
1667
+ }
1668
+ // The move is a physical send like any other, so the root workflow is charged too.
1669
+ chargeWorkflowSends(args.options.workflowRootId, 1);
1670
+ }
1594
1671
  noteAttemptSend(logCtx.activeAttempt, passthroughEstimate);
1595
1672
  try {
1596
1673
  upstreamResponse = await fetchWithHeaderTimeout(
@@ -1874,6 +1951,12 @@ export interface HandleResponsesOptions {
1874
1951
  onStoredPool401ReplayDispatched?: () => void;
1875
1952
  /** Caller-owned for Chat/Claude replay; omitted only at genuine Responses ingress. */
1876
1953
  translatorBudget?: TranslatorBudget;
1954
+ /**
1955
+ * Transient sends already spent by this logical request. Combo children inherit the parent's
1956
+ * holder through the options spread, so a fan-out shares one allowance instead of taking a
1957
+ * fresh one per target (#4546).
1958
+ */
1959
+ sendBudget?: TransientSendBudget;
1877
1960
  /**
1878
1961
  * Terminal vision-describe marker (roadmap 180): true when the inbound
1879
1962
  * request IS the vision sidecar's own loopback describe call. The plan site
@@ -3420,6 +3503,9 @@ export async function handleResponses(
3420
3503
  visionDescribeTerminal: options.visionDescribeTerminal === true
3421
3504
  || req.headers.get("x-opencodex-vision-describe") === "1",
3422
3505
  translatorBudget,
3506
+ // Created once at genuine ingress; a combo child arrives with the parent's holder already
3507
+ // in options and must not start a fresh allowance.
3508
+ sendBudget: options.sendBudget ?? createRequestExecutionBudget(),
3423
3509
  });
3424
3510
  return ownsBudget ? finalizeOwnedTranslatorBudget(response, translatorBudget) : response;
3425
3511
  } catch (error) {
@@ -4187,6 +4273,13 @@ async function handleResponsesInner(
4187
4273
  ? `${route.providerName}-${route.codexAccountNamespace}`
4188
4274
  : formatCodexProviderForLog(route.providerName, codexLogAccountId(authCtx), config);
4189
4275
  logCtx.accountLogLabel = codexAuthContextLogLabel(authCtx, config);
4276
+ // A move is the expensive event: it discards the prefix warmed on the previous account. Record
4277
+ // it as an event with its cause, so the operator reads it off one line instead of inferring it
4278
+ // from account labels across many (#4546).
4279
+ if (authCtx.kind === "pool" && authCtx.affinityDecision) {
4280
+ logCtx.affinity = authCtx.affinityDecision.move;
4281
+ logCtx.affinityReason = authCtx.affinityDecision.reason;
4282
+ }
4190
4283
  // Seed an account-derived scope before final adapter binding. Cursor never treats it as
4191
4284
  // authoritative: bindRouteReasoningReplayScope replaces it with the exact route owner or a
4192
4285
  // per-request fail-closed sentinel after the final provider and credential are known.
@@ -4769,7 +4862,7 @@ async function handleResponsesInner(
4769
4862
  const routedCompaction = parsed._compactionRequest === true
4770
4863
  && !isCanonicalOpenAiForwardProvider(route.provider);
4771
4864
  const needsOpenAiVision = !visionDescribeTerminal
4772
- && shouldResolveOpenAiVisionSidecar(config, route.provider, route.modelId, parsed);
4865
+ && shouldResolveOpenAiVisionSidecar(config, route.provider, route.modelId, parsed, route.providerName);
4773
4866
  const needsOpenAiSearch = !routedCompaction && !adapter.runTurn
4774
4867
  && (shouldResolveOpenAiWebSearchSidecar(config, parsed, isPassthrough)
4775
4868
  || shouldResolveOpenAiPassthroughWebSearchBridge(route.provider, parsed, isPassthrough));
@@ -4839,7 +4932,7 @@ async function handleResponsesInner(
4839
4932
  const visionPlan = visionDescribeTerminal
4840
4933
  ? undefined
4841
4934
  : planVisionSidecar(config, route.provider, route.modelId, parsed, openAiSidecar, {
4842
- admission: options.admission, codexAuthPolicy: options.codexAuthPolicy,
4935
+ admission: options.admission, codexAuthPolicy: options.codexAuthPolicy, providerName: route.providerName,
4843
4936
  });
4844
4937
  const recordSidecarOutcome = openAiSidecar?.recordOutcome;
4845
4938
  if (visionPlan) {
@@ -4851,9 +4944,9 @@ async function handleResponsesInner(
4851
4944
  recordSidecarOutcome,
4852
4945
  translatorBudget,
4853
4946
  );
4854
- } else if (isModelTextOnly(route.provider, route.modelId)) {
4855
- // Sidecar-covered model but NO plan (no forward provider / missing forwarded auth / sidecar
4856
- // disabled): fail closed — never forward raw images to a text-only upstream.
4947
+ } else if (requiresVisionPreprocessing(config, route.provider, route.modelId, route.providerName)) {
4948
+ // Image capability is not positively proven but no sidecar plan is dispatchable: fail closed.
4949
+ // Never forward raw image bytes to an unverified upstream.
4857
4950
  stripImagesInPlace(parsed, translatorBudget);
4858
4951
  }
4859
4952
 
@@ -4937,6 +5030,73 @@ async function handleResponsesInner(
4937
5030
  routedMuseToolNameAliases = builtRequest.convertedMuseToolNameAliases ?? new Map();
4938
5031
  };
4939
5032
 
5033
+ // One transient-retry budget for the whole LOGICAL request, read ABOVE the passthrough branch
5034
+ // so that branch shares it too. It used to be a local declared below, which put it in the
5035
+ // temporal dead zone for the passthrough sends and left each recovery leg taking the helper's
5036
+ // fresh default of 3. It is now a holder carried on options, so a combo child inherits the
5037
+ // parent's spend instead of starting over per target -- both halves of the measured
5038
+ // amplification in #4546.
5039
+ const sendBudget = options.sendBudget ?? createRequestExecutionBudget();
5040
+ // The root workflow is the user-visible task. A per-request cap cannot bound a fan-out that
5041
+ // sends once per child seven hundred times, so every send charged to the request is charged
5042
+ // to the root as well (#4546).
5043
+ const workflowRootId = req.headers.get("x-codex-parent-thread-id")?.trim() || undefined;
5044
+ const noteTransientSends = (used: number): void => {
5045
+ const charged = Math.max(0, used);
5046
+ sendBudget.used += charged;
5047
+ chargeWorkflowSends(workflowRootId, charged);
5048
+ };
5049
+ // Refused before any dispatch, and deliberately not by evicting the root's ledger entry:
5050
+ // dropping the record to make room would hand the fan-out a fresh allowance, which is the
5051
+ // laundering this ceiling exists to stop. The client is told the task needs a new grant
5052
+ // rather than being given a synthetic upstream error.
5053
+ if (workflowSendCeilingReached(workflowRootId)) {
5054
+ return formatErrorResponse(
5055
+ 429,
5056
+ "workflow_budget_exhausted",
5057
+ "This task has used its whole send budget, so no further upstream request was made. Requests already in flight settle as they finish.",
5058
+ );
5059
+ }
5060
+ // No floor. Math.max(1, ...) meant an exhausted request still funded one send on every
5061
+ // recovery leg, so a bounded per-leg allowance never became a bounded per-request one.
5062
+ const remainingTransientSendBudget = (budget: number): number =>
5063
+ isRequestExecutionBudget(sendBudget)
5064
+ ? sendBudget.remainingBaseSends(budget)
5065
+ : Math.max(0, budget - sendBudget.used);
5066
+ // The adapter contract needs the full budget, not just the counter. options.sendBudget is
5067
+ // typed as the narrow holder so a caller that predates this can still pass one, so narrow it
5068
+ // once here rather than asserting at each adapter call site.
5069
+ const adapterSendBudget = isRequestExecutionBudget(sendBudget) ? sendBudget : undefined;
5070
+ const sendBudgetExhausted = (): boolean =>
5071
+ remainingTransientSendBudget(TRANSIENT_RETRY_MAX_ATTEMPTS) === 0;
5072
+ /**
5073
+ * How many sends a recovery leg may make, and the permit that authorises the last one.
5074
+ *
5075
+ * The base allowance is spent first. Once it is gone a recovery class may still draw the
5076
+ * single shared final-recovery reserve -- which is what keeps the validated sanitized rebuild
5077
+ * after a 5xx streak alive at four total sends -- but an account move and a rebuild cannot
5078
+ * each take one. `countedExternally` is set because these legs run through the retry helper,
5079
+ * which reports the same send again through `onSendsConsumed`.
5080
+ */
5081
+ const recoverySendAllowance = (
5082
+ cap: number,
5083
+ sendClass: SendClass,
5084
+ targetKey: string,
5085
+ ): { attempts: number; permit?: SingleUseDispatchPermit } => {
5086
+ const base = remainingTransientSendBudget(cap);
5087
+ if (base > 0) return { attempts: base };
5088
+ if (!isRequestExecutionBudget(sendBudget)) return { attempts: 0 };
5089
+ const decision = sendBudget.reserveDispatch({ sendClass, targetKey, countedExternally: true });
5090
+ return decision.allowed ? { attempts: 1, permit: decision.permit } : { attempts: 0 };
5091
+ };
5092
+ /**
5093
+ * Both classes share the one reserve, so this only changes what the decision is called --
5094
+ * but a recovery event that says "repair" when a credential refresh drove it is the kind of
5095
+ * mislabelled evidence #4592 existed to stop.
5096
+ */
5097
+ const recoveryClassFor = (recovery: AttemptRecoveryKind): SendClass =>
5098
+ /401|429|oauth|rate-limit|key/.test(recovery) ? "auth-recovery" : "repair";
5099
+
4940
5100
  if ("passthrough" in adapter && adapter.passthrough && !routedCompaction) {
4941
5101
  let hostAdmissionLease = pendingHostAdmissionLease;
4942
5102
  pendingHostAdmissionLease = null;
@@ -5409,6 +5569,15 @@ async function handleResponsesInner(
5409
5569
  releaseCodexAuthContextProbeLease(authCtx);
5410
5570
  return clientCancelledResponse();
5411
5571
  }
5572
+ // A budget refusal is a proxy decision, not an upstream fault. Reporting it as
5573
+ // 502 upstream_error would blame the provider for a limit this process applied, and
5574
+ // would record a fake reachability failure against the account's health.
5575
+ if (err instanceof SendBudgetExhaustedError) {
5576
+ releaseUpstreamHostAdmission(hostAdmissionLease);
5577
+ hostAdmissionLease = null;
5578
+ releaseCodexAuthContextProbeLease(authCtx);
5579
+ return formatErrorResponse(429, "request_send_budget_exhausted", err.message);
5580
+ }
5412
5581
  const localRefusal = mapCodexAuthContextErrorToResponse(unwrapUpstreamRetryEvidenceError(err), {
5413
5582
  now: Date.now(), accountSelector: route.codexAccountNamespace,
5414
5583
  });
@@ -5479,7 +5648,7 @@ async function handleResponsesInner(
5479
5648
  // retry wrapper replaces — proves the host was reached (#914 review).
5480
5649
  .then(adoptObservedResponse);
5481
5650
  },
5482
- { abortSignal: upstream.signal, label: safeHostLabel(request.url) },
5651
+ { abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(TRANSIENT_RETRY_MAX_ATTEMPTS), onSendsConsumed: noteTransientSends },
5483
5652
  );
5484
5653
  } catch (err) {
5485
5654
  return transportFailureResponse(err);
@@ -5540,8 +5709,21 @@ async function handleResponsesInner(
5540
5709
  const rebuiltBodyRefusal = refuseOversizedOutboundBody(request);
5541
5710
  if (rebuiltBodyRefusal) return { failed: rebuiltBodyRefusal };
5542
5711
  try {
5712
+ // The base allowance is spent first; once it is gone this leg may still draw the one
5713
+ // shared final-recovery reserve, which is what keeps a validated sanitized rebuild
5714
+ // after a 5xx streak alive at four total sends instead of dying at three.
5715
+ const allowance = recoverySendAllowance(
5716
+ TRANSIENT_RETRY_MAX_ATTEMPTS,
5717
+ recoveryClassFor(recovery),
5718
+ `${route.providerName}|${route.modelId}|${recovery}`,
5719
+ );
5543
5720
  return await fetchWithTransientRetry(
5544
5721
  innerRecovery => {
5722
+ // Gated on the return, not fire-and-forget: a consumed permit means this leg
5723
+ // already sent once, and letting the second call through would be a free send.
5724
+ if (allowance.permit && !allowance.permit.use()) {
5725
+ throw new SendBudgetExhaustedError(safeHostLabel(request.url));
5726
+ }
5545
5727
  noteAttemptSend(logCtx.activeAttempt, passthroughEstimate, innerRecovery ?? recovery);
5546
5728
  return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
5547
5729
  method: request.method,
@@ -5559,7 +5741,7 @@ async function handleResponsesInner(
5559
5741
  route.provider.authMode === "forward")
5560
5742
  .then(adoptObservedResponse);
5561
5743
  },
5562
- { abortSignal: upstream.signal, label: safeHostLabel(request.url) },
5744
+ { abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: allowance.attempts, onSendsConsumed: noteTransientSends },
5563
5745
  );
5564
5746
  } catch (err) {
5565
5747
  return { failed: transportFailureResponse(err) };
@@ -5682,6 +5864,10 @@ async function handleResponsesInner(
5682
5864
  && isOAuth401ReplayProvider
5683
5865
  && sentOAuthSnapshot
5684
5866
  && !oauth401ReplayAttempted
5867
+ // Refused here, before the 401 body is cancelled: once it is gone the request can only
5868
+ // answer with a synthetic 502, which would report a proxy budget decision as an upstream
5869
+ // fault and throw away the credential evidence the client needs.
5870
+ && !sendBudgetExhausted()
5685
5871
  ) {
5686
5872
  oauth401ReplayAttempted = true;
5687
5873
  try { void upstreamResponse.body?.cancel().catch(() => {}); } catch { /* already consumed/closed */ }
@@ -5779,7 +5965,7 @@ async function handleResponsesInner(
5779
5965
  route.provider.authMode === "forward")
5780
5966
  .then(adoptObservedResponse);
5781
5967
  },
5782
- { abortSignal: upstream.signal, label: safeHostLabel(request.url) },
5968
+ { abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(TRANSIENT_RETRY_MAX_ATTEMPTS), onSendsConsumed: noteTransientSends },
5783
5969
  );
5784
5970
  } catch (err) {
5785
5971
  return transportFailureResponse(err);
@@ -5832,6 +6018,10 @@ async function handleResponsesInner(
5832
6018
  upstreamResponse.status === 429
5833
6019
  && rateLimitPolicy !== null
5834
6020
  && rateLimitRetries < rateLimitPolicy.attempts
6021
+ // Checked here rather than inside the helper: prepareSameTarget429Wait releases the 429
6022
+ // body, so a refusal discovered after the wait can no longer return the real rate-limit
6023
+ // answer and would surface a synthetic 502 instead.
6024
+ && !sendBudgetExhausted()
5835
6025
  ) {
5836
6026
  rateLimitRetries += 1;
5837
6027
  // Release unread body + deliberate wait via the shared same-target helper.
@@ -5876,7 +6066,7 @@ async function handleResponsesInner(
5876
6066
  route.provider.authMode === "forward")
5877
6067
  .then(adoptObservedResponse);
5878
6068
  },
5879
- { abortSignal: upstream.signal, label: safeHostLabel(request.url) },
6069
+ { abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(TRANSIENT_RETRY_MAX_ATTEMPTS), onSendsConsumed: noteTransientSends },
5880
6070
  );
5881
6071
  } catch (err) {
5882
6072
  return transportFailureResponse(err);
@@ -5943,7 +6133,7 @@ async function handleResponsesInner(
5943
6133
  route,
5944
6134
  parsed,
5945
6135
  logCtx,
5946
- options,
6136
+ options: { ...options, workflowRootId },
5947
6137
  firstAuthCtx: authCtx,
5948
6138
  firstResponse: upstreamResponse,
5949
6139
  outcomeStatus: poolRetryOutcome,
@@ -6251,6 +6441,7 @@ async function handleResponsesInner(
6251
6441
  openAiSidecar,
6252
6442
  );
6253
6443
  const webSearchBridgePlan = planPassthroughWebSearchBridge(parsed, route.provider, {
6444
+ providerName: route.providerName,
6254
6445
  isPassthrough: true,
6255
6446
  stream: parsed.stream === true,
6256
6447
  auth: webSearchBridgeAuth,
@@ -6291,7 +6482,7 @@ async function handleResponsesInner(
6291
6482
  providerApiKey: route.provider.apiKey ?? "",
6292
6483
  auth: webSearchBridgeAuth,
6293
6484
  hostedTool: parsed._webSearch,
6294
- describeImages: isModelTextOnly(route.provider, route.modelId),
6485
+ describeImages: requiresVisionPreprocessing(config, route.provider, route.modelId, route.providerName),
6295
6486
  sidecar: config.webSearchSidecar,
6296
6487
  }),
6297
6488
  // Appending a search result can push the continuation past the ceiling the first leg
@@ -6817,7 +7008,7 @@ async function handleResponsesInner(
6817
7008
  // can proceed for web-search-only turns
6818
7009
  const wsPlan = !routedCompaction
6819
7010
  ? planWebSearch(config, parsed, false, route.provider, route.modelId, openAiSidecar, {
6820
- admission: options.admission, codexAuthPolicy: options.codexAuthPolicy,
7011
+ admission: options.admission, codexAuthPolicy: options.codexAuthPolicy, providerName: route.providerName,
6821
7012
  })
6822
7013
  : undefined;
6823
7014
  const imgPlan = !routedCompaction ? await planImageBridge(config, parsed, route.provider) : undefined;
@@ -7516,13 +7707,6 @@ async function handleResponsesInner(
7516
7707
  notifyResponseComplete(json);
7517
7708
  return new Response(JSON.stringify(json), { headers: { "Content-Type": "application/json" } });
7518
7709
  }
7519
- // One request-scoped transient-retry budget owner, declared here so BOTH the initial send
7520
- // and the later recovery refetches (429, key/account rotation, OAuth replay) share it. A
7521
- // per-leg budget would let a request that recovers several times multiply upstream load.
7522
- let transientSendsUsed = 0;
7523
- const noteTransientSends = (used: number): void => { transientSendsUsed += Math.max(0, used); };
7524
- const remainingTransientSendBudget = (budget: number): number =>
7525
- Math.max(1, budget - transientSendsUsed);
7526
7710
  try {
7527
7711
  initialRequest = await activeAdapter.buildRequest(parsed, { headers: selectedForwardHeaders, translatorBudget });
7528
7712
  refreshRequestToolAliases(initialRequest);
@@ -7565,6 +7749,7 @@ async function handleResponsesInner(
7565
7749
  upstreamResponse = await activeAdapter.fetchResponse(builtInitialRequest, {
7566
7750
  abortSignal: upstream.signal,
7567
7751
  timeoutMs: connectMs,
7752
+ sendBudget: adapterSendBudget,
7568
7753
  stream: parsed.stream,
7569
7754
  executor: providerFetch(route.provider, options.codexWsRuntimeIdentity, {
7570
7755
  dispatchOverride: oauthDispatch(builtInitialRequest),
@@ -7602,7 +7787,13 @@ async function handleResponsesInner(
7602
7787
  abortSignal: upstream.signal,
7603
7788
  label: safeHostLabel(builtInitialRequest.url),
7604
7789
  ...(transientPolicy
7605
- ? { attempts: transientPolicy.attempts, onSendsConsumed: noteTransientSends }
7790
+ // Draws the remainder, not the raw policy. A combo child inherits the parent's
7791
+ // holder but used to take a fresh full allowance on its own first send, so the
7792
+ // shared counter was inherited without ever being read as a limit.
7793
+ ? {
7794
+ attempts: remainingTransientSendBudget(transientPolicy.attempts),
7795
+ onSendsConsumed: noteTransientSends,
7796
+ }
7606
7797
  : {}),
7607
7798
  },
7608
7799
  );
@@ -7693,6 +7884,7 @@ async function handleResponsesInner(
7693
7884
  return await activeAdapter.fetchResponse(retryRequest, {
7694
7885
  abortSignal: upstream.signal,
7695
7886
  timeoutMs: connectMs,
7887
+ sendBudget: adapterSendBudget,
7696
7888
  stream: parsed.stream,
7697
7889
  executor: providerFetch(route.provider, options.codexWsRuntimeIdentity, {
7698
7890
  dispatchOverride: oauthDispatch(retryRequest),
@@ -7711,8 +7903,22 @@ async function handleResponsesInner(
7711
7903
  const refetchWithPolicy = (route.provider.adapter === "google" || refetchTransientPolicy)
7712
7904
  ? fetchWithTransientRetry
7713
7905
  : fetchWithResetRetry;
7906
+ // Same rule as the passthrough rebuild: spend the base allowance first, then the one
7907
+ // shared final-recovery reserve, so a recovery that follows a spent streak still gets
7908
+ // its single send instead of dying at three.
7909
+ const refetchAllowance = refetchTransientPolicy
7910
+ ? recoverySendAllowance(
7911
+ refetchTransientPolicy.attempts,
7912
+ recoveryClassFor(recovery),
7913
+ `${route.providerName}|${route.modelId}|${recovery}`,
7914
+ )
7915
+ : undefined;
7714
7916
  return await refetchWithPolicy(
7715
- recoveryKind => fetchWithHeaderTimeout(retryRequest.url,
7917
+ recoveryKind => {
7918
+ if (refetchAllowance?.permit && !refetchAllowance.permit.use()) {
7919
+ throw new SendBudgetExhaustedError(safeHostLabel(retryRequest.url));
7920
+ }
7921
+ return fetchWithHeaderTimeout(retryRequest.url,
7716
7922
  applyUpstreamRecoveryInit({
7717
7923
  method: retryRequest.method, headers: retryRequest.headers, body: retryRequest.body,
7718
7924
  }, recoveryKind), upstream.signal, connectMs, parsed.stream,
@@ -7720,13 +7926,14 @@ async function handleResponsesInner(
7720
7926
  dispatchOverride: oauthDispatch(retryRequest),
7721
7927
  providerName: route.providerName,
7722
7928
  modelId: route.modelId,
7723
- })),
7929
+ }));
7930
+ },
7724
7931
  {
7725
7932
  abortSignal: upstream.signal,
7726
7933
  label: safeHostLabel(retryRequest.url),
7727
- ...(refetchTransientPolicy
7934
+ ...(refetchAllowance
7728
7935
  ? {
7729
- attempts: remainingTransientSendBudget(refetchTransientPolicy.attempts),
7936
+ attempts: refetchAllowance.attempts,
7730
7937
  onSendsConsumed: noteTransientSends,
7731
7938
  }
7732
7939
  : {}),
@@ -7752,6 +7959,7 @@ async function handleResponsesInner(
7752
7959
  && isOAuth401ReplayProvider
7753
7960
  && sentOAuthSnapshot
7754
7961
  && !oauth401ReplayAttempted
7962
+ && !sendBudgetExhausted()
7755
7963
  ) {
7756
7964
  oauth401ReplayAttempted = true;
7757
7965
  try { void upstreamResponse.body?.cancel().catch(() => {}); } catch { /* already consumed/closed */ }
@@ -7846,6 +8054,7 @@ async function handleResponsesInner(
7846
8054
  upstreamResponse.status === 429
7847
8055
  && rateLimitPolicy !== null
7848
8056
  && rateLimitRetries < rateLimitPolicy.attempts
8057
+ && !sendBudgetExhausted()
7849
8058
  ) {
7850
8059
  rateLimitRetries += 1;
7851
8060
  // Release unread body + deliberate wait via the shared same-target helper.
@@ -8233,6 +8442,7 @@ async function handleResponsesInner(
8233
8442
  return await activeAdapter.fetchResponse(builtContinuationRequest, {
8234
8443
  abortSignal: upstream.signal,
8235
8444
  timeoutMs: connectMs,
8445
+ sendBudget: adapterSendBudget,
8236
8446
  stream: nextParsed.stream,
8237
8447
  executor: providerFetch(route.provider, options.codexWsRuntimeIdentity, {
8238
8448
  dispatchOverride: oauthDispatch(builtContinuationRequest, nextParsed),
@@ -4,9 +4,11 @@
4
4
  * codex-rs's built-in search client executes CLIENT-SIDE: it POSTs `alpha/search` against the
5
5
  * configured base_url with the same ChatGPT bearer auth used for model requests. Under Design B
6
6
  * injection base_url is this proxy, so the request otherwise dies on the /v1/* JSON-404 guard.
7
- * The endpoint is private to the ChatGPT Codex backend, so routed providers and OpenAI API-key
8
- * providers cannot serve it. Relay the JSON request and response verbatim through the configured
9
- * ChatGPT forward provider.
7
+ * The endpoint is private to the ChatGPT Codex backend, so the honest answer while a forward
8
+ * provider is configured is to copy bytes. When none is, a configured web-search sidecar
9
+ * (anthropic / xai / gemini / exa) can still answer — see src/web-search/alpha-search.ts.
10
+ * That fallback never runs while a forward candidate exists, and never borrows a different
11
+ * paid backend than the one the operator named.
10
12
  */
11
13
  import { formatErrorResponse } from "../bridge";
12
14
  import {
@@ -34,6 +36,7 @@ import {
34
36
  type ExactOpenAiSidecarAccount,
35
37
  } from "../providers/openai-sidecar";
36
38
  import { routeModel } from "../router";
39
+ import { handleAlphaSearchSidecarFallback } from "../web-search/alpha-search";
37
40
  import { readJsonRequestBody, resolveInboundBodyLimitBytes } from "./request-decompress";
38
41
  import { ForwardAdmissionCredentialError, validateForwardAdmissionCredential } from "./auth-cors";
39
42
  import type { RequestLogContext } from "./request-log";
@@ -105,12 +108,7 @@ export async function handleSearch(
105
108
  }
106
109
  const candidates = listOpenAiForwardSidecarCandidates(config);
107
110
  if (candidates.length === 0) {
108
- return formatErrorResponse(
109
- 400,
110
- "invalid_request_error",
111
- "Built-in web search needs a ChatGPT forward provider, but none is configured in opencodex. "
112
- + "Routed and OpenAI API-key providers cannot serve /v1/alpha/search.",
113
- );
111
+ return handleAlphaSearchSidecarFallback(body, config, req.signal, logCtx);
114
112
  }
115
113
 
116
114
  let upstream: Awaited<ReturnType<typeof resolveFirstUsableOpenAiSidecar>>;
@@ -668,6 +668,16 @@ export interface OcxConfig {
668
668
  * so absence is the only default state this feature has.
669
669
  */
670
670
  quotaResetNotify?: OcxQuotaResetNotifyConfig;
671
+ /**
672
+ * Periodic provider model-catalog refresh (issue #3630). Absent means off: no timer, no
673
+ * refresh pass, no outcome record.
674
+ *
675
+ * Off by default for the same reason every optional subsystem here is: a refresh spends a
676
+ * live /models call against every enabled provider, and this repository's rule is that a
677
+ * default install runs no detection code and starts no live timer work. Not in
678
+ * `getDefaultConfig()` — absence is the only default state this feature has.
679
+ */
680
+ catalogAutoRefresh?: OcxCatalogAutoRefreshConfig;
671
681
  /** Active provider context limits; native long windows remain within their supported ceilings. */
672
682
  providerContextCaps?: Record<string, number>;
673
683
  /** Last selected provider caps; retained while a cap is switched off. Not an active limit. */
@@ -850,14 +860,26 @@ export interface OcxConfig {
850
860
  pool?: {
851
861
  kernel?: boolean;
852
862
  /**
853
- * Opt-in cache-affinity ordering, off by default.
863
+ * Cache-affinity ordering for bound Codex threads. **On unless set to `false`.**
854
864
  *
855
- * With it on, a bound Codex thread keeps its account until that account genuinely cannot
856
- * serve, instead of moving the moment usage crosses `autoSwitchThreshold`. Moving a live
865
+ * A bound Codex thread keeps its account until that account genuinely cannot serve,
866
+ * instead of moving the moment usage crosses `autoSwitchThreshold`. Moving a live
857
867
  * conversation throws away the prompt cache warmed on that account, and a threshold
858
- * crossing is a hint rather than evidence the account is spent. Separate from `kernel`
859
- * on purpose: that one governs the generic OAuth strategy consumer, and one switch
860
- * carrying two unrelated meanings cannot be turned on alone.
868
+ * crossing is a hint rather than evidence the account is spent.
869
+ *
870
+ * This shipped as an opt-in (#4292) and then #4546 measured what the opt-in default
871
+ * costs: a pool whose accounts all sit in the 80-99% band hands a conversation from
872
+ * account to account, re-sending the whole prefix every turn, and the install that gets
873
+ * hurt is precisely the one that never heard of this setting. `false` restores
874
+ * capacity-first routing for operators who want it.
875
+ *
876
+ * Separate from `kernel` on purpose: that one governs the generic OAuth strategy
877
+ * consumer, and one switch carrying two unrelated meanings cannot be turned on alone.
878
+ *
879
+ * Note what this does NOT govern. Unbound placement still follows
880
+ * `autoSwitchThreshold` and the configured strategy. A bound thread's destination must
881
+ * have real headroom under either setting, and a transient failure streak holds the
882
+ * binding under either setting -- neither is a cache-affinity preference.
861
883
  */
862
884
  cacheAffinity?: boolean;
863
885
  };
@@ -1303,3 +1325,26 @@ export interface OcxQuotaResetNotifyConfig {
1303
1325
  */
1304
1326
  command?: string[];
1305
1327
  }
1328
+
1329
+ /**
1330
+ * Periodic model-catalog auto-refresh settings (issue #3630).
1331
+ *
1332
+ * Every field is optional and the whole section defaults to off. Each tick converges the
1333
+ * served catalog the same way `ocx sync` does, which costs a live /models call against
1334
+ * every enabled provider — so an install that never asked for this must run no refresh
1335
+ * code and start no timer, matching the optional-subsystem rule the rest of this file
1336
+ * follows.
1337
+ */
1338
+ export interface OcxCatalogAutoRefreshConfig {
1339
+ /** Master switch. Default false — no scheduler, no tick, no upstream calls. */
1340
+ enabled?: boolean;
1341
+ /**
1342
+ * Minutes between refresh ticks. Default 60, floor 15, and 0 keeps the timer dormant
1343
+ * while leaving the section configured.
1344
+ *
1345
+ * The floor exists for the same reason src/quota/reset-poller.ts has MIN_INTERVAL_MS:
1346
+ * provider catalogs are cached upstream for minutes, so a faster cadence buys no
1347
+ * freshness and only risks a rate limit against every enabled provider at once.
1348
+ */
1349
+ intervalMinutes?: number;
1350
+ }