@bitkyc08/opencodex 2.49.0 → 2.50.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/AGENTS_INSTALL.md +9 -1
  2. package/README.md +3 -0
  3. package/gui/dist/assets/index-C39tnjXO.js +115 -0
  4. package/gui/dist/index.html +1 -1
  5. package/package.json +1 -1
  6. package/src/claude/inbound.ts +17 -5
  7. package/src/cli/account-api.ts +18 -3
  8. package/src/cli/account-auth.ts +8 -1
  9. package/src/cli/account-extended.ts +2 -1
  10. package/src/cli/account.ts +1 -0
  11. package/src/cli/capabilities.ts +15 -1
  12. package/src/cli/index.ts +5 -1
  13. package/src/cli/models-runtime.ts +8 -3
  14. package/src/cli/observe.ts +13 -3
  15. package/src/clients/config-export/zcode.ts +24 -0
  16. package/src/codex/account-runtime-state.ts +6 -1
  17. package/src/codex/account-store.ts +72 -9
  18. package/src/codex/account-usability.ts +3 -2
  19. package/src/codex/auth-api.ts +107 -23
  20. package/src/codex/auth-context.ts +21 -0
  21. package/src/codex/catalog/parsing.ts +23 -0
  22. package/src/codex/catalog/provider-fetch.ts +71 -2
  23. package/src/codex/catalog/sync.ts +14 -0
  24. package/src/codex/inject.ts +3 -2
  25. package/src/codex/quota-auto-refresh.ts +6 -1
  26. package/src/codex/quota.ts +54 -8
  27. package/src/combos/index.ts +2 -0
  28. package/src/combos/resolve.ts +52 -0
  29. package/src/config.ts +58 -0
  30. package/src/generated/compatibility-version.json +72 -60
  31. package/src/lib/errors.ts +8 -0
  32. package/src/lib/privacy.ts +25 -0
  33. package/src/oauth/health.ts +47 -12
  34. package/src/oauth/index.ts +46 -8
  35. package/src/oauth/token-guardian.ts +32 -6
  36. package/src/providers/google-ai-studio-model-discovery.ts +74 -0
  37. package/src/providers/opencode-zen-rate-limit.ts +75 -0
  38. package/src/providers/quota.ts +15 -0
  39. package/src/providers/registry.ts +1 -1
  40. package/src/server/auth-cors.ts +6 -0
  41. package/src/server/chat-completions.ts +4 -4
  42. package/src/server/chat-native.ts +10 -1
  43. package/src/server/claude-messages.ts +5 -5
  44. package/src/server/images.ts +2 -2
  45. package/src/server/index.ts +25 -2
  46. package/src/server/management/logs-usage-routes.ts +4 -1
  47. package/src/server/management/model-rows.ts +16 -1
  48. package/src/server/management/oauth-account-routes.ts +6 -2
  49. package/src/server/management/provider-routes.ts +9 -2
  50. package/src/server/management/request-history-routes.ts +4 -2
  51. package/src/server/management/route-registry.ts +5 -4
  52. package/src/server/management/shared.ts +66 -3
  53. package/src/server/management-api.ts +1 -1
  54. package/src/server/request-decompress.ts +91 -3
  55. package/src/server/request-log.ts +10 -0
  56. package/src/server/responses/codex-ws-wire.ts +1 -1
  57. package/src/server/responses/compact.ts +8 -2
  58. package/src/server/responses/context-overflow.ts +11 -0
  59. package/src/server/responses/core.ts +144 -38
  60. package/src/server/responses/policy-fallback.ts +6 -2
  61. package/src/server/search.ts +2 -2
  62. package/src/service.ts +92 -7
  63. package/src/types/accounts.ts +18 -0
  64. package/src/types/config.ts +36 -0
  65. package/src/types/provider.ts +56 -0
  66. package/src/types.ts +4 -0
  67. package/src/web-search/ollama-executor.ts +127 -0
  68. package/src/web-search/passthrough-bridge.ts +761 -0
  69. package/gui/dist/assets/index-BtyONQrZ.js +0 -115
@@ -102,7 +102,7 @@ import {
102
102
  } from "../../lib/errors";
103
103
  import { injectionDebugLog } from "../../lib/injection-debug-log";
104
104
  import { resolveClientRetryAfter } from "../../lib/retry-after";
105
- import { enrichOpenCodeZenRateLimitMessage } from "../../providers/opencode-zen-rate-limit";
105
+ import { enrichOpenCodeZenUpstreamMessage } from "../../providers/opencode-zen-rate-limit";
106
106
  import { CODE_MODE_EXEC_TOOL_NAME, modelInList, namespacedToolName } from "../../types";
107
107
  import type {
108
108
  AdapterEvent,
@@ -150,6 +150,11 @@ import {
150
150
  } from "../../oauth/generic-account-failover";
151
151
  import { resolveCopilotApiBaseUrl } from "../../oauth/github-copilot";
152
152
  import { buildWebSearchTool, planWebSearch, runWithWebSearch, shouldResolveOpenAiWebSearchSidecar } from "../../web-search";
153
+ import {
154
+ createOllamaBridgeExecutor,
155
+ createPassthroughWebSearchBridgeStream,
156
+ planPassthroughWebSearchBridge,
157
+ } from "../../web-search/passthrough-bridge";
153
158
  import { buildImageTool, buildVideoTool, planImageBridge, planVideoBridge, runWithImageBridge, clampImageMaxRounds, IMAGE_GEN_TOOL_NAME, VIDEO_GEN_TOOL_NAME } from "../../images";
154
159
  import { describeImagesInPlace, isModelTextOnly, planVisionSidecar, resolveOpenAiVisionModel, shouldResolveOpenAiVisionSidecar, stripImagesInPlace } from "../../vision";
155
160
  import { createAdapterEventQueue, preflightAdapterEvents, type AdapterEventQueue } from "../../adapters/run-turn-queue";
@@ -243,7 +248,13 @@ import { hasPassiveAccountQuota, recordAnthropicAccountQuotaFromHeaders, recordP
243
248
  import { captureConfigGeneration } from "../../lib/state-store-sweeper";
244
249
  import { applyOpenAiVirtualModel, resolveOpenAiCompactModel } from "../../providers/openai-virtual-models";
245
250
  import { isUsageDebugEnabled } from "../../usage/debug";
246
- import { readJsonRequestBody, DecompressedBodyTooLargeError, UnsupportedContentEncodingError } from "../request-decompress";
251
+ import {
252
+ readJsonRequestBody,
253
+ describeInboundBodyRefusal,
254
+ resolveInboundBodyLimitBytes,
255
+ DecompressedBodyTooLargeError,
256
+ UnsupportedContentEncodingError,
257
+ } from "../request-decompress";
247
258
  import { resolveAdapter, resolveWireProtocolOverride } from "../adapter-resolve";
248
259
  import {
249
260
  providerModelResponsesTerminalRepair,
@@ -424,7 +435,7 @@ import {
424
435
  } from "../responses-undeclared-tool-guard";
425
436
  import { createGithubCopilotResponsesBlockRewrite } from "../github-copilot-responses-repair";
426
437
  import { responsesJsonToSseStream } from "../responses-json-events";
427
- import { streamingContextOverflowResponse } from "./context-overflow";
438
+ import { jsonContextOverflowResponse, streamingContextOverflowResponse } from "./context-overflow";
428
439
  import { guardTerminalEventStream } from "./terminal-guard";
429
440
  import {
430
441
  emptyCompletionRetryEnabled,
@@ -1012,14 +1023,14 @@ export function usesCodexForwardPoolAuth(
1012
1023
  && provider.authMode === "forward" && provider.adapter === "openai-responses";
1013
1024
  }
1014
1025
 
1015
- function codexWsQuotaObserver(authCtx: CodexAuthContext, provider: OcxProviderConfig): CodexWsQuotaObserver | undefined {
1026
+ function codexWsQuotaObserver(authCtx: CodexAuthContext, provider: OcxProviderConfig, modelId?: string): CodexWsQuotaObserver | undefined {
1016
1027
  if (!isCanonicalOpenAiForwardProvider(provider) || !usesCodexForwardPoolAuth(authCtx, provider)) return undefined;
1017
1028
  const { accountId, writerGeneration } = authCtx;
1018
1029
  const credentialGeneration = authCtx.kind === "pool" ? authCtx.generation : undefined;
1019
1030
  const mainWriter = authCtx.kind === "main-pool" ? authCtx.mainQuotaWriter : undefined;
1020
1031
  return headers => {
1021
1032
  if (credentialGeneration !== undefined && !isCodexAccountGenerationLive(accountId, credentialGeneration)) return;
1022
- applyCapturedCodexQuota(accountId, headers, writerGeneration, mainWriter);
1033
+ applyCapturedCodexQuota(accountId, headers, writerGeneration, mainWriter, { modelId });
1023
1034
  };
1024
1035
  }
1025
1036
 
@@ -1382,6 +1393,7 @@ async function retryCodexPoolOnAlternateAccount(
1382
1393
  firstResponse.headers,
1383
1394
  firstAuthCtx.writerGeneration,
1384
1395
  firstAuthCtx.kind === "main-pool" ? firstAuthCtx.mainQuotaWriter : undefined,
1396
+ { modelId: route.modelId },
1385
1397
  );
1386
1398
  }
1387
1399
  const deferFirstOutcome = shouldDeferCodexResetDerivedCooldown(
@@ -1473,7 +1485,7 @@ async function retryCodexPoolOnAlternateAccount(
1473
1485
  providerFetch(route.provider, options.codexWsRuntimeIdentity, {
1474
1486
  providerName: route.providerName,
1475
1487
  modelId: route.modelId,
1476
- onCodexWsQuota: codexWsQuotaObserver(retryAuthCtx, route.provider),
1488
+ onCodexWsQuota: codexWsQuotaObserver(retryAuthCtx, route.provider, route.modelId),
1477
1489
  beforeDispatch: isCanonicalOpenAiForwardProvider(route.provider)
1478
1490
  ? createCodexReserveDispatchGuard(retryAuthCtx, options.codexAuthPolicy ?? config, route.modelId, options.admission, options.visionDescribeTerminal === true) : undefined,
1479
1491
  }),
@@ -1607,7 +1619,7 @@ export function decodeRequestErrorResponse(err: unknown, label: string): Respons
1607
1619
  return formatErrorResponse(415, "invalid_request_error", err.message);
1608
1620
  }
1609
1621
  if (err instanceof DecompressedBodyTooLargeError) {
1610
- return formatErrorResponse(413, "invalid_request_error", err.message);
1622
+ return formatErrorResponse(413, "inbound_body_too_large", describeInboundBodyRefusal(err));
1611
1623
  }
1612
1624
  console.warn(`[${label}] request body decode/parse failed: ${err instanceof Error ? `${err.name}: ${err.message}` : String(err)}`);
1613
1625
  return formatErrorResponse(400, "invalid_request_error", "Invalid JSON body");
@@ -2779,6 +2791,10 @@ export async function handleComboResponses(
2779
2791
  logCtx.routeDecision = comboRouteDecisionTrace(config, comboId, pick, requestedModel);
2780
2792
 
2781
2793
  let lastFailure: Response | null = null;
2794
+ // The exhausted-combo mapping below runs outside the loop, where `failure.upstreamCode`
2795
+ // is gone, so carry the loop's own classification decision instead of re-deriving a
2796
+ // weaker one from the status alone (#4149).
2797
+ let lastFailureClassifiesOverflow = false;
2782
2798
  while (pick) {
2783
2799
  if (options.abortSignal?.aborted) return clientCancelledResponse();
2784
2800
  const childLog: RequestLogContext = {
@@ -2986,6 +3002,12 @@ export async function handleComboResponses(
2986
3002
  const failureDecision = comboFailureDecision(failure.response.status, failure.classificationText, {
2987
3003
  code: failure.upstreamCode,
2988
3004
  });
3005
+ const wantsStream = (rawBody as { stream?: unknown } | null)?.stream === true;
3006
+ // Local byte admission has its own diagnostic; do not relabel it as an upstream refusal.
3007
+ const classifyOverflow = failure.response.status === 413
3008
+ && (wantsStream || (failure.upstreamCode !== "outbound_body_too_large"
3009
+ && failure.upstreamCode !== "translation_buffer_limit"));
3010
+ lastFailureClassifiesOverflow = classifyOverflow;
2989
3011
  if (storedPool401ReplayDispatched) {
2990
3012
  if (failureDecision === "hop" && unreadableEncryptedAgentTask && !comboPayloadReadable) {
2991
3013
  const recoveredTarget = await pickWithWait({
@@ -3010,15 +3032,19 @@ export async function handleComboResponses(
3010
3032
  // Keep the spent Pool budget sticky even after a recovered routed child:
3011
3033
  // no later failure may reopen ordinary combo/native account hopping.
3012
3034
  adoptFailedChildLog(childLog);
3035
+ if (classifyOverflow && failureDecision === "stop") {
3036
+ return wantsStream
3037
+ ? streamingContextOverflowResponse(requestedModel, options.translatorBudget)
3038
+ : jsonContextOverflowResponse();
3039
+ }
3013
3040
  return lastFailure;
3014
3041
  }
3015
3042
  if (failureDecision === "stop") {
3016
3043
  adoptFailedChildLog(childLog);
3017
- if (
3018
- failure.response.status === 413
3019
- && (rawBody as { stream?: unknown } | null)?.stream === true
3020
- ) {
3021
- return streamingContextOverflowResponse(requestedModel, options.translatorBudget);
3044
+ if (classifyOverflow) {
3045
+ return wantsStream
3046
+ ? streamingContextOverflowResponse(requestedModel, options.translatorBudget)
3047
+ : jsonContextOverflowResponse();
3022
3048
  }
3023
3049
  return lastFailure;
3024
3050
  }
@@ -3068,9 +3094,11 @@ export async function handleComboResponses(
3068
3094
  }
3069
3095
  if (
3070
3096
  lastFailure?.status === 413
3071
- && (rawBody as { stream?: unknown } | null)?.stream === true
3097
+ && lastFailureClassifiesOverflow
3072
3098
  ) {
3073
- return streamingContextOverflowResponse(requestedModel, options.translatorBudget);
3099
+ return (rawBody as { stream?: unknown } | null)?.stream === true
3100
+ ? streamingContextOverflowResponse(requestedModel, options.translatorBudget)
3101
+ : jsonContextOverflowResponse();
3074
3102
  }
3075
3103
  return lastFailure!;
3076
3104
  }
@@ -3222,7 +3250,7 @@ async function handleResponsesInner(
3222
3250
  const agentTaskRecovery = agentTaskRecoveryConfig(config);
3223
3251
  let body: unknown;
3224
3252
  try {
3225
- body = await readJsonRequestBody(req, translatorBudget);
3253
+ body = await readJsonRequestBody(req, translatorBudget, resolveInboundBodyLimitBytes(config.maxInboundBodyBytes));
3226
3254
  } catch (err) {
3227
3255
  if (options.abortSignal?.aborted || req.signal.aborted) {
3228
3256
  return clientCancelledResponse();
@@ -3275,6 +3303,30 @@ async function handleResponsesInner(
3275
3303
  }
3276
3304
  }
3277
3305
  }
3306
+ // A shadow-call replacement that names a COMBO is routing policy, not the identity of any
3307
+ // one pick. The late intercept site below resolves it through routeModel/tryPickComboModel,
3308
+ // which collapses the table to a single target while still tagging `routeKind: "combo"`, so
3309
+ // the combo gate on the next line never fires, handleComboResponses never runs, and 429/5xx
3310
+ // hops — which only exist inside that loop — are unreachable (#4129). Rewrite the selector
3311
+ // here instead, before comboIdFromRawBody reads `model`, and identify the combo by CONFIG
3312
+ // LOOKUP so the check can never observe a one-candidate collapse.
3313
+ if (!options.comboAttempt && body && typeof body === "object" && !Array.isArray(body)) {
3314
+ const shadowIntercept = config.shadowCallIntercept;
3315
+ const rawShadowModel = (body as { model?: unknown }).model;
3316
+ if (shadowIntercept?.enabled && shadowIntercept.model && typeof rawShadowModel === "string"
3317
+ && isShadowSourceModel(rawShadowModel, shadowIntercept.sourceModels)) {
3318
+ const shadowComboId = resolveComboId(config, shadowIntercept.model);
3319
+ if (shadowComboId && Object.hasOwn(config.combos ?? {}, shadowComboId)) {
3320
+ (body as Record<string, unknown>).model = shadowIntercept.model;
3321
+ // Same rule as the late intercept site: record the operator-configured prefix that
3322
+ // matched, never the caller's raw model string. Matching is by prefix, so the raw
3323
+ // value is caller-controlled and reaches usage.jsonl and /api/logs.
3324
+ logCtx.shadowCallRewrittenFrom = sanitizeLogMetadataString(
3325
+ shadowSourceModelPrefix(rawShadowModel, shadowIntercept.sourceModels),
3326
+ );
3327
+ }
3328
+ }
3329
+ }
3278
3330
  const comboId = !options.comboAttempt ? comboIdFromRawBody(body, config) : null;
3279
3331
  if (comboId && Object.hasOwn(config.combos ?? {}, comboId)) {
3280
3332
  options.onRequestBodyRead?.();
@@ -3612,10 +3664,19 @@ async function handleResponsesInner(
3612
3664
  let recoveryFailureReason: AgentTaskRecoveryFailureReason | undefined;
3613
3665
  // Native fallback and explicitly trusted direct Responses routes can consume ciphertext,
3614
3666
  // so recover only after final route selection.
3667
+ //
3668
+ // Deliberately NOT gated on `threadSpawn` (#4089). Switching a live thread from a native
3669
+ // ChatGPT model to a routed provider replays a backend-minted encrypted agent message on every
3670
+ // later turn, and a model switch is not a spawn, so the spawn requirement failed the thread
3671
+ // closed permanently without ever attempting recovery. The trust boundary is
3672
+ // `recoveryAdmission()` in ./agent-task-recovery -- Codex originator, live native ChatGPT
3673
+ // bearer, matching chatgpt-account-id, no inbound API key, no proxy-admission secret -- which
3674
+ // admits only the owner of the session that would be spent. `threadSpawn` narrowed which of
3675
+ // that owner's own requests could use their own session; it kept nobody else out. The combo
3676
+ // gate above keeps its spawn requirement: that path has its own native-target filtering and
3677
+ // per-attempt failover, and the reported defect is on this path.
3615
3678
  if (
3616
3679
  inboundWire === "responses"
3617
- &&
3618
- threadSpawn
3619
3680
  && agentTaskRecovery
3620
3681
  && !isCanonicalOpenAiForwardProvider(route.provider)
3621
3682
  && !options.comboAttempt
@@ -5050,7 +5111,7 @@ async function handleResponsesInner(
5050
5111
  dispatchOverride: oauthDispatch(request),
5051
5112
  providerName: route.providerName,
5052
5113
  modelId: route.modelId,
5053
- onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider),
5114
+ onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider, route.modelId),
5054
5115
  beforeDispatch: isCanonicalOpenAiForwardProvider(route.provider)
5055
5116
  ? createCodexReserveDispatchGuard(authCtx, options.codexAuthPolicy ?? config, route.modelId, options.admission, options.visionDescribeTerminal === true) : undefined,
5056
5117
  }),
@@ -5128,7 +5189,7 @@ async function handleResponsesInner(
5128
5189
  dispatchOverride: oauthDispatch(request),
5129
5190
  providerName: route.providerName,
5130
5191
  modelId: route.modelId,
5131
- onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider),
5192
+ onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider, route.modelId),
5132
5193
  beforeDispatch: isCanonicalOpenAiForwardProvider(route.provider)
5133
5194
  ? createCodexReserveDispatchGuard(authCtx, options.codexAuthPolicy ?? config, route.modelId, options.admission, options.visionDescribeTerminal === true) : undefined,
5134
5195
  }),
@@ -5234,7 +5295,7 @@ async function handleResponsesInner(
5234
5295
  dispatchOverride: oauthDispatch(request),
5235
5296
  providerName: route.providerName,
5236
5297
  modelId: route.modelId,
5237
- onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider),
5298
+ onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider, route.modelId),
5238
5299
  beforeDispatch: isCanonicalOpenAiForwardProvider(route.provider)
5239
5300
  ? createCodexReserveDispatchGuard(authCtx, options.codexAuthPolicy ?? config, route.modelId, options.admission, options.visionDescribeTerminal === true) : undefined,
5240
5301
  }),
@@ -5354,7 +5415,7 @@ async function handleResponsesInner(
5354
5415
  dispatchOverride: oauthDispatch(request),
5355
5416
  providerName: route.providerName,
5356
5417
  modelId: route.modelId,
5357
- onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider),
5418
+ onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider, route.modelId),
5358
5419
  beforeDispatch: isCanonicalOpenAiForwardProvider(route.provider)
5359
5420
  ? createCodexReserveDispatchGuard(authCtx, options.codexAuthPolicy ?? config, route.modelId, options.admission, options.visionDescribeTerminal === true) : undefined,
5360
5421
  }),
@@ -5454,7 +5515,7 @@ async function handleResponsesInner(
5454
5515
  dispatchOverride: oauthDispatch(request),
5455
5516
  providerName: route.providerName,
5456
5517
  modelId: route.modelId,
5457
- onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider),
5518
+ onCodexWsQuota: codexWsQuotaObserver(authCtx, route.provider, route.modelId),
5458
5519
  beforeDispatch: isCanonicalOpenAiForwardProvider(route.provider)
5459
5520
  ? createCodexReserveDispatchGuard(authCtx, options.codexAuthPolicy ?? config, route.modelId, options.admission, options.visionDescribeTerminal === true) : undefined,
5460
5521
  }),
@@ -5658,7 +5719,8 @@ async function handleResponsesInner(
5658
5719
  const { applyAccountQuotaFromUpstreamHeaders } = await import("../../codex/auth-api");
5659
5720
  if (!isCodexWsQuotaObservedResponse(upstreamResponse)) {
5660
5721
  applyAccountQuotaFromUpstreamHeaders(authCtx.accountId, upstreamResponse.headers,
5661
- authCtx.writerGeneration, authCtx.kind === "main-pool" ? authCtx.mainQuotaWriter : undefined);
5722
+ authCtx.writerGeneration, authCtx.kind === "main-pool" ? authCtx.mainQuotaWriter : undefined,
5723
+ { modelId: route.modelId });
5662
5724
  }
5663
5725
  if (terminalBodyWillRecord) {
5664
5726
  options.setTerminalOutcomeRecorder?.((status, httpStatusOverride) => {
@@ -5699,7 +5761,8 @@ async function handleResponsesInner(
5699
5761
 
5700
5762
  // Non-2xx passthrough failures must never reach Codex as an empty body —
5701
5763
  // Codex renders that as the opaque "Unknown error" (#452). Combo attempts
5702
- // keep their typed failure envelope. Non-empty bodies are relayed verbatim
5764
+ // keep their typed failure envelope. Except for the classified 413 below,
5765
+ // non-empty bodies are relayed verbatim
5703
5766
  // (headers included) so pool-retry Activation B/D and client diagnostics stay intact.
5704
5767
  // Manual-redirect policy (#914): a 3xx is relayed as-is (Location preserved
5705
5768
  // through sanitizePassthroughHeaders) so a redirect to a dead host can never
@@ -5727,11 +5790,10 @@ async function handleResponsesInner(
5727
5790
  // The bounded reader owns the original body, deadline, abort settlement, and lock.
5728
5791
  // Unsafe partial data falls back to #452's non-empty status-only JSON.
5729
5792
  const errorText = await readDisplaySafeErrorText(upstreamResponse, upstream.signal, "");
5730
- if (upstreamResponse.status === 413 && clientRequestedStream) {
5731
- return streamingContextOverflowResponse(
5732
- parsed._responseModelId ?? parsed.modelId,
5733
- translatorBudget,
5734
- );
5793
+ if (upstreamResponse.status === 413) {
5794
+ return clientRequestedStream
5795
+ ? streamingContextOverflowResponse(parsed._responseModelId ?? parsed.modelId, translatorBudget)
5796
+ : jsonContextOverflowResponse();
5735
5797
  }
5736
5798
  return formatPassthroughUpstreamError(upstreamResponse.status, errorText, {
5737
5799
  statusText: upstreamResponse.statusText,
@@ -5760,15 +5822,60 @@ async function handleResponsesInner(
5760
5822
  route.provider,
5761
5823
  route.modelId,
5762
5824
  );
5825
+ // #3761: opt-in hosted-web-search bridge. Codex always declares the hosted web_search tool,
5826
+ // and this branch relays that declaration on the assumption the destination executes it.
5827
+ // A KEY-auth gateway that does not (Ollama Cloud GLM) answers with a function_call named
5828
+ // web_search that nothing runs, and the undeclared-tool guard below ends the turn. When the
5829
+ // provider opts in, the bridge intercepts that one call, runs the search, continues the
5830
+ // conversation upstream, and hands back ordinary Responses SSE — so every rewrite below,
5831
+ // including the guard itself, still inspects the client-facing stream. Default OFF: without
5832
+ // the opt-in this is one planner call and the relay is byte-identical to before.
5833
+ const webSearchBridgePlan = planPassthroughWebSearchBridge(parsed, route.provider, {
5834
+ isPassthrough: true,
5835
+ stream: parsed.stream === true,
5836
+ });
5837
+ // The bridge wraps the RAW upstream body, so terminal repair below still owns the single
5838
+ // client-facing terminal — the bridge drops the terminal of every intercepted leg.
5839
+ const upstreamSseBody = webSearchBridgePlan
5840
+ ? createPassthroughWebSearchBridgeStream({
5841
+ plan: webSearchBridgePlan,
5842
+ firstLeg: upstreamResponse.body,
5843
+ requestBody: request.body,
5844
+ // Continuation legs replay the same built request with the executed search appended.
5845
+ // The first leg already passed the recovery ladder, the outbound size ceiling, and the
5846
+ // host circuit; a KEY-auth destination has no OAuth refresh to replay on a later leg.
5847
+ send: (continuationBody: string) => fetchWithHeaderTimeout(
5848
+ request.url,
5849
+ { method: request.method, headers: request.headers, body: continuationBody },
5850
+ upstream.signal,
5851
+ connectMs,
5852
+ true,
5853
+ providerFetch(route.provider, options.codexWsRuntimeIdentity, {
5854
+ dispatchOverride: oauthDispatch(request),
5855
+ providerName: route.providerName,
5856
+ modelId: route.modelId,
5857
+ }),
5858
+ false,
5859
+ ),
5860
+ execute: createOllamaBridgeExecutor(webSearchBridgePlan, route.provider.apiKey ?? ""),
5861
+ // Appending a search result can push the continuation past the ceiling the first leg
5862
+ // was admitted under, so the same limit is re-applied before every later send.
5863
+ checkOutboundBody: (continuationBody: string) => {
5864
+ const result = checkOutboundBodySize(continuationBody, config.maxUpstreamBodyBytes);
5865
+ return result.admitted ? undefined : describeOutboundBodyRefusal(result);
5866
+ },
5867
+ signal: upstream.signal,
5868
+ })
5869
+ : upstreamResponse.body;
5763
5870
  const passthroughSseBody = terminalRepairPolicy
5764
5871
  ? relayResponsesSseWithTerminalRepair(
5765
- upstreamResponse.body,
5872
+ upstreamSseBody,
5766
5873
  upstream,
5767
5874
  terminalRepairPolicy,
5768
5875
  translatorBudget,
5769
5876
  options.responsesTerminalRepairScheduler,
5770
5877
  )
5771
- : upstreamResponse.body;
5878
+ : upstreamSseBody;
5772
5879
  const repairConfig = route.provider.responsesItemIdRepair;
5773
5880
  // Grok Build renders deltas live but reconstructs its durable assistant
5774
5881
  // turn from the completed response snapshot. Native Responses streams
@@ -7463,11 +7570,10 @@ async function handleResponsesInner(
7463
7570
  } finally {
7464
7571
  cleanupUpstreamAbort();
7465
7572
  }
7466
- if (upstreamResponse.status === 413 && clientRequestedStream && !options.comboAttempt) {
7467
- return streamingContextOverflowResponse(
7468
- parsed._responseModelId ?? parsed.modelId,
7469
- translatorBudget,
7470
- );
7573
+ if (upstreamResponse.status === 413) {
7574
+ return clientRequestedStream
7575
+ ? streamingContextOverflowResponse(parsed._responseModelId ?? parsed.modelId, translatorBudget)
7576
+ : jsonContextOverflowResponse();
7471
7577
  }
7472
7578
  if (!isFixedCodexAccount(authCtx)) {
7473
7579
  recordSubagentQuotaFailureForThreadSpawn(
@@ -7487,7 +7593,7 @@ async function handleResponsesInner(
7487
7593
  const message = normalized.cyberPolicy
7488
7594
  ? normalized.message
7489
7595
  ?? (isCyberPolicyCode(normalized.code) ? CYBER_POLICY_FALLBACK_MESSAGE : normalized.safeText)
7490
- : enrichOpenCodeZenRateLimitMessage(
7596
+ : enrichOpenCodeZenUpstreamMessage(
7491
7597
  `Provider error ${upstreamResponse.status}: ${normalized.safeText}`,
7492
7598
  {
7493
7599
  status: upstreamResponse.status,
@@ -7498,7 +7604,7 @@ async function handleResponsesInner(
7498
7604
  hasApiKey: Boolean(route.provider.apiKey?.trim()),
7499
7605
  upstreamRetryAfter,
7500
7606
  // This recovery path is the HTTP Responses wire; custom runTurn transports
7501
- // never reach enrichOpenCodeZenRateLimitMessage here.
7607
+ // never reach enrichOpenCodeZenUpstreamMessage here.
7502
7608
  supportsHttpSameKeyRetry: true,
7503
7609
  },
7504
7610
  );
@@ -1,6 +1,6 @@
1
1
  import { comboFailureDecision } from "../../combos/failover";
2
2
  import { readBoundedResponseBody } from "../../lib/bounded-body";
3
- import { readJsonRequestBody } from "../request-decompress";
3
+ import { readJsonRequestBody, resolveInboundBodyLimitBytes } from "../request-decompress";
4
4
  import { finishRequestAttempt, type RequestLogContext } from "../request-log";
5
5
  import type { OcxConfig } from "../../types";
6
6
  import type { RouteCandidateTrace, RouteDecisionTraceV1 } from "../../routing/trace";
@@ -145,7 +145,11 @@ export async function handleResponsesWithPolicyFallback(
145
145
  };
146
146
  let rawBody: Record<string, unknown> | null = null;
147
147
  try {
148
- const parsed = await readJsonRequestBody(req.clone());
148
+ const parsed = await readJsonRequestBody(
149
+ req.clone(),
150
+ undefined,
151
+ resolveInboundBodyLimitBytes(config.maxInboundBodyBytes),
152
+ );
149
153
  if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) rawBody = parsed as Record<string, unknown>;
150
154
  } catch {
151
155
  // Core owns the client-facing parse/decompression error.
@@ -33,7 +33,7 @@ import {
33
33
  type ExactOpenAiSidecarAccount,
34
34
  } from "../providers/openai-sidecar";
35
35
  import { routeModel } from "../router";
36
- import { readJsonRequestBody } from "./request-decompress";
36
+ import { readJsonRequestBody, resolveInboundBodyLimitBytes } from "./request-decompress";
37
37
  import { ForwardAdmissionCredentialError, validateForwardAdmissionCredential } from "./auth-cors";
38
38
  import type { RequestLogContext } from "./request-log";
39
39
  import { codexLogAccountId, decodeRequestErrorResponse } from "./responses";
@@ -64,7 +64,7 @@ export async function handleSearch(
64
64
  }
65
65
  let body: unknown;
66
66
  try {
67
- body = await readJsonRequestBody(req);
67
+ body = await readJsonRequestBody(req, undefined, resolveInboundBodyLimitBytes(config.maxInboundBodyBytes));
68
68
  } catch (err) {
69
69
  return decodeRequestErrorResponse(err, "search");
70
70
  }
package/src/service.ts CHANGED
@@ -865,9 +865,51 @@ export function resolvedProxyEnv(env: NodeJS.ProcessEnv = process.env): { name:
865
865
  }
866
866
 
867
867
  function sh(cmd: string): string {
868
+ assertLiveServiceManagerAllowed(cmd);
868
869
  return execSync(cmd, { encoding: "utf8", stdio: ["pipe", "pipe", "pipe"] }).trim();
869
870
  }
870
871
 
872
+ /**
873
+ * Service-manager invocations that only observe. Everything else changes a job that
874
+ * launchd or the systemd user manager is running right now.
875
+ */
876
+ const READ_ONLY_SERVICE_MANAGER = new RegExp(
877
+ "^(?:launchctl\\s+(?:list|print|print-disabled|blame|managerpid|manageruid)\\b"
878
+ + "|systemctl\\s+(?:--user\\s+)?(?:show|show-environment|status|is-active|is-enabled|is-failed|cat|list-units|list-unit-files|--version)\\b)",
879
+ );
880
+
881
+ const SERVICE_MANAGER_COMMAND = /^(?:launchctl|systemctl)\b/;
882
+
883
+ /**
884
+ * Refuse to mutate a live service manager from an armed test process.
885
+ *
886
+ * The test preload isolates HOME, OPENCODEX_HOME and CODEX_HOME, and that is enough for
887
+ * anything addressed by path. It is not enough here. `systemctl --user stop
888
+ * opencodex-proxy.service` is addressed by job NAME and talks to the user manager that is
889
+ * already running, so it stops the proxy the developer is actually using no matter what
890
+ * HOME says. `launchctl bootout gui/<uid>/com.opencodex.proxy` has the same shape.
891
+ *
892
+ * Windows already had this guard: `querySchtasks` refuses every non-query call while the
893
+ * test-home guard is armed, after a partially-faked test replaced a real scheduled task
894
+ * with a launcher inside a temporary test home. macOS and Linux were left without the
895
+ * equivalent, which means the person most likely to run this suite - someone running
896
+ * opencodex on the machine they are developing it on - is the person it can disrupt.
897
+ *
898
+ * Read-only verbs stay allowed: probing what the manager reports is the whole point of
899
+ * the diagnostics, and observation cannot take a service down.
900
+ */
901
+ export function assertLiveServiceManagerAllowed(command: string): void {
902
+ if (!isTestHomeGuardArmed()) return;
903
+ const trimmed = command.trim();
904
+ if (!SERVICE_MANAGER_COMMAND.test(trimmed)) return;
905
+ if (READ_ONLY_SERVICE_MANAGER.test(trimmed)) return;
906
+ throw new Error(
907
+ `refusing to run \`${trimmed}\` from an armed test process: launchd and the systemd user `
908
+ + "manager address a job by name, not by HOME, so this reaches the service the developer is "
909
+ + "actually running. Inject the service operation instead of calling the live manager.",
910
+ );
911
+ }
912
+
871
913
  /**
872
914
  * Run `launchctl` and report BOTH streams regardless of exit status.
873
915
  *
@@ -887,6 +929,9 @@ export function runLaunchctl(
887
929
  deps: { run?: typeof spawnSync } = {},
888
930
  ): { ok: boolean; stdout: string; stderr: string; status: number | null } {
889
931
  const run = deps.run ?? spawnSync;
932
+ // Only the real runner is guarded. Tests that inject a spawnSync stand-in are
933
+ // exercising the parsing, not reaching launchd, and must keep working.
934
+ if (run === spawnSync) assertLiveServiceManagerAllowed(`launchctl ${args.join(" ")}`);
890
935
  const result = run("/bin/launchctl", args, { encoding: "utf8", windowsHide: true });
891
936
  // `error` is set when the spawn itself failed (ENOENT off macOS) and `status` is
892
937
  // null for a signalled child; neither may be reported as success.
@@ -2293,7 +2338,23 @@ export function readWindowsSchedulerXmlState(
2293
2338
  }
2294
2339
 
2295
2340
  // ── macOS (launchd) ──
2296
- function installLaunchd(): void {
2341
+ /**
2342
+ * Deps follow {@link startLaunchd}: `launchctl` replaces the LAYER, returning a
2343
+ * {@link runLaunchctl} result, not a spawnSync result. It is optional so this stays
2344
+ * assignable to `ServiceOps.install` and `RepairServiceDeps.repairLaunchd`
2345
+ * (`() => void`), and so `platformOps` wires the same function the tests exercise.
2346
+ *
2347
+ * The seam is what makes the eviction below testable at all. The live-service-manager
2348
+ * guard refuses every mutating verb from an armed test process and `bootout` is not on
2349
+ * its read-only list, so a test reaching the real runner would fail closed on the guard
2350
+ * instead of exercising the sequence.
2351
+ *
2352
+ * No `matches` dep: unlike `startLaunchd`, this function never consults
2353
+ * {@link launchdJobMatchesPlist}. It has just rewritten the plist, so a live job is stale
2354
+ * by construction and there is nothing to compare against.
2355
+ */
2356
+ export function installLaunchd(deps: { launchctl?: typeof runLaunchctl } = {}): void {
2357
+ const run = deps.launchctl ?? runLaunchctl;
2297
2358
  const dir = join(homedir(), "Library", "LaunchAgents");
2298
2359
  if (!existsSync(dir)) mkdirSync(dir, { recursive: true });
2299
2360
  recordOwnedConfigPath(getConfigDir(), serviceStatePath());
@@ -2307,17 +2368,41 @@ function installLaunchd(): void {
2307
2368
  // so the staleness diagnostic judges exactly what launchd runs.
2308
2369
  const launcher = stableLauncherEntry();
2309
2370
  writeServiceDefinitionFile(p, buildPlist(resolvedProxyEnv(), { launcher }), "utf8");
2310
- // Best-effort: an absent job is fine here, and a failed unload is caught by the
2311
- // load verification below with a better message than a raw unload error.
2312
- runLaunchctl(["unload", p]);
2313
- const loaded = runLaunchctl(["load", "-w", p]);
2371
+ // `unload` is the legacy verb and it does not evict a job bootstrapped into the GUI
2372
+ // domain — which is precisely the state that could not repair itself. Modern launchd
2373
+ // answers `load -w` for an already-bootstrapped job with "Load failed: 5:
2374
+ // Input/output error" AND exits 0, so `ocx update` replaced the binary, ran repair,
2375
+ // and left launchd running the PREVIOUS job while the fresh plist sat unused (#4141).
2376
+ //
2377
+ // This EVICTS the running job. That is the repair being asked for, and it is why it
2378
+ // lives here and nowhere else: `installLaunchd` has already rewritten the plist, so
2379
+ // whatever is loaded is stale by construction. `ocx service start` must never do
2380
+ // this, and `startLaunchd` accordingly still refuses to.
2381
+ //
2382
+ // Absence is fine: booting out a job that is not there is a no-op, and a real failure
2383
+ // is reported by the load verification below with a better message than a raw
2384
+ // eviction error would carry.
2385
+ const bootoutTarget = `${launchdGuiDomain()}/${LABEL}`;
2386
+ run(["bootout", bootoutTarget]);
2387
+ let loaded = run(["load", "-w", p]);
2388
+ if (launchctlLoadFailed(loaded.stderr)) {
2389
+ // Still bootstrapped after an eviction: the job re-registered between the two calls,
2390
+ // or the first `bootout` raced a job that had not finished exiting. Evict and load
2391
+ // once more — ONCE. A bounded retry recovers the race; a loop would turn a genuinely
2392
+ // wedged domain into a hang instead of the diagnosable throw below.
2393
+ run(["bootout", bootoutTarget]);
2394
+ loaded = run(["load", "-w", p]);
2395
+ }
2314
2396
  if (!loaded.ok || launchctlLoadFailed(loaded.stderr)) {
2315
2397
  // Do NOT write install state for a load that did not take: state describing an
2316
2398
  // unused plist is what made this failure invisible.
2317
2399
  throw new Error(
2318
2400
  `launchctl could not load ${p}: ${loaded.stderr || "load reported failure"}\n`
2319
- + "A previous job may still be bootstrapped. Try:\n"
2320
- + ` launchctl bootout ${launchdGuiDomain()}/${LABEL}\n`
2401
+ // The hint used to tell the operator to run `bootout` by hand. It now runs twice
2402
+ // above, so naming it as an untried remedy would send someone to repeat what just
2403
+ // failed. Report what was attempted instead.
2404
+ + `A previous job is still bootstrapped after two attempts to boot it out of ${launchdGuiDomain()}.\n`
2405
+ + `Inspect it with:\n launchctl print ${bootoutTarget}\n`
2321
2406
  // macOS `service repair` delegates straight to installLaunchd, so this fires for
2322
2407
  // an already-installed service too; repair reloads it without re-registering.
2323
2408
  + `then re-run '${wasInstalled ? "ocx service repair" : "ocx service install"}'.`,
@@ -34,4 +34,22 @@ export interface CodexAccountCredentialRecord {
34
34
  lastCodexValidatedAt?: number;
35
35
  lastCodexValidationStatus?: "ok" | "failed";
36
36
  lastCodexValidationError?: string;
37
+ /** OAuth succeeded while quota was exhausted; never route until deferred validation succeeds. */
38
+ codexValidationPending?: boolean;
39
+ /**
40
+ * Set when the recorded failure is TERMINAL: the OAuth grant itself was revoked or has
41
+ * expired, so no retry can recover it and only a re-login will. It distinguishes a dead
42
+ * credential from a transient warmup or probe failure that may clear on its own.
43
+ *
44
+ * Deliberately a separate optional key rather than a third value in
45
+ * `lastCodexValidationStatus`: `isCredentialRecord` admits only `"ok" | "failed"`, so a
46
+ * record carrying an unrecognized status fails validation and is DROPPED from the store
47
+ * on load. An unknown extra key is carried through untouched instead, which keeps a
48
+ * downgrade from deleting the account entry and its credential.
49
+ *
50
+ * Cleared by `markCodexAccountValidated` and — because it is absent from
51
+ * `preservedValidationMetadata` — by every credential write. A refresh that succeeds
52
+ * disproves "the grant was revoked", so the verdict must not outlive it.
53
+ */
54
+ lastCodexValidationTerminal?: boolean;
37
55
  }
@@ -288,6 +288,26 @@ export interface OcxRemoteGuiConfig {
288
288
 
289
289
  export type OcxConnectedClientId = "codex" | "claude";
290
290
 
291
+ /**
292
+ * Redaction policy for management and CLI projections (#3859).
293
+ *
294
+ * `privacy` rather than `dashboard`: `ocx status` and `ocx account` are not the dashboard, and
295
+ * they read the same projections.
296
+ */
297
+ export interface OcxPrivacyConfig {
298
+ /**
299
+ * Mask stored account emails before they leave the proxy. Omitted or `true` is the historical
300
+ * behaviour and the default.
301
+ *
302
+ * Setting this to `false` is a real disclosure decision, not a display preference. Management
303
+ * is not always loopback — under `remoteGui` the unmasked address reaches every management
304
+ * principal that can reach the hub, not only someone sitting at the machine. The default
305
+ * therefore stays masked, and turning it off is an explicit opt-in by the operator who owns
306
+ * those accounts.
307
+ */
308
+ maskEmails?: boolean;
309
+ }
310
+
291
311
  export interface OcxClientConnectionConfig {
292
312
  serverUrl: string;
293
313
  managementUrl: string;
@@ -335,6 +355,8 @@ export interface OcxConfig {
335
355
  remoteGui?: OcxRemoteGuiConfig;
336
356
  /** Remote-hub client state. The admission secret is stored only in service-api-token. */
337
357
  client?: OcxClientConnectionConfig;
358
+ /** Operator-facing redaction policy for management and CLI projections. */
359
+ privacy?: OcxPrivacyConfig;
338
360
  /** Opt in to one identical-turn retry when a Responses completion has no text or tool call. */
339
361
  emptyCompletionRetry?: boolean;
340
362
  /**
@@ -801,6 +823,20 @@ export interface OcxConfig {
801
823
  * that work today — on Azure and custom Responses gateways as well, whose limits are unknown.
802
824
  */
803
825
  maxUpstreamBodyBytes?: number;
826
+ /**
827
+ * Opt-in ceiling, in bytes, on a decompressed INBOUND data-plane request body (#3573).
828
+ *
829
+ * Omitted or 0 = the built-in 256 MiB default. The lever exists because a session on the
830
+ * 922k-token opt-in window serializes its full history past that default, and the request
831
+ * that crosses it is Codex's own remote-compaction request — so the session hits 413 on the
832
+ * one operation that would have shrunk it and cannot recover.
833
+ *
834
+ * Bounded on purpose. `resolveInboundBodyLimitBytes()` clamps to
835
+ * [1 MiB, 512 MiB]; an unbounded inbound cap is a memory DoS because the reader materializes
836
+ * the body several times over. The Bun listener's own `maxRequestBodySize` is fixed when the
837
+ * server starts, so raising this takes effect on restart.
838
+ */
839
+ maxInboundBodyBytes?: number;
804
840
  /**
805
841
  * Opt-in Anthropic OAuth PROACTIVE routing (#294). Default OFF.
806
842
  * Sticky session affinity; new sessions may pick lowest known 5h usage.