@bitkyc08/opencodex 2.49.0 → 2.50.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/AGENTS_INSTALL.md +9 -1
  2. package/README.md +3 -0
  3. package/gui/dist/assets/index-C39tnjXO.js +115 -0
  4. package/gui/dist/index.html +1 -1
  5. package/package.json +1 -1
  6. package/src/claude/inbound.ts +17 -5
  7. package/src/cli/account-api.ts +18 -3
  8. package/src/cli/account-auth.ts +8 -1
  9. package/src/cli/account-extended.ts +2 -1
  10. package/src/cli/account.ts +1 -0
  11. package/src/cli/capabilities.ts +15 -1
  12. package/src/cli/index.ts +5 -1
  13. package/src/cli/models-runtime.ts +8 -3
  14. package/src/cli/observe.ts +13 -3
  15. package/src/clients/config-export/zcode.ts +24 -0
  16. package/src/codex/account-runtime-state.ts +6 -1
  17. package/src/codex/account-store.ts +72 -9
  18. package/src/codex/account-usability.ts +3 -2
  19. package/src/codex/auth-api.ts +107 -23
  20. package/src/codex/auth-context.ts +21 -0
  21. package/src/codex/catalog/parsing.ts +23 -0
  22. package/src/codex/catalog/provider-fetch.ts +71 -2
  23. package/src/codex/catalog/sync.ts +14 -0
  24. package/src/codex/inject.ts +3 -2
  25. package/src/codex/quota-auto-refresh.ts +6 -1
  26. package/src/codex/quota.ts +54 -8
  27. package/src/combos/index.ts +2 -0
  28. package/src/combos/resolve.ts +52 -0
  29. package/src/config.ts +58 -0
  30. package/src/generated/compatibility-version.json +72 -60
  31. package/src/lib/errors.ts +8 -0
  32. package/src/lib/privacy.ts +25 -0
  33. package/src/oauth/health.ts +47 -12
  34. package/src/oauth/index.ts +46 -8
  35. package/src/oauth/token-guardian.ts +32 -6
  36. package/src/providers/google-ai-studio-model-discovery.ts +74 -0
  37. package/src/providers/opencode-zen-rate-limit.ts +75 -0
  38. package/src/providers/quota.ts +15 -0
  39. package/src/providers/registry.ts +1 -1
  40. package/src/server/auth-cors.ts +6 -0
  41. package/src/server/chat-completions.ts +4 -4
  42. package/src/server/chat-native.ts +10 -1
  43. package/src/server/claude-messages.ts +5 -5
  44. package/src/server/images.ts +2 -2
  45. package/src/server/index.ts +25 -2
  46. package/src/server/management/logs-usage-routes.ts +4 -1
  47. package/src/server/management/model-rows.ts +16 -1
  48. package/src/server/management/oauth-account-routes.ts +6 -2
  49. package/src/server/management/provider-routes.ts +9 -2
  50. package/src/server/management/request-history-routes.ts +4 -2
  51. package/src/server/management/route-registry.ts +5 -4
  52. package/src/server/management/shared.ts +66 -3
  53. package/src/server/management-api.ts +1 -1
  54. package/src/server/request-decompress.ts +91 -3
  55. package/src/server/request-log.ts +10 -0
  56. package/src/server/responses/codex-ws-wire.ts +1 -1
  57. package/src/server/responses/compact.ts +8 -2
  58. package/src/server/responses/context-overflow.ts +11 -0
  59. package/src/server/responses/core.ts +144 -38
  60. package/src/server/responses/policy-fallback.ts +6 -2
  61. package/src/server/search.ts +2 -2
  62. package/src/service.ts +92 -7
  63. package/src/types/accounts.ts +18 -0
  64. package/src/types/config.ts +36 -0
  65. package/src/types/provider.ts +56 -0
  66. package/src/types.ts +4 -0
  67. package/src/web-search/ollama-executor.ts +127 -0
  68. package/src/web-search/passthrough-bridge.ts +761 -0
  69. package/gui/dist/assets/index-BtyONQrZ.js +0 -115
@@ -54,6 +54,7 @@ import {
54
54
  readBoundedDiscoveryJson,
55
55
  resolveProviderModelDiscovery,
56
56
  } from "../../providers/model-discovery";
57
+ import { extractGoogleAiStudioModelItems } from "../../providers/google-ai-studio-model-discovery";
57
58
  import { routedSlug, slugEquals } from "../../providers/slug-codec";
58
59
  import { clearAccountQuotaCache, clearProviderQuotaCache, fetchProviderQuotaReports } from "../../providers/quota";
59
60
  import { clearKeyCooldowns } from "../../providers/key-failover";
@@ -1428,13 +1429,19 @@ export async function handleProviderRoutes(ctx: ManagementContext): Promise<Resp
1428
1429
  return jsonResponse({ ok: false, latencyMs, error: "upstream CCA model discovery returned an unexpected shape" });
1429
1430
  }
1430
1431
  // OpenAI-style lists (and Together top-level arrays) use the same validation/dedupe/filter
1431
- // as catalog discovery. Google's /v1beta/models uses `models[].name` and remains a
1432
- // connectivity-only count because it is not an authoritative catalog source.
1432
+ // as catalog discovery. Google AI Studio parses the native `models[]` envelope and filters to
1433
+ // `generateContent`, while other providers fall back to generic envelope rows if they return `models[]`.
1433
1434
  const record = bounded.value !== null && typeof bounded.value === "object" && !Array.isArray(bounded.value)
1434
1435
  ? bounded.value as Record<string, unknown>
1435
1436
  : undefined;
1437
+ const isAiStudio = effectiveGoogleMode(name, prov) === "ai-studio";
1438
+ const googleAiStudio = !ccaModels && isAiStudio
1439
+ ? extractGoogleAiStudioModelItems(bounded.value, discovery.maxModels)
1440
+ : undefined;
1436
1441
  const extracted = ccaModels
1437
1442
  ? undefined
1443
+ : googleAiStudio?.ok
1444
+ ? googleAiStudio
1438
1445
  : Array.isArray(bounded.value) || Array.isArray(record?.data)
1439
1446
  ? extractProviderModelItems(bounded.value, discovery)
1440
1447
  : extractModelEnvelopeRows(bounded.value, discovery.maxModels, ["models"]);
@@ -106,7 +106,9 @@ export async function handleRequestHistoryRoutes(ctx: ManagementContext): Promis
106
106
  to,
107
107
  }, cursor, limit);
108
108
  return jsonResponse({
109
- entries: page.rows.map(row => requestLogDto(requestLogEntryFromPersistedUsage(row))),
109
+ // The decode rate is a Logs-page metric; this endpoint shares the DTO but not its
110
+ // contract, so it opts out rather than silently widening its own response shape (#4038).
111
+ entries: page.rows.map(row => requestLogDto(requestLogEntryFromPersistedUsage(row), { includeDecodeRate: false })),
110
112
  ...(page.nextCursor ? { nextCursor: page.nextCursor } : {}),
111
113
  hasMore: page.hasMore,
112
114
  index: {
@@ -184,7 +186,7 @@ export async function handleRequestHistoryRoutes(ctx: ManagementContext): Promis
184
186
  if (!entry) {
185
187
  return jsonResponse({ error: { code: "not_found", message: "unknown request" } }, 404, req, config);
186
188
  }
187
- return jsonResponse(requestLogDto(requestLogEntryFromPersistedUsage(entry)), 200, req, config);
189
+ return jsonResponse(requestLogDto(requestLogEntryFromPersistedUsage(entry), { includeDecodeRate: false }), 200, req, config);
188
190
  }
189
191
 
190
192
  return null;
@@ -95,6 +95,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
95
95
  { method: "PATCH", path: "/api/codex-auth/pool-strategy", module: "codex/auth-api", mutates: true },
96
96
  { method: "POST", path: "/api/codex-auth/accounts", module: "codex/auth-api", mutates: true },
97
97
  { method: "POST", path: "/api/codex-auth/accounts/clear-cooldown", module: "codex/auth-api", mutates: true },
98
+ { method: "POST", path: "/api/codex-auth/accounts/refresh", module: "codex/auth-api", mutates: true },
98
99
  { method: "POST", path: "/api/codex-auth/login", module: "codex/auth-api", mutates: true },
99
100
  { method: "POST", path: "/api/codex-auth/login/cancel", module: "codex/auth-api", mutates: true },
100
101
  { method: "POST", path: "/api/codex-auth/login/code", module: "codex/auth-api", mutates: true },
@@ -148,9 +149,9 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
148
149
  { method: "GET", path: "/api/client-integrations/aside/profiles/{profileId}", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
149
150
  { method: "PUT", path: "/api/client-integrations/aside/profiles/{profileId}", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode" },
150
151
  { method: "GET", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
151
- { method: "DELETE", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside history deletion uses the dashboard journal cleanup; the CLI has history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_plan/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
152
+ { method: "DELETE", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside history deletion uses the dashboard journal cleanup; the CLI has history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
152
153
  { method: "GET", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
153
- { method: "DELETE", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside profile history deletion uses the dashboard journal cleanup; the CLI has scoped history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_plan/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
154
+ { method: "DELETE", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside profile history deletion uses the dashboard journal cleanup; the CLI has scoped history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
154
155
  { method: "POST", path: "/api/client-integrations/aside/profiles/{profileId}/restore", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode" },
155
156
  // server/management/codex-prompt-routes
156
157
  { method: "GET", path: "/api/codex-prompt", module: "server/management/codex-prompt-routes", mutates: false },
@@ -186,7 +187,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
186
187
  // server/management/integration-routes
187
188
  { method: "GET", path: "/api/client-integrations", module: "server/management/integration-routes", mutates: false },
188
189
  { method: "GET", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: false },
189
- { method: "DELETE", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Retiring one rollback row is a dashboard-local cleanup; the CLI verb that would drive it is owed by a later work-phase and is not implemented here.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_plan/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
190
+ { method: "DELETE", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Retiring one rollback row is a dashboard-local cleanup; the CLI verb that would drive it is owed by a later work-phase and is not implemented here.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
190
191
  { method: "POST", path: "/api/client-integrations/restore", module: "server/management/integration-routes", mutates: true },
191
192
  // server/management/lab-automation-routes
192
193
  { method: "GET", path: "/api/lab/automation", module: "server/management/lab-automation-routes", mutates: false, exempt: { reason: "local-transport", why: "ocx lab reads the same rows from the local SQLite projection; src/cli/lab.ts imports ../lab/query directly and never fetches /api/lab." } },
@@ -292,7 +293,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
292
293
  { method: "PATCH", path: "/api/providers", module: "server/management/provider-routes", mutates: true },
293
294
  { method: "POST", path: "/api/providers", module: "server/management/provider-routes", mutates: true },
294
295
  { method: "POST", path: "/api/providers/test", module: "server/management/provider-routes", mutates: true },
295
- { method: "PUT", path: "/api/providers", module: "server/management/provider-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Issue #3280 scopes this atomic batch endpoint to the GUI JSON editor; a matching CLI verb is outside wp5 and remains owed.", owner: "wp5-followup", ownerDoc: "devlog/_plan/260903_bug_drawdown_bcda/050_phase5.md" } },
296
+ { method: "PUT", path: "/api/providers", module: "server/management/provider-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Issue #3280 scopes this atomic batch endpoint to the GUI JSON editor; a matching CLI verb is outside wp5 and remains owed.", owner: "wp5-followup", ownerDoc: "devlog/_fin/260903_bug_drawdown_bcda/050_phase5.md" } },
296
297
  { method: "PUT", path: "/api/provider-context-caps", module: "server/management/provider-routes", mutates: true },
297
298
  // server/management/quota-reset-routes
298
299
  { method: "GET", path: "/api/quota-resets", module: "server/management/quota-reset-routes", mutates: false, mechanism: "negated-guard" },
@@ -79,7 +79,8 @@ export function parseDebugLogQuery(url: URL): { after: number; limit: number } {
79
79
  export type MetricUnavailableReason =
80
80
  | "usage_missing" | "usage_unsupported" | "output_missing" | "invalid_duration"
81
81
  | "price_unmatched" | "invalid_cache_breakdown"
82
- | "invalid_usage" | "combo_attempt_unavailable";
82
+ | "invalid_usage" | "combo_attempt_unavailable"
83
+ | "ttft_missing" | "decode_window_too_short";
83
84
 
84
85
  export type TokPerSecondResult =
85
86
  | { kind: "value"; value: number; estimated: boolean }
@@ -96,7 +97,7 @@ export type CostResult =
96
97
  | { kind: "value"; estimate: NonNullable<ReturnType<typeof estimateRequestCost>>; estimateReasons: CostEstimateReason[] }
97
98
  | { kind: "unavailable"; reason: MetricUnavailableReason };
98
99
 
99
- export type MetricSource = Pick<RequestLogEntry, "provider" | "model" | "durationMs" | "usageStatus" | "usage" | "requestedServiceTier" | "configuredServiceTier" | "responseServiceTier" | "tierOutcome" | "routeDecision"> & {
100
+ export type MetricSource = Pick<RequestLogEntry, "provider" | "model" | "durationMs" | "firstOutputMs" | "usageStatus" | "usage" | "requestedServiceTier" | "configuredServiceTier" | "responseServiceTier" | "tierOutcome" | "routeDecision"> & {
100
101
  attempts?: readonly PersistedUsageAttempt[];
101
102
  };
102
103
 
@@ -113,6 +114,51 @@ export function tokPerSecondResult(entry: Pick<MetricSource, "durationMs" | "usa
113
114
  return { kind: "value", value, estimated: entry.usageStatus === "estimated" || entry.usage.estimated === true };
114
115
  }
115
116
 
117
+ /**
118
+ * Shortest post-TTFT window that can carry a decode-rate estimate (#4038).
119
+ *
120
+ * Below one second the window is dominated by things that are not decoding: TTFT jitter, the
121
+ * proxy's own buffering, and the granularity of the timestamps themselves. A 240-token response
122
+ * whose first token arrived 50 ms before the last one is not a 4800 tok/s model, and printing
123
+ * that number is worse than printing nothing — which is precisely why the earlier attempt at
124
+ * this metric (#4040) was closed as an unreliable estimate.
125
+ */
126
+ export const MIN_DECODE_WINDOW_MS = 1_000;
127
+
128
+ /**
129
+ * Estimated DECODE throughput: output tokens over the window after the first token (#4038).
130
+ *
131
+ * Strictly additive. `tokensPerSecond`, `tokPerSecondResult`, `RequestLogEntry` and
132
+ * `usage.jsonl` are untouched, and the end-to-end rate beside it keeps meaning exactly what it
133
+ * has always meant — it is documented as end-to-end, so this is a missing metric rather than a
134
+ * miscalculated one.
135
+ *
136
+ * Always `estimated: true`. The proxy's TTFT is when the FIRST BYTE reached the proxy, which is
137
+ * not the provider's own generation start, so this can never be more than an estimate no matter
138
+ * how long the window is. Saying so in the payload is the honest half of the answer to #4040;
139
+ * MIN_DECODE_WINDOW_MS is the other half.
140
+ */
141
+ export function decodeTokPerSecondResult(
142
+ entry: Pick<MetricSource, "durationMs" | "firstOutputMs" | "usageStatus" | "usage">,
143
+ ): TokPerSecondResult {
144
+ if (!entry.usage) return { kind: "unavailable", reason: "usage_missing" };
145
+ if (entry.usageStatus === "unsupported") return { kind: "unavailable", reason: "usage_unsupported" };
146
+ if (entry.usage.outputTokens <= 0) return { kind: "unavailable", reason: "output_missing" };
147
+ // A row that predates TTFT capture, or a non-streaming turn that never recorded one, has no
148
+ // window to measure. That is a different fact from a bad duration, so it gets its own reason.
149
+ if (entry.firstOutputMs === undefined) return { kind: "unavailable", reason: "ttft_missing" };
150
+ if (!Number.isFinite(entry.firstOutputMs) || entry.firstOutputMs < 0 || !Number.isFinite(entry.durationMs)) {
151
+ return { kind: "unavailable", reason: "invalid_duration" };
152
+ }
153
+ const windowMs = entry.durationMs - entry.firstOutputMs;
154
+ // TTFT at or past the total duration means the two clocks disagree; there is no window.
155
+ if (windowMs <= 0) return { kind: "unavailable", reason: "invalid_duration" };
156
+ if (windowMs < MIN_DECODE_WINDOW_MS) return { kind: "unavailable", reason: "decode_window_too_short" };
157
+ const value = tokensPerSecond(entry.usage.outputTokens, windowMs);
158
+ if (value === null) return { kind: "unavailable", reason: "invalid_duration" };
159
+ return { kind: "value", value, estimated: true };
160
+ }
161
+
116
162
  export function unavailableCostReason(entry: MetricSource): MetricUnavailableReason {
117
163
  // Normalizer-first classification: the landed normalizer recovers legacy
118
164
  // cachedInputTokens=read+write rows via retry, so a raw read+write>input
@@ -152,11 +198,26 @@ export function costResult(entry: MetricSource): CostResult {
152
198
  return { kind: "value", estimate, estimateReasons };
153
199
  }
154
200
 
155
- export function requestLogDto(entry: RequestLogEntry): Record<string, unknown> {
201
+ /**
202
+ * `/api/logs` row projection.
203
+ *
204
+ * `includeDecodeRate` exists because `/api/request-history` shares this DTO but not its
205
+ * contract (#4038). The value would be meaningful there — `firstOutputMs` does survive into a
206
+ * persisted-usage row — so this is a scope decision, not a correctness one: the decode rate is
207
+ * a Logs-page metric, and widening a separate endpoint's response shape is not this change's
208
+ * business. Flipping it on later is one argument.
209
+ */
210
+ export function requestLogDto(
211
+ entry: RequestLogEntry,
212
+ { includeDecodeRate = true }: { includeDecodeRate?: boolean } = {},
213
+ ): Record<string, unknown> {
156
214
  return {
157
215
  ...entry,
158
216
  displayMetrics: {
159
217
  tokPerSecond: tokPerSecondResult(entry),
218
+ // The parent uses the REQUEST's own TTFT. A combo parent must not borrow an attempt's,
219
+ // which would measure a window the parent never had.
220
+ ...(includeDecodeRate ? { decodeTokPerSecond: decodeTokPerSecondResult(entry) } : {}),
160
221
  cost: costResult(entry),
161
222
  },
162
223
  ...(entry.attempts?.length
@@ -165,6 +226,8 @@ export function requestLogDto(entry: RequestLogEntry): Record<string, unknown> {
165
226
  ...attempt,
166
227
  displayMetrics: {
167
228
  tokPerSecond: tokPerSecondResult(attempt),
229
+ // Each attempt measures its own attempt-relative TTFT.
230
+ ...(includeDecodeRate ? { decodeTokPerSecond: decodeTokPerSecondResult(attempt) } : {}),
168
231
  cost: costResult({ ...attempt, attempts: undefined, routeDecision: entry.routeDecision, requestedServiceTier: entry.requestedServiceTier, configuredServiceTier: entry.configuredServiceTier, responseServiceTier: entry.responseServiceTier }),
169
232
  },
170
233
  })),
@@ -385,7 +385,7 @@ export async function handleManagementAPI(
385
385
  const { ConfigMutationLockError } = await import("../config");
386
386
  const { CodexCredentialRefreshLockTimeoutError } = await import("../codex/account-store");
387
387
  try {
388
- return await handleCodexAuthAPI(req, url, config, convergeCodexCatalog);
388
+ return await handleCodexAuthAPI(req, url, config, convergeCodexCatalog, principal);
389
389
  } catch (error) {
390
390
  // Credential writers remap ConfigMutationLockError to CodexCredentialRefreshLockTimeoutError;
391
391
  // treat both as the same retryable busy response.
@@ -21,6 +21,58 @@ import type { TranslatorBudget } from "../lib/translator-budget";
21
21
  */
22
22
  export const MAX_DECOMPRESSED_BODY_BYTES = 256 * 1024 * 1024;
23
23
 
24
+ /**
25
+ * Hard ceiling on the opt-in `maxInboundBodyBytes` (#3573).
26
+ *
27
+ * The opt-in exists because a 922k-token session serializes past the 256 MiB default, and the
28
+ * request that crosses it is the compaction request itself — so the session can no longer
29
+ * shrink and is stuck. An UNBOUNDED inbound cap is not an acceptable answer: this admission
30
+ * limit is the only thing standing between one request and the process heap, and
31
+ * `readBoundedJsonRequestBody` materializes the body several times over (retained wire bytes,
32
+ * decoded bytes, the decoded string, the re-encoded measurement copies, and the parsed object
33
+ * graph), so peak RSS is a MULTIPLE of whatever is admitted here. 512 MiB is the largest value
34
+ * that keeps that multiple survivable on an ordinary machine, and it is what #3573 asked for.
35
+ */
36
+ export const MAX_CONFIGURABLE_INBOUND_BODY_BYTES = 512 * 1024 * 1024;
37
+
38
+ /** Floor for the opt-in. Below this an ordinary multi-image turn cannot be admitted at all. */
39
+ export const MIN_CONFIGURABLE_INBOUND_BODY_BYTES = 1024 * 1024;
40
+
41
+ /**
42
+ * Resolve the configured inbound admission limit, clamped to the supported range.
43
+ *
44
+ * Pure and total on purpose: the schema in `src/config.ts` degrades an invalid hand edit to
45
+ * `undefined` rather than failing the parse, so the schema cannot be the place the ceiling is
46
+ * enforced. Every caller resolves through here, which makes this the single auditable bound
47
+ * regardless of how the config object was produced.
48
+ *
49
+ * Omitted, zero, or non-finite = the 256 MiB default, so an unconfigured proxy admits exactly
50
+ * what it admits today.
51
+ */
52
+ export function resolveInboundBodyLimitBytes(configured: number | undefined): number {
53
+ if (configured === undefined || !Number.isFinite(configured) || configured <= 0) {
54
+ return MAX_DECOMPRESSED_BODY_BYTES;
55
+ }
56
+ return Math.min(
57
+ Math.max(Math.floor(configured), MIN_CONFIGURABLE_INBOUND_BODY_BYTES),
58
+ MAX_CONFIGURABLE_INBOUND_BODY_BYTES,
59
+ );
60
+ }
61
+
62
+ /**
63
+ * Render a byte count, or nothing at all. `DecompressedBodyTooLargeError` accepts non-finite
64
+ * and untyped values from legacy callers and deliberately keeps them out of its own message;
65
+ * the client-facing message inherits that rule rather than printing `NaN MB`.
66
+ */
67
+ function megabytes(bytes: number): string | null {
68
+ return Number.isFinite(bytes) && bytes >= 0 && bytes <= Number.MAX_SAFE_INTEGER
69
+ ? (bytes / (1024 * 1024)).toFixed(1)
70
+ : null;
71
+ }
72
+
73
+ const INBOUND_CEILING_MB = (MAX_CONFIGURABLE_INBOUND_BODY_BYTES / (1024 * 1024)).toFixed(1);
74
+
75
+
24
76
  export class UnsupportedContentEncodingError extends Error {
25
77
  constructor(readonly encoding: string) {
26
78
  super(`Unsupported content-encoding: ${encoding}`);
@@ -54,6 +106,33 @@ export class DecompressedBodyTooLargeError extends Error {
54
106
  }
55
107
  }
56
108
 
109
+ /**
110
+ * Name OpenCodex as the refuser, and name the lever.
111
+ *
112
+ * #4112 gave the UPSTREAM context refusal on `/v1/responses` its own HTTP 413 with
113
+ * `context_length_exceeded`. That makes the two 413s on this surface look alike to a client
114
+ * while having opposite remedies: the upstream one means the provider will not take the turn,
115
+ * this one means the proxy never read it and a config key would have let it through. The
116
+ * wording deliberately avoids "context window"/"context length", which `classifyError` treats
117
+ * as evidence of an upstream context verdict.
118
+ */
119
+ export function describeInboundBodyRefusal(error: DecompressedBodyTooLargeError): string {
120
+ // A lower-bound measurement stopped counting at the cap; reporting it as exact would be a lie.
121
+ const approximate = error.measurement === "declared_wire" || error.measurement === "decoded_exact"
122
+ ? "" : "at least ";
123
+ const observed = megabytes(error.bytes);
124
+ const limit = megabytes(error.limit);
125
+ const sizes = limit === null
126
+ ? "the body is above the inbound admission limit"
127
+ : observed === null
128
+ ? `the body is above the ${limit} MB inbound admission limit`
129
+ : `the body is ${approximate}${observed} MB, above the ${limit} MB inbound admission limit`;
130
+ return `OpenCodex refused this request before reading it: ${sizes}. `
131
+ + "This is a local proxy limit, not a provider refusal. Raise \"maxInboundBodyBytes\" in "
132
+ + `config.json (ceiling ${INBOUND_CEILING_MB} MB) and restart the proxy, or compact the `
133
+ + "conversation earlier.";
134
+ }
135
+
57
136
  function assertBodySizeWithinLimit(
58
137
  body: Uint8Array,
59
138
  maxBytes: number,
@@ -259,7 +338,16 @@ export async function readBoundedJsonRequestBody(
259
338
  }
260
339
  }
261
340
 
262
- /** Parse a JSON data-plane body using the shared 256 MiB admission cap. */
263
- export function readJsonRequestBody(req: Request, budget?: TranslatorBudget): Promise<unknown> {
264
- return readBoundedJsonRequestBody(req, MAX_DECOMPRESSED_BODY_BYTES, budget);
341
+ /**
342
+ * Parse a JSON data-plane body using the shared admission cap.
343
+ *
344
+ * `maxBytes` is the resolved per-deployment limit from `resolveInboundBodyLimitBytes()`;
345
+ * omitting it keeps the 256 MiB default for callers with no config in scope.
346
+ */
347
+ export function readJsonRequestBody(
348
+ req: Request,
349
+ budget?: TranslatorBudget,
350
+ maxBytes: number = MAX_DECOMPRESSED_BODY_BYTES,
351
+ ): Promise<unknown> {
352
+ return readBoundedJsonRequestBody(req, maxBytes, budget);
265
353
  }
@@ -1119,6 +1119,16 @@ export function filterRequestLogs(logs: RequestLogEntry[], params: URLSearchPara
1119
1119
  filtered = filtered.filter(entry => entry.model === model
1120
1120
  || entry.attempts?.some(attempt => attempt.model === model));
1121
1121
  }
1122
+ // #4057: "which account served this request" is the first question asked when one provider
1123
+ // holds several accounts, and until now the only way to answer it was to grep usage.jsonl by
1124
+ // hand. Attempts are matched for the same reason `provider` and `model` match them: when a
1125
+ // request failed over between pool accounts, a search for the account that finally served it
1126
+ // has to find that request, not only the account that first refused it.
1127
+ const account = params.get("account")?.trim();
1128
+ if (account) {
1129
+ filtered = filtered.filter(entry => entry.accountLogLabel === account
1130
+ || entry.attempts?.some(attempt => attempt.accountLogLabel === account));
1131
+ }
1122
1132
  const status = params.get("status")?.trim().toLowerCase();
1123
1133
  if (status) {
1124
1134
  filtered = /^[1-5]xx$/.test(status)
@@ -2,7 +2,7 @@ import { MAX_CLIENT_SSE_FRAME_BYTES } from "../sse-frame-buffer";
2
2
  // If the 101 never arrives (network black hole), give SSE a chance well before
3
3
  // the caller's connect timeout (default 200s) would fire.
4
4
  export const UPGRADE_DEADLINE_MS = 10_000;
5
- export const CODEX_WS_RESPONSE_PRELUDE_TIMEOUT_MS = 30_000;
5
+ export const CODEX_WS_RESPONSE_PRELUDE_TIMEOUT_MS = 90_000;
6
6
  // Keep the push-based WS transport inside the same memory envelope as the
7
7
  // bounded SSE relays that consume this response. Unlike fetch response bodies,
8
8
  // a WebSocket cannot be paused when a ReadableStream applies backpressure, so
@@ -104,7 +104,12 @@ import { fastPolicyForModel } from "../../providers/service-tier";
104
104
  import { parseFastOnlyRowId } from "../fast-row";
105
105
  import { applyOpenAiVirtualModel, resolveOpenAiCompactModel } from "../../providers/openai-virtual-models";
106
106
  import { isUsageDebugEnabled } from "../../usage/debug";
107
- import { readJsonRequestBody, DecompressedBodyTooLargeError, UnsupportedContentEncodingError } from "../request-decompress";
107
+ import {
108
+ readJsonRequestBody,
109
+ resolveInboundBodyLimitBytes,
110
+ DecompressedBodyTooLargeError,
111
+ UnsupportedContentEncodingError,
112
+ } from "../request-decompress";
108
113
  import { resolveAdapter, resolveWireProtocolOverride } from "../adapter-resolve";
109
114
  import { hasKeyPoolFailover, rotateProviderTransportOn429 } from "../../providers/key-failover";
110
115
  import { shouldAttemptImageTierRetry } from "../image-retry";
@@ -522,7 +527,7 @@ export async function handleResponsesCompact(
522
527
  ): Promise<Response> {
523
528
  let body: unknown;
524
529
  try {
525
- body = await readJsonRequestBody(req);
530
+ body = await readJsonRequestBody(req, undefined, resolveInboundBodyLimitBytes(config.maxInboundBodyBytes));
526
531
  } catch (err) {
527
532
  return decodeRequestErrorResponse(err, "responses-compact");
528
533
  }
@@ -1015,6 +1020,7 @@ export async function handleResponsesCompact(
1015
1020
  upstream.headers,
1016
1021
  authCtx.writerGeneration,
1017
1022
  authCtx.kind === "main-pool" ? authCtx.mainQuotaWriter : undefined,
1023
+ { modelId: route.modelId },
1018
1024
  );
1019
1025
  }
1020
1026
  recordCompactPoolOutcome(authCtx, upstream.status, {
@@ -5,6 +5,17 @@ import type { AdapterEvent } from "../../types";
5
5
  export const PROVIDER_INPUT_TOO_LARGE_MESSAGE =
6
6
  "The provider rejected this turn because its input exceeds the provider size or context limit. Reduce the current input or compact the conversation before retrying.";
7
7
 
8
+ /** Preserve non-streaming HTTP failure semantics without exposing an upstream body. */
9
+ export function jsonContextOverflowResponse(): Response {
10
+ return Response.json({
11
+ error: {
12
+ message: PROVIDER_INPUT_TOO_LARGE_MESSAGE,
13
+ type: "invalid_request_error",
14
+ code: "context_length_exceeded",
15
+ },
16
+ }, { status: 413, headers: { "Cache-Control": "no-store" } });
17
+ }
18
+
8
19
  async function* contextOverflowEvents(): AsyncGenerator<AdapterEvent> {
9
20
  yield {
10
21
  type: "error",