@bitkyc08/opencodex 2.49.0 → 2.50.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS_INSTALL.md +9 -1
- package/README.md +3 -0
- package/gui/dist/assets/index-C39tnjXO.js +115 -0
- package/gui/dist/index.html +1 -1
- package/package.json +1 -1
- package/src/claude/inbound.ts +17 -5
- package/src/cli/account-api.ts +18 -3
- package/src/cli/account-auth.ts +8 -1
- package/src/cli/account-extended.ts +2 -1
- package/src/cli/account.ts +1 -0
- package/src/cli/capabilities.ts +15 -1
- package/src/cli/index.ts +5 -1
- package/src/cli/models-runtime.ts +8 -3
- package/src/cli/observe.ts +13 -3
- package/src/clients/config-export/zcode.ts +24 -0
- package/src/codex/account-runtime-state.ts +6 -1
- package/src/codex/account-store.ts +72 -9
- package/src/codex/account-usability.ts +3 -2
- package/src/codex/auth-api.ts +107 -23
- package/src/codex/auth-context.ts +21 -0
- package/src/codex/catalog/parsing.ts +23 -0
- package/src/codex/catalog/provider-fetch.ts +71 -2
- package/src/codex/catalog/sync.ts +14 -0
- package/src/codex/inject.ts +3 -2
- package/src/codex/quota-auto-refresh.ts +6 -1
- package/src/codex/quota.ts +54 -8
- package/src/combos/index.ts +2 -0
- package/src/combos/resolve.ts +52 -0
- package/src/config.ts +58 -0
- package/src/generated/compatibility-version.json +72 -60
- package/src/lib/errors.ts +8 -0
- package/src/lib/privacy.ts +25 -0
- package/src/oauth/health.ts +47 -12
- package/src/oauth/index.ts +46 -8
- package/src/oauth/token-guardian.ts +32 -6
- package/src/providers/google-ai-studio-model-discovery.ts +74 -0
- package/src/providers/opencode-zen-rate-limit.ts +75 -0
- package/src/providers/quota.ts +15 -0
- package/src/providers/registry.ts +1 -1
- package/src/server/auth-cors.ts +6 -0
- package/src/server/chat-completions.ts +4 -4
- package/src/server/chat-native.ts +10 -1
- package/src/server/claude-messages.ts +5 -5
- package/src/server/images.ts +2 -2
- package/src/server/index.ts +25 -2
- package/src/server/management/logs-usage-routes.ts +4 -1
- package/src/server/management/model-rows.ts +16 -1
- package/src/server/management/oauth-account-routes.ts +6 -2
- package/src/server/management/provider-routes.ts +9 -2
- package/src/server/management/request-history-routes.ts +4 -2
- package/src/server/management/route-registry.ts +5 -4
- package/src/server/management/shared.ts +66 -3
- package/src/server/management-api.ts +1 -1
- package/src/server/request-decompress.ts +91 -3
- package/src/server/request-log.ts +10 -0
- package/src/server/responses/codex-ws-wire.ts +1 -1
- package/src/server/responses/compact.ts +8 -2
- package/src/server/responses/context-overflow.ts +11 -0
- package/src/server/responses/core.ts +144 -38
- package/src/server/responses/policy-fallback.ts +6 -2
- package/src/server/search.ts +2 -2
- package/src/service.ts +92 -7
- package/src/types/accounts.ts +18 -0
- package/src/types/config.ts +36 -0
- package/src/types/provider.ts +56 -0
- package/src/types.ts +4 -0
- package/src/web-search/ollama-executor.ts +127 -0
- package/src/web-search/passthrough-bridge.ts +761 -0
- package/gui/dist/assets/index-BtyONQrZ.js +0 -115
|
@@ -54,6 +54,7 @@ import {
|
|
|
54
54
|
readBoundedDiscoveryJson,
|
|
55
55
|
resolveProviderModelDiscovery,
|
|
56
56
|
} from "../../providers/model-discovery";
|
|
57
|
+
import { extractGoogleAiStudioModelItems } from "../../providers/google-ai-studio-model-discovery";
|
|
57
58
|
import { routedSlug, slugEquals } from "../../providers/slug-codec";
|
|
58
59
|
import { clearAccountQuotaCache, clearProviderQuotaCache, fetchProviderQuotaReports } from "../../providers/quota";
|
|
59
60
|
import { clearKeyCooldowns } from "../../providers/key-failover";
|
|
@@ -1428,13 +1429,19 @@ export async function handleProviderRoutes(ctx: ManagementContext): Promise<Resp
|
|
|
1428
1429
|
return jsonResponse({ ok: false, latencyMs, error: "upstream CCA model discovery returned an unexpected shape" });
|
|
1429
1430
|
}
|
|
1430
1431
|
// OpenAI-style lists (and Together top-level arrays) use the same validation/dedupe/filter
|
|
1431
|
-
// as catalog discovery. Google
|
|
1432
|
-
//
|
|
1432
|
+
// as catalog discovery. Google AI Studio parses the native `models[]` envelope and filters to
|
|
1433
|
+
// `generateContent`, while other providers fall back to generic envelope rows if they return `models[]`.
|
|
1433
1434
|
const record = bounded.value !== null && typeof bounded.value === "object" && !Array.isArray(bounded.value)
|
|
1434
1435
|
? bounded.value as Record<string, unknown>
|
|
1435
1436
|
: undefined;
|
|
1437
|
+
const isAiStudio = effectiveGoogleMode(name, prov) === "ai-studio";
|
|
1438
|
+
const googleAiStudio = !ccaModels && isAiStudio
|
|
1439
|
+
? extractGoogleAiStudioModelItems(bounded.value, discovery.maxModels)
|
|
1440
|
+
: undefined;
|
|
1436
1441
|
const extracted = ccaModels
|
|
1437
1442
|
? undefined
|
|
1443
|
+
: googleAiStudio?.ok
|
|
1444
|
+
? googleAiStudio
|
|
1438
1445
|
: Array.isArray(bounded.value) || Array.isArray(record?.data)
|
|
1439
1446
|
? extractProviderModelItems(bounded.value, discovery)
|
|
1440
1447
|
: extractModelEnvelopeRows(bounded.value, discovery.maxModels, ["models"]);
|
|
@@ -106,7 +106,9 @@ export async function handleRequestHistoryRoutes(ctx: ManagementContext): Promis
|
|
|
106
106
|
to,
|
|
107
107
|
}, cursor, limit);
|
|
108
108
|
return jsonResponse({
|
|
109
|
-
|
|
109
|
+
// The decode rate is a Logs-page metric; this endpoint shares the DTO but not its
|
|
110
|
+
// contract, so it opts out rather than silently widening its own response shape (#4038).
|
|
111
|
+
entries: page.rows.map(row => requestLogDto(requestLogEntryFromPersistedUsage(row), { includeDecodeRate: false })),
|
|
110
112
|
...(page.nextCursor ? { nextCursor: page.nextCursor } : {}),
|
|
111
113
|
hasMore: page.hasMore,
|
|
112
114
|
index: {
|
|
@@ -184,7 +186,7 @@ export async function handleRequestHistoryRoutes(ctx: ManagementContext): Promis
|
|
|
184
186
|
if (!entry) {
|
|
185
187
|
return jsonResponse({ error: { code: "not_found", message: "unknown request" } }, 404, req, config);
|
|
186
188
|
}
|
|
187
|
-
return jsonResponse(requestLogDto(requestLogEntryFromPersistedUsage(entry)), 200, req, config);
|
|
189
|
+
return jsonResponse(requestLogDto(requestLogEntryFromPersistedUsage(entry), { includeDecodeRate: false }), 200, req, config);
|
|
188
190
|
}
|
|
189
191
|
|
|
190
192
|
return null;
|
|
@@ -95,6 +95,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
|
|
|
95
95
|
{ method: "PATCH", path: "/api/codex-auth/pool-strategy", module: "codex/auth-api", mutates: true },
|
|
96
96
|
{ method: "POST", path: "/api/codex-auth/accounts", module: "codex/auth-api", mutates: true },
|
|
97
97
|
{ method: "POST", path: "/api/codex-auth/accounts/clear-cooldown", module: "codex/auth-api", mutates: true },
|
|
98
|
+
{ method: "POST", path: "/api/codex-auth/accounts/refresh", module: "codex/auth-api", mutates: true },
|
|
98
99
|
{ method: "POST", path: "/api/codex-auth/login", module: "codex/auth-api", mutates: true },
|
|
99
100
|
{ method: "POST", path: "/api/codex-auth/login/cancel", module: "codex/auth-api", mutates: true },
|
|
100
101
|
{ method: "POST", path: "/api/codex-auth/login/code", module: "codex/auth-api", mutates: true },
|
|
@@ -148,9 +149,9 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
|
|
|
148
149
|
{ method: "GET", path: "/api/client-integrations/aside/profiles/{profileId}", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
|
|
149
150
|
{ method: "PUT", path: "/api/client-integrations/aside/profiles/{profileId}", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode" },
|
|
150
151
|
{ method: "GET", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
|
|
151
|
-
{ method: "DELETE", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside history deletion uses the dashboard journal cleanup; the CLI has history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/
|
|
152
|
+
{ method: "DELETE", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside history deletion uses the dashboard journal cleanup; the CLI has history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
|
|
152
153
|
{ method: "GET", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
|
|
153
|
-
{ method: "DELETE", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside profile history deletion uses the dashboard journal cleanup; the CLI has scoped history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/
|
|
154
|
+
{ method: "DELETE", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside profile history deletion uses the dashboard journal cleanup; the CLI has scoped history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
|
|
154
155
|
{ method: "POST", path: "/api/client-integrations/aside/profiles/{profileId}/restore", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode" },
|
|
155
156
|
// server/management/codex-prompt-routes
|
|
156
157
|
{ method: "GET", path: "/api/codex-prompt", module: "server/management/codex-prompt-routes", mutates: false },
|
|
@@ -186,7 +187,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
|
|
|
186
187
|
// server/management/integration-routes
|
|
187
188
|
{ method: "GET", path: "/api/client-integrations", module: "server/management/integration-routes", mutates: false },
|
|
188
189
|
{ method: "GET", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: false },
|
|
189
|
-
{ method: "DELETE", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Retiring one rollback row is a dashboard-local cleanup; the CLI verb that would drive it is owed by a later work-phase and is not implemented here.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/
|
|
190
|
+
{ method: "DELETE", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Retiring one rollback row is a dashboard-local cleanup; the CLI verb that would drive it is owed by a later work-phase and is not implemented here.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
|
|
190
191
|
{ method: "POST", path: "/api/client-integrations/restore", module: "server/management/integration-routes", mutates: true },
|
|
191
192
|
// server/management/lab-automation-routes
|
|
192
193
|
{ method: "GET", path: "/api/lab/automation", module: "server/management/lab-automation-routes", mutates: false, exempt: { reason: "local-transport", why: "ocx lab reads the same rows from the local SQLite projection; src/cli/lab.ts imports ../lab/query directly and never fetches /api/lab." } },
|
|
@@ -292,7 +293,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
|
|
|
292
293
|
{ method: "PATCH", path: "/api/providers", module: "server/management/provider-routes", mutates: true },
|
|
293
294
|
{ method: "POST", path: "/api/providers", module: "server/management/provider-routes", mutates: true },
|
|
294
295
|
{ method: "POST", path: "/api/providers/test", module: "server/management/provider-routes", mutates: true },
|
|
295
|
-
{ method: "PUT", path: "/api/providers", module: "server/management/provider-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Issue #3280 scopes this atomic batch endpoint to the GUI JSON editor; a matching CLI verb is outside wp5 and remains owed.", owner: "wp5-followup", ownerDoc: "devlog/
|
|
296
|
+
{ method: "PUT", path: "/api/providers", module: "server/management/provider-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Issue #3280 scopes this atomic batch endpoint to the GUI JSON editor; a matching CLI verb is outside wp5 and remains owed.", owner: "wp5-followup", ownerDoc: "devlog/_fin/260903_bug_drawdown_bcda/050_phase5.md" } },
|
|
296
297
|
{ method: "PUT", path: "/api/provider-context-caps", module: "server/management/provider-routes", mutates: true },
|
|
297
298
|
// server/management/quota-reset-routes
|
|
298
299
|
{ method: "GET", path: "/api/quota-resets", module: "server/management/quota-reset-routes", mutates: false, mechanism: "negated-guard" },
|
|
@@ -79,7 +79,8 @@ export function parseDebugLogQuery(url: URL): { after: number; limit: number } {
|
|
|
79
79
|
export type MetricUnavailableReason =
|
|
80
80
|
| "usage_missing" | "usage_unsupported" | "output_missing" | "invalid_duration"
|
|
81
81
|
| "price_unmatched" | "invalid_cache_breakdown"
|
|
82
|
-
| "invalid_usage" | "combo_attempt_unavailable"
|
|
82
|
+
| "invalid_usage" | "combo_attempt_unavailable"
|
|
83
|
+
| "ttft_missing" | "decode_window_too_short";
|
|
83
84
|
|
|
84
85
|
export type TokPerSecondResult =
|
|
85
86
|
| { kind: "value"; value: number; estimated: boolean }
|
|
@@ -96,7 +97,7 @@ export type CostResult =
|
|
|
96
97
|
| { kind: "value"; estimate: NonNullable<ReturnType<typeof estimateRequestCost>>; estimateReasons: CostEstimateReason[] }
|
|
97
98
|
| { kind: "unavailable"; reason: MetricUnavailableReason };
|
|
98
99
|
|
|
99
|
-
export type MetricSource = Pick<RequestLogEntry, "provider" | "model" | "durationMs" | "usageStatus" | "usage" | "requestedServiceTier" | "configuredServiceTier" | "responseServiceTier" | "tierOutcome" | "routeDecision"> & {
|
|
100
|
+
export type MetricSource = Pick<RequestLogEntry, "provider" | "model" | "durationMs" | "firstOutputMs" | "usageStatus" | "usage" | "requestedServiceTier" | "configuredServiceTier" | "responseServiceTier" | "tierOutcome" | "routeDecision"> & {
|
|
100
101
|
attempts?: readonly PersistedUsageAttempt[];
|
|
101
102
|
};
|
|
102
103
|
|
|
@@ -113,6 +114,51 @@ export function tokPerSecondResult(entry: Pick<MetricSource, "durationMs" | "usa
|
|
|
113
114
|
return { kind: "value", value, estimated: entry.usageStatus === "estimated" || entry.usage.estimated === true };
|
|
114
115
|
}
|
|
115
116
|
|
|
117
|
+
/**
|
|
118
|
+
* Shortest post-TTFT window that can carry a decode-rate estimate (#4038).
|
|
119
|
+
*
|
|
120
|
+
* Below one second the window is dominated by things that are not decoding: TTFT jitter, the
|
|
121
|
+
* proxy's own buffering, and the granularity of the timestamps themselves. A 240-token response
|
|
122
|
+
* whose first token arrived 50 ms before the last one is not a 4800 tok/s model, and printing
|
|
123
|
+
* that number is worse than printing nothing — which is precisely why the earlier attempt at
|
|
124
|
+
* this metric (#4040) was closed as an unreliable estimate.
|
|
125
|
+
*/
|
|
126
|
+
export const MIN_DECODE_WINDOW_MS = 1_000;
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Estimated DECODE throughput: output tokens over the window after the first token (#4038).
|
|
130
|
+
*
|
|
131
|
+
* Strictly additive. `tokensPerSecond`, `tokPerSecondResult`, `RequestLogEntry` and
|
|
132
|
+
* `usage.jsonl` are untouched, and the end-to-end rate beside it keeps meaning exactly what it
|
|
133
|
+
* has always meant — it is documented as end-to-end, so this is a missing metric rather than a
|
|
134
|
+
* miscalculated one.
|
|
135
|
+
*
|
|
136
|
+
* Always `estimated: true`. The proxy's TTFT is when the FIRST BYTE reached the proxy, which is
|
|
137
|
+
* not the provider's own generation start, so this can never be more than an estimate no matter
|
|
138
|
+
* how long the window is. Saying so in the payload is the honest half of the answer to #4040;
|
|
139
|
+
* MIN_DECODE_WINDOW_MS is the other half.
|
|
140
|
+
*/
|
|
141
|
+
export function decodeTokPerSecondResult(
|
|
142
|
+
entry: Pick<MetricSource, "durationMs" | "firstOutputMs" | "usageStatus" | "usage">,
|
|
143
|
+
): TokPerSecondResult {
|
|
144
|
+
if (!entry.usage) return { kind: "unavailable", reason: "usage_missing" };
|
|
145
|
+
if (entry.usageStatus === "unsupported") return { kind: "unavailable", reason: "usage_unsupported" };
|
|
146
|
+
if (entry.usage.outputTokens <= 0) return { kind: "unavailable", reason: "output_missing" };
|
|
147
|
+
// A row that predates TTFT capture, or a non-streaming turn that never recorded one, has no
|
|
148
|
+
// window to measure. That is a different fact from a bad duration, so it gets its own reason.
|
|
149
|
+
if (entry.firstOutputMs === undefined) return { kind: "unavailable", reason: "ttft_missing" };
|
|
150
|
+
if (!Number.isFinite(entry.firstOutputMs) || entry.firstOutputMs < 0 || !Number.isFinite(entry.durationMs)) {
|
|
151
|
+
return { kind: "unavailable", reason: "invalid_duration" };
|
|
152
|
+
}
|
|
153
|
+
const windowMs = entry.durationMs - entry.firstOutputMs;
|
|
154
|
+
// TTFT at or past the total duration means the two clocks disagree; there is no window.
|
|
155
|
+
if (windowMs <= 0) return { kind: "unavailable", reason: "invalid_duration" };
|
|
156
|
+
if (windowMs < MIN_DECODE_WINDOW_MS) return { kind: "unavailable", reason: "decode_window_too_short" };
|
|
157
|
+
const value = tokensPerSecond(entry.usage.outputTokens, windowMs);
|
|
158
|
+
if (value === null) return { kind: "unavailable", reason: "invalid_duration" };
|
|
159
|
+
return { kind: "value", value, estimated: true };
|
|
160
|
+
}
|
|
161
|
+
|
|
116
162
|
export function unavailableCostReason(entry: MetricSource): MetricUnavailableReason {
|
|
117
163
|
// Normalizer-first classification: the landed normalizer recovers legacy
|
|
118
164
|
// cachedInputTokens=read+write rows via retry, so a raw read+write>input
|
|
@@ -152,11 +198,26 @@ export function costResult(entry: MetricSource): CostResult {
|
|
|
152
198
|
return { kind: "value", estimate, estimateReasons };
|
|
153
199
|
}
|
|
154
200
|
|
|
155
|
-
|
|
201
|
+
/**
|
|
202
|
+
* `/api/logs` row projection.
|
|
203
|
+
*
|
|
204
|
+
* `includeDecodeRate` exists because `/api/request-history` shares this DTO but not its
|
|
205
|
+
* contract (#4038). The value would be meaningful there — `firstOutputMs` does survive into a
|
|
206
|
+
* persisted-usage row — so this is a scope decision, not a correctness one: the decode rate is
|
|
207
|
+
* a Logs-page metric, and widening a separate endpoint's response shape is not this change's
|
|
208
|
+
* business. Flipping it on later is one argument.
|
|
209
|
+
*/
|
|
210
|
+
export function requestLogDto(
|
|
211
|
+
entry: RequestLogEntry,
|
|
212
|
+
{ includeDecodeRate = true }: { includeDecodeRate?: boolean } = {},
|
|
213
|
+
): Record<string, unknown> {
|
|
156
214
|
return {
|
|
157
215
|
...entry,
|
|
158
216
|
displayMetrics: {
|
|
159
217
|
tokPerSecond: tokPerSecondResult(entry),
|
|
218
|
+
// The parent uses the REQUEST's own TTFT. A combo parent must not borrow an attempt's,
|
|
219
|
+
// which would measure a window the parent never had.
|
|
220
|
+
...(includeDecodeRate ? { decodeTokPerSecond: decodeTokPerSecondResult(entry) } : {}),
|
|
160
221
|
cost: costResult(entry),
|
|
161
222
|
},
|
|
162
223
|
...(entry.attempts?.length
|
|
@@ -165,6 +226,8 @@ export function requestLogDto(entry: RequestLogEntry): Record<string, unknown> {
|
|
|
165
226
|
...attempt,
|
|
166
227
|
displayMetrics: {
|
|
167
228
|
tokPerSecond: tokPerSecondResult(attempt),
|
|
229
|
+
// Each attempt measures its own attempt-relative TTFT.
|
|
230
|
+
...(includeDecodeRate ? { decodeTokPerSecond: decodeTokPerSecondResult(attempt) } : {}),
|
|
168
231
|
cost: costResult({ ...attempt, attempts: undefined, routeDecision: entry.routeDecision, requestedServiceTier: entry.requestedServiceTier, configuredServiceTier: entry.configuredServiceTier, responseServiceTier: entry.responseServiceTier }),
|
|
169
232
|
},
|
|
170
233
|
})),
|
|
@@ -385,7 +385,7 @@ export async function handleManagementAPI(
|
|
|
385
385
|
const { ConfigMutationLockError } = await import("../config");
|
|
386
386
|
const { CodexCredentialRefreshLockTimeoutError } = await import("../codex/account-store");
|
|
387
387
|
try {
|
|
388
|
-
return await handleCodexAuthAPI(req, url, config, convergeCodexCatalog);
|
|
388
|
+
return await handleCodexAuthAPI(req, url, config, convergeCodexCatalog, principal);
|
|
389
389
|
} catch (error) {
|
|
390
390
|
// Credential writers remap ConfigMutationLockError to CodexCredentialRefreshLockTimeoutError;
|
|
391
391
|
// treat both as the same retryable busy response.
|
|
@@ -21,6 +21,58 @@ import type { TranslatorBudget } from "../lib/translator-budget";
|
|
|
21
21
|
*/
|
|
22
22
|
export const MAX_DECOMPRESSED_BODY_BYTES = 256 * 1024 * 1024;
|
|
23
23
|
|
|
24
|
+
/**
|
|
25
|
+
* Hard ceiling on the opt-in `maxInboundBodyBytes` (#3573).
|
|
26
|
+
*
|
|
27
|
+
* The opt-in exists because a 922k-token session serializes past the 256 MiB default, and the
|
|
28
|
+
* request that crosses it is the compaction request itself — so the session can no longer
|
|
29
|
+
* shrink and is stuck. An UNBOUNDED inbound cap is not an acceptable answer: this admission
|
|
30
|
+
* limit is the only thing standing between one request and the process heap, and
|
|
31
|
+
* `readBoundedJsonRequestBody` materializes the body several times over (retained wire bytes,
|
|
32
|
+
* decoded bytes, the decoded string, the re-encoded measurement copies, and the parsed object
|
|
33
|
+
* graph), so peak RSS is a MULTIPLE of whatever is admitted here. 512 MiB is the largest value
|
|
34
|
+
* that keeps that multiple survivable on an ordinary machine, and it is what #3573 asked for.
|
|
35
|
+
*/
|
|
36
|
+
export const MAX_CONFIGURABLE_INBOUND_BODY_BYTES = 512 * 1024 * 1024;
|
|
37
|
+
|
|
38
|
+
/** Floor for the opt-in. Below this an ordinary multi-image turn cannot be admitted at all. */
|
|
39
|
+
export const MIN_CONFIGURABLE_INBOUND_BODY_BYTES = 1024 * 1024;
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Resolve the configured inbound admission limit, clamped to the supported range.
|
|
43
|
+
*
|
|
44
|
+
* Pure and total on purpose: the schema in `src/config.ts` degrades an invalid hand edit to
|
|
45
|
+
* `undefined` rather than failing the parse, so the schema cannot be the place the ceiling is
|
|
46
|
+
* enforced. Every caller resolves through here, which makes this the single auditable bound
|
|
47
|
+
* regardless of how the config object was produced.
|
|
48
|
+
*
|
|
49
|
+
* Omitted, zero, or non-finite = the 256 MiB default, so an unconfigured proxy admits exactly
|
|
50
|
+
* what it admits today.
|
|
51
|
+
*/
|
|
52
|
+
export function resolveInboundBodyLimitBytes(configured: number | undefined): number {
|
|
53
|
+
if (configured === undefined || !Number.isFinite(configured) || configured <= 0) {
|
|
54
|
+
return MAX_DECOMPRESSED_BODY_BYTES;
|
|
55
|
+
}
|
|
56
|
+
return Math.min(
|
|
57
|
+
Math.max(Math.floor(configured), MIN_CONFIGURABLE_INBOUND_BODY_BYTES),
|
|
58
|
+
MAX_CONFIGURABLE_INBOUND_BODY_BYTES,
|
|
59
|
+
);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Render a byte count, or nothing at all. `DecompressedBodyTooLargeError` accepts non-finite
|
|
64
|
+
* and untyped values from legacy callers and deliberately keeps them out of its own message;
|
|
65
|
+
* the client-facing message inherits that rule rather than printing `NaN MB`.
|
|
66
|
+
*/
|
|
67
|
+
function megabytes(bytes: number): string | null {
|
|
68
|
+
return Number.isFinite(bytes) && bytes >= 0 && bytes <= Number.MAX_SAFE_INTEGER
|
|
69
|
+
? (bytes / (1024 * 1024)).toFixed(1)
|
|
70
|
+
: null;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const INBOUND_CEILING_MB = (MAX_CONFIGURABLE_INBOUND_BODY_BYTES / (1024 * 1024)).toFixed(1);
|
|
74
|
+
|
|
75
|
+
|
|
24
76
|
export class UnsupportedContentEncodingError extends Error {
|
|
25
77
|
constructor(readonly encoding: string) {
|
|
26
78
|
super(`Unsupported content-encoding: ${encoding}`);
|
|
@@ -54,6 +106,33 @@ export class DecompressedBodyTooLargeError extends Error {
|
|
|
54
106
|
}
|
|
55
107
|
}
|
|
56
108
|
|
|
109
|
+
/**
|
|
110
|
+
* Name OpenCodex as the refuser, and name the lever.
|
|
111
|
+
*
|
|
112
|
+
* #4112 gave the UPSTREAM context refusal on `/v1/responses` its own HTTP 413 with
|
|
113
|
+
* `context_length_exceeded`. That makes the two 413s on this surface look alike to a client
|
|
114
|
+
* while having opposite remedies: the upstream one means the provider will not take the turn,
|
|
115
|
+
* this one means the proxy never read it and a config key would have let it through. The
|
|
116
|
+
* wording deliberately avoids "context window"/"context length", which `classifyError` treats
|
|
117
|
+
* as evidence of an upstream context verdict.
|
|
118
|
+
*/
|
|
119
|
+
export function describeInboundBodyRefusal(error: DecompressedBodyTooLargeError): string {
|
|
120
|
+
// A lower-bound measurement stopped counting at the cap; reporting it as exact would be a lie.
|
|
121
|
+
const approximate = error.measurement === "declared_wire" || error.measurement === "decoded_exact"
|
|
122
|
+
? "" : "at least ";
|
|
123
|
+
const observed = megabytes(error.bytes);
|
|
124
|
+
const limit = megabytes(error.limit);
|
|
125
|
+
const sizes = limit === null
|
|
126
|
+
? "the body is above the inbound admission limit"
|
|
127
|
+
: observed === null
|
|
128
|
+
? `the body is above the ${limit} MB inbound admission limit`
|
|
129
|
+
: `the body is ${approximate}${observed} MB, above the ${limit} MB inbound admission limit`;
|
|
130
|
+
return `OpenCodex refused this request before reading it: ${sizes}. `
|
|
131
|
+
+ "This is a local proxy limit, not a provider refusal. Raise \"maxInboundBodyBytes\" in "
|
|
132
|
+
+ `config.json (ceiling ${INBOUND_CEILING_MB} MB) and restart the proxy, or compact the `
|
|
133
|
+
+ "conversation earlier.";
|
|
134
|
+
}
|
|
135
|
+
|
|
57
136
|
function assertBodySizeWithinLimit(
|
|
58
137
|
body: Uint8Array,
|
|
59
138
|
maxBytes: number,
|
|
@@ -259,7 +338,16 @@ export async function readBoundedJsonRequestBody(
|
|
|
259
338
|
}
|
|
260
339
|
}
|
|
261
340
|
|
|
262
|
-
/**
|
|
263
|
-
|
|
264
|
-
|
|
341
|
+
/**
|
|
342
|
+
* Parse a JSON data-plane body using the shared admission cap.
|
|
343
|
+
*
|
|
344
|
+
* `maxBytes` is the resolved per-deployment limit from `resolveInboundBodyLimitBytes()`;
|
|
345
|
+
* omitting it keeps the 256 MiB default for callers with no config in scope.
|
|
346
|
+
*/
|
|
347
|
+
export function readJsonRequestBody(
|
|
348
|
+
req: Request,
|
|
349
|
+
budget?: TranslatorBudget,
|
|
350
|
+
maxBytes: number = MAX_DECOMPRESSED_BODY_BYTES,
|
|
351
|
+
): Promise<unknown> {
|
|
352
|
+
return readBoundedJsonRequestBody(req, maxBytes, budget);
|
|
265
353
|
}
|
|
@@ -1119,6 +1119,16 @@ export function filterRequestLogs(logs: RequestLogEntry[], params: URLSearchPara
|
|
|
1119
1119
|
filtered = filtered.filter(entry => entry.model === model
|
|
1120
1120
|
|| entry.attempts?.some(attempt => attempt.model === model));
|
|
1121
1121
|
}
|
|
1122
|
+
// #4057: "which account served this request" is the first question asked when one provider
|
|
1123
|
+
// holds several accounts, and until now the only way to answer it was to grep usage.jsonl by
|
|
1124
|
+
// hand. Attempts are matched for the same reason `provider` and `model` match them: when a
|
|
1125
|
+
// request failed over between pool accounts, a search for the account that finally served it
|
|
1126
|
+
// has to find that request, not only the account that first refused it.
|
|
1127
|
+
const account = params.get("account")?.trim();
|
|
1128
|
+
if (account) {
|
|
1129
|
+
filtered = filtered.filter(entry => entry.accountLogLabel === account
|
|
1130
|
+
|| entry.attempts?.some(attempt => attempt.accountLogLabel === account));
|
|
1131
|
+
}
|
|
1122
1132
|
const status = params.get("status")?.trim().toLowerCase();
|
|
1123
1133
|
if (status) {
|
|
1124
1134
|
filtered = /^[1-5]xx$/.test(status)
|
|
@@ -2,7 +2,7 @@ import { MAX_CLIENT_SSE_FRAME_BYTES } from "../sse-frame-buffer";
|
|
|
2
2
|
// If the 101 never arrives (network black hole), give SSE a chance well before
|
|
3
3
|
// the caller's connect timeout (default 200s) would fire.
|
|
4
4
|
export const UPGRADE_DEADLINE_MS = 10_000;
|
|
5
|
-
export const CODEX_WS_RESPONSE_PRELUDE_TIMEOUT_MS =
|
|
5
|
+
export const CODEX_WS_RESPONSE_PRELUDE_TIMEOUT_MS = 90_000;
|
|
6
6
|
// Keep the push-based WS transport inside the same memory envelope as the
|
|
7
7
|
// bounded SSE relays that consume this response. Unlike fetch response bodies,
|
|
8
8
|
// a WebSocket cannot be paused when a ReadableStream applies backpressure, so
|
|
@@ -104,7 +104,12 @@ import { fastPolicyForModel } from "../../providers/service-tier";
|
|
|
104
104
|
import { parseFastOnlyRowId } from "../fast-row";
|
|
105
105
|
import { applyOpenAiVirtualModel, resolveOpenAiCompactModel } from "../../providers/openai-virtual-models";
|
|
106
106
|
import { isUsageDebugEnabled } from "../../usage/debug";
|
|
107
|
-
import {
|
|
107
|
+
import {
|
|
108
|
+
readJsonRequestBody,
|
|
109
|
+
resolveInboundBodyLimitBytes,
|
|
110
|
+
DecompressedBodyTooLargeError,
|
|
111
|
+
UnsupportedContentEncodingError,
|
|
112
|
+
} from "../request-decompress";
|
|
108
113
|
import { resolveAdapter, resolveWireProtocolOverride } from "../adapter-resolve";
|
|
109
114
|
import { hasKeyPoolFailover, rotateProviderTransportOn429 } from "../../providers/key-failover";
|
|
110
115
|
import { shouldAttemptImageTierRetry } from "../image-retry";
|
|
@@ -522,7 +527,7 @@ export async function handleResponsesCompact(
|
|
|
522
527
|
): Promise<Response> {
|
|
523
528
|
let body: unknown;
|
|
524
529
|
try {
|
|
525
|
-
body = await readJsonRequestBody(req);
|
|
530
|
+
body = await readJsonRequestBody(req, undefined, resolveInboundBodyLimitBytes(config.maxInboundBodyBytes));
|
|
526
531
|
} catch (err) {
|
|
527
532
|
return decodeRequestErrorResponse(err, "responses-compact");
|
|
528
533
|
}
|
|
@@ -1015,6 +1020,7 @@ export async function handleResponsesCompact(
|
|
|
1015
1020
|
upstream.headers,
|
|
1016
1021
|
authCtx.writerGeneration,
|
|
1017
1022
|
authCtx.kind === "main-pool" ? authCtx.mainQuotaWriter : undefined,
|
|
1023
|
+
{ modelId: route.modelId },
|
|
1018
1024
|
);
|
|
1019
1025
|
}
|
|
1020
1026
|
recordCompactPoolOutcome(authCtx, upstream.status, {
|
|
@@ -5,6 +5,17 @@ import type { AdapterEvent } from "../../types";
|
|
|
5
5
|
export const PROVIDER_INPUT_TOO_LARGE_MESSAGE =
|
|
6
6
|
"The provider rejected this turn because its input exceeds the provider size or context limit. Reduce the current input or compact the conversation before retrying.";
|
|
7
7
|
|
|
8
|
+
/** Preserve non-streaming HTTP failure semantics without exposing an upstream body. */
|
|
9
|
+
export function jsonContextOverflowResponse(): Response {
|
|
10
|
+
return Response.json({
|
|
11
|
+
error: {
|
|
12
|
+
message: PROVIDER_INPUT_TOO_LARGE_MESSAGE,
|
|
13
|
+
type: "invalid_request_error",
|
|
14
|
+
code: "context_length_exceeded",
|
|
15
|
+
},
|
|
16
|
+
}, { status: 413, headers: { "Cache-Control": "no-store" } });
|
|
17
|
+
}
|
|
18
|
+
|
|
8
19
|
async function* contextOverflowEvents(): AsyncGenerator<AdapterEvent> {
|
|
9
20
|
yield {
|
|
10
21
|
type: "error",
|