@bitkyc08/opencodex 2.56.0 → 2.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/ocx.mjs +10 -0
- package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
- package/gui/dist/assets/{index-D4zuyIxQ.js → index-Cz7CLdif.js} +21 -21
- package/gui/dist/index.html +2 -2
- package/package.json +3 -3
- package/src/adapters/codebuddy/adapter.ts +2 -1
- package/src/adapters/codebuddy/scaffold-guard.ts +248 -0
- package/src/adapters/command-code.ts +1 -1
- package/src/adapters/cursor/envelope-echo.ts +8 -2
- package/src/adapters/google.ts +7 -7
- package/src/adapters/kiro/payload.ts +17 -3
- package/src/adapters/kiro/reasoning.ts +70 -7
- package/src/adapters/kiro/stream.ts +8 -2
- package/src/adapters/kiro/wire.ts +2 -1
- package/src/adapters/kiro-events.ts +21 -13
- package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
- package/src/adapters/openai-chat/tool-schema.ts +25 -7
- package/src/adapters/openai-chat.ts +8 -8
- package/src/adapters/openai-responses/passthrough.ts +32 -1
- package/src/bridge/errors.ts +26 -2
- package/src/bridge/response-json.ts +7 -1
- package/src/bridge/sse.ts +19 -1
- package/src/claude/desktop-profile.ts +66 -9
- package/src/claude/outbound.ts +18 -0
- package/src/cli/account-main.ts +1 -1
- package/src/cli/capabilities.ts +2 -2
- package/src/cli/combo.ts +10 -1
- package/src/cli/index.ts +48 -5
- package/src/cli/registry.ts +2 -1
- package/src/cli/system-command.ts +4 -4
- package/src/clients/config-export.ts +7 -3
- package/src/codex/account-label.ts +14 -3
- package/src/codex/account-store.ts +113 -26
- package/src/codex/account-usability.ts +21 -0
- package/src/codex/auth-api/login-flow.ts +14 -2
- package/src/codex/auth-api/reset-credit-service.ts +11 -2
- package/src/codex/auth-context.ts +157 -7
- package/src/codex/catalog/aggregation.ts +80 -1
- package/src/codex/catalog/model-visibility.ts +1 -0
- package/src/codex/catalog/remote.ts +30 -0
- package/src/codex/catalog/retained-sync.ts +9 -1
- package/src/codex/catalog/routed-gather.ts +38 -1
- package/src/codex/cli-install-provenance.ts +7 -1
- package/src/codex/convergence.ts +7 -2
- package/src/codex/desktop-app/types.ts +11 -2
- package/src/codex/desktop-app/windows.ts +5 -5
- package/src/codex/inject/restore.ts +29 -2
- package/src/codex/inject.ts +9 -9
- package/src/codex/model-entitlements.ts +152 -15
- package/src/codex/pool-refresh-backoff.ts +12 -3
- package/src/codex/quota-rejection.ts +104 -15
- package/src/codex/routing/cache-affinity.ts +70 -0
- package/src/codex/routing/cooldown-math.ts +10 -0
- package/src/codex/routing/selection.ts +79 -2
- package/src/codex/routing/thread-affinity.ts +50 -2
- package/src/codex/routing/transient-hold-dispatch.ts +141 -0
- package/src/codex/routing.ts +29 -49
- package/src/codex/warmup.ts +1 -1
- package/src/combos/failover.ts +85 -0
- package/src/combos/request.ts +17 -10
- package/src/combos/types.ts +23 -2
- package/src/config/pending-teardown.ts +31 -0
- package/src/generated/compatibility-version.json +163 -135
- package/src/images/loop.ts +1 -1
- package/src/lib/errors.ts +17 -0
- package/src/lib/request-execution-budget.ts +147 -21
- package/src/lib/spend-reservation-ledger.ts +18 -0
- package/src/lib/state-store-registrations.ts +6 -2
- package/src/lib/test-home-guard.ts +85 -1
- package/src/lib/upstream-retry.ts +77 -10
- package/src/lib/windows-elevation.ts +76 -14
- package/src/oauth/index.ts +2 -2
- package/src/oauth/key-providers.ts +2 -2
- package/src/providers/kiro-models.ts +4 -3
- package/src/providers/label.ts +19 -1
- package/src/providers/model-discovery.ts +16 -0
- package/src/providers/registry/entries-core.ts +7 -0
- package/src/providers/registry/entries-extended.ts +9 -0
- package/src/providers/registry/model-seeds.ts +4 -0
- package/src/responses/reasoning-envelope.ts +6 -3
- package/src/routing/identity-domains.ts +21 -14
- package/src/routing/probe-lease.ts +103 -1
- package/src/server/chat-completions.ts +3 -1
- package/src/server/chat-native.ts +37 -9
- package/src/server/index/live-sideband.ts +37 -1
- package/src/server/index/websocket-handler.ts +6 -2
- package/src/server/index.ts +5 -5
- package/src/server/inspection-tee.ts +107 -0
- package/src/server/live.ts +46 -1
- package/src/server/management/combo-routes.ts +10 -1
- package/src/server/relay-eager.ts +2 -0
- package/src/server/relay.ts +14 -19
- package/src/server/request-log.ts +127 -3
- package/src/server/response-log-body.ts +153 -0
- package/src/server/responses/account-change-state.ts +74 -0
- package/src/server/responses/adapter-continuation.ts +33 -7
- package/src/server/responses/adapter-delivery.ts +5 -11
- package/src/server/responses/adapter-dispatch.ts +84 -13
- package/src/server/responses/codex-ws-wire.ts +5 -0
- package/src/server/responses/collaboration.ts +74 -4
- package/src/server/responses/combo-session-recall.ts +68 -8
- package/src/server/responses/compact.ts +54 -13
- package/src/server/responses/core-auth.ts +2 -0
- package/src/server/responses/core-codex-account.ts +51 -3
- package/src/server/responses/core-combo.ts +103 -23
- package/src/server/responses/core-errors.ts +18 -0
- package/src/server/responses/core-replay.ts +105 -32
- package/src/server/responses/core.ts +3 -3
- package/src/server/responses/encrypted-payload.ts +0 -1
- package/src/server/responses/input-admission.ts +126 -6
- package/src/server/responses/passthrough-delivery.ts +19 -6
- package/src/server/responses/passthrough-dispatch.ts +28 -10
- package/src/server/responses/passthrough-error.ts +38 -2
- package/src/server/responses/request-prepare.ts +132 -22
- package/src/server/responses/request-send-budget.ts +97 -2
- package/src/server/responses/request-spend.ts +147 -0
- package/src/server/responses/request-transport.ts +62 -3
- package/src/server/responses/run-turn-execution.ts +59 -31
- package/src/server/responses/sidecar-execution.ts +7 -13
- package/src/server/responses/terminal-guard.ts +65 -4
- package/src/server/responses-undeclared-tool-guard.ts +9 -5
- package/src/service/windows-ops.ts +210 -16
- package/src/service/windows-scheduler.ts +28 -21
- package/src/service.ts +1 -1
- package/src/types/config.ts +4 -1
- package/src/types/request.ts +8 -5
- package/src/types/tools.ts +24 -0
- package/src/types.ts +2 -0
- package/src/update/index.ts +10 -0
- package/src/update/stop-contract.d.mts +1 -0
- package/src/update/stop-contract.mjs +19 -0
- package/src/update/stop-decision.d.mts +1 -1
- package/src/update/stop-decision.mjs +12 -3
- package/src/usage/log.ts +1 -1
- package/src/vision/anthropic-describe.ts +1 -1
- package/src/vision/describe.ts +5 -5
- package/src/web-search/anthropic-executor.ts +1 -1
- package/src/web-search/exa-executor.ts +1 -1
- package/src/web-search/executor.ts +1 -1
- package/src/web-search/gemini-executor.ts +1 -1
- package/src/web-search/loop.ts +1 -1
- package/src/web-search/ollama-executor.ts +1 -1
- package/src/web-search/parse.ts +67 -14
- package/src/web-search/passthrough-bridge.ts +64 -31
- package/src/web-search/xai-executor.ts +1 -1
|
@@ -10,7 +10,13 @@
|
|
|
10
10
|
* catches the pathological case and stays out of the way otherwise. Every uncertainty
|
|
11
11
|
* resolves toward admitting.
|
|
12
12
|
*/
|
|
13
|
-
import {
|
|
13
|
+
import {
|
|
14
|
+
nativeOpenAiContextWindow,
|
|
15
|
+
nativeOpenAiMaxInputTokens,
|
|
16
|
+
nativeOpenAiMaxOutputTokens,
|
|
17
|
+
type NativeContextLimitsInput,
|
|
18
|
+
} from "../../codex/catalog/metadata";
|
|
19
|
+
import { getModelMetadata } from "../../generated/model-metadata";
|
|
14
20
|
import { estimateTokens } from "../../lib/token-estimate";
|
|
15
21
|
import { isCanonicalOpenAiForwardProvider, OPENAI_CODEX_PROVIDER_ID } from "../../providers/openai-tiers";
|
|
16
22
|
import { modelRecordValue } from "../../reasoning-effort";
|
|
@@ -54,6 +60,8 @@ export interface InputAdmissionResult {
|
|
|
54
60
|
estimatedTokens: number;
|
|
55
61
|
/** Resolved ceiling, or null when nothing could be resolved (=> always admitted). */
|
|
56
62
|
ceiling: number | null;
|
|
63
|
+
/** Output space reserved by the combo preflight; absent on the loose direct gate. */
|
|
64
|
+
requiredOutputHeadroom?: number;
|
|
57
65
|
}
|
|
58
66
|
|
|
59
67
|
function positive(value: unknown): number | null {
|
|
@@ -135,14 +143,19 @@ export function estimateInputTokens(parsed: OcxParsedRequest, modelId: string):
|
|
|
135
143
|
* reject a user-defined provider that merely shares a built-in name using limits that
|
|
136
144
|
* belong to a different service.
|
|
137
145
|
*/
|
|
138
|
-
|
|
146
|
+
interface ResolvedContextLimits {
|
|
147
|
+
/** The target's total context window: input and output share it. */
|
|
148
|
+
window: number | null;
|
|
149
|
+
/** Largest admissible input, which input-only caps may tighten below the window. */
|
|
150
|
+
ceiling: number | null;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
function resolveContextLimits(
|
|
139
154
|
provider: OcxProviderConfig,
|
|
140
155
|
providerName: string,
|
|
141
156
|
modelId: string,
|
|
142
|
-
// Operator cap for the canonical native provider. Passed in rather than read from a
|
|
143
|
-
// config here so this stays pure: no filesystem, no catalog, no registry scan.
|
|
144
157
|
nativeContextCap?: NativeContextLimitsInput,
|
|
145
|
-
):
|
|
158
|
+
): ResolvedContextLimits {
|
|
146
159
|
// `modelRecordValue`, not a bare lookup: the catalog resolves these same two maps that
|
|
147
160
|
// way, so a `gpt-oss` entry covers `gpt-oss:120b`. Reading raw here made the gate fall
|
|
148
161
|
// back to the provider-wide window and refuse turns the model can plainly hold.
|
|
@@ -168,13 +181,120 @@ export function resolveInputCeiling(
|
|
|
168
181
|
: null;
|
|
169
182
|
const nativeMaxInput = canonicalNativeBare ? positive(nativeOpenAiMaxInputTokens(modelId, nativeLimits)) : null;
|
|
170
183
|
|
|
171
|
-
const window = canonicalNativeBare ? native : configured;
|
|
184
|
+
const window = canonicalNativeBare ? (native ?? generatedNativeWindow(modelId, configured, nativeContextCap)) : configured;
|
|
172
185
|
// modelMaxInputTokens is an input-only cap, so it can only tighten the window.
|
|
173
186
|
const configuredMaxInput = positive(modelRecordValue(provider.modelMaxInputTokens, modelId));
|
|
174
187
|
const limits = [window, configuredMaxInput, nativeMaxInput].filter((v): v is number => v !== null);
|
|
188
|
+
return { window, ceiling: limits.length === 0 ? null : Math.min(...limits) };
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Generated-catalog keys, not routing provider names. `OPENAI_CODEX_PROVIDER_ID` is the string
|
|
193
|
+
* `"openai"` -- the canonical Codex forward route -- so using it to index the generated bundle
|
|
194
|
+
* would silently skip the native Codex rows and read the public API rows instead.
|
|
195
|
+
*/
|
|
196
|
+
const NATIVE_METADATA_CATALOGS = ["openai-codex", "openai"] as const;
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Static in-tree metadata for a canonical native slug the narrower override and pinned-native
|
|
200
|
+
* tables do not carry. Falling through to null made input admission completely blind for
|
|
201
|
+
* exactly those models, which is how a 128k target accepted a turn it could not finish.
|
|
202
|
+
*
|
|
203
|
+
* This deliberately covers slugs that are no longer offered in the picker: a retired slug is
|
|
204
|
+
* still dispatchable when an operator names it explicitly in a combo target, and that is the
|
|
205
|
+
* configuration where the gate was inert. This is a generated bundle compiled into the binary,
|
|
206
|
+
* not a live catalog read, so it adds no I/O. Explicit provider and operator caps may only
|
|
207
|
+
* narrow the result, never widen it.
|
|
208
|
+
*/
|
|
209
|
+
function generatedNativeWindow(
|
|
210
|
+
modelId: string,
|
|
211
|
+
configured: number | null,
|
|
212
|
+
nativeContextCap: NativeContextLimitsInput | undefined,
|
|
213
|
+
): number | null {
|
|
214
|
+
let generated: number | null = null;
|
|
215
|
+
for (const catalog of NATIVE_METADATA_CATALOGS) {
|
|
216
|
+
generated = positive(getModelMetadata(catalog, modelId)?.contextWindow);
|
|
217
|
+
if (generated !== null) break;
|
|
218
|
+
}
|
|
219
|
+
if (generated === null) return null;
|
|
220
|
+
const cap = typeof nativeContextCap === "number"
|
|
221
|
+
? positive(nativeContextCap)
|
|
222
|
+
: positive(nativeContextCap?.cap);
|
|
223
|
+
return Math.min(generated, configured ?? generated, cap ?? generated);
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
export function resolveInputCeiling(
|
|
227
|
+
provider: OcxProviderConfig,
|
|
228
|
+
providerName: string,
|
|
229
|
+
modelId: string,
|
|
230
|
+
// Operator cap for the canonical native provider. Passed in rather than read from a
|
|
231
|
+
// config here so this stays pure: no filesystem, no catalog, no registry scan.
|
|
232
|
+
nativeContextCap?: NativeContextLimitsInput,
|
|
233
|
+
): number | null {
|
|
234
|
+
return resolveContextLimits(provider, providerName, modelId, nativeContextCap).ceiling;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
/**
|
|
238
|
+
* Largest output the concrete target can emit. Used only to avoid reserving MORE than the
|
|
239
|
+
* target could ever produce when a client asks for a bigger allowance than the model has.
|
|
240
|
+
* Unknown stays unknown rather than inventing a capability.
|
|
241
|
+
*/
|
|
242
|
+
export function resolveOutputCeiling(
|
|
243
|
+
provider: OcxProviderConfig,
|
|
244
|
+
providerName: string,
|
|
245
|
+
modelId: string,
|
|
246
|
+
): number | null {
|
|
247
|
+
const configured = positive(modelRecordValue(provider.modelMaxOutputTokens, modelId))
|
|
248
|
+
?? positive(provider.defaultMaxOutputTokens);
|
|
249
|
+
const canonicalNativeBare = providerName === OPENAI_CODEX_PROVIDER_ID
|
|
250
|
+
&& isCanonicalOpenAiForwardProvider(provider)
|
|
251
|
+
&& !modelId.includes("/");
|
|
252
|
+
const native = canonicalNativeBare ? positive(nativeOpenAiMaxOutputTokens(modelId)) : null;
|
|
253
|
+
const limits = [configured, native].filter((v): v is number => v !== null);
|
|
175
254
|
return limits.length === 0 ? null : Math.min(...limits);
|
|
176
255
|
}
|
|
177
256
|
|
|
257
|
+
/**
|
|
258
|
+
* Combo-only admission. A fallback must be able to satisfy the caller's declared output
|
|
259
|
+
* allowance inside its OWN context window. Otherwise it returns 200, emits a few hundred
|
|
260
|
+
* tokens, and terminates on `finish_reason: length` — which the Anthropic surface renders as
|
|
261
|
+
* "response exceeded the output token maximum" even though the real cause was the total
|
|
262
|
+
* window. By then the next target cannot be tried, because output has already committed.
|
|
263
|
+
*
|
|
264
|
+
* Two budgets are checked separately so the reserve is counted exactly once. `ceiling` is an
|
|
265
|
+
* input-only budget once `modelMaxInputTokens` tightens it below the window, so the output
|
|
266
|
+
* reserve belongs against `window`, not against `ceiling`.
|
|
267
|
+
*
|
|
268
|
+
* Direct and single-target requests keep the deliberately loose 2.5x pathological-input gate.
|
|
269
|
+
* This stricter rule applies only to synthetic combo children, where skipping one known-small
|
|
270
|
+
* target is safe and the ladder continues before any upstream bytes are sent. Unknown context
|
|
271
|
+
* stays fail-open, and a caller that declared no output allowance is unaffected.
|
|
272
|
+
*/
|
|
273
|
+
export function checkComboTargetInputAdmission(
|
|
274
|
+
parsed: OcxParsedRequest,
|
|
275
|
+
provider: OcxProviderConfig,
|
|
276
|
+
providerName: string,
|
|
277
|
+
modelId: string,
|
|
278
|
+
nativeContextCap?: NativeContextLimitsInput,
|
|
279
|
+
): InputAdmissionResult {
|
|
280
|
+
const { window, ceiling } = resolveContextLimits(provider, providerName, modelId, nativeContextCap);
|
|
281
|
+
const requestedOutput = positive(parsed.options.maxOutputTokens);
|
|
282
|
+
if (window === null || ceiling === null || requestedOutput === null) {
|
|
283
|
+
return checkInputAdmission(parsed, provider, providerName, modelId, nativeContextCap);
|
|
284
|
+
}
|
|
285
|
+
const targetOutput = resolveOutputCeiling(provider, providerName, modelId);
|
|
286
|
+
const requiredOutputHeadroom = targetOutput === null
|
|
287
|
+
? requestedOutput
|
|
288
|
+
: Math.min(requestedOutput, targetOutput);
|
|
289
|
+
const estimatedTokens = estimateInputTokens(parsed, modelId);
|
|
290
|
+
return {
|
|
291
|
+
admitted: estimatedTokens <= ceiling && estimatedTokens + requiredOutputHeadroom <= window,
|
|
292
|
+
estimatedTokens,
|
|
293
|
+
ceiling,
|
|
294
|
+
requiredOutputHeadroom,
|
|
295
|
+
};
|
|
296
|
+
}
|
|
297
|
+
|
|
178
298
|
/**
|
|
179
299
|
* Fail-open when no ceiling is known; refuse only past `ceiling * ADMISSION_TOLERANCE`.
|
|
180
300
|
*
|
|
@@ -15,6 +15,8 @@ import {
|
|
|
15
15
|
relayWithAbort,
|
|
16
16
|
} from "../relay";
|
|
17
17
|
import { isUsageDebugEnabled } from "../../usage/debug";
|
|
18
|
+
import { isReplayRefusalResponse } from "../../lib/upstream-retry";
|
|
19
|
+
import { teeWithBoundedInspection } from "../inspection-tee";
|
|
18
20
|
import {
|
|
19
21
|
codexForwardTerminalOutcomeRecorder,
|
|
20
22
|
usesCodexForwardPoolAuth,
|
|
@@ -27,7 +29,7 @@ import type { ResponsesTerminalStatus } from "../../bridge";
|
|
|
27
29
|
import { isCodexWsQuotaObservedResponse, isCodexWsUpstreamResponse } from "./ws-upstream";
|
|
28
30
|
import { recordSubagentQuotaFailureForThreadSpawn } from "../../codex/subagent-model-fallback";
|
|
29
31
|
import { recordCodexUpstreamOutcome } from "../../codex/routing";
|
|
30
|
-
import { codexProbeLeaseId, codexProbeQuotaScope } from "../../codex/auth-context";
|
|
32
|
+
import { codexProbeLeaseId, codexProbeQuotaScope, codexTransientProbeGrant } from "../../codex/auth-context";
|
|
31
33
|
import { consumeComboFailure } from "./core-combo-failure";
|
|
32
34
|
import { readDisplaySafeErrorText } from "./core-errors";
|
|
33
35
|
import { streamingContextOverflowResponse, jsonContextOverflowResponse } from "./context-overflow";
|
|
@@ -243,7 +245,12 @@ export async function deliverPassthroughResponse(
|
|
|
243
245
|
} else if (!shouldDeferCodexResetDerivedCooldown(
|
|
244
246
|
upstreamResponse,
|
|
245
247
|
options.deferCodexResetDerivedCooldown,
|
|
246
|
-
)) {
|
|
248
|
+
) && !isReplayRefusalResponse(upstreamResponse)) {
|
|
249
|
+
// A refusal this proxy made is not evidence about the account. Recording it would
|
|
250
|
+
// classify the synthetic 429 as quota exhaustion and write a default cooldown against
|
|
251
|
+
// a credential the request may never have reached, and that false signal outlives the
|
|
252
|
+
// request. The sibling recorders on this path already decline: the terminal recorder
|
|
253
|
+
// needs an ok streaming body, and the quota-header snapshot finds no quota headers.
|
|
247
254
|
recordCodexUpstreamOutcome(config, admissionState.authCtx.accountId, upstreamResponse.status, {
|
|
248
255
|
...quotaMeta,
|
|
249
256
|
threadId: admissionState.authCtx.affinityKey,
|
|
@@ -251,6 +258,7 @@ export async function deliverPassthroughResponse(
|
|
|
251
258
|
modelId: route.modelId,
|
|
252
259
|
probeLeaseId: codexProbeLeaseId(admissionState.authCtx),
|
|
253
260
|
probeQuotaScope: codexProbeQuotaScope(admissionState.authCtx),
|
|
261
|
+
transientProbe: codexTransientProbeGrant(admissionState.authCtx),
|
|
254
262
|
writerGeneration: admissionState.authCtx.writerGeneration,
|
|
255
263
|
// Includes a replay's second 401, which is the case that actually retires the
|
|
256
264
|
// account — fence it on the credential the request was holding.
|
|
@@ -298,6 +306,9 @@ export async function deliverPassthroughResponse(
|
|
|
298
306
|
return formatPassthroughUpstreamError(upstreamResponse.status, errorText, {
|
|
299
307
|
statusText: upstreamResponse.statusText,
|
|
300
308
|
headers,
|
|
309
|
+
// Provenance, not inference: `errorText` is empty when the bounded read finds nothing
|
|
310
|
+
// display-safe, and an empty body is exactly what the retryable-429 default fires on.
|
|
311
|
+
replayRefusal: isReplayRefusalResponse(upstreamResponse),
|
|
301
312
|
});
|
|
302
313
|
}
|
|
303
314
|
|
|
@@ -593,16 +604,18 @@ export async function deliverPassthroughResponse(
|
|
|
593
604
|
})),
|
|
594
605
|
);
|
|
595
606
|
}
|
|
596
|
-
const [nativeBody, inspectBody] = passthroughSseBody.tee();
|
|
597
607
|
const turnAc = new AbortController();
|
|
598
608
|
const clientGone = new AbortController();
|
|
609
|
+
const clientGoneSignal = options.abortSignal
|
|
610
|
+
? AbortSignal.any([clientGone.signal, options.abortSignal])
|
|
611
|
+
: clientGone.signal;
|
|
612
|
+
// Pace against raw bytes before rewrites, without detaching terminal ownership.
|
|
613
|
+
const [nativeBody, inspectBody] = teeWithBoundedInspection(passthroughSseBody, { clientGoneSignal });
|
|
599
614
|
linkAbortSignal(upstream, turnAc.signal);
|
|
600
615
|
registerTurn(turnAc, options.turnAdmissionLease);
|
|
601
616
|
const inspectionConsumerOptions = {
|
|
602
617
|
// Request abort can reject the fetch body before the response cancel hook runs.
|
|
603
|
-
clientGoneSignal
|
|
604
|
-
? AbortSignal.any([clientGone.signal, options.abortSignal])
|
|
605
|
-
: clientGone.signal,
|
|
618
|
+
clientGoneSignal,
|
|
606
619
|
drainBounds: { ms: 15_000, bytes: 32 * 1024 * 1024 },
|
|
607
620
|
upstream,
|
|
608
621
|
pinCompletedResponseIdToFirstSeen: githubCopilotRepairEnabled,
|
|
@@ -29,6 +29,7 @@ import {
|
|
|
29
29
|
unwrapUpstreamRetryEvidenceError,
|
|
30
30
|
codexProbeLeaseId,
|
|
31
31
|
codexProbeQuotaScope,
|
|
32
|
+
codexTransientProbeGrant,
|
|
32
33
|
createCodexReserveDispatchGuard,
|
|
33
34
|
} from "../../codex/auth-context";
|
|
34
35
|
import {
|
|
@@ -60,7 +61,6 @@ import { restorePlaintextV2AgentMessageCalls } from "../../responses/plaintext-v
|
|
|
60
61
|
import {
|
|
61
62
|
recordAdapterReasoning,
|
|
62
63
|
recordAdapterTier,
|
|
63
|
-
noteAttemptSend,
|
|
64
64
|
sealRequestAttemptIdentity,
|
|
65
65
|
recordAttemptCredentialSource,
|
|
66
66
|
} from "../request-log";
|
|
@@ -80,6 +80,7 @@ import {
|
|
|
80
80
|
safeHostLabel,
|
|
81
81
|
storedPoolReplayDispatchNotifier,
|
|
82
82
|
} from "./fetch-helpers";
|
|
83
|
+
import { classifyPoolRecoveryDispatch } from "../../routing/probe-lease";
|
|
83
84
|
import { clientCancelledResponse } from "./core-errors";
|
|
84
85
|
import {
|
|
85
86
|
upstreamHostCircuitOpenResponse,
|
|
@@ -101,6 +102,7 @@ import {
|
|
|
101
102
|
fetchWithTransientRetry,
|
|
102
103
|
applyUpstreamRecoveryInit,
|
|
103
104
|
TRANSIENT_RETRY_MAX_ATTEMPTS,
|
|
105
|
+
isNonReplayableResponse,
|
|
104
106
|
prepareSameTarget429Wait,
|
|
105
107
|
sleepWithAbort,
|
|
106
108
|
} from "../../lib/upstream-retry";
|
|
@@ -171,6 +173,7 @@ export async function preparePassthroughExchange(
|
|
|
171
173
|
| "replayOAuthCredentialSnapshot"
|
|
172
174
|
| "genericFailovers"
|
|
173
175
|
| "applyFailoverSnapshot"
|
|
176
|
+
| "noteRoutedAttemptSend"
|
|
174
177
|
>,
|
|
175
178
|
responseEffects: Pick<
|
|
176
179
|
ResponsesEffects,
|
|
@@ -346,10 +349,13 @@ export async function preparePassthroughExchange(
|
|
|
346
349
|
const declaredWireToolNames = new Set<string>();
|
|
347
350
|
const declaredBareWireToolNames = new Set<string>();
|
|
348
351
|
const declaredNamelessClientCallTypes = new Set<string>();
|
|
349
|
-
// `buildToolBridgeMaps`
|
|
350
|
-
//
|
|
351
|
-
//
|
|
352
|
-
//
|
|
352
|
+
// `buildToolBridgeMaps` adds each eligible bare alias to `declaredToolNames` and `toolNsMap`
|
|
353
|
+
// (one authorized identity claims the bare name). `refreshUndeclaredToolGuard` normally copies
|
|
354
|
+
// those entries into `declaredWireToolNames`, but passthrough restoration runs before the
|
|
355
|
+
// undeclared-tool guard, so restore that request-bounded identity here, before authorization
|
|
356
|
+
// checks. `exec` uses separate handling: its bridge alias is copied into the declared set only
|
|
357
|
+
// when the client itself declared bare `exec`, because otherwise code-mode normalization could
|
|
358
|
+
// authorize the unrelated code-mode helper names.
|
|
353
359
|
const authorizedBareNamespaceToolAliases: RoutedNamespaceToolAliases = new Map(
|
|
354
360
|
[...toolBridgeMaps.toolNsMap].flatMap(([alias, identity]) =>
|
|
355
361
|
alias === identity.name
|
|
@@ -736,6 +742,7 @@ export async function preparePassthroughExchange(
|
|
|
736
742
|
modelId: route.modelId,
|
|
737
743
|
probeLeaseId: codexProbeLeaseId(admissionState.authCtx),
|
|
738
744
|
probeQuotaScope: codexProbeQuotaScope(admissionState.authCtx),
|
|
745
|
+
transientProbe: codexTransientProbeGrant(admissionState.authCtx),
|
|
739
746
|
writerGeneration: admissionState.authCtx.writerGeneration,
|
|
740
747
|
});
|
|
741
748
|
}
|
|
@@ -752,7 +759,13 @@ export async function preparePassthroughExchange(
|
|
|
752
759
|
// Body is a replayable string; nothing has streamed to the client yet.
|
|
753
760
|
upstreamResponse = await fetchWithTransientRetry(
|
|
754
761
|
recovery => {
|
|
755
|
-
|
|
762
|
+
// The pool-wide recovery window measures recovery traffic against observed demand,
|
|
763
|
+
// and this is where demand is observed: `recovery === undefined` is a new request's
|
|
764
|
+
// first send, everything after it is the same request trying again. Without this the
|
|
765
|
+
// ratio has no denominator and the window collapses to its quiet-pool floor, which
|
|
766
|
+
// would throttle recovery on a busy proxy exactly as hard as on an idle one (#4701).
|
|
767
|
+
if (recovery === undefined) classifyPoolRecoveryDispatch("initial");
|
|
768
|
+
transportState.noteRoutedAttemptSend(passthroughEstimate, recovery);
|
|
756
769
|
return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
|
|
757
770
|
method: request.method,
|
|
758
771
|
headers: request.headers,
|
|
@@ -848,7 +861,7 @@ export async function preparePassthroughExchange(
|
|
|
848
861
|
if (allowance.permit && !allowance.permit.use()) {
|
|
849
862
|
throw new SendBudgetExhaustedError(safeHostLabel(request.url));
|
|
850
863
|
}
|
|
851
|
-
|
|
864
|
+
transportState.noteRoutedAttemptSend(passthroughEstimate, innerRecovery ?? recovery);
|
|
852
865
|
return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
|
|
853
866
|
method: request.method,
|
|
854
867
|
headers: request.headers,
|
|
@@ -946,7 +959,7 @@ export async function preparePassthroughExchange(
|
|
|
946
959
|
// every other build site; a replay is exactly when a grown payload reappears.
|
|
947
960
|
const replayBodyRefusal = refuseOversizedOutboundBody(request);
|
|
948
961
|
if (replayBodyRefusal) return replayBodyRefusal;
|
|
949
|
-
|
|
962
|
+
transportState.noteRoutedAttemptSend(passthroughEstimate, "oauth-401");
|
|
950
963
|
upstreamResponse = await fetchWithHeaderTimeout(
|
|
951
964
|
request.url,
|
|
952
965
|
{ method: request.method, headers: request.headers, body: request.body },
|
|
@@ -1075,7 +1088,7 @@ export async function preparePassthroughExchange(
|
|
|
1075
1088
|
try {
|
|
1076
1089
|
upstreamResponse = await fetchWithTransientRetry(
|
|
1077
1090
|
recovery => {
|
|
1078
|
-
|
|
1091
|
+
transportState.noteRoutedAttemptSend(passthroughEstimate, recovery ?? "oauth-401");
|
|
1079
1092
|
return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
|
|
1080
1093
|
method: request.method,
|
|
1081
1094
|
headers: request.headers,
|
|
@@ -1105,6 +1118,10 @@ export async function preparePassthroughExchange(
|
|
|
1105
1118
|
// the same quorum, cooldown and request budget here, before any client bytes flow.
|
|
1106
1119
|
if (
|
|
1107
1120
|
upstreamResponse.status === 429
|
|
1121
|
+
// Not a provider rate limit when this proxy synthesized it for a refused reset
|
|
1122
|
+
// replay; rotating accounts on it would re-send an inference that may already
|
|
1123
|
+
// have run and would cool down an account that refused nothing.
|
|
1124
|
+
&& !isNonReplayableResponse(upstreamResponse)
|
|
1108
1125
|
&& transportState.genericFailoverAccountId
|
|
1109
1126
|
&& transportState.genericFailovers < GENERIC_OAUTH_MAX_FAILOVERS_PER_REQUEST
|
|
1110
1127
|
&& isGenericOAuthFailoverEnabled(config, route.providerName)
|
|
@@ -1159,6 +1176,7 @@ export async function preparePassthroughExchange(
|
|
|
1159
1176
|
// keep their pool logic below (rateLimitRetryPolicyFor returns null for them).
|
|
1160
1177
|
while (
|
|
1161
1178
|
upstreamResponse.status === 429
|
|
1179
|
+
&& !isNonReplayableResponse(upstreamResponse)
|
|
1162
1180
|
&& rateLimitPolicy !== null
|
|
1163
1181
|
&& rateLimitRetries < rateLimitPolicy.attempts
|
|
1164
1182
|
// Checked here rather than inside the helper: prepareSameTarget429Wait releases the 429
|
|
@@ -1192,7 +1210,7 @@ export async function preparePassthroughExchange(
|
|
|
1192
1210
|
recovery => {
|
|
1193
1211
|
// The first send of every replay is itself a rate-limit retry; inner transient-5xx
|
|
1194
1212
|
// recoveries keep their own label (recovery is provided for those).
|
|
1195
|
-
|
|
1213
|
+
transportState.noteRoutedAttemptSend(passthroughEstimate, recovery ?? "rate-limit-429");
|
|
1196
1214
|
return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
|
|
1197
1215
|
method: request.method,
|
|
1198
1216
|
headers: request.headers,
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { formatErrorResponse } from "../../bridge";
|
|
2
2
|
import { isCyberPolicyCode, isCyberPolicyMessage } from "../../lib/errors";
|
|
3
|
+
import { isReplayRefusalCode, UPSTREAM_RESET_REPLAY_REFUSED_CODE } from "../../lib/upstream-retry";
|
|
3
4
|
import {
|
|
4
5
|
resolveClientRetryAfter,
|
|
5
6
|
validateClientRetryAfterHeader,
|
|
@@ -24,6 +25,26 @@ function isCyberPolicyBody(body: string): boolean {
|
|
|
24
25
|
return false;
|
|
25
26
|
}
|
|
26
27
|
|
|
28
|
+
/**
|
|
29
|
+
* True for a body this proxy wrote to refuse replaying an ambiguous pre-header reset.
|
|
30
|
+
*
|
|
31
|
+
* It is read off the body rather than a marker because this formatter is handed bytes, not
|
|
32
|
+
* the response they came from, and the refusal reaches it after the original body was read.
|
|
33
|
+
* The code is this proxy's own, so an upstream echoing it is not a case worth widening for.
|
|
34
|
+
*/
|
|
35
|
+
function isReplayRefusalBody(body: string): boolean {
|
|
36
|
+
if (!body.includes(UPSTREAM_RESET_REPLAY_REFUSED_CODE)) return false;
|
|
37
|
+
try {
|
|
38
|
+
const parsed = JSON.parse(body) as Record<string, unknown>;
|
|
39
|
+
const error = parsed.error && typeof parsed.error === "object" && !Array.isArray(parsed.error)
|
|
40
|
+
? parsed.error as Record<string, unknown>
|
|
41
|
+
: undefined;
|
|
42
|
+
return isReplayRefusalCode(error?.code) || isReplayRefusalCode(parsed.code);
|
|
43
|
+
} catch {
|
|
44
|
+
return false;
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
27
48
|
/**
|
|
28
49
|
* Passthrough adapters historically relayed upstream non-2xx bodies verbatim.
|
|
29
50
|
* Codex maps an *empty* body to the literal client string "Unknown error"
|
|
@@ -38,6 +59,9 @@ function isCyberPolicyBody(body: string): boolean {
|
|
|
38
59
|
* - missing/malformed values are replaced when resolveClientRetryAfter yields a value
|
|
39
60
|
* - malformed/expired values are removed when the resolver returns undefined
|
|
40
61
|
* (e.g. quota-exhausted 429s must not keep junk headers or get the synthetic "2")
|
|
62
|
+
* - a replay refusal this proxy wrote gets none and keeps none: the whole point of the
|
|
63
|
+
* refusal is that the turn may already be running, and the synthetic default for a
|
|
64
|
+
* retryable 429 is a direct instruction to the client to send it a second time
|
|
41
65
|
*/
|
|
42
66
|
export function formatPassthroughUpstreamError(
|
|
43
67
|
status: number,
|
|
@@ -46,6 +70,13 @@ export function formatPassthroughUpstreamError(
|
|
|
46
70
|
statusText?: string;
|
|
47
71
|
headers?: Headers;
|
|
48
72
|
now?: number;
|
|
73
|
+
/**
|
|
74
|
+
* Provenance from the caller that still holds the response: this body is a refusal this
|
|
75
|
+
* proxy synthesized. The body check below is the fallback for a re-wrapped body, and it
|
|
76
|
+
* cannot answer at all when the bounded read returned nothing display-safe -- which is
|
|
77
|
+
* precisely when the empty-body branch would invent the retryable-429 default.
|
|
78
|
+
*/
|
|
79
|
+
replayRefusal?: boolean;
|
|
49
80
|
},
|
|
50
81
|
): Response {
|
|
51
82
|
const trimmed = bodyText.trim();
|
|
@@ -53,7 +84,12 @@ export function formatPassthroughUpstreamError(
|
|
|
53
84
|
const upstreamRetryAfter = options?.headers?.get("retry-after")?.trim() || undefined;
|
|
54
85
|
const originalValid = validateClientRetryAfterHeader(upstreamRetryAfter, now);
|
|
55
86
|
const cyberPolicyFailure = isCyberPolicyBody(trimmed);
|
|
56
|
-
|
|
87
|
+
// Two different reasons to answer with no wait at all, handled the same way: a hard policy
|
|
88
|
+
// block will not become servable, and a refusal we made was never a rate limit.
|
|
89
|
+
const suppressRetryAfter = cyberPolicyFailure
|
|
90
|
+
|| options?.replayRefusal === true
|
|
91
|
+
|| isReplayRefusalBody(trimmed);
|
|
92
|
+
const resolved = suppressRetryAfter
|
|
57
93
|
? undefined
|
|
58
94
|
: resolveClientRetryAfter({
|
|
59
95
|
status,
|
|
@@ -64,7 +100,7 @@ export function formatPassthroughUpstreamError(
|
|
|
64
100
|
|
|
65
101
|
if (trimmed) {
|
|
66
102
|
const needsSet = resolved !== undefined && upstreamRetryAfter !== resolved;
|
|
67
|
-
const needsDelete = (
|
|
103
|
+
const needsDelete = (suppressRetryAfter && upstreamRetryAfter !== undefined)
|
|
68
104
|
|| (resolved === undefined
|
|
69
105
|
&& upstreamRetryAfter !== undefined
|
|
70
106
|
&& originalValid === undefined);
|