@bitkyc08/opencodex 2.58.0 → 2.59.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -10
- package/gui/dist/assets/index-C5IebErG.js +136 -0
- package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
- package/gui/dist/index.html +2 -2
- package/gui/dist/provider-icons/crusoe.svg +1 -0
- package/gui/dist/provider-icons/opper.svg +3 -0
- package/package.json +1 -1
- package/src/adapters/base.ts +11 -1
- package/src/adapters/cursor/catalog.ts +11 -0
- package/src/adapters/cursor/effort-map.ts +16 -2
- package/src/adapters/cursor/envelope-echo.ts +55 -2
- package/src/adapters/cursor/message-mapper.ts +3 -2
- package/src/adapters/cursor/protobuf-request.ts +8 -5
- package/src/adapters/cursor/request-builder.ts +14 -3
- package/src/adapters/cursor/thread-continuity.ts +105 -31
- package/src/adapters/cursor/tool-guidance.ts +5 -4
- package/src/adapters/cursor.ts +42 -1
- package/src/adapters/devin/cloud-direct/chat.ts +11 -2
- package/src/adapters/devin/cloud-direct/index.ts +7 -0
- package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
- package/src/adapters/devin.ts +75 -13
- package/src/adapters/google-antigravity-wire.ts +29 -2
- package/src/adapters/google-http.ts +8 -1
- package/src/adapters/google.ts +23 -4
- package/src/adapters/openai-chat/response-events.ts +61 -0
- package/src/adapters/openai-chat.ts +5 -10
- package/src/adapters/openai-responses/passthrough.ts +10 -1
- package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
- package/src/adapters/openai-responses/tool-schema.ts +19 -7
- package/src/adapters/responses-tool-schema.ts +76 -46
- package/src/adapters/run-turn-queue.ts +17 -4
- package/src/bridge/response-json.ts +1 -1
- package/src/bridge/sse.ts +165 -24
- package/src/claude/context-windows.ts +22 -0
- package/src/claude/outbound.ts +35 -4
- package/src/cli/account-api.ts +4 -3
- package/src/cli/account-extended.ts +22 -2
- package/src/cli/account-orca-import.ts +63 -0
- package/src/cli/account.ts +32 -4
- package/src/cli/capabilities.ts +40 -0
- package/src/cli/claude.ts +29 -1
- package/src/cli/codex-cli-update.ts +97 -2
- package/src/cli/dispatch.ts +54 -0
- package/src/cli/doctor.ts +197 -2
- package/src/cli/help.ts +4 -1
- package/src/cli/index.ts +88 -20
- package/src/cli/models-runtime.ts +33 -4
- package/src/cli/registry.ts +11 -1
- package/src/cli/runtime-api.ts +44 -0
- package/src/cli/start-args.ts +94 -0
- package/src/cli/system-command.ts +2 -0
- package/src/client/machine-api.ts +4 -3
- package/src/client/machine-listener.ts +14 -1
- package/src/clients/config-export/constants.ts +2 -3
- package/src/clients/config-export.ts +5 -5
- package/src/codex/account-store.ts +81 -5
- package/src/codex/auth-api/pool-quota-probe.ts +14 -3
- package/src/codex/auth-api/routes.ts +17 -2
- package/src/codex/auth-context.ts +16 -12
- package/src/codex/catalog/build-entries.ts +25 -4
- package/src/codex/catalog/derive-entry.ts +8 -1
- package/src/codex/catalog/effort.ts +10 -6
- package/src/codex/catalog/gather-capture.ts +1 -0
- package/src/codex/catalog/model-hints.ts +37 -5
- package/src/codex/catalog/parsing.ts +83 -5
- package/src/codex/catalog/reserve-warn.ts +96 -0
- package/src/codex/catalog/retained-sync.ts +19 -0
- package/src/codex/catalog/routed-gather.ts +42 -3
- package/src/codex/cli-installation-identity.ts +210 -0
- package/src/codex/cli-installation-targets.ts +158 -0
- package/src/codex/convergence.ts +5 -0
- package/src/codex/history-provider.ts +4 -1
- package/src/codex/history-state-open.ts +105 -0
- package/src/codex/inject/config-toml.ts +44 -2
- package/src/codex/inject.ts +3 -2
- package/src/codex/lineage.ts +83 -32
- package/src/codex/loopback-target.ts +31 -0
- package/src/codex/main-account-hard-lock.ts +2 -1
- package/src/codex/main-account.ts +10 -3
- package/src/codex/main-device-reauth.ts +17 -9
- package/src/codex/model-entitlements.ts +60 -1
- package/src/codex/observed-model-denials.ts +137 -0
- package/src/codex/orca-auth-source.ts +94 -0
- package/src/codex/orca-import.ts +219 -0
- package/src/codex/prompt-text-probe.ts +282 -12
- package/src/codex/quota-401-recovery.ts +12 -0
- package/src/codex/quota-types.ts +65 -0
- package/src/codex/quota.ts +24 -19
- package/src/codex/routing/cooldown-math.ts +8 -47
- package/src/codex/routing/pin-drain.ts +57 -0
- package/src/codex/routing.ts +13 -15
- package/src/codex/subagent-model-fallback.ts +94 -0
- package/src/codex/windows-installation-files.ts +224 -0
- package/src/combos/failover.ts +122 -5
- package/src/config/diagnostics.ts +21 -0
- package/src/config/load-degrade.ts +15 -0
- package/src/config/pending-teardown.ts +8 -0
- package/src/config/process-state.ts +36 -3
- package/src/config/provider-relative-send-path.ts +16 -0
- package/src/config/proxy-env.ts +23 -5
- package/src/config/schema/config-schema.ts +21 -0
- package/src/config/schema/leaf-validators.ts +64 -17
- package/src/generated/compatibility-version.json +235 -163
- package/src/generated/model-metadata.ts +1 -1
- package/src/lib/bounded-body.ts +4 -2
- package/src/lib/destination-policy.ts +48 -6
- package/src/lib/errors.ts +3 -15
- package/src/lib/local-destinations.ts +32 -5
- package/src/lib/provider-outbound.ts +3 -3
- package/src/lib/proxy-env.ts +70 -3
- package/src/lib/request-execution-budget.ts +11 -3
- package/src/lib/response-body-inactivity.ts +193 -0
- package/src/lib/retry-delay.ts +69 -0
- package/src/lib/socks5-fetch.ts +631 -0
- package/src/lib/spend-reservation-ledger.ts +115 -9
- package/src/lib/workflow-budget.ts +145 -8
- package/src/oauth/account-quota-rank.ts +72 -15
- package/src/oauth/generic-account-failover.ts +40 -27
- package/src/oauth/orcarouter.ts +15 -2
- package/src/oauth/store.ts +8 -0
- package/src/providers/codex-capacity.ts +9 -0
- package/src/providers/devin-provider-merge-migration.ts +33 -12
- package/src/providers/free-directory.ts +20 -2
- package/src/providers/key-failover.ts +261 -7
- package/src/providers/model-rename-migration.ts +1 -0
- package/src/providers/openai-sidecar.ts +4 -0
- package/src/providers/opencode-go-transport.ts +14 -5
- package/src/providers/quota/report-cache.ts +3 -0
- package/src/providers/registry/entries-extended.ts +96 -0
- package/src/providers/registry/model-seeds.ts +78 -21
- package/src/responses/apply-patch-envelope.ts +44 -11
- package/src/responses/bridge-search-replay-cache.ts +152 -0
- package/src/responses/code-mode-helper-compat.ts +26 -16
- package/src/responses/custom-tool-compat.ts +1 -1
- package/src/responses/hosted-tool-policy.ts +85 -2
- package/src/responses/schema.ts +9 -2
- package/src/server/auth-cors.ts +26 -0
- package/src/server/chat-completions.ts +9 -4
- package/src/server/chat-native-sse.ts +26 -9
- package/src/server/chat-native.ts +10 -4
- package/src/server/claude-messages.ts +24 -2
- package/src/server/gui-static.ts +36 -2
- package/src/server/inbound-body-admission.ts +187 -0
- package/src/server/index.ts +15 -19
- package/src/server/management/api-access.ts +3 -4
- package/src/server/management/config-routes.ts +31 -6
- package/src/server/management/provider-capability-config.ts +35 -7
- package/src/server/management/provider-routes.ts +70 -18
- package/src/server/proxy-liveness.ts +97 -2
- package/src/server/relay.ts +17 -24
- package/src/server/request-log.ts +25 -1
- package/src/server/responses/adapter-continuation.ts +71 -27
- package/src/server/responses/adapter-delivery.ts +39 -8
- package/src/server/responses/adapter-dispatch.ts +52 -24
- package/src/server/responses/compact.ts +60 -11
- package/src/server/responses/core-codex-account.ts +83 -22
- package/src/server/responses/core-normalize.ts +12 -5
- package/src/server/responses/fetch-helpers.ts +68 -2
- package/src/server/responses/passthrough-delivery.ts +10 -1
- package/src/server/responses/passthrough-dispatch.ts +113 -48
- package/src/server/responses/passthrough-execution.ts +11 -1
- package/src/server/responses/request-prepare.ts +29 -0
- package/src/server/responses/request-send-budget.ts +84 -7
- package/src/server/responses/request-sidecar-auth.ts +16 -8
- package/src/server/responses/request-spend.ts +38 -9
- package/src/server/responses/request-transport.ts +13 -10
- package/src/server/responses/run-turn-execution.ts +20 -5
- package/src/server/responses/sidecar-execution.ts +2 -0
- package/src/server/responses/ws-upstream.ts +2 -1
- package/src/server/responses-custom-tool-repair.ts +2 -2
- package/src/server/sse-frame-buffer.ts +12 -10
- package/src/server/sse-payload-rewrite.ts +36 -9
- package/src/server/system-env-shell.ts +5 -1
- package/src/server/system-env.ts +7 -1
- package/src/server/workflow-refusal.ts +56 -2
- package/src/service/cli.ts +16 -6
- package/src/service/guards.ts +10 -0
- package/src/service/health.ts +43 -0
- package/src/service/state.ts +7 -2
- package/src/types/accounts.ts +4 -0
- package/src/types/config.ts +100 -3
- package/src/types/provider.ts +19 -0
- package/src/types/request.ts +7 -1
- package/src/types/wire.ts +9 -1
- package/src/usage/expected-prices.ts +28 -0
- package/src/usage/log.ts +87 -4
- package/src/web-search/passthrough-bridge.ts +39 -5
- package/gui/dist/assets/index-BbrHOIY0.js +0 -128
|
@@ -11,6 +11,7 @@ import type { PreparedResponsesRequest } from "./request-prepare";
|
|
|
11
11
|
import type { ResponsesTransport } from "./request-transport";
|
|
12
12
|
import type { ResponsesEffects } from "./response-effects";
|
|
13
13
|
import type { ResponsesSendBudget } from "./request-send-budget";
|
|
14
|
+
import { transientSendCapFor } from "./request-send-budget";
|
|
14
15
|
import { isCanonicalOpenAiForwardProvider } from "../../providers/openai-tiers";
|
|
15
16
|
import { codexSafetyBufferingFilterOptions, terminalStatusFromParsed } from "../relay";
|
|
16
17
|
import { imageGenToolCallAliases } from "../responses-image-gen-repair";
|
|
@@ -68,6 +69,7 @@ import {
|
|
|
68
69
|
sealRequestAttemptIdentity,
|
|
69
70
|
recordAttemptCredentialSource,
|
|
70
71
|
} from "../request-log";
|
|
72
|
+
import { noteAttemptRecoveryWithheld } from "../request-log";
|
|
71
73
|
import {
|
|
72
74
|
upstreamHostHealthKey,
|
|
73
75
|
normalizeUpstreamHostCircuitThreshold,
|
|
@@ -91,11 +93,15 @@ import {
|
|
|
91
93
|
usesCodexForwardPoolAuth,
|
|
92
94
|
codexWsQuotaObserver,
|
|
93
95
|
isFixedCodexAccount,
|
|
94
|
-
|
|
96
|
+
codexPoolAccountModel400Denial,
|
|
95
97
|
shouldRetryCodexPoolAccountQuota,
|
|
96
98
|
shouldRetryCodexPoolAccountTransient,
|
|
97
99
|
retryCodexPoolOnAlternateAccount,
|
|
98
100
|
} from "./core-codex-account";
|
|
101
|
+
import {
|
|
102
|
+
clearCodexModelDenialEvidence,
|
|
103
|
+
recordCodexModelDenialEvidence,
|
|
104
|
+
} from "../../codex/model-entitlements";
|
|
99
105
|
import { readCodexWsStage } from "./codex-ws-wire";
|
|
100
106
|
import { linkAbortSignal } from "./core-lifetime";
|
|
101
107
|
import type { CodexAuthContext } from "../../codex/auth-context";
|
|
@@ -105,9 +111,8 @@ import {
|
|
|
105
111
|
SendBudgetExhaustedError,
|
|
106
112
|
fetchWithTransientRetry,
|
|
107
113
|
applyUpstreamRecoveryInit,
|
|
108
|
-
TRANSIENT_RETRY_MAX_ATTEMPTS,
|
|
109
114
|
isNonReplayableResponse,
|
|
110
|
-
|
|
115
|
+
prepareSameTarget429Wait,
|
|
111
116
|
sleepWithAbort,
|
|
112
117
|
} from "../../lib/upstream-retry";
|
|
113
118
|
import { mapCodexAuthContextErrorToResponse } from "./codex-auth-error";
|
|
@@ -115,7 +120,11 @@ import { classifyTransportFailureKind, transportErrorCode } from "../../lib/upst
|
|
|
115
120
|
import { recordCodexUpstreamOutcome } from "../../codex/routing";
|
|
116
121
|
import { describeUpstreamConnectFailure } from "./upstream-error";
|
|
117
122
|
import type { OpaqueBlobRecoveryGuard } from "./core-opaque-recovery";
|
|
118
|
-
import {
|
|
123
|
+
import {
|
|
124
|
+
rateLimitRetryPolicyFor,
|
|
125
|
+
rateLimitRetryDelayMs,
|
|
126
|
+
transientRetryPolicyFor,
|
|
127
|
+
} from "../../providers/key-failover";
|
|
119
128
|
import type { AttemptRecoveryKind } from "../../usage/log";
|
|
120
129
|
import { resolveWireProtocolOverride } from "../adapter-resolve";
|
|
121
130
|
import { refreshPoolForwardAuth, refreshNativeMainForwardAuth, withClaudeNativeSession } from "./core-auth";
|
|
@@ -198,6 +207,7 @@ export async function preparePassthroughExchange(
|
|
|
198
207
|
| "reserveCredentialHop"
|
|
199
208
|
| "pendingHopPermit"
|
|
200
209
|
| "workflowRootId"
|
|
210
|
+
| "sendsUsed"
|
|
201
211
|
>,
|
|
202
212
|
) {
|
|
203
213
|
const { config, logCtx, options, req } = requestContext;
|
|
@@ -665,6 +675,33 @@ export async function preparePassthroughExchange(
|
|
|
665
675
|
linkAbortSignal(upstream, options.abortSignal);
|
|
666
676
|
const connectMs = config.connectTimeoutMs ?? 200_000;
|
|
667
677
|
let upstreamResponse: Response;
|
|
678
|
+
/**
|
|
679
|
+
* This leg's transient-5xx ladder cap, from the provider's own `transientRetryOn5xx`.
|
|
680
|
+
*
|
|
681
|
+
* The lane used to pass `TRANSIENT_RETRY_MAX_ATTEMPTS` at every call site, so an operator who
|
|
682
|
+
* configured the option on a key-auth `openai-responses` provider changed nothing in either
|
|
683
|
+
* direction, while the same provider on `openai-chat` was tuned normally. That asymmetry is
|
|
684
|
+
* #4893. The gate in `transientRetryPolicyFor` returns null for OAuth and forward providers,
|
|
685
|
+
* so the ChatGPT pool keeps exactly the ladder it has always had.
|
|
686
|
+
*
|
|
687
|
+
* Read per call rather than captured once, for two reasons. `route.provider` is reassigned
|
|
688
|
+
* inside the recovery loop by credential rotation and transport resolution, so a hoisted
|
|
689
|
+
* policy could outlive the provider row it came from. And the configured value is a total for
|
|
690
|
+
* the whole request, so it has to be measured against what the request has already sent at
|
|
691
|
+
* the moment each leg asks.
|
|
692
|
+
*
|
|
693
|
+
* Still bounded by the request: every site feeds this to `remainingTransientSendBudget` or
|
|
694
|
+
* `recoverySendAllowance`, which intersect it with the request-wide base allowance. So a
|
|
695
|
+
* provider can narrow this request's sends exactly and cannot widen the bound that exists to
|
|
696
|
+
* stop per-request amplification (#4546).
|
|
697
|
+
*/
|
|
698
|
+
const transientSendPolicy = () => transientRetryPolicyFor(route.provider);
|
|
699
|
+
const transientSendAttempts = (): number => transientSendCapFor(
|
|
700
|
+
transientSendPolicy()?.attempts,
|
|
701
|
+
sendBudgetState.sendsUsed,
|
|
702
|
+
);
|
|
703
|
+
const configuredTransientSendBudgetExhausted = (): boolean =>
|
|
704
|
+
transientSendPolicy() !== null && transientSendAttempts() === 0;
|
|
668
705
|
/**
|
|
669
706
|
* Refuse a built body that exceeds the operator's configured ceiling, before it is sent.
|
|
670
707
|
*
|
|
@@ -804,7 +841,7 @@ export async function preparePassthroughExchange(
|
|
|
804
841
|
// retry wrapper replaces — proves the host was reached (#914 review).
|
|
805
842
|
.then(adoptObservedResponse);
|
|
806
843
|
},
|
|
807
|
-
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(
|
|
844
|
+
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(transientSendAttempts()), onSendsConsumed: noteTransientSends },
|
|
808
845
|
);
|
|
809
846
|
} catch (err) {
|
|
810
847
|
return transportFailureResponse(err);
|
|
@@ -864,14 +901,15 @@ export async function preparePassthroughExchange(
|
|
|
864
901
|
recordAttemptCredentialSource(logCtx.activeAttempt, route.providerName, route.provider, retryAdapter.name);
|
|
865
902
|
const rebuiltBodyRefusal = refuseOversizedOutboundBody(request);
|
|
866
903
|
if (rebuiltBodyRefusal) return { failed: rebuiltBodyRefusal };
|
|
867
|
-
// The base allowance is spent first
|
|
868
|
-
//
|
|
869
|
-
//
|
|
870
|
-
// outside the try so the finally can hand it back if the leg never
|
|
904
|
+
// The base allowance is spent first. An unconfigured provider may then draw the one shared
|
|
905
|
+
// final-recovery reserve, which keeps a validated sanitized rebuild after a 5xx streak alive
|
|
906
|
+
// at four total sends instead of dying at three. An explicit provider total cannot widen.
|
|
907
|
+
// Reserve outside the try so the finally can hand it back if the leg never reaches its send.
|
|
871
908
|
const allowance = recoverySendAllowance(
|
|
872
|
-
|
|
909
|
+
transientSendAttempts(),
|
|
873
910
|
recoveryClassFor(recovery),
|
|
874
911
|
`${route.providerName}|${route.modelId}|${recovery}`,
|
|
912
|
+
{ allowFinalRecoveryReserve: transientSendPolicy() === null },
|
|
875
913
|
);
|
|
876
914
|
try {
|
|
877
915
|
return await fetchWithTransientRetry(
|
|
@@ -1033,7 +1071,7 @@ export async function preparePassthroughExchange(
|
|
|
1033
1071
|
// Refused here, before the 401 body is cancelled: once it is gone the request can only
|
|
1034
1072
|
// answer with a synthetic 502, which would report a proxy budget decision as an upstream
|
|
1035
1073
|
// fault and throw away the credential evidence the client needs.
|
|
1036
|
-
&& !sendBudgetExhausted()
|
|
1074
|
+
&& !sendBudgetExhausted(transientSendAttempts())
|
|
1037
1075
|
) {
|
|
1038
1076
|
oauth401ReplayAttempted = true;
|
|
1039
1077
|
try { void upstreamResponse.body?.cancel().catch(() => {}); } catch { /* already consumed/closed */ }
|
|
@@ -1134,7 +1172,7 @@ export async function preparePassthroughExchange(
|
|
|
1134
1172
|
route.provider.authMode === "forward")
|
|
1135
1173
|
.then(adoptObservedResponse);
|
|
1136
1174
|
},
|
|
1137
|
-
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(
|
|
1175
|
+
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(transientSendAttempts()), onSendsConsumed: noteTransientSends },
|
|
1138
1176
|
);
|
|
1139
1177
|
} catch (err) {
|
|
1140
1178
|
return transportFailureResponse(err);
|
|
@@ -1145,13 +1183,13 @@ export async function preparePassthroughExchange(
|
|
|
1145
1183
|
|
|
1146
1184
|
// Native Responses returns before the generic adapter's OAuth rotation loop. Keep
|
|
1147
1185
|
// the same quorum, cooldown and request budget here, before any client bytes flow.
|
|
1148
|
-
|
|
1149
|
-
|
|
1186
|
+
if (
|
|
1187
|
+
upstreamResponse.status === 429
|
|
1150
1188
|
// Not a provider rate limit when this proxy synthesized it for a refused reset
|
|
1151
1189
|
// replay; rotating accounts on it would re-send an inference that may already
|
|
1152
1190
|
// have run and would cool down an account that refused nothing.
|
|
1153
1191
|
&& !isNonReplayableResponse(upstreamResponse)
|
|
1154
|
-
|
|
1192
|
+
&& transportState.genericFailoverAccountId
|
|
1155
1193
|
&& transportState.genericFailovers < GENERIC_OAUTH_MAX_FAILOVERS_PER_REQUEST
|
|
1156
1194
|
&& isGenericOAuthFailoverEnabled(config, route.providerName)
|
|
1157
1195
|
) {
|
|
@@ -1167,6 +1205,8 @@ export async function preparePassthroughExchange(
|
|
|
1167
1205
|
const nextAccountId = rotateGenericOAuthAccountOn429(
|
|
1168
1206
|
config, route.providerName, transportState.genericFailoverAccountId,
|
|
1169
1207
|
upstreamResponse.headers.get("retry-after"),
|
|
1208
|
+
Date.now(),
|
|
1209
|
+
route.modelId,
|
|
1170
1210
|
);
|
|
1171
1211
|
let snapshot: OAuthAccessSnapshot | undefined;
|
|
1172
1212
|
if (nextAccountId) {
|
|
@@ -1194,6 +1234,11 @@ export async function preparePassthroughExchange(
|
|
|
1194
1234
|
}
|
|
1195
1235
|
// No credential moved, so the reservation costs nothing.
|
|
1196
1236
|
hop.permit?.release();
|
|
1237
|
+
} else {
|
|
1238
|
+
// Rotation was available -- the roster cap above admitted it -- and the shared request
|
|
1239
|
+
// budget refused. Recorded so a one-send log is not read as "nothing was eligible",
|
|
1240
|
+
// which is the ambiguity this attribution exists to remove (#5044).
|
|
1241
|
+
noteAttemptRecoveryWithheld(logCtx.activeAttempt, "rotation-send-budget");
|
|
1197
1242
|
}
|
|
1198
1243
|
}
|
|
1199
1244
|
|
|
@@ -1203,15 +1248,15 @@ export async function preparePassthroughExchange(
|
|
|
1203
1248
|
// immediately with no same-key replay. Pre-stream only — nothing has been relayed yet, so
|
|
1204
1249
|
// the replay is lossless (same invariant as the recovery loop). Forward/OAuth providers
|
|
1205
1250
|
// keep their pool logic below (rateLimitRetryPolicyFor returns null for them).
|
|
1206
|
-
|
|
1207
|
-
|
|
1251
|
+
while (
|
|
1252
|
+
upstreamResponse.status === 429
|
|
1208
1253
|
&& !isNonReplayableResponse(upstreamResponse)
|
|
1209
|
-
|
|
1254
|
+
&& rateLimitPolicy !== null
|
|
1210
1255
|
&& rateLimitRetries < rateLimitPolicy.attempts
|
|
1211
1256
|
// Checked here rather than inside the helper: prepareSameTarget429Wait releases the 429
|
|
1212
1257
|
// body, so a refusal discovered after the wait can no longer return the real rate-limit
|
|
1213
1258
|
// answer and would surface a synthetic 502 instead.
|
|
1214
|
-
&& !sendBudgetExhausted()
|
|
1259
|
+
&& !sendBudgetExhausted(transientSendAttempts())
|
|
1215
1260
|
) {
|
|
1216
1261
|
rateLimitRetries += 1;
|
|
1217
1262
|
// Release unread body + deliberate wait via the shared same-target helper.
|
|
@@ -1259,7 +1304,7 @@ export async function preparePassthroughExchange(
|
|
|
1259
1304
|
route.provider.authMode === "forward")
|
|
1260
1305
|
.then(adoptObservedResponse);
|
|
1261
1306
|
},
|
|
1262
|
-
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(
|
|
1307
|
+
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(transientSendAttempts()), onSendsConsumed: noteTransientSends },
|
|
1263
1308
|
);
|
|
1264
1309
|
} catch (err) {
|
|
1265
1310
|
return transportFailureResponse(err);
|
|
@@ -1291,11 +1336,25 @@ export async function preparePassthroughExchange(
|
|
|
1291
1336
|
|
|
1292
1337
|
if (usesCodexForwardPoolAuth(admissionState.authCtx, route.provider)) {
|
|
1293
1338
|
let poolRetryOutcome: number | undefined;
|
|
1294
|
-
|
|
1339
|
+
// A success is the freshest evidence there is about this pair, and it outranks any earlier
|
|
1340
|
+
// refusal: whatever the entitlement was when upstream declined, it is not that now. Both
|
|
1341
|
+
// ids are cleared because the wire model can differ from the routed one.
|
|
1342
|
+
if (upstreamResponse.ok) {
|
|
1343
|
+
clearCodexModelDenialEvidence(admissionState.authCtx.accountId, route.modelId);
|
|
1344
|
+
clearCodexModelDenialEvidence(admissionState.authCtx.accountId, parsed.modelId);
|
|
1345
|
+
}
|
|
1346
|
+
const model400Denial = await codexPoolAccountModel400Denial(
|
|
1295
1347
|
upstreamResponse,
|
|
1296
1348
|
route.modelId,
|
|
1297
1349
|
options.abortSignal,
|
|
1298
|
-
|
|
1350
|
+
parsed.modelId,
|
|
1351
|
+
);
|
|
1352
|
+
if (model400Denial !== undefined) {
|
|
1353
|
+
// Spend this refusal on more than one retry. It is the account's own authenticated
|
|
1354
|
+
// answer about this model, and the roster cache that selection otherwise reads expires
|
|
1355
|
+
// five minutes after a catalog sync fills it -- so without remembering this, the next
|
|
1356
|
+
// request selects the same account on quota alone and takes the same 400 (#4906).
|
|
1357
|
+
recordCodexModelDenialEvidence(admissionState.authCtx.accountId, model400Denial);
|
|
1299
1358
|
poolRetryOutcome = 400;
|
|
1300
1359
|
} else if (!admissionState.authCtx.fixedAccount && await shouldRetryCodexPoolAccountQuota(
|
|
1301
1360
|
upstreamResponse,
|
|
@@ -1365,18 +1424,20 @@ export async function preparePassthroughExchange(
|
|
|
1365
1424
|
// eviction, or an older transcript). Inspect only a bounded clone of a 4xx whose exact outbound
|
|
1366
1425
|
// Responses body still carries opaque state, then rebuild once through the ordinary adapter
|
|
1367
1426
|
// sanitation path. A second rejection falls through unchanged because the guard stays armed.
|
|
1368
|
-
|
|
1369
|
-
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
|
|
1375
|
-
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1427
|
+
if (!configuredTransientSendBudgetExhausted()) {
|
|
1428
|
+
const opaqueBlobRecovery = await attemptOpaqueBlobRecovery({
|
|
1429
|
+
response: upstreamResponse,
|
|
1430
|
+
outboundBody: request.body,
|
|
1431
|
+
adapterName: transportState.adapter.name,
|
|
1432
|
+
parsed,
|
|
1433
|
+
guard: opaqueBlobRecoveryGuard,
|
|
1434
|
+
signal: upstream.signal,
|
|
1435
|
+
}, rebuildAndRefetch);
|
|
1436
|
+
if (opaqueBlobRecovery.kind === "failed") return opaqueBlobRecovery.response;
|
|
1437
|
+
if (opaqueBlobRecovery.kind === "recovered") {
|
|
1438
|
+
upstreamResponse = opaqueBlobRecovery.response;
|
|
1439
|
+
continue passthroughRecovery;
|
|
1440
|
+
}
|
|
1380
1441
|
}
|
|
1381
1442
|
|
|
1382
1443
|
const recoveryContentType = upstreamResponse.headers.get("content-type")?.toLowerCase() ?? "";
|
|
@@ -1384,6 +1445,7 @@ export async function preparePassthroughExchange(
|
|
|
1384
1445
|
&& !!upstreamResponse.body
|
|
1385
1446
|
&& (recoveryContentType.includes("text/event-stream") || (!recoveryContentType && parsed.stream))
|
|
1386
1447
|
&& !opaqueBlobRecoveryGuard.attempted
|
|
1448
|
+
&& !configuredTransientSendBudgetExhausted()
|
|
1387
1449
|
&& outboundResponsesBodyCarriesEncryptedFunctionOutput(request.body);
|
|
1388
1450
|
if (streamedFunctionOutputCandidate) {
|
|
1389
1451
|
const preflightLog: RequestLogContext = { model: logCtx.model, provider: logCtx.provider };
|
|
@@ -1400,19 +1462,21 @@ export async function preparePassthroughExchange(
|
|
|
1400
1462
|
if (options.abortSignal?.aborted) return transportFailureResponse(options.abortSignal.reason);
|
|
1401
1463
|
upstreamResponse = preflight.response;
|
|
1402
1464
|
if (preflight.kind === "failed") {
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1465
|
+
if (!configuredTransientSendBudgetExhausted()) {
|
|
1466
|
+
const streamedOpaqueRecovery = await attemptOpaqueBlobRecovery({
|
|
1467
|
+
response: upstreamResponse,
|
|
1468
|
+
outboundBody: request.body,
|
|
1469
|
+
adapterName: transportState.adapter.name,
|
|
1470
|
+
parsed,
|
|
1471
|
+
guard: opaqueBlobRecoveryGuard,
|
|
1472
|
+
signal: upstream.signal,
|
|
1473
|
+
}, rebuildAndRefetch);
|
|
1474
|
+
if (streamedOpaqueRecovery.kind === "failed") return streamedOpaqueRecovery.response;
|
|
1475
|
+
if (streamedOpaqueRecovery.kind === "recovered") {
|
|
1476
|
+
resetStreamedOpaqueBlobLogContext(logCtx);
|
|
1477
|
+
upstreamResponse = streamedOpaqueRecovery.response;
|
|
1478
|
+
continue passthroughRecovery;
|
|
1479
|
+
}
|
|
1416
1480
|
}
|
|
1417
1481
|
logCtx.upstreamError = preflightLog.upstreamError;
|
|
1418
1482
|
logCtx.terminalHttpStatus = preflightLog.terminalHttpStatus;
|
|
@@ -1431,6 +1495,7 @@ export async function preparePassthroughExchange(
|
|
|
1431
1495
|
upstream.signal,
|
|
1432
1496
|
);
|
|
1433
1497
|
if (uploadRejectionBody !== undefined
|
|
1498
|
+
&& !configuredTransientSendBudgetExhausted()
|
|
1434
1499
|
&& isTransientConsoleGoUploadRejection({
|
|
1435
1500
|
status: upstreamResponse.status,
|
|
1436
1501
|
errorBody: uploadRejectionBody,
|
|
@@ -1469,7 +1534,7 @@ export async function preparePassthroughExchange(
|
|
|
1469
1534
|
requested: parsed.options.reasoning,
|
|
1470
1535
|
rejectionText,
|
|
1471
1536
|
});
|
|
1472
|
-
if (downgrade) {
|
|
1537
|
+
if (downgrade && !configuredTransientSendBudgetExhausted()) {
|
|
1473
1538
|
reasoningEffortDowngradeGuard.attempted = true;
|
|
1474
1539
|
parsed.options.reasoning = downgrade.effort;
|
|
1475
1540
|
try { void upstreamResponse.body?.cancel().catch(() => {}); } catch { /* already consumed/closed */ }
|
|
@@ -10,6 +10,8 @@ import type { ResponsesEffects } from "./response-effects";
|
|
|
10
10
|
import type { ResponsesSendBudget } from "./request-send-budget";
|
|
11
11
|
import { preparePassthroughExchange } from "./passthrough-dispatch";
|
|
12
12
|
import { deliverPassthroughResponse } from "./passthrough-delivery";
|
|
13
|
+
import { guardDirectPassthroughBodyInactivity } from "../../lib/response-body-inactivity";
|
|
14
|
+
import { resolveStallTimeoutSec } from "../../stall-timeout";
|
|
13
15
|
import { releaseUpstreamHostAdmission } from "../../codex/upstream-host-health";
|
|
14
16
|
import { releaseCodexAuthContextProbeLease } from "../../codex/auth-context";
|
|
15
17
|
|
|
@@ -36,7 +38,7 @@ export async function executePassthroughResponse(
|
|
|
36
38
|
sendBudgetState,
|
|
37
39
|
);
|
|
38
40
|
if (nativeExchange instanceof Response) return nativeExchange;
|
|
39
|
-
|
|
41
|
+
const response = await deliverPassthroughResponse(
|
|
40
42
|
requestContext,
|
|
41
43
|
admissionState,
|
|
42
44
|
requestState,
|
|
@@ -45,6 +47,14 @@ export async function executePassthroughResponse(
|
|
|
45
47
|
responseEffects,
|
|
46
48
|
nativeExchange,
|
|
47
49
|
);
|
|
50
|
+
// Delivery has already classified SSE (including a missing upstream content type)
|
|
51
|
+
// and consumed bounded JSON/errors. Guard only its remaining direct body, outside
|
|
52
|
+
// the relay/lifetime wrappers so their prefetch cannot arm our inactivity clock.
|
|
53
|
+
return guardDirectPassthroughBodyInactivity(
|
|
54
|
+
response,
|
|
55
|
+
nativeExchange.upstream.signal,
|
|
56
|
+
resolveStallTimeoutSec(requestContext.config.stallTimeoutSec) * 1000,
|
|
57
|
+
);
|
|
48
58
|
} finally {
|
|
49
59
|
if (nativeHostState.lease) {
|
|
50
60
|
releaseUpstreamHostAdmission(nativeHostState.lease);
|
|
@@ -105,7 +105,9 @@ import { hasUnmappedRoutedCustomToolOutput } from "../../responses/custom-tool-c
|
|
|
105
105
|
import { PROVIDER_OWNED_CONTINUATION_WIRES, resolvedAdapterWire } from "../../responses/continuation-ownership";
|
|
106
106
|
import {
|
|
107
107
|
isCodexReserveHelperUnsupported,
|
|
108
|
+
isCodexReserveOptInMissing,
|
|
108
109
|
CODEX_RESERVE_HELPER_UNSUPPORTED_MESSAGE,
|
|
110
|
+
CODEX_RESERVE_OPT_IN_REQUIRED_MESSAGE,
|
|
109
111
|
} from "../../codex/loopback-target";
|
|
110
112
|
import { checkComboTargetInputAdmission, checkInputAdmission } from "./input-admission";
|
|
111
113
|
import { nativeContextLimits } from "../../codex/catalog";
|
|
@@ -231,6 +233,10 @@ export async function prepareResponsesRequest(
|
|
|
231
233
|
(body as { input?: unknown } | undefined)?.input,
|
|
232
234
|
);
|
|
233
235
|
const inboundClientThreadId = req.headers.get("x-codex-parent-thread-id")?.trim() || undefined;
|
|
236
|
+
// The request's OWN thread, which `x-codex-parent-thread-id` is not: parallel children of one
|
|
237
|
+
// parent all present the same parent id. `codexConversationIdentity` already reads this header
|
|
238
|
+
// for the same reason, and a surface that must tell siblings apart needs it too (#5033).
|
|
239
|
+
const inboundOwnThreadId = req.headers.get("thread-id")?.trim() || undefined;
|
|
234
240
|
const cursorClientThreadId = codexPoolAffinityKey(req.headers);
|
|
235
241
|
const originalBody = body;
|
|
236
242
|
if (options.comboReplaySnapshot) {
|
|
@@ -303,6 +309,7 @@ export async function prepareResponsesRequest(
|
|
|
303
309
|
? options.comboReplaySnapshot.providerContinuation
|
|
304
310
|
: previousResponseProviderState(parsed.previousResponseId);
|
|
305
311
|
if (providerContinuationCandidate) parsed._providerContinuationCandidate = providerContinuationCandidate;
|
|
312
|
+
if (inboundOwnThreadId) parsed._codexOwnThreadId = inboundOwnThreadId;
|
|
306
313
|
if (inboundClientThreadId) {
|
|
307
314
|
parsed._clientThreadId = inboundClientThreadId;
|
|
308
315
|
} else if (
|
|
@@ -699,6 +706,7 @@ export async function prepareResponsesRequest(
|
|
|
699
706
|
"_providerContinuationOwner",
|
|
700
707
|
"_cursorConversationId",
|
|
701
708
|
"_clientThreadId",
|
|
709
|
+
"_codexOwnThreadId",
|
|
702
710
|
"_promptCacheKeyIsSharedCohort",
|
|
703
711
|
"_cursorClientThreadId",
|
|
704
712
|
"_reasoningReplayScope",
|
|
@@ -976,6 +984,27 @@ export async function prepareResponsesRequest(
|
|
|
976
984
|
options.admission, options.visionDescribeTerminal === true)) {
|
|
977
985
|
return formatErrorResponse(400, "invalid_request_error", CODEX_RESERVE_HELPER_UNSUPPORTED_MESSAGE);
|
|
978
986
|
}
|
|
987
|
+
// #4940: the opt-in is off, so every Reserve affordance in this process is inert -- no catalog
|
|
988
|
+
// row, no main-credential substitution, no authorization handshake, and no `luna-reserve` header
|
|
989
|
+
// on the send. Forwarding `gpt-reserve` as an ordinary native model therefore buys nothing but a
|
|
990
|
+
// 429 "The usage limit has been reached", which names neither the real cause nor the setting the
|
|
991
|
+
// operator would have to change. Refuse here instead, on the same terms and in the same place as
|
|
992
|
+
// the helper refusal above: after alias/combo resolution, before auth, host-circuit budget or any
|
|
993
|
+
// upstream byte.
|
|
994
|
+
//
|
|
995
|
+
// Two narrowings beyond the predicate, both about not answering a question this refusal cannot
|
|
996
|
+
// answer correctly. A terminal vision/search helper is excluded because enabling the opt-in would
|
|
997
|
+
// not make it work -- it would produce the helper refusal above instead, so telling that caller to
|
|
998
|
+
// enable the flag is advice that does not hold. Non-native inbound wires are excluded because a
|
|
999
|
+
// `gpt-reserve` selector reaching us over Chat or Anthropic Messages is an operator-authored
|
|
1000
|
+
// route (a `claudeCode.modelMap` entry, say), not a Codex client that was forced onto Reserve by
|
|
1001
|
+
// its own usage snapshot, and that route keeps whatever behavior it has today.
|
|
1002
|
+
if (inboundWire === "responses"
|
|
1003
|
+
&& options.visionDescribeTerminal !== true
|
|
1004
|
+
&& isCanonicalOpenAiForwardProvider(route.provider)
|
|
1005
|
+
&& isCodexReserveOptInMissing(options.codexAuthPolicy ?? config, route.modelId, options.admission)) {
|
|
1006
|
+
return formatErrorResponse(400, "invalid_request_error", CODEX_RESERVE_OPT_IN_REQUIRED_MESSAGE);
|
|
1007
|
+
}
|
|
979
1008
|
// Refuse an input that cannot plausibly fit the model context window before spending auth,
|
|
980
1009
|
// circuit budget, or upstream bandwidth on a turn the provider will reject anyway (#1412).
|
|
981
1010
|
//
|
|
@@ -1,9 +1,13 @@
|
|
|
1
1
|
import type { ResponsesRequestContext } from "./core-options";
|
|
2
2
|
import { createRequestExecutionBudget, isRequestExecutionBudget } from "../../lib/request-execution-budget";
|
|
3
|
-
import {
|
|
3
|
+
import {
|
|
4
|
+
chargeWorkflowSends,
|
|
5
|
+
workflowSendCeilingReached,
|
|
6
|
+
workflowSpendCeilingReached,
|
|
7
|
+
} from "../../lib/workflow-budget";
|
|
4
8
|
import { workflowRefusalResponse } from "../workflow-refusal";
|
|
5
|
-
import type { AttemptRecoveryKind } from "../../usage/log";
|
|
6
|
-
import { noteAttemptSend } from "../request-log";
|
|
9
|
+
import type { AttemptRecoveryKind, AttemptRecoveryWithheld } from "../../usage/log";
|
|
10
|
+
import { noteAttemptRecoveryWithheld, noteAttemptSend } from "../request-log";
|
|
7
11
|
import { TRANSIENT_RETRY_MAX_ATTEMPTS } from "../../lib/upstream-retry";
|
|
8
12
|
import type {
|
|
9
13
|
DispatchDecision,
|
|
@@ -13,6 +17,28 @@ import type {
|
|
|
13
17
|
SingleUseDispatchPermit,
|
|
14
18
|
} from "../../lib/request-execution-budget";
|
|
15
19
|
|
|
20
|
+
/**
|
|
21
|
+
* The transient-5xx ladder cap for ONE leg, given the provider's configured request total.
|
|
22
|
+
*
|
|
23
|
+
* `remainingTransientSendBudget(cap)` treats `cap` as a ceiling on what REMAINS, which is the
|
|
24
|
+
* right shape for the fixed constant: every leg may ask for up to three, and the request-wide
|
|
25
|
+
* base allowance is what actually bounds the total. A configured `transientRetryOn5xx.attempts`
|
|
26
|
+
* is documented as the total for one request including the first send, so it has to be reduced
|
|
27
|
+
* by what the request already sent before it is intersected with that allowance. Passing it
|
|
28
|
+
* straight through would make it a per-leg ceiling instead, and a request configured at one send
|
|
29
|
+
* could still reach upstream again on a recovery leg (#4893).
|
|
30
|
+
*
|
|
31
|
+
* An absent policy returns the constant unchanged, so a provider that configures nothing behaves
|
|
32
|
+
* exactly as it does today at every call site.
|
|
33
|
+
*/
|
|
34
|
+
export function transientSendCapFor(
|
|
35
|
+
configuredAttempts: number | undefined,
|
|
36
|
+
sendsUsed: number,
|
|
37
|
+
): number {
|
|
38
|
+
if (configuredAttempts === undefined) return TRANSIENT_RETRY_MAX_ATTEMPTS;
|
|
39
|
+
return Math.max(0, configuredAttempts - Math.max(0, sendsUsed));
|
|
40
|
+
}
|
|
41
|
+
|
|
16
42
|
/** Owns the shared request send counter and recovery permits. */
|
|
17
43
|
export function createResponsesSendBudget(
|
|
18
44
|
requestContext: Pick<ResponsesRequestContext, "options" | "req" | "logCtx">,
|
|
@@ -45,6 +71,22 @@ export function createResponsesSendBudget(
|
|
|
45
71
|
// marked synthetic rather than reading as a request that vanished with zero sends.
|
|
46
72
|
return workflowRefusalResponse("workflow-sends-exhausted", logCtx, undefined, workflowRootId);
|
|
47
73
|
}
|
|
74
|
+
// The token ceiling asked at the same seam, for the same reason the count one is asked here.
|
|
75
|
+
// Without it a spent root reaches the dispatch ladder, is refused by the ledger at the first
|
|
76
|
+
// physical send, and answers with the generic send-budget error every exhausted request
|
|
77
|
+
// returns -- a refusal an operator cannot tell from an ordinary budget exhaustion, on a
|
|
78
|
+
// ceiling they configured themselves. Asked before dispatch, it names the scope and the
|
|
79
|
+
// number instead. Returns undefined and touches no ledger when no ceiling is configured.
|
|
80
|
+
const spentCeiling = workflowSpendCeilingReached(workflowRootId);
|
|
81
|
+
if (spentCeiling) {
|
|
82
|
+
return workflowRefusalResponse(
|
|
83
|
+
"workflow-spend-exhausted",
|
|
84
|
+
logCtx,
|
|
85
|
+
undefined,
|
|
86
|
+
workflowRootId,
|
|
87
|
+
spentCeiling,
|
|
88
|
+
);
|
|
89
|
+
}
|
|
48
90
|
// No floor. Math.max(1, ...) meant an exhausted request still funded one send on every
|
|
49
91
|
// recovery leg, so a bounded per-leg allowance never became a bounded per-request one.
|
|
50
92
|
const remainingTransientSendBudget = (budget: number): number =>
|
|
@@ -71,8 +113,28 @@ export function createResponsesSendBudget(
|
|
|
71
113
|
if (send.ordinal <= 1) return;
|
|
72
114
|
noteAttemptSend(logCtx.activeAttempt, inputTokens, send.recovery);
|
|
73
115
|
};
|
|
74
|
-
|
|
75
|
-
|
|
116
|
+
/**
|
|
117
|
+
* Records a recovery an adapter was ready to make and the budget refused.
|
|
118
|
+
*
|
|
119
|
+
* No send happened, so this deliberately does not touch `sendCount`. It is the other half of
|
|
120
|
+
* the pair that makes a one-send log readable: no recovery kind AND no withheld reason means
|
|
121
|
+
* nothing was eligible; a withheld reason means something was (#5044).
|
|
122
|
+
*/
|
|
123
|
+
const noteAdapterRecoveryWithheld = (withheld: { reason: AttemptRecoveryWithheld }): void => {
|
|
124
|
+
noteAttemptRecoveryWithheld(logCtx.activeAttempt, withheld.reason);
|
|
125
|
+
};
|
|
126
|
+
/**
|
|
127
|
+
* Whether this request has any base send left under `cap`.
|
|
128
|
+
*
|
|
129
|
+
* The cap is a parameter because a lane that reads a provider's configured
|
|
130
|
+
* `transientRetryOn5xx` ladder has to ask this question at the SAME cap its sends use.
|
|
131
|
+
* Asking at the constant while dispatching at a configured value lets a provider with
|
|
132
|
+
* headroom be told it is exhausted, and lets one configured below the constant pass this
|
|
133
|
+
* check and then be refused at the send (#4893). Defaulted, so every existing caller keeps
|
|
134
|
+
* the constant it already used.
|
|
135
|
+
*/
|
|
136
|
+
const sendBudgetExhausted = (cap: number = TRANSIENT_RETRY_MAX_ATTEMPTS): boolean =>
|
|
137
|
+
remainingTransientSendBudget(cap) === 0;
|
|
76
138
|
/**
|
|
77
139
|
* A credential hop reserves the send its own replay will make, and that replay is a recovery
|
|
78
140
|
* leg. The leg must SPEND the hop's reservation instead of taking a second one: the
|
|
@@ -113,16 +175,22 @@ export function createResponsesSendBudget(
|
|
|
113
175
|
* The base allowance is spent first. Once it is gone a recovery class may still draw the
|
|
114
176
|
* single shared final-recovery reserve -- which is what keeps the validated sanitized rebuild
|
|
115
177
|
* after a 5xx streak alive at four total sends -- but an account move and a rebuild cannot
|
|
116
|
-
* each take one.
|
|
117
|
-
*
|
|
178
|
+
* each take one. A caller with an exact provider total can suppress that reserve. The
|
|
179
|
+
* `countedExternally` flag exists because these legs run through the retry helper, which reports
|
|
180
|
+
* the same send again through `onSendsConsumed`.
|
|
118
181
|
*/
|
|
119
182
|
const recoverySendAllowance = (
|
|
120
183
|
cap: number,
|
|
121
184
|
sendClass: SendClass,
|
|
122
185
|
targetKey: string,
|
|
186
|
+
options: { allowFinalRecoveryReserve?: boolean } = {},
|
|
123
187
|
): { attempts: number; permit?: SingleUseDispatchPermit } => {
|
|
124
188
|
const base = remainingTransientSendBudget(cap);
|
|
125
189
|
if (base > 0) return { attempts: base };
|
|
190
|
+
// A provider-configured transient total is an exact physical-send ceiling. Once it is
|
|
191
|
+
// exhausted, the request-wide recovery reserve must not silently widen it. The default stays
|
|
192
|
+
// permissive so unconfigured providers retain the guarded profile's fourth recovery send.
|
|
193
|
+
if (options.allowFinalRecoveryReserve === false) return { attempts: 0 };
|
|
126
194
|
if (pendingHopPermit) {
|
|
127
195
|
const hopPermit = pendingHopPermit;
|
|
128
196
|
pendingHopPermit = undefined;
|
|
@@ -178,9 +246,18 @@ export function createResponsesSendBudget(
|
|
|
178
246
|
workflowRootId,
|
|
179
247
|
noteTransientSends,
|
|
180
248
|
remainingTransientSendBudget,
|
|
249
|
+
/**
|
|
250
|
+
* Physical sends this logical request has already made.
|
|
251
|
+
*
|
|
252
|
+
* A live getter, not a snapshot: it is read once per dispatch leg to resolve a configured
|
|
253
|
+
* ladder, and a value frozen at construction would answer for a request that had sent
|
|
254
|
+
* nothing.
|
|
255
|
+
*/
|
|
256
|
+
get sendsUsed(): number { return sendBudget.used; },
|
|
181
257
|
adapterSendBudget,
|
|
182
258
|
adapterDispatchBudget,
|
|
183
259
|
noteAdapterPhysicalSend,
|
|
260
|
+
noteAdapterRecoveryWithheld,
|
|
184
261
|
sendBudgetExhausted,
|
|
185
262
|
get pendingHopPermit(): SingleUseDispatchPermit | undefined {
|
|
186
263
|
return pendingHopPermit;
|
|
@@ -126,18 +126,26 @@ export async function prepareResponsesSidecarAuth(
|
|
|
126
126
|
});
|
|
127
127
|
const recordSidecarOutcome = openAiSidecar?.recordOutcome;
|
|
128
128
|
if (visionPlan) {
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
129
|
+
try {
|
|
130
|
+
await describeImagesInPlace(
|
|
131
|
+
parsed,
|
|
132
|
+
visionPlan,
|
|
133
|
+
openAiSidecar?.headers ?? requestState.selectedForwardHeaders,
|
|
134
|
+
options.abortSignal,
|
|
135
|
+
recordSidecarOutcome,
|
|
136
|
+
translatorBudget,
|
|
137
|
+
);
|
|
138
|
+
} finally {
|
|
139
|
+
// Local validation can reject every image before the sidecar fetch records an outcome.
|
|
140
|
+
// Vision-only turns must hand that unused cooldown probe back; when a fetch did run the
|
|
141
|
+
// outcome already consumed it, so this generation-bound release is a safe no-op.
|
|
142
|
+
if (!needsOpenAiSearch) openAiSidecar?.releaseProbeLease?.();
|
|
143
|
+
}
|
|
137
144
|
} else if (requiresVisionPreprocessing(config, route.provider, route.modelId, route.providerName)) {
|
|
138
145
|
// Image capability is not positively proven but no sidecar plan is dispatchable: fail closed.
|
|
139
146
|
// Never forward raw image bytes to an unverified upstream.
|
|
140
147
|
stripImagesInPlace(parsed, translatorBudget);
|
|
148
|
+
if (!needsOpenAiSearch) openAiSidecar?.releaseProbeLease?.();
|
|
141
149
|
}
|
|
142
150
|
|
|
143
151
|
return {
|