@bitkyc08/opencodex 2.54.0-preview.20260914 → 2.55.0-preview.20260914
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/gui/dist/assets/{index-B4VYfZcY.js → index-DH2PUHqr.js} +10 -10
- package/gui/dist/index.html +1 -1
- package/package.json +1 -1
- package/src/adapters/anthropic-image-codec.ts +57 -0
- package/src/adapters/anthropic-image-normalize.ts +28 -1
- package/src/adapters/anthropic.ts +68 -6
- package/src/adapters/base.ts +8 -0
- package/src/adapters/coding-agent/protocol.ts +41 -16
- package/src/adapters/cursor/cursor-errors.ts +1 -1
- package/src/adapters/cursor/live-transport.ts +5 -1
- package/src/adapters/cursor/native-exec-fs.ts +10 -10
- package/src/adapters/cursor/native-exec-network.ts +2 -2
- package/src/adapters/cursor/native-exec-shell.ts +13 -12
- package/src/adapters/cursor/native-exec.ts +51 -10
- package/src/adapters/cursor/policy-error.ts +75 -0
- package/src/adapters/cursor/protobuf-request.ts +105 -1
- package/src/adapters/devin/cloud-direct/catalog.ts +34 -2
- package/src/adapters/devin/live-models.ts +33 -2
- package/src/adapters/google-wire-compiler.ts +8 -0
- package/src/adapters/google.ts +46 -0
- package/src/adapters/input-media-guard.ts +45 -0
- package/src/adapters/kiro/adapter.ts +8 -0
- package/src/adapters/kiro/payload.ts +28 -6
- package/src/adapters/kiro-events.ts +25 -6
- package/src/adapters/kiro-images.ts +30 -0
- package/src/adapters/kiro-retry.ts +8 -0
- package/src/adapters/openai-chat.ts +33 -4
- package/src/adapters/openai-responses.ts +26 -0
- package/src/adapters/registry.ts +4 -0
- package/src/bridge.ts +163 -116
- package/src/chat/image-parts.ts +151 -0
- package/src/chat/inbound.ts +70 -33
- package/src/cli/connect.ts +30 -9
- package/src/cli/dispatch.ts +7 -3
- package/src/cli/index.ts +3 -0
- package/src/cli/runtime-api.ts +25 -0
- package/src/cli/status.ts +21 -19
- package/src/cli/system-restart-client.ts +25 -0
- package/src/clients/config-export.ts +14 -4
- package/src/codex/app-server-processes.ts +25 -0
- package/src/codex/auth-context.ts +8 -0
- package/src/codex/autostart-health.ts +36 -2
- package/src/codex/catalog/provider-fetch.ts +41 -0
- package/src/codex/catalog-auto-refresh.ts +182 -0
- package/src/codex/catalog-refresh-status.ts +93 -0
- package/src/codex/history-provider.ts +55 -0
- package/src/codex/model-entitlements.ts +78 -0
- package/src/codex/native-profile-processes.ts +114 -15
- package/src/codex/prompt-text-probe.ts +274 -41
- package/src/codex/routing-adoption.ts +189 -0
- package/src/codex/routing.ts +520 -48
- package/src/codex/runtime.ts +249 -7
- package/src/combos/failover.ts +45 -0
- package/src/config.ts +124 -4
- package/src/generated/compatibility-version.json +110 -74
- package/src/generated/model-metadata.ts +1 -0
- package/src/lib/request-execution-budget.ts +202 -0
- package/src/lib/upstream-retry.ts +95 -8
- package/src/lib/workflow-budget.ts +172 -0
- package/src/oauth/devin.ts +57 -12
- package/src/providers/quota.ts +37 -6
- package/src/providers/registry.ts +53 -6
- package/src/responses/input-media.ts +65 -0
- package/src/responses/parser-content.ts +42 -0
- package/src/responses/schema.ts +12 -2
- package/src/server/audio-live.ts +1 -2
- package/src/server/audio-transcriptions.ts +1 -2
- package/src/server/auth-cors.ts +1 -1
- package/src/server/background-lifecycle.ts +18 -0
- package/src/server/chat-completions.ts +23 -8
- package/src/server/chat-native.ts +17 -17
- package/src/server/index.ts +24 -0
- package/src/server/management/request-history-routes.ts +5 -0
- package/src/server/request-log.ts +8 -2
- package/src/server/responses/compact.ts +51 -3
- package/src/server/responses/core.ts +238 -28
- package/src/server/search.ts +7 -9
- package/src/types/config.ts +51 -6
- package/src/usage/log.ts +37 -0
- package/src/vision/eligibility.ts +37 -4
- package/src/vision/index.ts +1 -0
- package/src/vision/plan.ts +45 -10
- package/src/web-search/alpha-search.ts +324 -0
- package/src/web-search/index.ts +13 -22
- package/src/web-search/passthrough-bridge.ts +195 -22
- package/src/web-search/sidecar-providers.ts +22 -0
|
@@ -165,7 +165,7 @@ import {
|
|
|
165
165
|
shouldResolveOpenAiPassthroughWebSearchBridge,
|
|
166
166
|
} from "../../web-search/passthrough-bridge";
|
|
167
167
|
import { buildImageTool, buildVideoTool, planImageBridge, planVideoBridge, runWithImageBridge, clampImageMaxRounds, IMAGE_GEN_TOOL_NAME, VIDEO_GEN_TOOL_NAME } from "../../images";
|
|
168
|
-
import { describeImagesInPlace,
|
|
168
|
+
import { describeImagesInPlace, planVisionSidecar, requiresVisionPreprocessing, resolveOpenAiVisionModel, shouldResolveOpenAiVisionSidecar, stripImagesInPlace } from "../../vision";
|
|
169
169
|
import { createAdapterEventQueue, preflightAdapterEvents, type AdapterEventQueue } from "../../adapters/run-turn-queue";
|
|
170
170
|
import {
|
|
171
171
|
applyCodexAuthContextToProvider,
|
|
@@ -223,7 +223,20 @@ import {
|
|
|
223
223
|
isTransientUpstreamStatus,
|
|
224
224
|
prepareSameTarget429Wait,
|
|
225
225
|
sleepWithAbort,
|
|
226
|
+
TRANSIENT_RETRY_MAX_ATTEMPTS,
|
|
227
|
+
SendBudgetExhaustedError,
|
|
228
|
+
type TransientSendBudget,
|
|
226
229
|
} from "../../lib/upstream-retry";
|
|
230
|
+
import {
|
|
231
|
+
createRequestExecutionBudget,
|
|
232
|
+
isRequestExecutionBudget,
|
|
233
|
+
type SendClass,
|
|
234
|
+
type SingleUseDispatchPermit,
|
|
235
|
+
} from "../../lib/request-execution-budget";
|
|
236
|
+
import {
|
|
237
|
+
chargeWorkflowSends,
|
|
238
|
+
workflowSendCeilingReached,
|
|
239
|
+
} from "../../lib/workflow-budget";
|
|
227
240
|
import {
|
|
228
241
|
ForwardAdmissionCredentialError,
|
|
229
242
|
hasForwardableCodexBearer,
|
|
@@ -781,6 +794,18 @@ function isEncryptedFunctionOutputRejection(bodyText: string): boolean {
|
|
|
781
794
|
}
|
|
782
795
|
}
|
|
783
796
|
|
|
797
|
+
/**
|
|
798
|
+
* #4469: reasoning encrypted_content is minted per caller identity, so replaying it under a
|
|
799
|
+
* different caller is rejected with "reasoning `encrypted_content` was not issued to this
|
|
800
|
+
* caller". Substring checks tolerate the optional backticks and a leading or trailing
|
|
801
|
+
* sentence, while the "was not issued to this caller" anchor plus an encrypted-content or
|
|
802
|
+
* reasoning subject keep unrelated invalid_request_error prose from gaining a hidden resend.
|
|
803
|
+
*/
|
|
804
|
+
function isReasoningBlobCallerMismatchMessage(message: string): boolean {
|
|
805
|
+
if (!message.includes("was not issued to this caller")) return false;
|
|
806
|
+
return message.includes("encrypted_content") || message.includes("reasoning");
|
|
807
|
+
}
|
|
808
|
+
|
|
784
809
|
function isSelfIdentifiedOpaqueBlobRejection(bodyText: string): boolean {
|
|
785
810
|
if (isEncryptedFunctionOutputRejection(bodyText)) return true;
|
|
786
811
|
try {
|
|
@@ -793,7 +818,7 @@ function isSelfIdentifiedOpaqueBlobRejection(bodyText: string): boolean {
|
|
|
793
818
|
try {
|
|
794
819
|
const payload = JSON.parse(bodyText) as unknown;
|
|
795
820
|
if (!payload || typeof payload !== "object" || Array.isArray(payload)) return false;
|
|
796
|
-
const record = payload as { code?: unknown; error?: unknown };
|
|
821
|
+
const record = payload as { code?: unknown; type?: unknown; message?: unknown; error?: unknown };
|
|
797
822
|
|
|
798
823
|
if (record.error && typeof record.error === "object" && !Array.isArray(record.error)) {
|
|
799
824
|
const error = record.error as { type?: unknown; code?: unknown; message?: unknown };
|
|
@@ -807,9 +832,23 @@ function isSelfIdentifiedOpaqueBlobRejection(bodyText: string): boolean {
|
|
|
807
832
|
" could not be verified. Reason: Encrypted content could not be decrypted or parsed.",
|
|
808
833
|
)
|
|
809
834
|
) return true;
|
|
835
|
+
// #4469: the caller-mismatch wording arrives without a dedicated code, so the
|
|
836
|
+
// message itself is the identity. It is not gated on code being null — the upstream
|
|
837
|
+
// may attach a generic code — because the anchored phrase is already specific.
|
|
838
|
+
if (typeof error.message === "string" && isReasoningBlobCallerMismatchMessage(error.message)) {
|
|
839
|
+
return true;
|
|
840
|
+
}
|
|
810
841
|
}
|
|
811
842
|
}
|
|
812
843
|
|
|
844
|
+
// The flat stream-error envelope carries type/message at the top level rather than under
|
|
845
|
+
// an error object; the same anchored identity applies there.
|
|
846
|
+
if (
|
|
847
|
+
record.type === "invalid_request_error"
|
|
848
|
+
&& typeof record.message === "string"
|
|
849
|
+
&& isReasoningBlobCallerMismatchMessage(record.message)
|
|
850
|
+
) return true;
|
|
851
|
+
|
|
813
852
|
if (record.code !== "invalid-argument" || typeof record.error !== "string") return false;
|
|
814
853
|
return record.error.startsWith("Could not decode the compaction blob")
|
|
815
854
|
|| record.error.startsWith("Could not decrypt the provided encrypted_content");
|
|
@@ -824,8 +863,9 @@ function isSelfIdentifiedOpaqueBlobRejection(bodyText: string): boolean {
|
|
|
824
863
|
* The outbound-body check is intentional: the inbound transcript may contain a proxy envelope or
|
|
825
864
|
* compaction blob that the adapter already lowered, in which case a replay would be byte-identical.
|
|
826
865
|
* OpenAI usually exposes a dedicated nested code; ChatGPT also emits one exact code-less
|
|
827
|
-
* unverifiable-ciphertext message
|
|
828
|
-
*
|
|
866
|
+
* unverifiable-ciphertext message, and #4469 added the anchored caller-mismatch wording for
|
|
867
|
+
* reasoning blobs minted under a different caller. xAI's code is generic, so its two concrete
|
|
868
|
+
* decoder error identities are also required. Unrelated error prose must never gain a hidden resend.
|
|
829
869
|
*/
|
|
830
870
|
export function shouldAttemptOpaqueBlobRecovery(args: {
|
|
831
871
|
status: number;
|
|
@@ -1259,6 +1299,10 @@ interface CodexPoolAccountRetryArgs {
|
|
|
1259
1299
|
translatorBudget: TranslatorBudget;
|
|
1260
1300
|
turnAdmissionLease?: AdmissionLease;
|
|
1261
1301
|
resolveCodexModelEntitlements?: typeof resolveCodexModelEntitlements;
|
|
1302
|
+
/** The logical request's execution budget: the account move is its fourth send. */
|
|
1303
|
+
sendBudget?: TransientSendBudget;
|
|
1304
|
+
/** Root workflow this turn belongs to, so the move is charged there as well. */
|
|
1305
|
+
workflowRootId?: string;
|
|
1262
1306
|
};
|
|
1263
1307
|
firstAuthCtx: Extract<CodexAuthContext, { kind: "pool" | "main-pool" }>;
|
|
1264
1308
|
firstResponse: Response;
|
|
@@ -1453,6 +1497,27 @@ async function retryCodexPoolOnAlternateAccount(
|
|
|
1453
1497
|
recordUnmovedTransientOutcome();
|
|
1454
1498
|
return { kind: "no-alternate" };
|
|
1455
1499
|
}
|
|
1500
|
+
// An account move is the guarded profile's fourth send and draws the single shared
|
|
1501
|
+
// final-recovery reserve. Nothing bounded it per request before: `excludeAccountId` excludes
|
|
1502
|
+
// only the account that just failed, and the caller's recovery loop can return here after the
|
|
1503
|
+
// alternate fails too, so one request could walk the pool an account at a time. The permit is
|
|
1504
|
+
// consumed immediately before the physical send, so a resolution that finds no alternate
|
|
1505
|
+
// costs nothing.
|
|
1506
|
+
const executionBudget = isRequestExecutionBudget(args.options.sendBudget)
|
|
1507
|
+
? args.options.sendBudget
|
|
1508
|
+
: undefined;
|
|
1509
|
+
let accountMovePermit: SingleUseDispatchPermit | undefined;
|
|
1510
|
+
if (!retryAuthCtx && executionBudget) {
|
|
1511
|
+
const decision = executionBudget.reserveDispatch({
|
|
1512
|
+
sendClass: "account-failover",
|
|
1513
|
+
targetKey: `${route.providerName}|${route.modelId}|alternate-account`,
|
|
1514
|
+
});
|
|
1515
|
+
if (!decision.allowed) {
|
|
1516
|
+
recordUnmovedTransientOutcome();
|
|
1517
|
+
return { kind: "no-alternate" };
|
|
1518
|
+
}
|
|
1519
|
+
accountMovePermit = decision.permit;
|
|
1520
|
+
}
|
|
1456
1521
|
try {
|
|
1457
1522
|
retryAuthCtx ??= await resolveCodexAuthContext(
|
|
1458
1523
|
callerAuthHeaders,
|
|
@@ -1591,6 +1656,18 @@ async function retryCodexPoolOnAlternateAccount(
|
|
|
1591
1656
|
let upstreamResponse: Response;
|
|
1592
1657
|
try {
|
|
1593
1658
|
while (true) {
|
|
1659
|
+
// The same-account gated-model 400 ladder below keeps its own `maxRetrySends` bound and
|
|
1660
|
+
// does not take the reserve again; only the move itself does.
|
|
1661
|
+
if (accountMovePermit) {
|
|
1662
|
+
const charged = accountMovePermit.use();
|
|
1663
|
+
accountMovePermit = undefined;
|
|
1664
|
+
if (!charged) {
|
|
1665
|
+
recordUnmovedTransientOutcome();
|
|
1666
|
+
return { kind: "no-alternate" };
|
|
1667
|
+
}
|
|
1668
|
+
// The move is a physical send like any other, so the root workflow is charged too.
|
|
1669
|
+
chargeWorkflowSends(args.options.workflowRootId, 1);
|
|
1670
|
+
}
|
|
1594
1671
|
noteAttemptSend(logCtx.activeAttempt, passthroughEstimate);
|
|
1595
1672
|
try {
|
|
1596
1673
|
upstreamResponse = await fetchWithHeaderTimeout(
|
|
@@ -1874,6 +1951,12 @@ export interface HandleResponsesOptions {
|
|
|
1874
1951
|
onStoredPool401ReplayDispatched?: () => void;
|
|
1875
1952
|
/** Caller-owned for Chat/Claude replay; omitted only at genuine Responses ingress. */
|
|
1876
1953
|
translatorBudget?: TranslatorBudget;
|
|
1954
|
+
/**
|
|
1955
|
+
* Transient sends already spent by this logical request. Combo children inherit the parent's
|
|
1956
|
+
* holder through the options spread, so a fan-out shares one allowance instead of taking a
|
|
1957
|
+
* fresh one per target (#4546).
|
|
1958
|
+
*/
|
|
1959
|
+
sendBudget?: TransientSendBudget;
|
|
1877
1960
|
/**
|
|
1878
1961
|
* Terminal vision-describe marker (roadmap 180): true when the inbound
|
|
1879
1962
|
* request IS the vision sidecar's own loopback describe call. The plan site
|
|
@@ -3420,6 +3503,9 @@ export async function handleResponses(
|
|
|
3420
3503
|
visionDescribeTerminal: options.visionDescribeTerminal === true
|
|
3421
3504
|
|| req.headers.get("x-opencodex-vision-describe") === "1",
|
|
3422
3505
|
translatorBudget,
|
|
3506
|
+
// Created once at genuine ingress; a combo child arrives with the parent's holder already
|
|
3507
|
+
// in options and must not start a fresh allowance.
|
|
3508
|
+
sendBudget: options.sendBudget ?? createRequestExecutionBudget(),
|
|
3423
3509
|
});
|
|
3424
3510
|
return ownsBudget ? finalizeOwnedTranslatorBudget(response, translatorBudget) : response;
|
|
3425
3511
|
} catch (error) {
|
|
@@ -4187,6 +4273,13 @@ async function handleResponsesInner(
|
|
|
4187
4273
|
? `${route.providerName}-${route.codexAccountNamespace}`
|
|
4188
4274
|
: formatCodexProviderForLog(route.providerName, codexLogAccountId(authCtx), config);
|
|
4189
4275
|
logCtx.accountLogLabel = codexAuthContextLogLabel(authCtx, config);
|
|
4276
|
+
// A move is the expensive event: it discards the prefix warmed on the previous account. Record
|
|
4277
|
+
// it as an event with its cause, so the operator reads it off one line instead of inferring it
|
|
4278
|
+
// from account labels across many (#4546).
|
|
4279
|
+
if (authCtx.kind === "pool" && authCtx.affinityDecision) {
|
|
4280
|
+
logCtx.affinity = authCtx.affinityDecision.move;
|
|
4281
|
+
logCtx.affinityReason = authCtx.affinityDecision.reason;
|
|
4282
|
+
}
|
|
4190
4283
|
// Seed an account-derived scope before final adapter binding. Cursor never treats it as
|
|
4191
4284
|
// authoritative: bindRouteReasoningReplayScope replaces it with the exact route owner or a
|
|
4192
4285
|
// per-request fail-closed sentinel after the final provider and credential are known.
|
|
@@ -4769,7 +4862,7 @@ async function handleResponsesInner(
|
|
|
4769
4862
|
const routedCompaction = parsed._compactionRequest === true
|
|
4770
4863
|
&& !isCanonicalOpenAiForwardProvider(route.provider);
|
|
4771
4864
|
const needsOpenAiVision = !visionDescribeTerminal
|
|
4772
|
-
&& shouldResolveOpenAiVisionSidecar(config, route.provider, route.modelId, parsed);
|
|
4865
|
+
&& shouldResolveOpenAiVisionSidecar(config, route.provider, route.modelId, parsed, route.providerName);
|
|
4773
4866
|
const needsOpenAiSearch = !routedCompaction && !adapter.runTurn
|
|
4774
4867
|
&& (shouldResolveOpenAiWebSearchSidecar(config, parsed, isPassthrough)
|
|
4775
4868
|
|| shouldResolveOpenAiPassthroughWebSearchBridge(route.provider, parsed, isPassthrough));
|
|
@@ -4839,7 +4932,7 @@ async function handleResponsesInner(
|
|
|
4839
4932
|
const visionPlan = visionDescribeTerminal
|
|
4840
4933
|
? undefined
|
|
4841
4934
|
: planVisionSidecar(config, route.provider, route.modelId, parsed, openAiSidecar, {
|
|
4842
|
-
admission: options.admission, codexAuthPolicy: options.codexAuthPolicy,
|
|
4935
|
+
admission: options.admission, codexAuthPolicy: options.codexAuthPolicy, providerName: route.providerName,
|
|
4843
4936
|
});
|
|
4844
4937
|
const recordSidecarOutcome = openAiSidecar?.recordOutcome;
|
|
4845
4938
|
if (visionPlan) {
|
|
@@ -4851,9 +4944,9 @@ async function handleResponsesInner(
|
|
|
4851
4944
|
recordSidecarOutcome,
|
|
4852
4945
|
translatorBudget,
|
|
4853
4946
|
);
|
|
4854
|
-
} else if (
|
|
4855
|
-
//
|
|
4856
|
-
//
|
|
4947
|
+
} else if (requiresVisionPreprocessing(config, route.provider, route.modelId, route.providerName)) {
|
|
4948
|
+
// Image capability is not positively proven but no sidecar plan is dispatchable: fail closed.
|
|
4949
|
+
// Never forward raw image bytes to an unverified upstream.
|
|
4857
4950
|
stripImagesInPlace(parsed, translatorBudget);
|
|
4858
4951
|
}
|
|
4859
4952
|
|
|
@@ -4937,6 +5030,73 @@ async function handleResponsesInner(
|
|
|
4937
5030
|
routedMuseToolNameAliases = builtRequest.convertedMuseToolNameAliases ?? new Map();
|
|
4938
5031
|
};
|
|
4939
5032
|
|
|
5033
|
+
// One transient-retry budget for the whole LOGICAL request, read ABOVE the passthrough branch
|
|
5034
|
+
// so that branch shares it too. It used to be a local declared below, which put it in the
|
|
5035
|
+
// temporal dead zone for the passthrough sends and left each recovery leg taking the helper's
|
|
5036
|
+
// fresh default of 3. It is now a holder carried on options, so a combo child inherits the
|
|
5037
|
+
// parent's spend instead of starting over per target -- both halves of the measured
|
|
5038
|
+
// amplification in #4546.
|
|
5039
|
+
const sendBudget = options.sendBudget ?? createRequestExecutionBudget();
|
|
5040
|
+
// The root workflow is the user-visible task. A per-request cap cannot bound a fan-out that
|
|
5041
|
+
// sends once per child seven hundred times, so every send charged to the request is charged
|
|
5042
|
+
// to the root as well (#4546).
|
|
5043
|
+
const workflowRootId = req.headers.get("x-codex-parent-thread-id")?.trim() || undefined;
|
|
5044
|
+
const noteTransientSends = (used: number): void => {
|
|
5045
|
+
const charged = Math.max(0, used);
|
|
5046
|
+
sendBudget.used += charged;
|
|
5047
|
+
chargeWorkflowSends(workflowRootId, charged);
|
|
5048
|
+
};
|
|
5049
|
+
// Refused before any dispatch, and deliberately not by evicting the root's ledger entry:
|
|
5050
|
+
// dropping the record to make room would hand the fan-out a fresh allowance, which is the
|
|
5051
|
+
// laundering this ceiling exists to stop. The client is told the task needs a new grant
|
|
5052
|
+
// rather than being given a synthetic upstream error.
|
|
5053
|
+
if (workflowSendCeilingReached(workflowRootId)) {
|
|
5054
|
+
return formatErrorResponse(
|
|
5055
|
+
429,
|
|
5056
|
+
"workflow_budget_exhausted",
|
|
5057
|
+
"This task has used its whole send budget, so no further upstream request was made. Requests already in flight settle as they finish.",
|
|
5058
|
+
);
|
|
5059
|
+
}
|
|
5060
|
+
// No floor. Math.max(1, ...) meant an exhausted request still funded one send on every
|
|
5061
|
+
// recovery leg, so a bounded per-leg allowance never became a bounded per-request one.
|
|
5062
|
+
const remainingTransientSendBudget = (budget: number): number =>
|
|
5063
|
+
isRequestExecutionBudget(sendBudget)
|
|
5064
|
+
? sendBudget.remainingBaseSends(budget)
|
|
5065
|
+
: Math.max(0, budget - sendBudget.used);
|
|
5066
|
+
// The adapter contract needs the full budget, not just the counter. options.sendBudget is
|
|
5067
|
+
// typed as the narrow holder so a caller that predates this can still pass one, so narrow it
|
|
5068
|
+
// once here rather than asserting at each adapter call site.
|
|
5069
|
+
const adapterSendBudget = isRequestExecutionBudget(sendBudget) ? sendBudget : undefined;
|
|
5070
|
+
const sendBudgetExhausted = (): boolean =>
|
|
5071
|
+
remainingTransientSendBudget(TRANSIENT_RETRY_MAX_ATTEMPTS) === 0;
|
|
5072
|
+
/**
|
|
5073
|
+
* How many sends a recovery leg may make, and the permit that authorises the last one.
|
|
5074
|
+
*
|
|
5075
|
+
* The base allowance is spent first. Once it is gone a recovery class may still draw the
|
|
5076
|
+
* single shared final-recovery reserve -- which is what keeps the validated sanitized rebuild
|
|
5077
|
+
* after a 5xx streak alive at four total sends -- but an account move and a rebuild cannot
|
|
5078
|
+
* each take one. `countedExternally` is set because these legs run through the retry helper,
|
|
5079
|
+
* which reports the same send again through `onSendsConsumed`.
|
|
5080
|
+
*/
|
|
5081
|
+
const recoverySendAllowance = (
|
|
5082
|
+
cap: number,
|
|
5083
|
+
sendClass: SendClass,
|
|
5084
|
+
targetKey: string,
|
|
5085
|
+
): { attempts: number; permit?: SingleUseDispatchPermit } => {
|
|
5086
|
+
const base = remainingTransientSendBudget(cap);
|
|
5087
|
+
if (base > 0) return { attempts: base };
|
|
5088
|
+
if (!isRequestExecutionBudget(sendBudget)) return { attempts: 0 };
|
|
5089
|
+
const decision = sendBudget.reserveDispatch({ sendClass, targetKey, countedExternally: true });
|
|
5090
|
+
return decision.allowed ? { attempts: 1, permit: decision.permit } : { attempts: 0 };
|
|
5091
|
+
};
|
|
5092
|
+
/**
|
|
5093
|
+
* Both classes share the one reserve, so this only changes what the decision is called --
|
|
5094
|
+
* but a recovery event that says "repair" when a credential refresh drove it is the kind of
|
|
5095
|
+
* mislabelled evidence #4592 existed to stop.
|
|
5096
|
+
*/
|
|
5097
|
+
const recoveryClassFor = (recovery: AttemptRecoveryKind): SendClass =>
|
|
5098
|
+
/401|429|oauth|rate-limit|key/.test(recovery) ? "auth-recovery" : "repair";
|
|
5099
|
+
|
|
4940
5100
|
if ("passthrough" in adapter && adapter.passthrough && !routedCompaction) {
|
|
4941
5101
|
let hostAdmissionLease = pendingHostAdmissionLease;
|
|
4942
5102
|
pendingHostAdmissionLease = null;
|
|
@@ -5409,6 +5569,15 @@ async function handleResponsesInner(
|
|
|
5409
5569
|
releaseCodexAuthContextProbeLease(authCtx);
|
|
5410
5570
|
return clientCancelledResponse();
|
|
5411
5571
|
}
|
|
5572
|
+
// A budget refusal is a proxy decision, not an upstream fault. Reporting it as
|
|
5573
|
+
// 502 upstream_error would blame the provider for a limit this process applied, and
|
|
5574
|
+
// would record a fake reachability failure against the account's health.
|
|
5575
|
+
if (err instanceof SendBudgetExhaustedError) {
|
|
5576
|
+
releaseUpstreamHostAdmission(hostAdmissionLease);
|
|
5577
|
+
hostAdmissionLease = null;
|
|
5578
|
+
releaseCodexAuthContextProbeLease(authCtx);
|
|
5579
|
+
return formatErrorResponse(429, "request_send_budget_exhausted", err.message);
|
|
5580
|
+
}
|
|
5412
5581
|
const localRefusal = mapCodexAuthContextErrorToResponse(unwrapUpstreamRetryEvidenceError(err), {
|
|
5413
5582
|
now: Date.now(), accountSelector: route.codexAccountNamespace,
|
|
5414
5583
|
});
|
|
@@ -5479,7 +5648,7 @@ async function handleResponsesInner(
|
|
|
5479
5648
|
// retry wrapper replaces — proves the host was reached (#914 review).
|
|
5480
5649
|
.then(adoptObservedResponse);
|
|
5481
5650
|
},
|
|
5482
|
-
{ abortSignal: upstream.signal, label: safeHostLabel(request.url) },
|
|
5651
|
+
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(TRANSIENT_RETRY_MAX_ATTEMPTS), onSendsConsumed: noteTransientSends },
|
|
5483
5652
|
);
|
|
5484
5653
|
} catch (err) {
|
|
5485
5654
|
return transportFailureResponse(err);
|
|
@@ -5540,8 +5709,21 @@ async function handleResponsesInner(
|
|
|
5540
5709
|
const rebuiltBodyRefusal = refuseOversizedOutboundBody(request);
|
|
5541
5710
|
if (rebuiltBodyRefusal) return { failed: rebuiltBodyRefusal };
|
|
5542
5711
|
try {
|
|
5712
|
+
// The base allowance is spent first; once it is gone this leg may still draw the one
|
|
5713
|
+
// shared final-recovery reserve, which is what keeps a validated sanitized rebuild
|
|
5714
|
+
// after a 5xx streak alive at four total sends instead of dying at three.
|
|
5715
|
+
const allowance = recoverySendAllowance(
|
|
5716
|
+
TRANSIENT_RETRY_MAX_ATTEMPTS,
|
|
5717
|
+
recoveryClassFor(recovery),
|
|
5718
|
+
`${route.providerName}|${route.modelId}|${recovery}`,
|
|
5719
|
+
);
|
|
5543
5720
|
return await fetchWithTransientRetry(
|
|
5544
5721
|
innerRecovery => {
|
|
5722
|
+
// Gated on the return, not fire-and-forget: a consumed permit means this leg
|
|
5723
|
+
// already sent once, and letting the second call through would be a free send.
|
|
5724
|
+
if (allowance.permit && !allowance.permit.use()) {
|
|
5725
|
+
throw new SendBudgetExhaustedError(safeHostLabel(request.url));
|
|
5726
|
+
}
|
|
5545
5727
|
noteAttemptSend(logCtx.activeAttempt, passthroughEstimate, innerRecovery ?? recovery);
|
|
5546
5728
|
return fetchWithHeaderTimeout(request.url, applyUpstreamRecoveryInit({
|
|
5547
5729
|
method: request.method,
|
|
@@ -5559,7 +5741,7 @@ async function handleResponsesInner(
|
|
|
5559
5741
|
route.provider.authMode === "forward")
|
|
5560
5742
|
.then(adoptObservedResponse);
|
|
5561
5743
|
},
|
|
5562
|
-
{ abortSignal: upstream.signal, label: safeHostLabel(request.url) },
|
|
5744
|
+
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: allowance.attempts, onSendsConsumed: noteTransientSends },
|
|
5563
5745
|
);
|
|
5564
5746
|
} catch (err) {
|
|
5565
5747
|
return { failed: transportFailureResponse(err) };
|
|
@@ -5682,6 +5864,10 @@ async function handleResponsesInner(
|
|
|
5682
5864
|
&& isOAuth401ReplayProvider
|
|
5683
5865
|
&& sentOAuthSnapshot
|
|
5684
5866
|
&& !oauth401ReplayAttempted
|
|
5867
|
+
// Refused here, before the 401 body is cancelled: once it is gone the request can only
|
|
5868
|
+
// answer with a synthetic 502, which would report a proxy budget decision as an upstream
|
|
5869
|
+
// fault and throw away the credential evidence the client needs.
|
|
5870
|
+
&& !sendBudgetExhausted()
|
|
5685
5871
|
) {
|
|
5686
5872
|
oauth401ReplayAttempted = true;
|
|
5687
5873
|
try { void upstreamResponse.body?.cancel().catch(() => {}); } catch { /* already consumed/closed */ }
|
|
@@ -5779,7 +5965,7 @@ async function handleResponsesInner(
|
|
|
5779
5965
|
route.provider.authMode === "forward")
|
|
5780
5966
|
.then(adoptObservedResponse);
|
|
5781
5967
|
},
|
|
5782
|
-
{ abortSignal: upstream.signal, label: safeHostLabel(request.url) },
|
|
5968
|
+
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(TRANSIENT_RETRY_MAX_ATTEMPTS), onSendsConsumed: noteTransientSends },
|
|
5783
5969
|
);
|
|
5784
5970
|
} catch (err) {
|
|
5785
5971
|
return transportFailureResponse(err);
|
|
@@ -5832,6 +6018,10 @@ async function handleResponsesInner(
|
|
|
5832
6018
|
upstreamResponse.status === 429
|
|
5833
6019
|
&& rateLimitPolicy !== null
|
|
5834
6020
|
&& rateLimitRetries < rateLimitPolicy.attempts
|
|
6021
|
+
// Checked here rather than inside the helper: prepareSameTarget429Wait releases the 429
|
|
6022
|
+
// body, so a refusal discovered after the wait can no longer return the real rate-limit
|
|
6023
|
+
// answer and would surface a synthetic 502 instead.
|
|
6024
|
+
&& !sendBudgetExhausted()
|
|
5835
6025
|
) {
|
|
5836
6026
|
rateLimitRetries += 1;
|
|
5837
6027
|
// Release unread body + deliberate wait via the shared same-target helper.
|
|
@@ -5876,7 +6066,7 @@ async function handleResponsesInner(
|
|
|
5876
6066
|
route.provider.authMode === "forward")
|
|
5877
6067
|
.then(adoptObservedResponse);
|
|
5878
6068
|
},
|
|
5879
|
-
{ abortSignal: upstream.signal, label: safeHostLabel(request.url) },
|
|
6069
|
+
{ abortSignal: upstream.signal, label: safeHostLabel(request.url), attempts: remainingTransientSendBudget(TRANSIENT_RETRY_MAX_ATTEMPTS), onSendsConsumed: noteTransientSends },
|
|
5880
6070
|
);
|
|
5881
6071
|
} catch (err) {
|
|
5882
6072
|
return transportFailureResponse(err);
|
|
@@ -5943,7 +6133,7 @@ async function handleResponsesInner(
|
|
|
5943
6133
|
route,
|
|
5944
6134
|
parsed,
|
|
5945
6135
|
logCtx,
|
|
5946
|
-
options,
|
|
6136
|
+
options: { ...options, workflowRootId },
|
|
5947
6137
|
firstAuthCtx: authCtx,
|
|
5948
6138
|
firstResponse: upstreamResponse,
|
|
5949
6139
|
outcomeStatus: poolRetryOutcome,
|
|
@@ -6251,6 +6441,7 @@ async function handleResponsesInner(
|
|
|
6251
6441
|
openAiSidecar,
|
|
6252
6442
|
);
|
|
6253
6443
|
const webSearchBridgePlan = planPassthroughWebSearchBridge(parsed, route.provider, {
|
|
6444
|
+
providerName: route.providerName,
|
|
6254
6445
|
isPassthrough: true,
|
|
6255
6446
|
stream: parsed.stream === true,
|
|
6256
6447
|
auth: webSearchBridgeAuth,
|
|
@@ -6291,7 +6482,7 @@ async function handleResponsesInner(
|
|
|
6291
6482
|
providerApiKey: route.provider.apiKey ?? "",
|
|
6292
6483
|
auth: webSearchBridgeAuth,
|
|
6293
6484
|
hostedTool: parsed._webSearch,
|
|
6294
|
-
describeImages:
|
|
6485
|
+
describeImages: requiresVisionPreprocessing(config, route.provider, route.modelId, route.providerName),
|
|
6295
6486
|
sidecar: config.webSearchSidecar,
|
|
6296
6487
|
}),
|
|
6297
6488
|
// Appending a search result can push the continuation past the ceiling the first leg
|
|
@@ -6817,7 +7008,7 @@ async function handleResponsesInner(
|
|
|
6817
7008
|
// can proceed for web-search-only turns
|
|
6818
7009
|
const wsPlan = !routedCompaction
|
|
6819
7010
|
? planWebSearch(config, parsed, false, route.provider, route.modelId, openAiSidecar, {
|
|
6820
|
-
admission: options.admission, codexAuthPolicy: options.codexAuthPolicy,
|
|
7011
|
+
admission: options.admission, codexAuthPolicy: options.codexAuthPolicy, providerName: route.providerName,
|
|
6821
7012
|
})
|
|
6822
7013
|
: undefined;
|
|
6823
7014
|
const imgPlan = !routedCompaction ? await planImageBridge(config, parsed, route.provider) : undefined;
|
|
@@ -7516,13 +7707,6 @@ async function handleResponsesInner(
|
|
|
7516
7707
|
notifyResponseComplete(json);
|
|
7517
7708
|
return new Response(JSON.stringify(json), { headers: { "Content-Type": "application/json" } });
|
|
7518
7709
|
}
|
|
7519
|
-
// One request-scoped transient-retry budget owner, declared here so BOTH the initial send
|
|
7520
|
-
// and the later recovery refetches (429, key/account rotation, OAuth replay) share it. A
|
|
7521
|
-
// per-leg budget would let a request that recovers several times multiply upstream load.
|
|
7522
|
-
let transientSendsUsed = 0;
|
|
7523
|
-
const noteTransientSends = (used: number): void => { transientSendsUsed += Math.max(0, used); };
|
|
7524
|
-
const remainingTransientSendBudget = (budget: number): number =>
|
|
7525
|
-
Math.max(1, budget - transientSendsUsed);
|
|
7526
7710
|
try {
|
|
7527
7711
|
initialRequest = await activeAdapter.buildRequest(parsed, { headers: selectedForwardHeaders, translatorBudget });
|
|
7528
7712
|
refreshRequestToolAliases(initialRequest);
|
|
@@ -7565,6 +7749,7 @@ async function handleResponsesInner(
|
|
|
7565
7749
|
upstreamResponse = await activeAdapter.fetchResponse(builtInitialRequest, {
|
|
7566
7750
|
abortSignal: upstream.signal,
|
|
7567
7751
|
timeoutMs: connectMs,
|
|
7752
|
+
sendBudget: adapterSendBudget,
|
|
7568
7753
|
stream: parsed.stream,
|
|
7569
7754
|
executor: providerFetch(route.provider, options.codexWsRuntimeIdentity, {
|
|
7570
7755
|
dispatchOverride: oauthDispatch(builtInitialRequest),
|
|
@@ -7602,7 +7787,13 @@ async function handleResponsesInner(
|
|
|
7602
7787
|
abortSignal: upstream.signal,
|
|
7603
7788
|
label: safeHostLabel(builtInitialRequest.url),
|
|
7604
7789
|
...(transientPolicy
|
|
7605
|
-
|
|
7790
|
+
// Draws the remainder, not the raw policy. A combo child inherits the parent's
|
|
7791
|
+
// holder but used to take a fresh full allowance on its own first send, so the
|
|
7792
|
+
// shared counter was inherited without ever being read as a limit.
|
|
7793
|
+
? {
|
|
7794
|
+
attempts: remainingTransientSendBudget(transientPolicy.attempts),
|
|
7795
|
+
onSendsConsumed: noteTransientSends,
|
|
7796
|
+
}
|
|
7606
7797
|
: {}),
|
|
7607
7798
|
},
|
|
7608
7799
|
);
|
|
@@ -7693,6 +7884,7 @@ async function handleResponsesInner(
|
|
|
7693
7884
|
return await activeAdapter.fetchResponse(retryRequest, {
|
|
7694
7885
|
abortSignal: upstream.signal,
|
|
7695
7886
|
timeoutMs: connectMs,
|
|
7887
|
+
sendBudget: adapterSendBudget,
|
|
7696
7888
|
stream: parsed.stream,
|
|
7697
7889
|
executor: providerFetch(route.provider, options.codexWsRuntimeIdentity, {
|
|
7698
7890
|
dispatchOverride: oauthDispatch(retryRequest),
|
|
@@ -7711,8 +7903,22 @@ async function handleResponsesInner(
|
|
|
7711
7903
|
const refetchWithPolicy = (route.provider.adapter === "google" || refetchTransientPolicy)
|
|
7712
7904
|
? fetchWithTransientRetry
|
|
7713
7905
|
: fetchWithResetRetry;
|
|
7906
|
+
// Same rule as the passthrough rebuild: spend the base allowance first, then the one
|
|
7907
|
+
// shared final-recovery reserve, so a recovery that follows a spent streak still gets
|
|
7908
|
+
// its single send instead of dying at three.
|
|
7909
|
+
const refetchAllowance = refetchTransientPolicy
|
|
7910
|
+
? recoverySendAllowance(
|
|
7911
|
+
refetchTransientPolicy.attempts,
|
|
7912
|
+
recoveryClassFor(recovery),
|
|
7913
|
+
`${route.providerName}|${route.modelId}|${recovery}`,
|
|
7914
|
+
)
|
|
7915
|
+
: undefined;
|
|
7714
7916
|
return await refetchWithPolicy(
|
|
7715
|
-
recoveryKind =>
|
|
7917
|
+
recoveryKind => {
|
|
7918
|
+
if (refetchAllowance?.permit && !refetchAllowance.permit.use()) {
|
|
7919
|
+
throw new SendBudgetExhaustedError(safeHostLabel(retryRequest.url));
|
|
7920
|
+
}
|
|
7921
|
+
return fetchWithHeaderTimeout(retryRequest.url,
|
|
7716
7922
|
applyUpstreamRecoveryInit({
|
|
7717
7923
|
method: retryRequest.method, headers: retryRequest.headers, body: retryRequest.body,
|
|
7718
7924
|
}, recoveryKind), upstream.signal, connectMs, parsed.stream,
|
|
@@ -7720,13 +7926,14 @@ async function handleResponsesInner(
|
|
|
7720
7926
|
dispatchOverride: oauthDispatch(retryRequest),
|
|
7721
7927
|
providerName: route.providerName,
|
|
7722
7928
|
modelId: route.modelId,
|
|
7723
|
-
}))
|
|
7929
|
+
}));
|
|
7930
|
+
},
|
|
7724
7931
|
{
|
|
7725
7932
|
abortSignal: upstream.signal,
|
|
7726
7933
|
label: safeHostLabel(retryRequest.url),
|
|
7727
|
-
...(
|
|
7934
|
+
...(refetchAllowance
|
|
7728
7935
|
? {
|
|
7729
|
-
attempts:
|
|
7936
|
+
attempts: refetchAllowance.attempts,
|
|
7730
7937
|
onSendsConsumed: noteTransientSends,
|
|
7731
7938
|
}
|
|
7732
7939
|
: {}),
|
|
@@ -7752,6 +7959,7 @@ async function handleResponsesInner(
|
|
|
7752
7959
|
&& isOAuth401ReplayProvider
|
|
7753
7960
|
&& sentOAuthSnapshot
|
|
7754
7961
|
&& !oauth401ReplayAttempted
|
|
7962
|
+
&& !sendBudgetExhausted()
|
|
7755
7963
|
) {
|
|
7756
7964
|
oauth401ReplayAttempted = true;
|
|
7757
7965
|
try { void upstreamResponse.body?.cancel().catch(() => {}); } catch { /* already consumed/closed */ }
|
|
@@ -7846,6 +8054,7 @@ async function handleResponsesInner(
|
|
|
7846
8054
|
upstreamResponse.status === 429
|
|
7847
8055
|
&& rateLimitPolicy !== null
|
|
7848
8056
|
&& rateLimitRetries < rateLimitPolicy.attempts
|
|
8057
|
+
&& !sendBudgetExhausted()
|
|
7849
8058
|
) {
|
|
7850
8059
|
rateLimitRetries += 1;
|
|
7851
8060
|
// Release unread body + deliberate wait via the shared same-target helper.
|
|
@@ -8233,6 +8442,7 @@ async function handleResponsesInner(
|
|
|
8233
8442
|
return await activeAdapter.fetchResponse(builtContinuationRequest, {
|
|
8234
8443
|
abortSignal: upstream.signal,
|
|
8235
8444
|
timeoutMs: connectMs,
|
|
8445
|
+
sendBudget: adapterSendBudget,
|
|
8236
8446
|
stream: nextParsed.stream,
|
|
8237
8447
|
executor: providerFetch(route.provider, options.codexWsRuntimeIdentity, {
|
|
8238
8448
|
dispatchOverride: oauthDispatch(builtContinuationRequest, nextParsed),
|
package/src/server/search.ts
CHANGED
|
@@ -4,9 +4,11 @@
|
|
|
4
4
|
* codex-rs's built-in search client executes CLIENT-SIDE: it POSTs `alpha/search` against the
|
|
5
5
|
* configured base_url with the same ChatGPT bearer auth used for model requests. Under Design B
|
|
6
6
|
* injection base_url is this proxy, so the request otherwise dies on the /v1/* JSON-404 guard.
|
|
7
|
-
* The endpoint is private to the ChatGPT Codex backend, so
|
|
8
|
-
*
|
|
9
|
-
*
|
|
7
|
+
* The endpoint is private to the ChatGPT Codex backend, so the honest answer while a forward
|
|
8
|
+
* provider is configured is to copy bytes. When none is, a configured web-search sidecar
|
|
9
|
+
* (anthropic / xai / gemini / exa) can still answer — see src/web-search/alpha-search.ts.
|
|
10
|
+
* That fallback never runs while a forward candidate exists, and never borrows a different
|
|
11
|
+
* paid backend than the one the operator named.
|
|
10
12
|
*/
|
|
11
13
|
import { formatErrorResponse } from "../bridge";
|
|
12
14
|
import {
|
|
@@ -34,6 +36,7 @@ import {
|
|
|
34
36
|
type ExactOpenAiSidecarAccount,
|
|
35
37
|
} from "../providers/openai-sidecar";
|
|
36
38
|
import { routeModel } from "../router";
|
|
39
|
+
import { handleAlphaSearchSidecarFallback } from "../web-search/alpha-search";
|
|
37
40
|
import { readJsonRequestBody, resolveInboundBodyLimitBytes } from "./request-decompress";
|
|
38
41
|
import { ForwardAdmissionCredentialError, validateForwardAdmissionCredential } from "./auth-cors";
|
|
39
42
|
import type { RequestLogContext } from "./request-log";
|
|
@@ -105,12 +108,7 @@ export async function handleSearch(
|
|
|
105
108
|
}
|
|
106
109
|
const candidates = listOpenAiForwardSidecarCandidates(config);
|
|
107
110
|
if (candidates.length === 0) {
|
|
108
|
-
return
|
|
109
|
-
400,
|
|
110
|
-
"invalid_request_error",
|
|
111
|
-
"Built-in web search needs a ChatGPT forward provider, but none is configured in opencodex. "
|
|
112
|
-
+ "Routed and OpenAI API-key providers cannot serve /v1/alpha/search.",
|
|
113
|
-
);
|
|
111
|
+
return handleAlphaSearchSidecarFallback(body, config, req.signal, logCtx);
|
|
114
112
|
}
|
|
115
113
|
|
|
116
114
|
let upstream: Awaited<ReturnType<typeof resolveFirstUsableOpenAiSidecar>>;
|
package/src/types/config.ts
CHANGED
|
@@ -668,6 +668,16 @@ export interface OcxConfig {
|
|
|
668
668
|
* so absence is the only default state this feature has.
|
|
669
669
|
*/
|
|
670
670
|
quotaResetNotify?: OcxQuotaResetNotifyConfig;
|
|
671
|
+
/**
|
|
672
|
+
* Periodic provider model-catalog refresh (issue #3630). Absent means off: no timer, no
|
|
673
|
+
* refresh pass, no outcome record.
|
|
674
|
+
*
|
|
675
|
+
* Off by default for the same reason every optional subsystem here is: a refresh spends a
|
|
676
|
+
* live /models call against every enabled provider, and this repository's rule is that a
|
|
677
|
+
* default install runs no detection code and starts no live timer work. Not in
|
|
678
|
+
* `getDefaultConfig()` — absence is the only default state this feature has.
|
|
679
|
+
*/
|
|
680
|
+
catalogAutoRefresh?: OcxCatalogAutoRefreshConfig;
|
|
671
681
|
/** Active provider context limits; native long windows remain within their supported ceilings. */
|
|
672
682
|
providerContextCaps?: Record<string, number>;
|
|
673
683
|
/** Last selected provider caps; retained while a cap is switched off. Not an active limit. */
|
|
@@ -850,14 +860,26 @@ export interface OcxConfig {
|
|
|
850
860
|
pool?: {
|
|
851
861
|
kernel?: boolean;
|
|
852
862
|
/**
|
|
853
|
-
*
|
|
863
|
+
* Cache-affinity ordering for bound Codex threads. **On unless set to `false`.**
|
|
854
864
|
*
|
|
855
|
-
*
|
|
856
|
-
*
|
|
865
|
+
* A bound Codex thread keeps its account until that account genuinely cannot serve,
|
|
866
|
+
* instead of moving the moment usage crosses `autoSwitchThreshold`. Moving a live
|
|
857
867
|
* conversation throws away the prompt cache warmed on that account, and a threshold
|
|
858
|
-
* crossing is a hint rather than evidence the account is spent.
|
|
859
|
-
*
|
|
860
|
-
*
|
|
868
|
+
* crossing is a hint rather than evidence the account is spent.
|
|
869
|
+
*
|
|
870
|
+
* This shipped as an opt-in (#4292) and then #4546 measured what the opt-in default
|
|
871
|
+
* costs: a pool whose accounts all sit in the 80-99% band hands a conversation from
|
|
872
|
+
* account to account, re-sending the whole prefix every turn, and the install that gets
|
|
873
|
+
* hurt is precisely the one that never heard of this setting. `false` restores
|
|
874
|
+
* capacity-first routing for operators who want it.
|
|
875
|
+
*
|
|
876
|
+
* Separate from `kernel` on purpose: that one governs the generic OAuth strategy
|
|
877
|
+
* consumer, and one switch carrying two unrelated meanings cannot be turned on alone.
|
|
878
|
+
*
|
|
879
|
+
* Note what this does NOT govern. Unbound placement still follows
|
|
880
|
+
* `autoSwitchThreshold` and the configured strategy. A bound thread's destination must
|
|
881
|
+
* have real headroom under either setting, and a transient failure streak holds the
|
|
882
|
+
* binding under either setting -- neither is a cache-affinity preference.
|
|
861
883
|
*/
|
|
862
884
|
cacheAffinity?: boolean;
|
|
863
885
|
};
|
|
@@ -1303,3 +1325,26 @@ export interface OcxQuotaResetNotifyConfig {
|
|
|
1303
1325
|
*/
|
|
1304
1326
|
command?: string[];
|
|
1305
1327
|
}
|
|
1328
|
+
|
|
1329
|
+
/**
|
|
1330
|
+
* Periodic model-catalog auto-refresh settings (issue #3630).
|
|
1331
|
+
*
|
|
1332
|
+
* Every field is optional and the whole section defaults to off. Each tick converges the
|
|
1333
|
+
* served catalog the same way `ocx sync` does, which costs a live /models call against
|
|
1334
|
+
* every enabled provider — so an install that never asked for this must run no refresh
|
|
1335
|
+
* code and start no timer, matching the optional-subsystem rule the rest of this file
|
|
1336
|
+
* follows.
|
|
1337
|
+
*/
|
|
1338
|
+
export interface OcxCatalogAutoRefreshConfig {
|
|
1339
|
+
/** Master switch. Default false — no scheduler, no tick, no upstream calls. */
|
|
1340
|
+
enabled?: boolean;
|
|
1341
|
+
/**
|
|
1342
|
+
* Minutes between refresh ticks. Default 60, floor 15, and 0 keeps the timer dormant
|
|
1343
|
+
* while leaving the section configured.
|
|
1344
|
+
*
|
|
1345
|
+
* The floor exists for the same reason src/quota/reset-poller.ts has MIN_INTERVAL_MS:
|
|
1346
|
+
* provider catalogs are cached upstream for minutes, so a faster cadence buys no
|
|
1347
|
+
* freshness and only risks a rate limit against every enabled provider at once.
|
|
1348
|
+
*/
|
|
1349
|
+
intervalMinutes?: number;
|
|
1350
|
+
}
|