@bitkyc08/opencodex 2.56.0 → 2.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/ocx.mjs +10 -0
- package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
- package/gui/dist/assets/{index-D4zuyIxQ.js → index-Cz7CLdif.js} +21 -21
- package/gui/dist/index.html +2 -2
- package/package.json +3 -3
- package/src/adapters/codebuddy/adapter.ts +2 -1
- package/src/adapters/codebuddy/scaffold-guard.ts +248 -0
- package/src/adapters/command-code.ts +1 -1
- package/src/adapters/cursor/envelope-echo.ts +8 -2
- package/src/adapters/google.ts +7 -7
- package/src/adapters/kiro/payload.ts +17 -3
- package/src/adapters/kiro/reasoning.ts +70 -7
- package/src/adapters/kiro/stream.ts +8 -2
- package/src/adapters/kiro/wire.ts +2 -1
- package/src/adapters/kiro-events.ts +21 -13
- package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
- package/src/adapters/openai-chat/tool-schema.ts +25 -7
- package/src/adapters/openai-chat.ts +8 -8
- package/src/adapters/openai-responses/passthrough.ts +32 -1
- package/src/bridge/errors.ts +26 -2
- package/src/bridge/response-json.ts +7 -1
- package/src/bridge/sse.ts +19 -1
- package/src/claude/desktop-profile.ts +66 -9
- package/src/claude/outbound.ts +18 -0
- package/src/cli/account-main.ts +1 -1
- package/src/cli/capabilities.ts +2 -2
- package/src/cli/combo.ts +10 -1
- package/src/cli/index.ts +48 -5
- package/src/cli/registry.ts +2 -1
- package/src/cli/system-command.ts +4 -4
- package/src/clients/config-export.ts +7 -3
- package/src/codex/account-label.ts +14 -3
- package/src/codex/account-store.ts +113 -26
- package/src/codex/account-usability.ts +21 -0
- package/src/codex/auth-api/login-flow.ts +14 -2
- package/src/codex/auth-api/reset-credit-service.ts +11 -2
- package/src/codex/auth-context.ts +157 -7
- package/src/codex/catalog/aggregation.ts +80 -1
- package/src/codex/catalog/model-visibility.ts +1 -0
- package/src/codex/catalog/remote.ts +30 -0
- package/src/codex/catalog/retained-sync.ts +9 -1
- package/src/codex/catalog/routed-gather.ts +38 -1
- package/src/codex/cli-install-provenance.ts +7 -1
- package/src/codex/convergence.ts +7 -2
- package/src/codex/desktop-app/types.ts +11 -2
- package/src/codex/desktop-app/windows.ts +5 -5
- package/src/codex/inject/restore.ts +29 -2
- package/src/codex/inject.ts +9 -9
- package/src/codex/model-entitlements.ts +152 -15
- package/src/codex/pool-refresh-backoff.ts +12 -3
- package/src/codex/quota-rejection.ts +104 -15
- package/src/codex/routing/cache-affinity.ts +70 -0
- package/src/codex/routing/cooldown-math.ts +10 -0
- package/src/codex/routing/selection.ts +79 -2
- package/src/codex/routing/thread-affinity.ts +50 -2
- package/src/codex/routing/transient-hold-dispatch.ts +141 -0
- package/src/codex/routing.ts +29 -49
- package/src/codex/warmup.ts +1 -1
- package/src/combos/failover.ts +85 -0
- package/src/combos/request.ts +17 -10
- package/src/combos/types.ts +23 -2
- package/src/config/pending-teardown.ts +31 -0
- package/src/generated/compatibility-version.json +163 -135
- package/src/images/loop.ts +1 -1
- package/src/lib/errors.ts +17 -0
- package/src/lib/request-execution-budget.ts +147 -21
- package/src/lib/spend-reservation-ledger.ts +18 -0
- package/src/lib/state-store-registrations.ts +6 -2
- package/src/lib/test-home-guard.ts +85 -1
- package/src/lib/upstream-retry.ts +77 -10
- package/src/lib/windows-elevation.ts +76 -14
- package/src/oauth/index.ts +2 -2
- package/src/oauth/key-providers.ts +2 -2
- package/src/providers/kiro-models.ts +4 -3
- package/src/providers/label.ts +19 -1
- package/src/providers/model-discovery.ts +16 -0
- package/src/providers/registry/entries-core.ts +7 -0
- package/src/providers/registry/entries-extended.ts +9 -0
- package/src/providers/registry/model-seeds.ts +4 -0
- package/src/responses/reasoning-envelope.ts +6 -3
- package/src/routing/identity-domains.ts +21 -14
- package/src/routing/probe-lease.ts +103 -1
- package/src/server/chat-completions.ts +3 -1
- package/src/server/chat-native.ts +37 -9
- package/src/server/index/live-sideband.ts +37 -1
- package/src/server/index/websocket-handler.ts +6 -2
- package/src/server/index.ts +5 -5
- package/src/server/inspection-tee.ts +107 -0
- package/src/server/live.ts +46 -1
- package/src/server/management/combo-routes.ts +10 -1
- package/src/server/relay-eager.ts +2 -0
- package/src/server/relay.ts +14 -19
- package/src/server/request-log.ts +127 -3
- package/src/server/response-log-body.ts +153 -0
- package/src/server/responses/account-change-state.ts +74 -0
- package/src/server/responses/adapter-continuation.ts +33 -7
- package/src/server/responses/adapter-delivery.ts +5 -11
- package/src/server/responses/adapter-dispatch.ts +84 -13
- package/src/server/responses/codex-ws-wire.ts +5 -0
- package/src/server/responses/collaboration.ts +74 -4
- package/src/server/responses/combo-session-recall.ts +68 -8
- package/src/server/responses/compact.ts +54 -13
- package/src/server/responses/core-auth.ts +2 -0
- package/src/server/responses/core-codex-account.ts +51 -3
- package/src/server/responses/core-combo.ts +103 -23
- package/src/server/responses/core-errors.ts +18 -0
- package/src/server/responses/core-replay.ts +105 -32
- package/src/server/responses/core.ts +3 -3
- package/src/server/responses/encrypted-payload.ts +0 -1
- package/src/server/responses/input-admission.ts +126 -6
- package/src/server/responses/passthrough-delivery.ts +19 -6
- package/src/server/responses/passthrough-dispatch.ts +28 -10
- package/src/server/responses/passthrough-error.ts +38 -2
- package/src/server/responses/request-prepare.ts +132 -22
- package/src/server/responses/request-send-budget.ts +97 -2
- package/src/server/responses/request-spend.ts +147 -0
- package/src/server/responses/request-transport.ts +62 -3
- package/src/server/responses/run-turn-execution.ts +59 -31
- package/src/server/responses/sidecar-execution.ts +7 -13
- package/src/server/responses/terminal-guard.ts +65 -4
- package/src/server/responses-undeclared-tool-guard.ts +9 -5
- package/src/service/windows-ops.ts +210 -16
- package/src/service/windows-scheduler.ts +28 -21
- package/src/service.ts +1 -1
- package/src/types/config.ts +4 -1
- package/src/types/request.ts +8 -5
- package/src/types/tools.ts +24 -0
- package/src/types.ts +2 -0
- package/src/update/index.ts +10 -0
- package/src/update/stop-contract.d.mts +1 -0
- package/src/update/stop-contract.mjs +19 -0
- package/src/update/stop-decision.d.mts +1 -1
- package/src/update/stop-decision.mjs +12 -3
- package/src/usage/log.ts +1 -1
- package/src/vision/anthropic-describe.ts +1 -1
- package/src/vision/describe.ts +5 -5
- package/src/web-search/anthropic-executor.ts +1 -1
- package/src/web-search/exa-executor.ts +1 -1
- package/src/web-search/executor.ts +1 -1
- package/src/web-search/gemini-executor.ts +1 -1
- package/src/web-search/loop.ts +1 -1
- package/src/web-search/ollama-executor.ts +1 -1
- package/src/web-search/parse.ts +67 -14
- package/src/web-search/passthrough-bridge.ts +64 -31
- package/src/web-search/xai-executor.ts +1 -1
|
@@ -250,6 +250,21 @@ export const OCX_ELEVATED_PROTOCOL_FAILED = 13;
|
|
|
250
250
|
/** Windows ERROR_CANCELLED — reserved for UAC denial; never emitted by the elevated script. */
|
|
251
251
|
export const OCX_ELEVATED_UAC_CANCELLED = 1223;
|
|
252
252
|
|
|
253
|
+
/**
|
|
254
|
+
* The elevated process could not read a staged payload (#4692).
|
|
255
|
+
*
|
|
256
|
+
* `hardenSecretPath` grants the staging account and strips inheritance, so a split-token
|
|
257
|
+
* elevation of the same user reads the file and an elevation answered with a DIFFERENT
|
|
258
|
+
* administrator's credentials does not. The elevated side cannot explain that itself: it
|
|
259
|
+
* runs hidden, so its stderr goes nowhere and only the exit code survives the boundary.
|
|
260
|
+
* Without a code of its own the operator would be told "exit code 1" for a cause that
|
|
261
|
+
* names its own remedy — the same undiagnosable failure this change set exists to remove.
|
|
262
|
+
*
|
|
263
|
+
* Deliberately outside OCX_ELEVATED_PROTOCOL_CODES: that list is the create-and-run
|
|
264
|
+
* transaction's alphabet, and this code belongs to the registration path.
|
|
265
|
+
*/
|
|
266
|
+
export const OCX_ELEVATED_STAGING_UNREADABLE = 14;
|
|
267
|
+
|
|
253
268
|
export const OCX_ELEVATED_PROTOCOL_CODES = [
|
|
254
269
|
OCX_ELEVATED_SUCCESS,
|
|
255
270
|
OCX_ELEVATED_CREATE_FAILED,
|
|
@@ -645,36 +660,83 @@ export function runWindowsElevated(file: string, args: string[]): Promise<number
|
|
|
645
660
|
}
|
|
646
661
|
|
|
647
662
|
/**
|
|
648
|
-
*
|
|
649
|
-
*
|
|
650
|
-
*
|
|
663
|
+
* A task definition staged for the elevated process.
|
|
664
|
+
*
|
|
665
|
+
* The bytes live in a freshly created, ACL-hardened private directory, and the digest is
|
|
666
|
+
* taken over exactly those bytes by the caller that validated them. The elevated script
|
|
667
|
+
* reads the file once, hashes what it read, and refuses unless the digest matches, so a
|
|
668
|
+
* pathname is no longer a promise about content — it is a claim the receiver checks.
|
|
669
|
+
*/
|
|
670
|
+
export interface StagedWindowsTaskXml {
|
|
671
|
+
/** Path inside the caller's hardened staging directory. */
|
|
672
|
+
readonly path: string;
|
|
673
|
+
/** Lowercase hex SHA-256 of the staged bytes (UTF-16LE, no BOM). */
|
|
674
|
+
readonly sha256: string;
|
|
675
|
+
}
|
|
676
|
+
|
|
677
|
+
/**
|
|
678
|
+
* Read a staged payload, prove it is the one that was validated, and decode it.
|
|
679
|
+
*
|
|
680
|
+
* One read: the bytes that are hashed are the same array that is decoded and registered.
|
|
681
|
+
* Hashing a path and then reopening it would reintroduce the swap window this check
|
|
682
|
+
* exists to close.
|
|
683
|
+
*/
|
|
684
|
+
const READ_STAGED_TASK_XML = "function Read-OcxStagedTaskXml([string]$path, [string]$expectedHash) {"
|
|
685
|
+
// An unreadable payload is a diagnosable condition, not a generic throw: a hidden
|
|
686
|
+
// elevated process has nowhere to print, so the cause has to ride the exit code.
|
|
687
|
+
+ " try { $bytes = [IO.File]::ReadAllBytes($path) }"
|
|
688
|
+
+ " catch [System.UnauthorizedAccessException] { exit " + OCX_ELEVATED_STAGING_UNREADABLE + " }"
|
|
689
|
+
+ " catch [System.Security.SecurityException] { exit " + OCX_ELEVATED_STAGING_UNREADABLE + " };"
|
|
690
|
+
+ " $sha = [Security.Cryptography.SHA256]::Create();"
|
|
691
|
+
+ " try { $actual = [BitConverter]::ToString($sha.ComputeHash($bytes)).Replace('-', '').ToLowerInvariant() } finally { $sha.Dispose() };"
|
|
692
|
+
+ " if ($actual -cne $expectedHash) { throw 'Task Scheduler staged payload failed its integrity check.' };"
|
|
693
|
+
+ " return [Text.Encoding]::Unicode.GetString($bytes) }";
|
|
694
|
+
|
|
695
|
+
/**
|
|
696
|
+
* Register one scheduled-task definition from staged, digest-verified bytes.
|
|
697
|
+
*
|
|
698
|
+
* The payloads used to be embedded as base64(utf16le) inside an inner PowerShell script
|
|
699
|
+
* that was itself base64(utf16le)-encoded into `-EncodedCommand`. Two layers of base64
|
|
700
|
+
* over UTF-16 cost about 14.2 command-line characters per XML character, and a
|
|
701
|
+
* replacement carries two payloads, so a ~2 KB task definition pushed the outer command
|
|
702
|
+
* past the Windows limit and the spawn failed with ENAMETOOLONG before UAC ever
|
|
703
|
+
* appeared (#4692). On a host where the trigger scope exports as an account name the
|
|
704
|
+
* re-register path runs on every repair, so repair could never succeed.
|
|
705
|
+
*
|
|
706
|
+
* The command now carries two paths and two 64-character digests, so its length no
|
|
707
|
+
* longer depends on the size of the XML at all.
|
|
708
|
+
*
|
|
709
|
+
* The original design goal was "immutable bytes, never a caller-writable pathname".
|
|
710
|
+
* That goal is kept by different means rather than abandoned: the staging directory is
|
|
711
|
+
* private and ACL-hardened, the files are created exclusively so nothing can be waiting
|
|
712
|
+
* at the path, and the digest makes a same-account swap during the UAC prompt fail
|
|
713
|
+
* closed instead of registering something else. An ACL alone could not do that last
|
|
714
|
+
* part, because a process running as the same user has the same SID.
|
|
715
|
+
*
|
|
716
|
+
* The replacement precondition is unchanged: the elevated process still re-queries the
|
|
717
|
+
* live registration and compares it to the captured predecessor before passing -Force.
|
|
651
718
|
*/
|
|
652
719
|
export function runWindowsElevatedScheduledTaskRegistration(
|
|
653
720
|
taskName: string,
|
|
654
|
-
xml:
|
|
721
|
+
xml: StagedWindowsTaskXml,
|
|
655
722
|
replace = false,
|
|
656
|
-
|
|
723
|
+
expectedExisting?: StagedWindowsTaskXml,
|
|
657
724
|
): Promise<number> {
|
|
658
|
-
if (replace && !
|
|
725
|
+
if (replace && !expectedExisting) {
|
|
659
726
|
throw new Error("Elevated Task Scheduler replacement requires a captured existing definition.");
|
|
660
727
|
}
|
|
661
|
-
const xmlBase64 = Buffer.from(xml, "utf16le").toString("base64");
|
|
662
|
-
const expectedExistingBase64 = expectedExistingXml === undefined
|
|
663
|
-
? null
|
|
664
|
-
: Buffer.from(expectedExistingXml, "utf16le").toString("base64");
|
|
665
728
|
const powerShellPath = windowsPowerShell();
|
|
666
729
|
const powerShellDirectory = powerShellPath.replace(/[\\/][^\\/]+$/, "");
|
|
667
730
|
const scheduledTasksModule = `${powerShellDirectory}\\Modules\\ScheduledTasks\\ScheduledTasks.psd1`;
|
|
668
731
|
const inner = [
|
|
669
732
|
`$taskName = ${psSingleQuote(taskName)}`,
|
|
670
|
-
|
|
671
|
-
|
|
733
|
+
READ_STAGED_TASK_XML,
|
|
734
|
+
`$xml = Read-OcxStagedTaskXml ${psSingleQuote(xml.path)} ${psSingleQuote(xml.sha256)}`,
|
|
672
735
|
`$module = Microsoft.PowerShell.Core\\Import-Module -Name ${psSingleQuote(scheduledTasksModule)} -PassThru -Force -ErrorAction Stop`,
|
|
673
736
|
"$registerTask = $module.ExportedCommands['Register-ScheduledTask']",
|
|
674
737
|
"if ($null -eq $registerTask) { throw 'Trusted ScheduledTasks module does not export Register-ScheduledTask.' }",
|
|
675
738
|
...(replace ? [
|
|
676
|
-
`$
|
|
677
|
-
"$expectedXml = [Text.Encoding]::Unicode.GetString([Convert]::FromBase64String($expectedBase64))",
|
|
739
|
+
`$expectedXml = Read-OcxStagedTaskXml ${psSingleQuote(expectedExisting!.path)} ${psSingleQuote(expectedExisting!.sha256)}`,
|
|
678
740
|
`$schtasks = ${psSingleQuote(resolveTrustedWindowsSchtasksExe())}`,
|
|
679
741
|
"$currentXml = & $schtasks /query /tn $taskName /xml 2>$null | Out-String",
|
|
680
742
|
"if ($LASTEXITCODE -ne 0) { throw 'Task Scheduler replacement precondition could not be read.' }",
|
package/src/oauth/index.ts
CHANGED
|
@@ -47,7 +47,7 @@ import { ANTIGRAVITY_REQUEST_UA } from "../adapters/google-antigravity-wire";
|
|
|
47
47
|
import { deriveOAuthDefaultModel, deriveOAuthProviderConfig } from "../providers/derive";
|
|
48
48
|
import { apiKeyPoolEntryId, sanitizeApiKeyValue } from "../providers/api-keys";
|
|
49
49
|
import { effectiveGoogleMode, getProviderRegistryEntry, mergeRegistryStaticHeaders, providerMatchesRegistryTransport } from "../providers/registry";
|
|
50
|
-
import { resolveProviderModelDiscoveryUrl } from "../providers/model-discovery";
|
|
50
|
+
import { providerModelsUrl, resolveProviderModelDiscoveryUrl } from "../providers/model-discovery";
|
|
51
51
|
import { resolveProviderTransport } from "../providers/xai-transport";
|
|
52
52
|
import { detectClaudeCodeToken, detectGrokCliToken, hasComparableGrokIdentity, isSameGrokIdentity, shouldAdoptGrokGeneration } from "./local-token-detect";
|
|
53
53
|
import { logOAuthEvent } from "./log";
|
|
@@ -1228,7 +1228,7 @@ export function buildModelsRequest(
|
|
|
1228
1228
|
return { url: discoveryUrl(`${base}/v1/models?limit=1000`), headers };
|
|
1229
1229
|
}
|
|
1230
1230
|
if (apiKey) headers["Authorization"] = `Bearer ${apiKey}`;
|
|
1231
|
-
return { url: discoveryUrl(
|
|
1231
|
+
return { url: discoveryUrl(providerModelsUrl(effectiveProvider.baseUrl)), headers };
|
|
1232
1232
|
}
|
|
1233
1233
|
|
|
1234
1234
|
/**
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { OcxProviderConfig } from "../types";
|
|
2
2
|
import { deriveKeyLoginMap, enrichProviderFromRegistry, type DerivedKeyLoginProvider } from "../providers/derive";
|
|
3
|
-
import { resolveProviderModelDiscoveryUrl } from "../providers/model-discovery";
|
|
3
|
+
import { providerModelsUrl, resolveProviderModelDiscoveryUrl } from "../providers/model-discovery";
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* API-key "login" providers: not OAuth — the flow opens the provider's dashboard so the user can
|
|
@@ -117,7 +117,7 @@ export async function validateApiKey(
|
|
|
117
117
|
providerName,
|
|
118
118
|
configuredProvider,
|
|
119
119
|
provider.baseUrl,
|
|
120
|
-
|
|
120
|
+
providerModelsUrl(provider.baseUrl),
|
|
121
121
|
);
|
|
122
122
|
const res = await fetch(modelsUrl, {
|
|
123
123
|
headers: { Authorization: `Bearer ${key}` },
|
|
@@ -47,9 +47,10 @@ export const KIRO_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
|
|
|
47
47
|
|
|
48
48
|
const KIRO_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
49
49
|
|
|
50
|
-
//
|
|
51
|
-
// (`reasoning.effort`
|
|
52
|
-
// thinking instructions until their native
|
|
50
|
+
// The GPT-5.6 family (sol/terra/luna) and claude-opus-5 send these values through Kiro's verified
|
|
51
|
+
// native effort fields (`reasoning.effort` for the GPT-5.6 models, `output_config.effort` for
|
|
52
|
+
// claude-opus-5). Other models map them to bounded thinking instructions until their native
|
|
53
|
+
// effort support is verified.
|
|
53
54
|
export const KIRO_MODEL_REASONING_EFFORTS: Record<string, string[]> = Object.fromEntries(
|
|
54
55
|
KIRO_MODELS.map(id => [id, KIRO_REASONING_EFFORTS]),
|
|
55
56
|
);
|
package/src/providers/label.ts
CHANGED
|
@@ -1,10 +1,28 @@
|
|
|
1
|
-
import { CODEX_ACCOUNT_LOG_LABEL_RE, oauthAccountLogLabel } from "../codex/account-label";
|
|
1
|
+
import { CODEX_ACCOUNT_LOG_LABEL_RE, KEY_ACCOUNT_LOG_LABEL_RE, apiKeyAccountLogLabel, oauthAccountLogLabel } from "../codex/account-label";
|
|
2
2
|
import type { OcxProviderConfig } from "../types";
|
|
3
3
|
|
|
4
4
|
export function canonicalUsageProviderLabel(provider: string): string {
|
|
5
5
|
return provider === "chatgpt" || provider === "openai-multi" ? "openai" : provider;
|
|
6
6
|
}
|
|
7
7
|
|
|
8
|
+
export function usesApiKeyAccount(provider: Pick<OcxProviderConfig, "authMode" | "_apiKeyAttempt">): boolean {
|
|
9
|
+
return provider.authMode === "key"
|
|
10
|
+
|| (provider.authMode === undefined && !!provider._apiKeyAttempt?.reference);
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
/** Key identity comes from the captured selection, before env/keychain resolution. */
|
|
14
|
+
export function stampApiKeyAccountLabel(
|
|
15
|
+
logCtx: { accountLogLabel?: string },
|
|
16
|
+
providerName: string,
|
|
17
|
+
provider: Pick<OcxProviderConfig, "authMode" | "_apiKeyAttempt">,
|
|
18
|
+
): void {
|
|
19
|
+
if (usesApiKeyAccount(provider)) {
|
|
20
|
+
logCtx.accountLogLabel = apiKeyAccountLogLabel(providerName, provider._apiKeyAttempt);
|
|
21
|
+
} else if (KEY_ACCOUNT_LOG_LABEL_RE.test(logCtx.accountLogLabel ?? "")) {
|
|
22
|
+
delete logCtx.accountLogLabel;
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
8
26
|
export function baseProviderLabel(provider: string): string {
|
|
9
27
|
const canonical = canonicalUsageProviderLabel(provider);
|
|
10
28
|
if (canonical !== provider) return canonical;
|
|
@@ -23,6 +23,8 @@ import {
|
|
|
23
23
|
|
|
24
24
|
const MODEL_DISCOVERY_MAX_FILTER_VALUES = 256;
|
|
25
25
|
const MODEL_DISCOVERY_MAX_FILTER_STRING_LENGTH = 1_024;
|
|
26
|
+
const TRAILING_SLASHES = /\/+$/;
|
|
27
|
+
const TRAILING_MODELS = /\/models$/;
|
|
26
28
|
|
|
27
29
|
export interface ResolvedProviderModelDiscovery {
|
|
28
30
|
spec?: ProviderModelDiscoverySpec;
|
|
@@ -50,6 +52,20 @@ export type ModelEnvelopeRowsResult =
|
|
|
50
52
|
| { ok: true; rows: unknown[] }
|
|
51
53
|
| { ok: false; reason: "invalid_shape" | "too_many_models" };
|
|
52
54
|
|
|
55
|
+
/**
|
|
56
|
+
* Build the default OpenAI-compatible model-discovery URL from a configured baseUrl.
|
|
57
|
+
*
|
|
58
|
+
* `baseUrl` is required on both `OcxProviderConfig` and the persisted-config schema, so a row
|
|
59
|
+
* without one is not a state configuration loading can produce. It is deliberately not tolerated
|
|
60
|
+
* here: the old template-literal join silently produced `"undefined/models"`, which is not a usable
|
|
61
|
+
* fallback either — it only ever survived because a static row returns before the URL is parsed.
|
|
62
|
+
*/
|
|
63
|
+
export function providerModelsUrl(baseUrl: string): string {
|
|
64
|
+
const trimmed = baseUrl.trim().replace(TRAILING_SLASHES, "");
|
|
65
|
+
const withoutEndpoint = trimmed.replace(TRAILING_MODELS, "");
|
|
66
|
+
return `${withoutEndpoint}/models`;
|
|
67
|
+
}
|
|
68
|
+
|
|
53
69
|
function positiveIntegerAtMost(value: number | undefined, hardLimit: number): number {
|
|
54
70
|
if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return hardLimit;
|
|
55
71
|
return Math.min(Math.floor(value), hardLimit);
|
|
@@ -17,6 +17,7 @@ import type { ProviderRegistryEntry } from "./types";
|
|
|
17
17
|
import {
|
|
18
18
|
ANTHROPIC_MODELS,
|
|
19
19
|
ANTHROPIC_MODEL_CONTEXT_WINDOWS,
|
|
20
|
+
ANTHROPIC_MODEL_INPUT_MODALITIES,
|
|
20
21
|
ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
|
|
21
22
|
ANTHROPIC_MODEL_REASONING_EFFORTS,
|
|
22
23
|
ZAI_GLM_52_REASONING_EFFORTS,
|
|
@@ -383,6 +384,7 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
|
383
384
|
note: "Log in with your Claude account",
|
|
384
385
|
models: [...ANTHROPIC_MODELS],
|
|
385
386
|
modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
|
|
387
|
+
modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
|
|
386
388
|
modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
|
|
387
389
|
// Codex omits max_output_tokens; without a provider budget the Anthropic adapter
|
|
388
390
|
// falls back to 8192, which truncates long answers with stop_reason=max_tokens.
|
|
@@ -403,6 +405,7 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
|
403
405
|
models: [...ANTHROPIC_MODELS],
|
|
404
406
|
liveModels: true,
|
|
405
407
|
modelContextWindows: { ...ANTHROPIC_MODEL_CONTEXT_WINDOWS },
|
|
408
|
+
modelInputModalities: { ...ANTHROPIC_MODEL_INPUT_MODALITIES },
|
|
406
409
|
modelReasoningEfforts: { ...ANTHROPIC_MODEL_REASONING_EFFORTS },
|
|
407
410
|
defaultMaxOutputTokens: ANTHROPIC_DEFAULT_MAX_OUTPUT_TOKENS,
|
|
408
411
|
defaultModel: "claude-sonnet-5",
|
|
@@ -420,6 +423,10 @@ export const PROVIDER_REGISTRY_CORE: readonly ProviderRegistryEntry[] = [
|
|
|
420
423
|
// or the one the Claude /v1/messages inbound derives); the adapter itself never invents one.
|
|
421
424
|
// Evidence: https://platform.kimi.com/docs/api/chat
|
|
422
425
|
promptCacheKey: true,
|
|
426
|
+
// Kimi's Responses endpoint rejects hook-provided context between a tool call and
|
|
427
|
+
// its matching result (#4726), the same strict shape DeepSeek exposed in #1292.
|
|
428
|
+
// The flag is inert while this preset uses the Chat wire.
|
|
429
|
+
requiresAdjacentResponsesToolResults: true,
|
|
423
430
|
featured: true,
|
|
424
431
|
oauthId: "kimi",
|
|
425
432
|
jawcodeBundle: "moonshot",
|
|
@@ -110,6 +110,13 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
110
110
|
// Baseten says models outside its reasoning table do not support reasoning. Keep
|
|
111
111
|
// unknown/new live slugs conservative until an official-docs registry refresh proves it.
|
|
112
112
|
reasoningEfforts: [],
|
|
113
|
+
// `text.verbosity` is an OpenAI Responses parameter. Baseten documents its Model
|
|
114
|
+
// APIs as Chat Completions compatible, so there is nothing on that wire for it to
|
|
115
|
+
// become, and a routed row must not inherit the Codex template's verbosity picker
|
|
116
|
+
// (#4630: Codex sent `text: { verbosity: "low" }` and the turn 400'd before any
|
|
117
|
+
// model output). Provider-wide rather than per-model because this catalog is live-
|
|
118
|
+
// discovered: a slug that arrives tomorrow supports it no more than the seeded ones.
|
|
119
|
+
supportsVerbosity: false,
|
|
113
120
|
modelReasoningEfforts: BASETEN_MODEL_REASONING_EFFORTS,
|
|
114
121
|
modelReasoningEffortMap: BASETEN_MODEL_REASONING_EFFORT_MAP,
|
|
115
122
|
modelDefaultReasoningEfforts: BASETEN_MODEL_DEFAULT_REASONING_EFFORTS,
|
|
@@ -883,6 +890,8 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
|
|
|
883
890
|
modelSuffixBracketStrip: true,
|
|
884
891
|
// API-key form of the same Kimi Code Plan transport; keep cache affinity identical to OAuth.
|
|
885
892
|
promptCacheKey: true,
|
|
893
|
+
// Keep Responses tool-result adjacency aligned with the OAuth preset (#4726).
|
|
894
|
+
requiresAdjacentResponsesToolResults: true,
|
|
886
895
|
models: KIMI_CODING_MODELS,
|
|
887
896
|
modelContextWindows: KIMI_CODING_MODEL_CONTEXT_WINDOWS,
|
|
888
897
|
modelInputModalities: KIMI_CODING_MODEL_INPUT_MODALITIES,
|
|
@@ -8,6 +8,10 @@ import type { ProviderModelDiscoverySpec } from "./types";
|
|
|
8
8
|
// always on, per the official models overview and pricing page (platform.claude.com).
|
|
9
9
|
export const ANTHROPIC_MODELS = ["claude-fable-5-1", "claude-fable-5", "claude-sonnet-5", "claude-opus-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-sonnet-4-6", "claude-haiku-4-5"];
|
|
10
10
|
export const ANTHROPIC_MODEL_CONTEXT_WINDOWS: Record<string, number> = { "claude-fable-5-1": 1_000_000, "claude-sonnet-5": 1_000_000, "claude-fable-5": 1_000_000, "claude-opus-5": 1_000_000, "claude-opus-4-8": 1_000_000, "claude-opus-4-7": 1_000_000, "claude-opus-4-6": 1_000_000, "claude-sonnet-4-6": 1_000_000, "claude-haiku-4-5": 200_000 };
|
|
11
|
+
// All seeded Claude models support vision: https://platform.claude.com/docs/en/models/overview
|
|
12
|
+
export const ANTHROPIC_MODEL_INPUT_MODALITIES: Record<string, string[]> = Object.fromEntries(
|
|
13
|
+
ANTHROPIC_MODELS.map(id => [id, ["text", "image"]]),
|
|
14
|
+
);
|
|
11
15
|
// Every current Claude family accepts at least 64k output tokens (Haiku 4.5 / Sonnet 4.x
|
|
12
16
|
// through Opus 5 and Fable 5). Anthropic caps max_tokens per model server-side, so a
|
|
13
17
|
// larger request never over-allocates; it only stops the 8192 truncation.
|
|
@@ -28,9 +28,12 @@ export interface ReasoningEnvelope {
|
|
|
28
28
|
*/
|
|
29
29
|
txt?: string;
|
|
30
30
|
/**
|
|
31
|
-
* Kiro `reasoningContentEvent
|
|
32
|
-
* the
|
|
33
|
-
*
|
|
31
|
+
* Kiro's reasoning blob from `reasoningContentEvent`: a KMS-encrypted value that is opaque to the
|
|
32
|
+
* proxy (the GPT-5.6 family sends it as `signature`, other models as the base64
|
|
33
|
+
* `redactedContent`, and the value carries a tag naming which one — see
|
|
34
|
+
* src/adapters/kiro/reasoning.ts). Kiro's own CLI replays it on the matching
|
|
35
|
+
* `assistantResponseMessage` to preserve model reasoning across turns, so it round-trips here the
|
|
36
|
+
* same way a signature does.
|
|
34
37
|
*/
|
|
35
38
|
krc?: string;
|
|
36
39
|
}
|
|
@@ -37,13 +37,16 @@ export type IdentityDomainProvenance = "operator-declared" | "provider-documente
|
|
|
37
37
|
* What a domain key is evidence FOR, which is two facts rather than one.
|
|
38
38
|
*
|
|
39
39
|
* Proven SEPARATION and proven SHARING are different claims, and a provider routinely
|
|
40
|
-
* gives the first without the second. OpenAI
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
40
|
+
* gives the first without the second. OpenAI's prompt-caching guide states the separating
|
|
41
|
+
* half outright -- "Caches are not shared across organizations and cannot be reused across
|
|
42
|
+
* regional processing boundaries" -- and never states a sharing half at all. The page does
|
|
43
|
+
* not discuss two API keys inside one organization, and what it does say about keys is that
|
|
44
|
+
* they "influence routing; they do not pin requests to a machine or guarantee a cache hit."
|
|
45
|
+
* So a different org-or-region key proves two domains, while an identical one proves
|
|
46
|
+
* nothing. A positive cache inference needs the provider to promise the hit, and no such
|
|
47
|
+
* promise exists here -- the silence is the evidence, not a documented denial. Inferring
|
|
48
|
+
* "shared" from an equal key would be the same guess this module exists to refuse, only
|
|
49
|
+
* pointed the other way.
|
|
47
50
|
*
|
|
48
51
|
* "separates" therefore means two different keys are two different domains while two
|
|
49
52
|
* identical keys stay "unknown". "separates-and-shares" means the same source also
|
|
@@ -130,7 +133,9 @@ export interface DeclaredCredentialGroup {
|
|
|
130
133
|
* then classifies "unknown" rather than extrapolating.
|
|
131
134
|
*
|
|
132
135
|
* - OpenAI: rate limits are per organization and project, with model groups sharing a
|
|
133
|
-
* limit
|
|
136
|
+
* limit ("Rate limits are defined at the organization level and at the project level, not
|
|
137
|
+
* user level", plus the documented shared limit across a model family); prompt caches are
|
|
138
|
+
* not shared across organizations or regional processing boundaries.
|
|
134
139
|
* - Anthropic: prompt cache is isolated per workspace even inside one organization.
|
|
135
140
|
* (Cache-read tokens are also excluded from input TPM there, which is quota
|
|
136
141
|
* accounting, not domain shape, so it does not appear here.)
|
|
@@ -158,12 +163,14 @@ const PROVIDER_DOCUMENTED_DOMAINS: Record<string, {
|
|
|
158
163
|
evidence: "separates-and-shares",
|
|
159
164
|
},
|
|
160
165
|
cache: {
|
|
161
|
-
// Separation only
|
|
162
|
-
//
|
|
163
|
-
//
|
|
164
|
-
//
|
|
165
|
-
//
|
|
166
|
-
//
|
|
166
|
+
// Separation only, and the asymmetry is in the source. The prompt-caching guide says
|
|
167
|
+
// "Caches are not shared across organizations and cannot be reused across regional
|
|
168
|
+
// processing boundaries", which settles a DIFFERENT org or region as distinct. It
|
|
169
|
+
// states no counterpart for an identical one: the guide never discusses two keys in
|
|
170
|
+
// one organization, and a key is documented to "influence routing" without
|
|
171
|
+
// guaranteeing a hit. Same org and region therefore stays "unknown" -- claiming
|
|
172
|
+
// "shared" would assert a warm prefix the provider never promised, and the caller
|
|
173
|
+
// would pay for the guess by replaying a long prompt that misses.
|
|
167
174
|
key: (ref) => ref.organizationId !== undefined && ref.region !== undefined
|
|
168
175
|
? `openai:org:${ref.organizationId}:region:${ref.region}`
|
|
169
176
|
: undefined,
|
|
@@ -349,7 +349,14 @@ export function resolveHeldAccountDispatch(input: {
|
|
|
349
349
|
kind: "withheld",
|
|
350
350
|
boundAccountId: input.boundAccountId,
|
|
351
351
|
...(input.detourAccountId !== undefined ? { detourAccountId: input.detourAccountId } : {}),
|
|
352
|
-
|
|
352
|
+
// Both bounds, not just the probe pacing. A request refused by the RATIO has no probe state
|
|
353
|
+
// of its own yet, so `nextProbeAt` answered `now` and the refusal told the caller to try
|
|
354
|
+
// again immediately -- a withheld dispatch that busy-loops is the same load as the dispatch
|
|
355
|
+
// it refused. The limiter is the only thing that knows when its window moves.
|
|
356
|
+
retryAt: Math.max(
|
|
357
|
+
nextProbeAt(input.boundAccountId, now, input.minProbeIntervalMs),
|
|
358
|
+
limiter.nextRecoveryAt(now),
|
|
359
|
+
),
|
|
353
360
|
};
|
|
354
361
|
}
|
|
355
362
|
|
|
@@ -408,6 +415,16 @@ export interface PoolBackpressureLimiter {
|
|
|
408
415
|
tryPermitRetryDispatch(now?: number): boolean;
|
|
409
416
|
/** Admit one probe dispatch under the same shared recovery budget. */
|
|
410
417
|
tryPermitProbeDispatch(now?: number): boolean;
|
|
418
|
+
/**
|
|
419
|
+
* Earliest moment this limiter could admit another recovery dispatch.
|
|
420
|
+
*
|
|
421
|
+
* A refusal has to hand back a time, or the caller has nothing to wait on and busy-loops
|
|
422
|
+
* against a pool that is already failing -- which is the load this limiter exists to remove.
|
|
423
|
+
* `now` when the allowance is not spent; otherwise the moment the oldest bucket still inside
|
|
424
|
+
* the window falls out of it, which is strictly in the future and is a real change point
|
|
425
|
+
* rather than a guess.
|
|
426
|
+
*/
|
|
427
|
+
nextRecoveryAt(now?: number): number;
|
|
411
428
|
state(now?: number): PoolBackpressureState;
|
|
412
429
|
}
|
|
413
430
|
|
|
@@ -461,6 +478,19 @@ export function createPoolBackpressureLimiter(
|
|
|
461
478
|
return true;
|
|
462
479
|
}
|
|
463
480
|
|
|
481
|
+
function nextRecoveryAt(now: number): number {
|
|
482
|
+
const { initials, recoveries } = totals(now);
|
|
483
|
+
if (recoveries + 1 <= allowanceFor(initials)) return now;
|
|
484
|
+
// The window has to move before another recovery fits. The earliest that can happen is the
|
|
485
|
+
// moment the oldest bucket still inside it leaves, and every such bucket started after
|
|
486
|
+
// `now - windowMs`, so the answer is always strictly in the future.
|
|
487
|
+
for (const bucket of buckets) {
|
|
488
|
+
if (bucket.start <= now - policy.windowMs) continue;
|
|
489
|
+
return bucket.start + policy.windowMs;
|
|
490
|
+
}
|
|
491
|
+
return now + policy.windowMs;
|
|
492
|
+
}
|
|
493
|
+
|
|
464
494
|
return {
|
|
465
495
|
recordInitialSend(now = Date.now()): void {
|
|
466
496
|
bucketFor(now).initials += 1;
|
|
@@ -471,6 +501,9 @@ export function createPoolBackpressureLimiter(
|
|
|
471
501
|
tryPermitProbeDispatch(now = Date.now()): boolean {
|
|
472
502
|
return tryPermit(now);
|
|
473
503
|
},
|
|
504
|
+
nextRecoveryAt(now = Date.now()): number {
|
|
505
|
+
return nextRecoveryAt(now);
|
|
506
|
+
},
|
|
474
507
|
state(now = Date.now()): PoolBackpressureState {
|
|
475
508
|
const { initials, recoveries } = totals(now);
|
|
476
509
|
return {
|
|
@@ -509,3 +542,72 @@ export function configureSharedPoolBackpressure(policy: PoolBackpressurePolicy):
|
|
|
509
542
|
export function resetSharedPoolBackpressureForTests(): void {
|
|
510
543
|
sharedLimiter = undefined;
|
|
511
544
|
}
|
|
545
|
+
|
|
546
|
+
/**
|
|
547
|
+
* Forget every account's probe pacing AND the shared recovery window.
|
|
548
|
+
*
|
|
549
|
+
* Called when the pool's routing state is reset wholesale -- a roster change, a config reload,
|
|
550
|
+
* an account removal. Both halves describe a pool that no longer exists: pacing is keyed on
|
|
551
|
+
* account ids that may be gone, and the window's buckets count sends made by a roster that
|
|
552
|
+
* changed underneath them. Keeping either across such a reset lets one context's recovery
|
|
553
|
+
* decisions govern the next one, which is also how it leaks between test files.
|
|
554
|
+
*
|
|
555
|
+
* This is the production reset. The two `ForTests` seams above stay separate because a test
|
|
556
|
+
* frequently wants exactly one half of it.
|
|
557
|
+
*/
|
|
558
|
+
export function clearPoolRecoveryState(): void {
|
|
559
|
+
probeStates.clear();
|
|
560
|
+
sharedLimiter = undefined;
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
/**
|
|
564
|
+
* What one physical send IS, as far as the recovery window is concerned.
|
|
565
|
+
*
|
|
566
|
+
* The window measures recovery traffic against observed demand, so it needs the distinction
|
|
567
|
+
* made where the send happens -- and the transport wrapper cannot make it. That layer sees a
|
|
568
|
+
* URL and an init; whether this is a conversation's first attempt, its third retry, or the one
|
|
569
|
+
* trial admitted against a held account is knowledge only the caller has. So the caller names
|
|
570
|
+
* it, and the classification lives here with the window rather than in the transport, which
|
|
571
|
+
* owns no routing policy and has an enforced import boundary saying so.
|
|
572
|
+
*
|
|
573
|
+
* - `initial`: a new request's first send. Recorded, never refused -- it is the denominator,
|
|
574
|
+
* and refusing it would make this a throughput cap rather than a recovery bound.
|
|
575
|
+
* - `retry`: a re-send of a request that already reached upstream once. Admitted only while
|
|
576
|
+
* recovery traffic stays under its ratio of observed demand.
|
|
577
|
+
* - `probe`: the half-open trial against a held account. It ALREADY paid at selection, inside
|
|
578
|
+
* {@link resolveHeldAccountDispatch}; charging it again would bill one send twice and shrink
|
|
579
|
+
* the very budget it was admitted from.
|
|
580
|
+
*/
|
|
581
|
+
export type PoolRecoveryDispatchClass = "initial" | "retry" | "probe";
|
|
582
|
+
|
|
583
|
+
export interface PoolRecoveryDispatchDecision {
|
|
584
|
+
readonly admitted: boolean;
|
|
585
|
+
/**
|
|
586
|
+
* Earliest moment another recovery dispatch could be admitted. `now` when the send was
|
|
587
|
+
* admitted; otherwise a real change point strictly in the future, so a refused caller has
|
|
588
|
+
* something to wait on instead of busy-looping against a pool that is already failing.
|
|
589
|
+
*/
|
|
590
|
+
readonly retryAt: number;
|
|
591
|
+
}
|
|
592
|
+
|
|
593
|
+
/**
|
|
594
|
+
* Admit one physical send against the process-wide recovery window.
|
|
595
|
+
*
|
|
596
|
+
* Per-request send budgets cannot see a storm: thousands of requests each staying inside their
|
|
597
|
+
* own allowance still compose into an unbounded rate against one failing upstream. This is the
|
|
598
|
+
* layer above them, and it is shared by construction.
|
|
599
|
+
*/
|
|
600
|
+
export function classifyPoolRecoveryDispatch(
|
|
601
|
+
dispatchClass: PoolRecoveryDispatchClass,
|
|
602
|
+
now = Date.now(),
|
|
603
|
+
limiter: PoolBackpressureLimiter = sharedPoolBackpressure(),
|
|
604
|
+
): PoolRecoveryDispatchDecision {
|
|
605
|
+
if (dispatchClass === "initial") {
|
|
606
|
+
limiter.recordInitialSend(now);
|
|
607
|
+
return { admitted: true, retryAt: now };
|
|
608
|
+
}
|
|
609
|
+
if (dispatchClass === "probe") return { admitted: true, retryAt: now };
|
|
610
|
+
return limiter.tryPermitRetryDispatch(now)
|
|
611
|
+
? { admitted: true, retryAt: now }
|
|
612
|
+
: { admitted: false, retryAt: limiter.nextRecoveryAt(now) };
|
|
613
|
+
}
|
|
@@ -170,7 +170,9 @@ async function handleChatCompletionsWithBudget(
|
|
|
170
170
|
if (chatBody.tools !== undefined) parts.push(JSON.stringify(chatBody.tools));
|
|
171
171
|
logCtx.usageLogInputTokens = Math.max(1, estimateTokens(parts.join("\n"), requestedModel));
|
|
172
172
|
}
|
|
173
|
-
|
|
173
|
+
// Combos must enter the Responses routing path so child selection, forced default
|
|
174
|
+
// effort, failover, and per-attempt telemetry run before any native Chat send.
|
|
175
|
+
if (!route.combo && !effortRow && isNativeChatRouteEligible(route, chatBody, config)) chatNativeRoute = route;
|
|
174
176
|
} catch (err) {
|
|
175
177
|
if (err instanceof UnknownRoutingPolicyError) {
|
|
176
178
|
logCtx.requestedModel = requestedModel;
|
|
@@ -25,8 +25,12 @@ import {
|
|
|
25
25
|
applyUpstreamRecoveryInit,
|
|
26
26
|
fetchWithResetRetry,
|
|
27
27
|
fetchWithTransientRetry,
|
|
28
|
+
isNonReplayableResponse,
|
|
29
|
+
isReplayRefusalCode,
|
|
30
|
+
isReplayRefusalResponse,
|
|
28
31
|
prepareSameTarget429Wait,
|
|
29
32
|
type UpstreamSendRecovery,
|
|
33
|
+
UPSTREAM_RESET_REPLAY_REFUSED_CODE,
|
|
30
34
|
} from "../lib/upstream-retry";
|
|
31
35
|
import {
|
|
32
36
|
isTranslatorBudgetExceededError,
|
|
@@ -51,7 +55,9 @@ import { linkAbortSignal } from "./responses";
|
|
|
51
55
|
import {
|
|
52
56
|
addFinalRequestLog,
|
|
53
57
|
beginRequestAttempt,
|
|
54
|
-
|
|
58
|
+
noteProviderAttemptSend,
|
|
59
|
+
recordKeyAttemptFailure,
|
|
60
|
+
recordKeyWireAttemptUsage,
|
|
55
61
|
recordFirstOutput,
|
|
56
62
|
recordAttemptCredentialSource,
|
|
57
63
|
sealRequestAttemptIdentity,
|
|
@@ -344,10 +350,12 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
|
|
|
344
350
|
const encoding = new Headers(init.headers).get("accept-encoding");
|
|
345
351
|
if (!headers.has("accept-encoding") && encoding) headers.set("accept-encoding", encoding);
|
|
346
352
|
if (init.signal?.aborted) throw init.signal.reason;
|
|
347
|
-
|
|
348
|
-
|
|
353
|
+
noteProviderAttemptSend(logCtx, route.providerName, activeProvider, logCtx.usageLogInputTokens, transportRecovery ?? recovery);
|
|
354
|
+
const dispatched = await ((activeProvider as OcxProviderTransport).fetch ?? execute)(request.url, applyUpstreamRecoveryInit({
|
|
349
355
|
...init, method: request.method, headers, body: request.body,
|
|
350
356
|
}, transportRecovery));
|
|
357
|
+
if (!dispatched.ok) await recordKeyAttemptFailure(logCtx, dispatched, init.signal ?? upstream.signal);
|
|
358
|
+
return dispatched;
|
|
351
359
|
},
|
|
352
360
|
}),
|
|
353
361
|
);
|
|
@@ -375,6 +383,11 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
|
|
|
375
383
|
let retries = 0;
|
|
376
384
|
while (
|
|
377
385
|
response.status === 429
|
|
386
|
+
// A 429 this proxy synthesized for a refused reset replay is not a provider rate
|
|
387
|
+
// limit: waiting and re-sending here is exactly the duplicate inference the refusal
|
|
388
|
+
// exists to stop. It kept the same shape under the old 502 only because 502 never
|
|
389
|
+
// matched this branch.
|
|
390
|
+
&& !isNonReplayableResponse(response)
|
|
378
391
|
&& retryPolicy
|
|
379
392
|
&& retries < retryPolicy.attempts
|
|
380
393
|
&& transientSendAvailable()
|
|
@@ -388,7 +401,9 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
|
|
|
388
401
|
if (upstream.signal.aborted) throw upstream.signal.reason;
|
|
389
402
|
response = await send(activeRequest, "rate-limit-429");
|
|
390
403
|
}
|
|
391
|
-
|
|
404
|
+
// Same reason as above, plus a second one: rotating here would write a cooldown against
|
|
405
|
+
// a key that rate-limited nothing, and that false signal outlives the request.
|
|
406
|
+
while (response.status === 429 && !isNonReplayableResponse(response) && hasKeyPoolFailover(activeProvider)) {
|
|
392
407
|
const rotated = rotateProviderTransportOn429(config, route.providerName, activeProvider, {
|
|
393
408
|
retryAfter: response.headers.get("retry-after"),
|
|
394
409
|
now: Date.now(),
|
|
@@ -474,6 +489,12 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
|
|
|
474
489
|
if (isCyberPolicyCode(upstreamCode) || classified.code === CYBER_POLICY_ERROR_CODE) {
|
|
475
490
|
classified.code = CYBER_POLICY_ERROR_CODE;
|
|
476
491
|
classified.type = cyberPolicyErrorType(upstreamType);
|
|
492
|
+
} else if (isReplayRefusalResponse(response) || isReplayRefusalCode(upstreamCode)) {
|
|
493
|
+
// 429 classifies as a rate limit and a rate limit already carries a code, so the branch
|
|
494
|
+
// below -- which only fills an EMPTY code -- could never restore this one. Without it the
|
|
495
|
+
// client is told the provider throttled the turn, when what happened is that this proxy
|
|
496
|
+
// declined to send it a second time.
|
|
497
|
+
classified.code = UPSTREAM_RESET_REPLAY_REFUSED_CODE;
|
|
477
498
|
} else if (upstreamCode === "model_not_found") {
|
|
478
499
|
classified.code = "model_not_found";
|
|
479
500
|
classified.type = "invalid_request_error";
|
|
@@ -481,7 +502,10 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
|
|
|
481
502
|
classified.code = upstreamCode;
|
|
482
503
|
}
|
|
483
504
|
const status = isCyberPolicyCode(classified.code) ? 400 : response.status;
|
|
484
|
-
|
|
505
|
+
// A refusal this proxy made has no wait to report. Synthesizing one here would hand the
|
|
506
|
+
// client the default two-second retry for a rate limit that never happened, which is the
|
|
507
|
+
// duplicate send the refusal exists to prevent.
|
|
508
|
+
const retryAfter = isCyberPolicyCode(classified.code) || isReplayRefusalCode(classified.code)
|
|
485
509
|
? undefined
|
|
486
510
|
: resolveClientRetryAfter({
|
|
487
511
|
status: response.status,
|
|
@@ -509,8 +533,10 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
|
|
|
509
533
|
stallTimeoutSec: config.stallTimeoutSec,
|
|
510
534
|
onFirstOutput: logIds ? () => recordFirstOutput(logCtx, logIds.start) : undefined,
|
|
511
535
|
onUsage: usage => {
|
|
512
|
-
logCtx
|
|
513
|
-
|
|
536
|
+
if (!recordKeyWireAttemptUsage(logCtx, usage)) {
|
|
537
|
+
logCtx.usage = usage;
|
|
538
|
+
attempt.usage = usage;
|
|
539
|
+
}
|
|
514
540
|
},
|
|
515
541
|
onTerminal: (status: number, message?: string) => {
|
|
516
542
|
terminalStatus = status;
|
|
@@ -600,8 +626,10 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
|
|
|
600
626
|
if (!completion) return fail(502, "upstream response contained no choices", "upstream_error");
|
|
601
627
|
const usage = usageFromChat(completion.usage);
|
|
602
628
|
if (usage) {
|
|
603
|
-
logCtx
|
|
604
|
-
|
|
629
|
+
if (!recordKeyWireAttemptUsage(logCtx, usage)) {
|
|
630
|
+
logCtx.usage = usage;
|
|
631
|
+
attempt.usage = usage;
|
|
632
|
+
}
|
|
605
633
|
}
|
|
606
634
|
if (logIds) recordFirstOutput(logCtx, logIds.start);
|
|
607
635
|
try {
|