@bitkyc08/opencodex 2.58.0 → 2.59.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -10
- package/gui/dist/assets/index-C5IebErG.js +136 -0
- package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
- package/gui/dist/index.html +2 -2
- package/gui/dist/provider-icons/crusoe.svg +1 -0
- package/gui/dist/provider-icons/opper.svg +3 -0
- package/package.json +1 -1
- package/src/adapters/base.ts +11 -1
- package/src/adapters/cursor/catalog.ts +11 -0
- package/src/adapters/cursor/effort-map.ts +16 -2
- package/src/adapters/cursor/envelope-echo.ts +55 -2
- package/src/adapters/cursor/message-mapper.ts +3 -2
- package/src/adapters/cursor/protobuf-request.ts +8 -5
- package/src/adapters/cursor/request-builder.ts +14 -3
- package/src/adapters/cursor/thread-continuity.ts +105 -31
- package/src/adapters/cursor/tool-guidance.ts +5 -4
- package/src/adapters/cursor.ts +42 -1
- package/src/adapters/devin/cloud-direct/chat.ts +11 -2
- package/src/adapters/devin/cloud-direct/index.ts +7 -0
- package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
- package/src/adapters/devin.ts +75 -13
- package/src/adapters/google-antigravity-wire.ts +29 -2
- package/src/adapters/google-http.ts +8 -1
- package/src/adapters/google.ts +23 -4
- package/src/adapters/openai-chat/response-events.ts +61 -0
- package/src/adapters/openai-chat.ts +5 -10
- package/src/adapters/openai-responses/passthrough.ts +10 -1
- package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
- package/src/adapters/openai-responses/tool-schema.ts +19 -7
- package/src/adapters/responses-tool-schema.ts +76 -46
- package/src/adapters/run-turn-queue.ts +17 -4
- package/src/bridge/response-json.ts +1 -1
- package/src/bridge/sse.ts +165 -24
- package/src/claude/context-windows.ts +22 -0
- package/src/claude/outbound.ts +35 -4
- package/src/cli/account-api.ts +4 -3
- package/src/cli/account-extended.ts +22 -2
- package/src/cli/account-orca-import.ts +63 -0
- package/src/cli/account.ts +32 -4
- package/src/cli/capabilities.ts +40 -0
- package/src/cli/claude.ts +29 -1
- package/src/cli/codex-cli-update.ts +97 -2
- package/src/cli/dispatch.ts +54 -0
- package/src/cli/doctor.ts +197 -2
- package/src/cli/help.ts +4 -1
- package/src/cli/index.ts +88 -20
- package/src/cli/models-runtime.ts +33 -4
- package/src/cli/registry.ts +11 -1
- package/src/cli/runtime-api.ts +44 -0
- package/src/cli/start-args.ts +94 -0
- package/src/cli/system-command.ts +2 -0
- package/src/client/machine-api.ts +4 -3
- package/src/client/machine-listener.ts +14 -1
- package/src/clients/config-export/constants.ts +2 -3
- package/src/clients/config-export.ts +5 -5
- package/src/codex/account-store.ts +81 -5
- package/src/codex/auth-api/pool-quota-probe.ts +14 -3
- package/src/codex/auth-api/routes.ts +17 -2
- package/src/codex/auth-context.ts +16 -12
- package/src/codex/catalog/build-entries.ts +25 -4
- package/src/codex/catalog/derive-entry.ts +8 -1
- package/src/codex/catalog/effort.ts +10 -6
- package/src/codex/catalog/gather-capture.ts +1 -0
- package/src/codex/catalog/model-hints.ts +37 -5
- package/src/codex/catalog/parsing.ts +83 -5
- package/src/codex/catalog/reserve-warn.ts +96 -0
- package/src/codex/catalog/retained-sync.ts +19 -0
- package/src/codex/catalog/routed-gather.ts +42 -3
- package/src/codex/cli-installation-identity.ts +210 -0
- package/src/codex/cli-installation-targets.ts +158 -0
- package/src/codex/convergence.ts +5 -0
- package/src/codex/history-provider.ts +4 -1
- package/src/codex/history-state-open.ts +105 -0
- package/src/codex/inject/config-toml.ts +44 -2
- package/src/codex/inject.ts +3 -2
- package/src/codex/lineage.ts +83 -32
- package/src/codex/loopback-target.ts +31 -0
- package/src/codex/main-account-hard-lock.ts +2 -1
- package/src/codex/main-account.ts +10 -3
- package/src/codex/main-device-reauth.ts +17 -9
- package/src/codex/model-entitlements.ts +60 -1
- package/src/codex/observed-model-denials.ts +137 -0
- package/src/codex/orca-auth-source.ts +94 -0
- package/src/codex/orca-import.ts +219 -0
- package/src/codex/prompt-text-probe.ts +282 -12
- package/src/codex/quota-401-recovery.ts +12 -0
- package/src/codex/quota-types.ts +65 -0
- package/src/codex/quota.ts +24 -19
- package/src/codex/routing/cooldown-math.ts +8 -47
- package/src/codex/routing/pin-drain.ts +57 -0
- package/src/codex/routing.ts +13 -15
- package/src/codex/subagent-model-fallback.ts +94 -0
- package/src/codex/windows-installation-files.ts +224 -0
- package/src/combos/failover.ts +122 -5
- package/src/config/diagnostics.ts +21 -0
- package/src/config/load-degrade.ts +15 -0
- package/src/config/pending-teardown.ts +8 -0
- package/src/config/process-state.ts +36 -3
- package/src/config/provider-relative-send-path.ts +16 -0
- package/src/config/proxy-env.ts +23 -5
- package/src/config/schema/config-schema.ts +21 -0
- package/src/config/schema/leaf-validators.ts +64 -17
- package/src/generated/compatibility-version.json +235 -163
- package/src/generated/model-metadata.ts +1 -1
- package/src/lib/bounded-body.ts +4 -2
- package/src/lib/destination-policy.ts +48 -6
- package/src/lib/errors.ts +3 -15
- package/src/lib/local-destinations.ts +32 -5
- package/src/lib/provider-outbound.ts +3 -3
- package/src/lib/proxy-env.ts +70 -3
- package/src/lib/request-execution-budget.ts +11 -3
- package/src/lib/response-body-inactivity.ts +193 -0
- package/src/lib/retry-delay.ts +69 -0
- package/src/lib/socks5-fetch.ts +631 -0
- package/src/lib/spend-reservation-ledger.ts +115 -9
- package/src/lib/workflow-budget.ts +145 -8
- package/src/oauth/account-quota-rank.ts +72 -15
- package/src/oauth/generic-account-failover.ts +40 -27
- package/src/oauth/orcarouter.ts +15 -2
- package/src/oauth/store.ts +8 -0
- package/src/providers/codex-capacity.ts +9 -0
- package/src/providers/devin-provider-merge-migration.ts +33 -12
- package/src/providers/free-directory.ts +20 -2
- package/src/providers/key-failover.ts +261 -7
- package/src/providers/model-rename-migration.ts +1 -0
- package/src/providers/openai-sidecar.ts +4 -0
- package/src/providers/opencode-go-transport.ts +14 -5
- package/src/providers/quota/report-cache.ts +3 -0
- package/src/providers/registry/entries-extended.ts +96 -0
- package/src/providers/registry/model-seeds.ts +78 -21
- package/src/responses/apply-patch-envelope.ts +44 -11
- package/src/responses/bridge-search-replay-cache.ts +152 -0
- package/src/responses/code-mode-helper-compat.ts +26 -16
- package/src/responses/custom-tool-compat.ts +1 -1
- package/src/responses/hosted-tool-policy.ts +85 -2
- package/src/responses/schema.ts +9 -2
- package/src/server/auth-cors.ts +26 -0
- package/src/server/chat-completions.ts +9 -4
- package/src/server/chat-native-sse.ts +26 -9
- package/src/server/chat-native.ts +10 -4
- package/src/server/claude-messages.ts +24 -2
- package/src/server/gui-static.ts +36 -2
- package/src/server/inbound-body-admission.ts +187 -0
- package/src/server/index.ts +15 -19
- package/src/server/management/api-access.ts +3 -4
- package/src/server/management/config-routes.ts +31 -6
- package/src/server/management/provider-capability-config.ts +35 -7
- package/src/server/management/provider-routes.ts +70 -18
- package/src/server/proxy-liveness.ts +97 -2
- package/src/server/relay.ts +17 -24
- package/src/server/request-log.ts +25 -1
- package/src/server/responses/adapter-continuation.ts +71 -27
- package/src/server/responses/adapter-delivery.ts +39 -8
- package/src/server/responses/adapter-dispatch.ts +52 -24
- package/src/server/responses/compact.ts +60 -11
- package/src/server/responses/core-codex-account.ts +83 -22
- package/src/server/responses/core-normalize.ts +12 -5
- package/src/server/responses/fetch-helpers.ts +68 -2
- package/src/server/responses/passthrough-delivery.ts +10 -1
- package/src/server/responses/passthrough-dispatch.ts +113 -48
- package/src/server/responses/passthrough-execution.ts +11 -1
- package/src/server/responses/request-prepare.ts +29 -0
- package/src/server/responses/request-send-budget.ts +84 -7
- package/src/server/responses/request-sidecar-auth.ts +16 -8
- package/src/server/responses/request-spend.ts +38 -9
- package/src/server/responses/request-transport.ts +13 -10
- package/src/server/responses/run-turn-execution.ts +20 -5
- package/src/server/responses/sidecar-execution.ts +2 -0
- package/src/server/responses/ws-upstream.ts +2 -1
- package/src/server/responses-custom-tool-repair.ts +2 -2
- package/src/server/sse-frame-buffer.ts +12 -10
- package/src/server/sse-payload-rewrite.ts +36 -9
- package/src/server/system-env-shell.ts +5 -1
- package/src/server/system-env.ts +7 -1
- package/src/server/workflow-refusal.ts +56 -2
- package/src/service/cli.ts +16 -6
- package/src/service/guards.ts +10 -0
- package/src/service/health.ts +43 -0
- package/src/service/state.ts +7 -2
- package/src/types/accounts.ts +4 -0
- package/src/types/config.ts +100 -3
- package/src/types/provider.ts +19 -0
- package/src/types/request.ts +7 -1
- package/src/types/wire.ts +9 -1
- package/src/usage/expected-prices.ts +28 -0
- package/src/usage/log.ts +87 -4
- package/src/web-search/passthrough-bridge.ts +39 -5
- package/gui/dist/assets/index-BbrHOIY0.js +0 -128
|
@@ -60,6 +60,9 @@ import { dirname, join } from "node:path";
|
|
|
60
60
|
// Definition-site import, not the ../config barrel -- same reasoning as
|
|
61
61
|
// src/quota/reset-seen-store.ts: the barrel pulls ~154 modules into a hot path.
|
|
62
62
|
import { getConfigDir } from "../config/paths";
|
|
63
|
+
// Type-only, so it is erased before this module has a runtime import graph at all. The
|
|
64
|
+
// config SHAPE is what this file needs; the config loader is what the note above keeps out.
|
|
65
|
+
import type { OcxSpendConfig, OcxSpendScopeConfig } from "../types/config";
|
|
63
66
|
import { assertNotRealHomeUnderTest } from "./test-home-guard";
|
|
64
67
|
// Windows chmod does not remove inherited ACEs; this is the repository's icacls path.
|
|
65
68
|
import { hardenSecretPath } from "./windows-secret-acl";
|
|
@@ -147,6 +150,24 @@ export interface SpendReservationRequest {
|
|
|
147
150
|
/** Enforceable output ceiling -- max_output_tokens or the model's documented cap. */
|
|
148
151
|
readonly outputCeilingTokens: number;
|
|
149
152
|
readonly at?: number;
|
|
153
|
+
/**
|
|
154
|
+
* This send has ALREADY left for upstream and is being recorded rather than admitted.
|
|
155
|
+
*
|
|
156
|
+
* Some transports report their physical sends after the fact -- the passthrough ladder
|
|
157
|
+
* reports through `onSendsConsumed`, and an adapter's inner retries are counted when they
|
|
158
|
+
* finish. For those, a ceiling cannot refuse anything: the tokens are spent. Refusing to
|
|
159
|
+
* BOOK them is the worse answer, and it is not hypothetical -- it is a fixpoint. The send
|
|
160
|
+
* that would cross the ceiling gets dropped from the total, the total stays just under the
|
|
161
|
+
* limit forever, the scope never reads as exhausted, and the ceiling never fires again for
|
|
162
|
+
* any request. So a recorded send skips the limit check and takes the scope over its
|
|
163
|
+
* ceiling, which is what makes the NEXT request refusable.
|
|
164
|
+
*
|
|
165
|
+
* It skips the durability refusal for the same reason: a journal that could not be written
|
|
166
|
+
* is a reason to report degradation, never a reason to forget spend that really happened.
|
|
167
|
+
* Identity, capacity and journal-integrity denials still apply -- those say the ledger
|
|
168
|
+
* cannot account for the send at all, which no flag here can change.
|
|
169
|
+
*/
|
|
170
|
+
readonly alreadySent?: boolean;
|
|
150
171
|
}
|
|
151
172
|
|
|
152
173
|
/**
|
|
@@ -468,6 +489,20 @@ export interface SpendReservationLedger {
|
|
|
468
489
|
prune(now?: number): void;
|
|
469
490
|
/** Whether this send id is already known, and therefore refused. */
|
|
470
491
|
knows(sendId: string): boolean;
|
|
492
|
+
/**
|
|
493
|
+
* Replace the live policy.
|
|
494
|
+
*
|
|
495
|
+
* Every figure already accounted survives: raising, lowering or clearing a ceiling changes
|
|
496
|
+
* what is REFUSED from here on and never what was spent. Rebuilding the ledger instead
|
|
497
|
+
* would replay the journal into a second set of maps while the first still holds this
|
|
498
|
+
* process's open reservations, and the two would then disagree about what is in flight.
|
|
499
|
+
*/
|
|
500
|
+
reconfigure(next: SpendReservationPolicy): void;
|
|
501
|
+
/**
|
|
502
|
+
* The policy in force. A live read, not a copy: a caller that formats a refusal has to name
|
|
503
|
+
* the ceiling this ledger would enforce on the NEXT request, not the one it was built with.
|
|
504
|
+
*/
|
|
505
|
+
readonly policy: SpendReservationPolicy;
|
|
471
506
|
/** Journal writes that failed; a nonzero count means durability is degraded. */
|
|
472
507
|
readonly persistFailures: number;
|
|
473
508
|
/**
|
|
@@ -495,13 +530,17 @@ export function createSpendReservationLedger(options: {
|
|
|
495
530
|
*/
|
|
496
531
|
readonly salt?: string;
|
|
497
532
|
} = {}): SpendReservationLedger {
|
|
498
|
-
|
|
533
|
+
// Mutable because the ceilings are operator configuration, and configuration is reloadable.
|
|
534
|
+
// The three bounds below are read through functions for the same reason: a value captured
|
|
535
|
+
// at construction would answer for the policy this ledger was BUILT with, and an operator
|
|
536
|
+
// who raised a bound would keep the old one until the process restarted.
|
|
537
|
+
let policy = options.policy ?? DEFAULT_SPEND_RESERVATION_POLICY;
|
|
499
538
|
const journal = options.journal;
|
|
500
539
|
const now = options.now ?? (() => Date.now());
|
|
501
540
|
const salt = options.salt ?? "";
|
|
502
|
-
const maxTrackedScopes = policy.maxTrackedScopes ?? DEFAULT_MAX_TRACKED_SCOPES;
|
|
503
|
-
const maxTrackedSends = policy.maxTrackedSends ?? DEFAULT_MAX_TRACKED_SENDS;
|
|
504
|
-
const compactAfterRecords = policy.compactAfterRecords ?? DEFAULT_COMPACT_AFTER_RECORDS;
|
|
541
|
+
const maxTrackedScopes = (): number => policy.maxTrackedScopes ?? DEFAULT_MAX_TRACKED_SCOPES;
|
|
542
|
+
const maxTrackedSends = (): number => policy.maxTrackedSends ?? DEFAULT_MAX_TRACKED_SENDS;
|
|
543
|
+
const compactAfterRecords = (): number => policy.compactAfterRecords ?? DEFAULT_COMPACT_AFTER_RECORDS;
|
|
505
544
|
const scopes = new Map<string, ScopeState>();
|
|
506
545
|
const reservations = new Map<string, Reservation>();
|
|
507
546
|
let persistFailures = 0;
|
|
@@ -753,7 +792,7 @@ export function createSpendReservationLedger(options: {
|
|
|
753
792
|
*/
|
|
754
793
|
const compact = (at: number): void => {
|
|
755
794
|
const rewrite = journal?.rewrite;
|
|
756
|
-
if (!journal || !rewrite || recordsOnDisk < compactAfterRecords) return;
|
|
795
|
+
if (!journal || !rewrite || recordsOnDisk < compactAfterRecords()) return;
|
|
757
796
|
const checkpoint: JournalRecord = {
|
|
758
797
|
v: 1,
|
|
759
798
|
kind: "checkpoint",
|
|
@@ -791,12 +830,12 @@ export function createSpendReservationLedger(options: {
|
|
|
791
830
|
const makeRoom = (refs: readonly ScopeRef[], at: number): SpendDenial | undefined => {
|
|
792
831
|
evictSends(at, false);
|
|
793
832
|
evictScopes(at, false);
|
|
794
|
-
while (reservations.size >= maxTrackedSends) {
|
|
833
|
+
while (reservations.size >= maxTrackedSends()) {
|
|
795
834
|
if (evictSends(at, true) === 0) return { reason: "tracking-capacity-exhausted" };
|
|
796
835
|
}
|
|
797
836
|
let fresh = 0;
|
|
798
837
|
for (const ref of refs) if (!scopes.has(scopeKey(ref.scope, ref.alias))) fresh += 1;
|
|
799
|
-
while (scopes.size + fresh > maxTrackedScopes) {
|
|
838
|
+
while (scopes.size + fresh > maxTrackedScopes()) {
|
|
800
839
|
if (evictScopes(at, true) === 0) {
|
|
801
840
|
return { reason: "tracking-capacity-exhausted", scope: refs[0]?.scope };
|
|
802
841
|
}
|
|
@@ -808,6 +847,7 @@ export function createSpendReservationLedger(options: {
|
|
|
808
847
|
get persistFailures() { return persistFailures; },
|
|
809
848
|
get corruptRecords() { return corruptRecords; },
|
|
810
849
|
get degraded() { return persistFailures > 0 || corruptRecords > 0; },
|
|
850
|
+
get policy() { return policy; },
|
|
811
851
|
|
|
812
852
|
reserve(request: SpendReservationRequest): SpendReservationDecision {
|
|
813
853
|
const tokens = sanitizeTokens(request.inputTokens) + sanitizeTokens(request.outputCeilingTokens);
|
|
@@ -834,7 +874,9 @@ export function createSpendReservationLedger(options: {
|
|
|
834
874
|
// reservation booked on the scopes that would have passed. Reading state without
|
|
835
875
|
// creating it matters here -- a denied request must not leave a tracked scope behind.
|
|
836
876
|
for (const ref of refs) {
|
|
837
|
-
|
|
877
|
+
// A recorded send has no limit to fail: it already happened, and the point of booking
|
|
878
|
+
// it is to let the total go OVER the ceiling so the next request can be refused.
|
|
879
|
+
const limit = request.alreadySent === true ? undefined : limitFor(ref.scope);
|
|
838
880
|
if (limit === undefined) continue;
|
|
839
881
|
const state = scopes.get(scopeKey(ref.scope, ref.alias));
|
|
840
882
|
const projected = (state ? state.settled + state.reserved + state.unresolved : 0) + tokens;
|
|
@@ -853,7 +895,7 @@ export function createSpendReservationLedger(options: {
|
|
|
853
895
|
// limit a failed write refuses the request rather than admitting one that a restart
|
|
854
896
|
// would forget -- which is exactly the disk-full and permission case durability is for.
|
|
855
897
|
const durable = append({ v: 1, kind: "reserve", send, targets: refs, tokens, at });
|
|
856
|
-
if (!durable && enforced) {
|
|
898
|
+
if (!durable && enforced && request.alreadySent !== true) {
|
|
857
899
|
return { reserved: false, denial: { reason: "reserve-not-durable", sendId: request.sendId } };
|
|
858
900
|
}
|
|
859
901
|
applyReserve(send, refs, tokens, at);
|
|
@@ -931,10 +973,72 @@ export function createSpendReservationLedger(options: {
|
|
|
931
973
|
evictSends(at, false);
|
|
932
974
|
evictScopes(at, false);
|
|
933
975
|
},
|
|
976
|
+
|
|
977
|
+
reconfigure(next: SpendReservationPolicy): void {
|
|
978
|
+
policy = next;
|
|
979
|
+
},
|
|
934
980
|
};
|
|
935
981
|
}
|
|
936
982
|
|
|
937
983
|
let sharedLedger: SpendReservationLedger | undefined;
|
|
984
|
+
/**
|
|
985
|
+
* The operator policy in effect. Held beside the ledger rather than inside it because the
|
|
986
|
+
* ledger is built lazily: a configured ceiling has to be remembered from startup until the
|
|
987
|
+
* first request that actually reserves, and an install that configures nothing must still
|
|
988
|
+
* open no journal.
|
|
989
|
+
*/
|
|
990
|
+
let sharedPolicy: SpendReservationPolicy = DEFAULT_SPEND_RESERVATION_POLICY;
|
|
991
|
+
|
|
992
|
+
/** Whether any scope carries a ceiling -- that is, whether anything at all can be refused. */
|
|
993
|
+
export function spendCeilingsConfigured(policy: SpendReservationPolicy = sharedPolicy): boolean {
|
|
994
|
+
return policy.root.maxTokens !== undefined
|
|
995
|
+
|| policy.identity.maxTokens !== undefined
|
|
996
|
+
|| policy.pool.maxTokens !== undefined;
|
|
997
|
+
}
|
|
998
|
+
|
|
999
|
+
/** The policy the process-wide ledger enforces right now. */
|
|
1000
|
+
export function sharedSpendPolicy(): SpendReservationPolicy {
|
|
1001
|
+
return sharedPolicy;
|
|
1002
|
+
}
|
|
1003
|
+
|
|
1004
|
+
const spendScopeLimitFromConfig = (scope: OcxSpendScopeConfig | undefined): SpendScopeLimit =>
|
|
1005
|
+
scope?.maxTokens !== undefined && Number.isFinite(scope.maxTokens) && scope.maxTokens > 0
|
|
1006
|
+
? { maxTokens: Math.trunc(scope.maxTokens) }
|
|
1007
|
+
: {};
|
|
1008
|
+
|
|
1009
|
+
/**
|
|
1010
|
+
* The ledger policy an operator's `spend` section asks for.
|
|
1011
|
+
*
|
|
1012
|
+
* An absent section, an empty one, and one whose every ceiling is absent all produce the
|
|
1013
|
+
* unconfigured default: observe-only accounting that refuses nothing. That equivalence is the
|
|
1014
|
+
* load-bearing part. This ledger is on and journaling by default, so shipping a default
|
|
1015
|
+
* ceiling would start refusing real traffic on the first upgrade that ran this code, against
|
|
1016
|
+
* a number nobody chose. There is deliberately no default figure here at all.
|
|
1017
|
+
*/
|
|
1018
|
+
export function spendPolicyFromConfig(spend: OcxSpendConfig | undefined): SpendReservationPolicy {
|
|
1019
|
+
const retentionDays = spend?.retentionDays;
|
|
1020
|
+
return {
|
|
1021
|
+
root: spendScopeLimitFromConfig(spend?.root),
|
|
1022
|
+
identity: spendScopeLimitFromConfig(spend?.identity),
|
|
1023
|
+
pool: spendScopeLimitFromConfig(spend?.pool),
|
|
1024
|
+
retentionMs: retentionDays !== undefined && Number.isFinite(retentionDays) && retentionDays > 0
|
|
1025
|
+
? Math.trunc(retentionDays) * 24 * 60 * 60_000
|
|
1026
|
+
: DEFAULT_SPEND_RESERVATION_POLICY.retentionMs,
|
|
1027
|
+
};
|
|
1028
|
+
}
|
|
1029
|
+
|
|
1030
|
+
/**
|
|
1031
|
+
* Apply an operator policy to the process-wide ledger.
|
|
1032
|
+
*
|
|
1033
|
+
* Startup calls this with the loaded config, and a reload may call it again: the ledger keeps
|
|
1034
|
+
* every figure it has already accounted, so changing a ceiling changes what is refused from
|
|
1035
|
+
* here on and never what was spent. It does not CREATE the ledger -- an install that
|
|
1036
|
+
* configures no ceiling must not open a journal merely because the server started.
|
|
1037
|
+
*/
|
|
1038
|
+
export function configureSharedSpendLedger(policy: SpendReservationPolicy): void {
|
|
1039
|
+
sharedPolicy = policy;
|
|
1040
|
+
sharedLedger?.reconfigure(policy);
|
|
1041
|
+
}
|
|
938
1042
|
|
|
939
1043
|
/**
|
|
940
1044
|
* Process-wide ledger backed by the journal under OPENCODEX_HOME. Created lazily so
|
|
@@ -947,6 +1051,7 @@ export function sharedSpendLedger(): SpendReservationLedger {
|
|
|
947
1051
|
sharedLedger = createSpendReservationLedger({
|
|
948
1052
|
journal: createFileSpendJournal(join(home, SPEND_LEDGER_JOURNAL_FILENAME)),
|
|
949
1053
|
salt: loadOrCreateSpendLedgerSalt(join(home, SPEND_LEDGER_SALT_FILENAME)),
|
|
1054
|
+
policy: sharedPolicy,
|
|
950
1055
|
});
|
|
951
1056
|
}
|
|
952
1057
|
return sharedLedger;
|
|
@@ -955,4 +1060,5 @@ export function sharedSpendLedger(): SpendReservationLedger {
|
|
|
955
1060
|
/** Test seam. Production never discards the ledger: that would reset a spent budget. */
|
|
956
1061
|
export function resetSharedSpendLedgerForTest(): void {
|
|
957
1062
|
sharedLedger = undefined;
|
|
1063
|
+
sharedPolicy = DEFAULT_SPEND_RESERVATION_POLICY;
|
|
958
1064
|
}
|
|
@@ -21,11 +21,47 @@
|
|
|
21
21
|
|
|
22
22
|
import {
|
|
23
23
|
sharedSpendLedger,
|
|
24
|
+
spendCeilingsConfigured,
|
|
24
25
|
type SpendReservationLedger,
|
|
25
26
|
type SpendScope,
|
|
26
27
|
type SpendUsage,
|
|
27
28
|
} from "./spend-reservation-ledger";
|
|
28
29
|
|
|
30
|
+
/**
|
|
31
|
+
* What a token-ceiling refusal has to be able to say.
|
|
32
|
+
*
|
|
33
|
+
* "Budget exhausted" on its own is the failure this repository keeps re-learning: a policy
|
|
34
|
+
* rejection wearing another error's clothing sends an operator to look at the provider. The
|
|
35
|
+
* scope says WHICH ceiling fired -- one task, one account, or the whole pool -- and the limit
|
|
36
|
+
* is the number they would otherwise have to read the journal to recover. The scope ID is
|
|
37
|
+
* deliberately not here: root ids are client thread headers and identity ids are credentials,
|
|
38
|
+
* and the ledger's rule is that neither is written down in the clear.
|
|
39
|
+
*/
|
|
40
|
+
export interface WorkflowSpendDenialDetail {
|
|
41
|
+
readonly scope: SpendScope;
|
|
42
|
+
readonly limit: number;
|
|
43
|
+
/** Tokens the refused reservation would have taken the scope to, where that is known. */
|
|
44
|
+
readonly projected?: number;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** Operator-facing name for each scope. What an operator calls it, not what the type calls it. */
|
|
48
|
+
const SPEND_SCOPE_LABEL: Record<SpendScope, string> = {
|
|
49
|
+
root: "task",
|
|
50
|
+
identity: "account",
|
|
51
|
+
pool: "provider pool",
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Thousands separators, done here rather than by `toLocaleString`.
|
|
56
|
+
*
|
|
57
|
+
* A ceiling is an eight- or nine-digit number and an unseparated one is genuinely hard to read
|
|
58
|
+
* against the figure beside it. `toLocaleString` would do this too, but its output depends on
|
|
59
|
+
* the ICU data the runtime happens to carry, and a message a test pins must not differ between
|
|
60
|
+
* a developer's machine and a CI image.
|
|
61
|
+
*/
|
|
62
|
+
const formatTokenCount = (tokens: number): string =>
|
|
63
|
+
Math.trunc(tokens).toString().replace(/\B(?=(\d{3})+(?!\d))/g, ",");
|
|
64
|
+
|
|
29
65
|
export interface WorkflowBudgetPolicy {
|
|
30
66
|
/** Children admitted concurrently under one root. */
|
|
31
67
|
readonly maxConcurrentChildren: number;
|
|
@@ -172,7 +208,10 @@ export type WorkflowDenial =
|
|
|
172
208
|
* therefore says which ceiling fired AND that no provider was contacted, because that is the
|
|
173
209
|
* first thing an operator needs and the only place left to put it.
|
|
174
210
|
*/
|
|
175
|
-
export function workflowDenialSummary(
|
|
211
|
+
export function workflowDenialSummary(
|
|
212
|
+
reason: WorkflowDenial,
|
|
213
|
+
spend?: WorkflowSpendDenialDetail,
|
|
214
|
+
): { code: string; message: string } {
|
|
176
215
|
switch (reason) {
|
|
177
216
|
case "workflow-sends-exhausted":
|
|
178
217
|
return {
|
|
@@ -197,8 +236,21 @@ export function workflowDenialSummary(reason: WorkflowDenial): { code: string; m
|
|
|
197
236
|
case "workflow-spend-exhausted":
|
|
198
237
|
return {
|
|
199
238
|
code: "workflow_spend_exhausted",
|
|
200
|
-
|
|
201
|
-
|
|
239
|
+
// With the denial in hand the sentence names the ceiling that fired and its number,
|
|
240
|
+
// because the alternative is an operator who can see that something refused and has
|
|
241
|
+
// no way to find out what. Without one -- a caller that knows only the reason -- the
|
|
242
|
+
// original sentence is kept unchanged.
|
|
243
|
+
message: spend
|
|
244
|
+
? "This proxy refused the request locally: the configured " + SPEND_SCOPE_LABEL[spend.scope]
|
|
245
|
+
+ " token ceiling of " + formatTokenCount(spend.limit) + " is spent"
|
|
246
|
+
+ (spend.projected !== undefined
|
|
247
|
+
? " (this send would have taken it to " + formatTokenCount(spend.projected) + ")"
|
|
248
|
+
: "")
|
|
249
|
+
+ ", so no provider was contacted. Spend is durable, so it does not roll forward"
|
|
250
|
+
+ " with the send window: raise or remove spend." + spend.scope
|
|
251
|
+
+ ".maxTokens in config.json to grant more."
|
|
252
|
+
: "This proxy refused the request locally: the task reached a configured token"
|
|
253
|
+
+ " ceiling, so no provider was contacted.",
|
|
202
254
|
};
|
|
203
255
|
case "workflow-tracking-exhausted":
|
|
204
256
|
return {
|
|
@@ -241,6 +293,10 @@ export interface WorkflowBudgetEvent {
|
|
|
241
293
|
readonly rootId: string;
|
|
242
294
|
/** The ceiling that fired. Present for `refused`, absent for `cleared`. */
|
|
243
295
|
readonly reason?: WorkflowDenial;
|
|
296
|
+
/** Which token scope refused, on a spend denial. Absent on every count denial. */
|
|
297
|
+
readonly spendScope?: SpendScope;
|
|
298
|
+
/** That scope's ceiling, so the event is readable without the config open beside it. */
|
|
299
|
+
readonly spendLimit?: number;
|
|
244
300
|
/** Windowed sends at the moment of the event. */
|
|
245
301
|
readonly sends: number;
|
|
246
302
|
/** Windowed distinct children at the moment of the event. */
|
|
@@ -289,6 +345,7 @@ export function recordWorkflowRefusalEvent(
|
|
|
289
345
|
rootId: string | undefined,
|
|
290
346
|
reason: WorkflowDenial,
|
|
291
347
|
now: number = Date.now(),
|
|
348
|
+
spend?: WorkflowSpendDenialDetail,
|
|
292
349
|
): void {
|
|
293
350
|
if (!rootId) return;
|
|
294
351
|
const state = roots.get(rootId);
|
|
@@ -297,6 +354,7 @@ export function recordWorkflowRefusalEvent(
|
|
|
297
354
|
kind: "refused",
|
|
298
355
|
rootId,
|
|
299
356
|
reason,
|
|
357
|
+
...(spend ? { spendScope: spend.scope, spendLimit: spend.limit } : {}),
|
|
300
358
|
sends: state ? windowedSends(state, now) : 0,
|
|
301
359
|
children: state ? windowedChildren(state, now) : 0,
|
|
302
360
|
});
|
|
@@ -323,6 +381,10 @@ export type WorkflowDecision =
|
|
|
323
381
|
rootId: string;
|
|
324
382
|
/** Which spend scope refused, when the denial came from the token ledger. */
|
|
325
383
|
spendScope?: SpendScope;
|
|
384
|
+
/** That scope's configured ceiling, so a caller can say what it was. */
|
|
385
|
+
spendLimit?: number;
|
|
386
|
+
/** Tokens the refused reservation would have taken the scope to, where known. */
|
|
387
|
+
spendProjected?: number;
|
|
326
388
|
};
|
|
327
389
|
|
|
328
390
|
/**
|
|
@@ -424,24 +486,40 @@ export function admitWorkflowTurn(
|
|
|
424
486
|
): WorkflowDecision | undefined {
|
|
425
487
|
if (!rootId) return undefined;
|
|
426
488
|
// An explicit ledger is consulted even without a spend request, so root eviction can
|
|
427
|
-
// still see spend-exhausted entries.
|
|
428
|
-
|
|
489
|
+
// still see spend-exhausted entries. The shared one is resolved whenever a ceiling is
|
|
490
|
+
// CONFIGURED, which is what lets admission refuse an already-spent scope before a body is
|
|
491
|
+
// parsed. An install that configured nothing resolves no ledger, opens no journal, and runs
|
|
492
|
+
// this function exactly as it did before -- the unconfigured path has to stay byte-identical
|
|
493
|
+
// because the ledger is on and journalling by default.
|
|
494
|
+
const ledger = spendLedger ?? (spend || spendCeilingsConfigured() ? sharedSpendLedger() : undefined);
|
|
429
495
|
let state = roots.get(rootId);
|
|
430
496
|
// Every refusal below goes on the record through this one seam. Recording at each return
|
|
431
497
|
// site instead of at the HTTP caller is what makes the record complete: the spend denials
|
|
432
498
|
// are decided inside the ledger branch and never surface as a distinct reason to the caller
|
|
433
499
|
// that formats the response.
|
|
434
|
-
const refuse = (reason: WorkflowDenial,
|
|
500
|
+
const refuse = (reason: WorkflowDenial, denial?: WorkflowSpendDenialDetail): WorkflowDecision => {
|
|
435
501
|
const current = roots.get(rootId);
|
|
436
502
|
recordBudgetEvent({
|
|
437
503
|
at: now,
|
|
438
504
|
kind: "refused",
|
|
439
505
|
rootId,
|
|
440
506
|
reason,
|
|
507
|
+
...(denial ? { spendScope: denial.scope, spendLimit: denial.limit } : {}),
|
|
441
508
|
sends: current ? windowedSends(current, now) : 0,
|
|
442
509
|
children: current ? windowedChildren(current, now) : 0,
|
|
443
510
|
});
|
|
444
|
-
return {
|
|
511
|
+
return {
|
|
512
|
+
admitted: false,
|
|
513
|
+
reason,
|
|
514
|
+
rootId,
|
|
515
|
+
...(denial
|
|
516
|
+
? {
|
|
517
|
+
spendScope: denial.scope,
|
|
518
|
+
spendLimit: denial.limit,
|
|
519
|
+
...(denial.projected !== undefined ? { spendProjected: denial.projected } : {}),
|
|
520
|
+
}
|
|
521
|
+
: {}),
|
|
522
|
+
};
|
|
445
523
|
};
|
|
446
524
|
if (!state) {
|
|
447
525
|
if (roots.size >= policy.maxTrackedRoots && !evictOneRoot(policy, ledger, now)) {
|
|
@@ -469,6 +547,26 @@ export function admitWorkflowTurn(
|
|
|
469
547
|
return refuse("workflow-concurrency-exhausted");
|
|
470
548
|
}
|
|
471
549
|
|
|
550
|
+
// The counts are checked first and the token ceiling second, and the order is deliberate
|
|
551
|
+
// rather than emergent. A count check reads two integers this process already holds; a token
|
|
552
|
+
// check may have to build the ledger and replay its journal. Checking the cheap bound first
|
|
553
|
+
// means the expensive one is never reached for a request the cheap one already refused.
|
|
554
|
+
//
|
|
555
|
+
// The two therefore CAN disagree, and the intersection is what is enforced: a request passes
|
|
556
|
+
// only when every count cap and every token ceiling admits it. A token denial happens before
|
|
557
|
+
// any count is charged, and a count denial happens before any reservation is booked, so
|
|
558
|
+
// neither leaves the other's accounting to unwind. Whichever refuses first is reported as
|
|
559
|
+
// itself -- one refusal is never relabelled as the other, because "sends exhausted" and
|
|
560
|
+
// "spend exhausted" send an operator to two different remedies.
|
|
561
|
+
//
|
|
562
|
+
// A scope whose ceiling is ALREADY spent is refused here rather than at the reservation. The
|
|
563
|
+
// reservation needs a token count, which is not known until the body is parsed and a route
|
|
564
|
+
// resolved; an exhausted scope needs neither and is the cheapest refusal available.
|
|
565
|
+
if (!spend && ledger) {
|
|
566
|
+
const reached = spentRootCeiling(rootId, ledger);
|
|
567
|
+
if (reached) return refuse("workflow-spend-exhausted", reached);
|
|
568
|
+
}
|
|
569
|
+
|
|
472
570
|
if (spend && ledger) {
|
|
473
571
|
const decision = ledger.reserve({
|
|
474
572
|
sendId: spend.sendId,
|
|
@@ -491,7 +589,9 @@ export function admitWorkflowTurn(
|
|
|
491
589
|
: "workflow-spend-exhausted";
|
|
492
590
|
return refuse(
|
|
493
591
|
reason,
|
|
494
|
-
denial.reason === "spend-limit-exceeded"
|
|
592
|
+
denial.reason === "spend-limit-exceeded"
|
|
593
|
+
? { scope: denial.scope, limit: denial.limit, projected: denial.projected }
|
|
594
|
+
: undefined,
|
|
495
595
|
);
|
|
496
596
|
}
|
|
497
597
|
}
|
|
@@ -600,6 +700,43 @@ export function workflowSendCeilingReached(
|
|
|
600
700
|
return state !== undefined && windowedSends(state, now) >= policy.maxPhysicalSends;
|
|
601
701
|
}
|
|
602
702
|
|
|
703
|
+
/**
|
|
704
|
+
* The root scope's ceiling when that scope is already spent, or undefined.
|
|
705
|
+
*
|
|
706
|
+
* Root only: identity and pool are not known until routing has picked an account, so those two
|
|
707
|
+
* refuse at the reservation itself. `exhausted` sums settled spend, open reservations and
|
|
708
|
+
* unresolved spend, which is the same total the reservation compares, so this answers the same
|
|
709
|
+
* question the reservation would -- just without needing the request's token count.
|
|
710
|
+
*/
|
|
711
|
+
function spentRootCeiling(
|
|
712
|
+
rootId: string,
|
|
713
|
+
ledger: SpendReservationLedger,
|
|
714
|
+
): WorkflowSpendDenialDetail | undefined {
|
|
715
|
+
const limit = ledger.policy.root.maxTokens;
|
|
716
|
+
if (limit === undefined) return undefined;
|
|
717
|
+
return ledger.exhausted("root", rootId) ? { scope: "root", limit } : undefined;
|
|
718
|
+
}
|
|
719
|
+
|
|
720
|
+
/**
|
|
721
|
+
* The token ceiling a root has already spent, or undefined when it has room or has none.
|
|
722
|
+
*
|
|
723
|
+
* The count-side twin of {@link workflowSendCeilingReached}, and the responses path calls both
|
|
724
|
+
* at the same seam for the same reason: a refusal decided before dispatch can be reported as
|
|
725
|
+
* ITSELF -- a named ceiling, a synthetic log row, a machine-readable header -- instead of
|
|
726
|
+
* surfacing later as a generic send-budget error from whichever leg happened to run out first.
|
|
727
|
+
*
|
|
728
|
+
* Returns undefined when no ceiling is configured, without resolving a ledger, so an install
|
|
729
|
+
* that never opted in neither pays for this check nor opens a journal because of it.
|
|
730
|
+
*/
|
|
731
|
+
export function workflowSpendCeilingReached(
|
|
732
|
+
rootId: string | undefined,
|
|
733
|
+
spendLedger?: SpendReservationLedger,
|
|
734
|
+
): WorkflowSpendDenialDetail | undefined {
|
|
735
|
+
if (!rootId) return undefined;
|
|
736
|
+
const ledger = spendLedger ?? (spendCeilingsConfigured() ? sharedSpendLedger() : undefined);
|
|
737
|
+
return ledger ? spentRootCeiling(rootId, ledger) : undefined;
|
|
738
|
+
}
|
|
739
|
+
|
|
603
740
|
export interface WorkflowBudgetSnapshot {
|
|
604
741
|
active: number;
|
|
605
742
|
/** Sends inside the window. This is the number the ceiling compares. */
|
|
@@ -13,6 +13,38 @@
|
|
|
13
13
|
import { getCachedProviderAccountQuota, hasPassiveAccountQuota } from "../providers/quota";
|
|
14
14
|
import { getKiroAccountExhaustion } from "../providers/kiro-usage";
|
|
15
15
|
|
|
16
|
+
/** Antigravity hosts Gemini and Claude windows on one account; ranking must not mix them. */
|
|
17
|
+
export type QuotaModelFamily = "gem" | "cla";
|
|
18
|
+
|
|
19
|
+
export function classifyModelFamilyForQuota(
|
|
20
|
+
provider: string,
|
|
21
|
+
modelId?: string | null,
|
|
22
|
+
): QuotaModelFamily | undefined {
|
|
23
|
+
if (provider !== "google-antigravity" || typeof modelId !== "string" || !modelId.trim()) {
|
|
24
|
+
return undefined;
|
|
25
|
+
}
|
|
26
|
+
const id = modelId.toLowerCase();
|
|
27
|
+
// Gemma is not Gemini: a substring/prefix match would poison Gemini ranking.
|
|
28
|
+
if (/(?:^|[^a-z])gemma(?:[^a-z]|$)/.test(id)) return undefined;
|
|
29
|
+
// Catalog ids are gemini-*, never a bare gem- token. Window labels still match Gem via
|
|
30
|
+
// windowMatchesFamily; this classifier is only for request model ids.
|
|
31
|
+
if (/(?:^|[^a-z])gemini(?:[^a-z]|$)/.test(id)) return "gem";
|
|
32
|
+
if (
|
|
33
|
+
/(?:^|[^a-z])claude(?:[^a-z]|$)/.test(id)
|
|
34
|
+
|| /(?:^|[^a-z])opus(?:[^a-z]|$)/.test(id)
|
|
35
|
+
|| /(?:^|[^a-z])sonnet(?:[^a-z]|$)/.test(id)
|
|
36
|
+
|| /(?:^|[^a-z])haiku(?:[^a-z]|$)/.test(id)
|
|
37
|
+
|| /(?:^|[^a-z])gpt[-_]oss(?:[^a-z]|$)/.test(id)
|
|
38
|
+
) return "cla";
|
|
39
|
+
return undefined;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function windowMatchesFamily(label: string, family: QuotaModelFamily): boolean {
|
|
43
|
+
const token = label.trim().split(/[\s(/]+/)[0] ?? "";
|
|
44
|
+
if (family === "gem") return /^gem(?:ini)?$/i.test(token);
|
|
45
|
+
return /^cla(?:ude)?$/i.test(token);
|
|
46
|
+
}
|
|
47
|
+
|
|
16
48
|
/** Lower sorts earlier. Unknown sits between measured-healthy and measured-empty. */
|
|
17
49
|
const RANK_HEALTHY = 0;
|
|
18
50
|
const RANK_UNKNOWN = 1;
|
|
@@ -48,15 +80,24 @@ const PASSIVE_HEADROOM_MAX_AGE_MS = 60 * 60_000;
|
|
|
48
80
|
/**
|
|
49
81
|
* Remaining headroom across every window the provider reports.
|
|
50
82
|
*
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
function headroomOf(provider: string, accountId: string): number | null {
|
|
83
|
+
* The minimum wins: an account at 5% of its five-hour window is unusable right now even if
|
|
84
|
+
* its monthly allowance is barely touched.
|
|
85
|
+
*/
|
|
86
|
+
function headroomOf(provider: string, accountId: string, requestedModelId?: string | null): number | null {
|
|
55
87
|
const quota = getCachedProviderAccountQuota(provider, accountId);
|
|
56
88
|
if (!quota) return null;
|
|
57
89
|
// Null, not a low rank: this must reproduce "no evidence" so a stale roster degrades to
|
|
58
90
|
// the unranked ring rather than to a differently wrong answer.
|
|
59
91
|
if (hasPassiveAccountQuota(provider) && Date.now() - quota.updatedAt > PASSIVE_HEADROOM_MAX_AGE_MS) return null;
|
|
92
|
+
const family = classifyModelFamilyForQuota(provider, requestedModelId);
|
|
93
|
+
if (family) {
|
|
94
|
+
const percents = (quota.customWindows ?? [])
|
|
95
|
+
.filter(window => windowMatchesFamily(window.label, family))
|
|
96
|
+
.map(window => window.percent)
|
|
97
|
+
.filter((value): value is number => typeof value === "number");
|
|
98
|
+
if (percents.length === 0) return null;
|
|
99
|
+
return 100 - Math.max(...percents);
|
|
100
|
+
}
|
|
60
101
|
const percents = [
|
|
61
102
|
quota.fiveHourPercent,
|
|
62
103
|
quota.weeklyPercent,
|
|
@@ -74,15 +115,23 @@ function headroomOf(provider: string, accountId: string): number | null {
|
|
|
74
115
|
* than an ordering. Null stays null all the way out: a caller must decide what "unmeasured"
|
|
75
116
|
* means for its own rule instead of being handed a fabricated 0 or 100.
|
|
76
117
|
*/
|
|
77
|
-
export function accountHeadroomPercent(
|
|
78
|
-
|
|
118
|
+
export function accountHeadroomPercent(
|
|
119
|
+
provider: string,
|
|
120
|
+
accountId: string,
|
|
121
|
+
requestedModelId?: string | null,
|
|
122
|
+
): number | null {
|
|
123
|
+
return headroomOf(provider, accountId, requestedModelId);
|
|
79
124
|
}
|
|
80
125
|
|
|
81
126
|
/** Unknown usage is not exhaustion; Kiro's explicit overage verdict is authoritative. */
|
|
82
|
-
export function isAccountQuotaExhausted(
|
|
127
|
+
export function isAccountQuotaExhausted(
|
|
128
|
+
provider: string,
|
|
129
|
+
accountId: string,
|
|
130
|
+
requestedModelId?: string | null,
|
|
131
|
+
): boolean {
|
|
83
132
|
const exhaustion = provider === "kiro" ? getKiroAccountExhaustion(`${provider}\u0000${accountId}`) : null;
|
|
84
133
|
if (exhaustion !== null) return exhaustion.exhausted;
|
|
85
|
-
const headroom = headroomOf(provider, accountId);
|
|
134
|
+
const headroom = headroomOf(provider, accountId, requestedModelId);
|
|
86
135
|
return headroom !== null && headroom <= 0;
|
|
87
136
|
}
|
|
88
137
|
|
|
@@ -92,24 +141,28 @@ export function isAccountQuotaExhausted(provider: string, accountId: string): bo
|
|
|
92
141
|
* Returns the input untouched when no candidate has quota evidence, which keeps every
|
|
93
142
|
* provider without per-account quota on exactly the behaviour it has today.
|
|
94
143
|
*/
|
|
95
|
-
export function rankAccountsByHeadroom(
|
|
144
|
+
export function rankAccountsByHeadroom(
|
|
145
|
+
provider: string,
|
|
146
|
+
ring: readonly string[],
|
|
147
|
+
requestedModelId?: string | null,
|
|
148
|
+
): string[] {
|
|
96
149
|
if (ring.length < 2) return [...ring];
|
|
97
150
|
|
|
98
151
|
let sawEvidence = false;
|
|
99
152
|
// Same rule as hasHeadroomEvidence: a passive provider's partial roster must not rank
|
|
100
153
|
// at all. The failover path calls this directly (selectFailoverAccount), so the guard
|
|
101
154
|
// cannot live only in the pre-dispatch predicate.
|
|
102
|
-
if (hasPassiveAccountQuota(provider) && !ring.every(id => headroomOf(provider, id) !== null)) {
|
|
155
|
+
if (hasPassiveAccountQuota(provider) && !ring.every(id => headroomOf(provider, id, requestedModelId) !== null)) {
|
|
103
156
|
return [...ring];
|
|
104
157
|
}
|
|
105
158
|
const ranked: Ranked[] = ring.map((id, index) => {
|
|
106
159
|
// A provider-declared exhaustion verdict outranks the percentage: an account may sit at
|
|
107
160
|
// 100% and still be servable when overage is enabled, and the verdict knows that.
|
|
108
161
|
const exhaustion = provider === "kiro" ? getKiroAccountExhaustion(`${provider}\u0000${id}`) : null;
|
|
109
|
-
const headroom = headroomOf(provider, id);
|
|
162
|
+
const headroom = headroomOf(provider, id, requestedModelId);
|
|
110
163
|
if (exhaustion !== null || headroom !== null) sawEvidence = true;
|
|
111
164
|
|
|
112
|
-
if (isAccountQuotaExhausted(provider, id)) return { id, bucket: RANK_EXHAUSTED, headroom: 0, index };
|
|
165
|
+
if (isAccountQuotaExhausted(provider, id, requestedModelId)) return { id, bucket: RANK_EXHAUSTED, headroom: 0, index };
|
|
113
166
|
if (headroom === null) return { id, bucket: RANK_UNKNOWN, headroom: 0, index };
|
|
114
167
|
return { id, bucket: RANK_HEALTHY, headroom, index };
|
|
115
168
|
});
|
|
@@ -129,7 +182,11 @@ export function rankAccountsByHeadroom(provider: string, ring: readonly string[]
|
|
|
129
182
|
* told "ranked" when nothing was measured. Pre-dispatch selection asks this first so it
|
|
130
183
|
* can decline to act on a roster it knows nothing about.
|
|
131
184
|
*/
|
|
132
|
-
export function hasHeadroomEvidence(
|
|
185
|
+
export function hasHeadroomEvidence(
|
|
186
|
+
provider: string,
|
|
187
|
+
ids: readonly string[],
|
|
188
|
+
requestedModelId?: string | null,
|
|
189
|
+
): boolean {
|
|
133
190
|
// A PASSIVE provider needs evidence for EVERY candidate, not any one of them.
|
|
134
191
|
//
|
|
135
192
|
// A probe fills the whole roster in one pass (fetchProviderAccountQuotas), so "any"
|
|
@@ -140,10 +197,10 @@ export function hasHeadroomEvidence(provider: string, ids: readonly string[]): b
|
|
|
140
197
|
// AWAY from an unmeasured account and TOWARD the one account known to be spent, which
|
|
141
198
|
// is the exact inversion of what ranking is for.
|
|
142
199
|
if (hasPassiveAccountQuota(provider)) {
|
|
143
|
-
return ids.length > 0 && ids.every(id => headroomOf(provider, id) !== null);
|
|
200
|
+
return ids.length > 0 && ids.every(id => headroomOf(provider, id, requestedModelId) !== null);
|
|
144
201
|
}
|
|
145
202
|
return ids.some(id =>
|
|
146
|
-
headroomOf(provider, id) !== null
|
|
203
|
+
headroomOf(provider, id, requestedModelId) !== null
|
|
147
204
|
|| (provider === "kiro" && getKiroAccountExhaustion(`${provider}\u0000${id}`) !== null));
|
|
148
205
|
}
|
|
149
206
|
/**
|