@bitkyc08/opencodex 2.58.0 → 2.59.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. package/README.md +28 -10
  2. package/gui/dist/assets/index-C5IebErG.js +136 -0
  3. package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
  4. package/gui/dist/index.html +2 -2
  5. package/gui/dist/provider-icons/crusoe.svg +1 -0
  6. package/gui/dist/provider-icons/opper.svg +3 -0
  7. package/package.json +1 -1
  8. package/src/adapters/base.ts +11 -1
  9. package/src/adapters/cursor/catalog.ts +11 -0
  10. package/src/adapters/cursor/effort-map.ts +16 -2
  11. package/src/adapters/cursor/envelope-echo.ts +55 -2
  12. package/src/adapters/cursor/message-mapper.ts +3 -2
  13. package/src/adapters/cursor/protobuf-request.ts +8 -5
  14. package/src/adapters/cursor/request-builder.ts +14 -3
  15. package/src/adapters/cursor/thread-continuity.ts +105 -31
  16. package/src/adapters/cursor/tool-guidance.ts +5 -4
  17. package/src/adapters/cursor.ts +42 -1
  18. package/src/adapters/devin/cloud-direct/chat.ts +11 -2
  19. package/src/adapters/devin/cloud-direct/index.ts +7 -0
  20. package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
  21. package/src/adapters/devin.ts +75 -13
  22. package/src/adapters/google-antigravity-wire.ts +29 -2
  23. package/src/adapters/google-http.ts +8 -1
  24. package/src/adapters/google.ts +23 -4
  25. package/src/adapters/openai-chat/response-events.ts +61 -0
  26. package/src/adapters/openai-chat.ts +5 -10
  27. package/src/adapters/openai-responses/passthrough.ts +10 -1
  28. package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
  29. package/src/adapters/openai-responses/tool-schema.ts +19 -7
  30. package/src/adapters/responses-tool-schema.ts +76 -46
  31. package/src/adapters/run-turn-queue.ts +17 -4
  32. package/src/bridge/response-json.ts +1 -1
  33. package/src/bridge/sse.ts +165 -24
  34. package/src/claude/context-windows.ts +22 -0
  35. package/src/claude/outbound.ts +35 -4
  36. package/src/cli/account-api.ts +4 -3
  37. package/src/cli/account-extended.ts +22 -2
  38. package/src/cli/account-orca-import.ts +63 -0
  39. package/src/cli/account.ts +32 -4
  40. package/src/cli/capabilities.ts +40 -0
  41. package/src/cli/claude.ts +29 -1
  42. package/src/cli/codex-cli-update.ts +97 -2
  43. package/src/cli/dispatch.ts +54 -0
  44. package/src/cli/doctor.ts +197 -2
  45. package/src/cli/help.ts +4 -1
  46. package/src/cli/index.ts +88 -20
  47. package/src/cli/models-runtime.ts +33 -4
  48. package/src/cli/registry.ts +11 -1
  49. package/src/cli/runtime-api.ts +44 -0
  50. package/src/cli/start-args.ts +94 -0
  51. package/src/cli/system-command.ts +2 -0
  52. package/src/client/machine-api.ts +4 -3
  53. package/src/client/machine-listener.ts +14 -1
  54. package/src/clients/config-export/constants.ts +2 -3
  55. package/src/clients/config-export.ts +5 -5
  56. package/src/codex/account-store.ts +81 -5
  57. package/src/codex/auth-api/pool-quota-probe.ts +14 -3
  58. package/src/codex/auth-api/routes.ts +17 -2
  59. package/src/codex/auth-context.ts +16 -12
  60. package/src/codex/catalog/build-entries.ts +25 -4
  61. package/src/codex/catalog/derive-entry.ts +8 -1
  62. package/src/codex/catalog/effort.ts +10 -6
  63. package/src/codex/catalog/gather-capture.ts +1 -0
  64. package/src/codex/catalog/model-hints.ts +37 -5
  65. package/src/codex/catalog/parsing.ts +83 -5
  66. package/src/codex/catalog/reserve-warn.ts +96 -0
  67. package/src/codex/catalog/retained-sync.ts +19 -0
  68. package/src/codex/catalog/routed-gather.ts +42 -3
  69. package/src/codex/cli-installation-identity.ts +210 -0
  70. package/src/codex/cli-installation-targets.ts +158 -0
  71. package/src/codex/convergence.ts +5 -0
  72. package/src/codex/history-provider.ts +4 -1
  73. package/src/codex/history-state-open.ts +105 -0
  74. package/src/codex/inject/config-toml.ts +44 -2
  75. package/src/codex/inject.ts +3 -2
  76. package/src/codex/lineage.ts +83 -32
  77. package/src/codex/loopback-target.ts +31 -0
  78. package/src/codex/main-account-hard-lock.ts +2 -1
  79. package/src/codex/main-account.ts +10 -3
  80. package/src/codex/main-device-reauth.ts +17 -9
  81. package/src/codex/model-entitlements.ts +60 -1
  82. package/src/codex/observed-model-denials.ts +137 -0
  83. package/src/codex/orca-auth-source.ts +94 -0
  84. package/src/codex/orca-import.ts +219 -0
  85. package/src/codex/prompt-text-probe.ts +282 -12
  86. package/src/codex/quota-401-recovery.ts +12 -0
  87. package/src/codex/quota-types.ts +65 -0
  88. package/src/codex/quota.ts +24 -19
  89. package/src/codex/routing/cooldown-math.ts +8 -47
  90. package/src/codex/routing/pin-drain.ts +57 -0
  91. package/src/codex/routing.ts +13 -15
  92. package/src/codex/subagent-model-fallback.ts +94 -0
  93. package/src/codex/windows-installation-files.ts +224 -0
  94. package/src/combos/failover.ts +122 -5
  95. package/src/config/diagnostics.ts +21 -0
  96. package/src/config/load-degrade.ts +15 -0
  97. package/src/config/pending-teardown.ts +8 -0
  98. package/src/config/process-state.ts +36 -3
  99. package/src/config/provider-relative-send-path.ts +16 -0
  100. package/src/config/proxy-env.ts +23 -5
  101. package/src/config/schema/config-schema.ts +21 -0
  102. package/src/config/schema/leaf-validators.ts +64 -17
  103. package/src/generated/compatibility-version.json +235 -163
  104. package/src/generated/model-metadata.ts +1 -1
  105. package/src/lib/bounded-body.ts +4 -2
  106. package/src/lib/destination-policy.ts +48 -6
  107. package/src/lib/errors.ts +3 -15
  108. package/src/lib/local-destinations.ts +32 -5
  109. package/src/lib/provider-outbound.ts +3 -3
  110. package/src/lib/proxy-env.ts +70 -3
  111. package/src/lib/request-execution-budget.ts +11 -3
  112. package/src/lib/response-body-inactivity.ts +193 -0
  113. package/src/lib/retry-delay.ts +69 -0
  114. package/src/lib/socks5-fetch.ts +631 -0
  115. package/src/lib/spend-reservation-ledger.ts +115 -9
  116. package/src/lib/workflow-budget.ts +145 -8
  117. package/src/oauth/account-quota-rank.ts +72 -15
  118. package/src/oauth/generic-account-failover.ts +40 -27
  119. package/src/oauth/orcarouter.ts +15 -2
  120. package/src/oauth/store.ts +8 -0
  121. package/src/providers/codex-capacity.ts +9 -0
  122. package/src/providers/devin-provider-merge-migration.ts +33 -12
  123. package/src/providers/free-directory.ts +20 -2
  124. package/src/providers/key-failover.ts +261 -7
  125. package/src/providers/model-rename-migration.ts +1 -0
  126. package/src/providers/openai-sidecar.ts +4 -0
  127. package/src/providers/opencode-go-transport.ts +14 -5
  128. package/src/providers/quota/report-cache.ts +3 -0
  129. package/src/providers/registry/entries-extended.ts +96 -0
  130. package/src/providers/registry/model-seeds.ts +78 -21
  131. package/src/responses/apply-patch-envelope.ts +44 -11
  132. package/src/responses/bridge-search-replay-cache.ts +152 -0
  133. package/src/responses/code-mode-helper-compat.ts +26 -16
  134. package/src/responses/custom-tool-compat.ts +1 -1
  135. package/src/responses/hosted-tool-policy.ts +85 -2
  136. package/src/responses/schema.ts +9 -2
  137. package/src/server/auth-cors.ts +26 -0
  138. package/src/server/chat-completions.ts +9 -4
  139. package/src/server/chat-native-sse.ts +26 -9
  140. package/src/server/chat-native.ts +10 -4
  141. package/src/server/claude-messages.ts +24 -2
  142. package/src/server/gui-static.ts +36 -2
  143. package/src/server/inbound-body-admission.ts +187 -0
  144. package/src/server/index.ts +15 -19
  145. package/src/server/management/api-access.ts +3 -4
  146. package/src/server/management/config-routes.ts +31 -6
  147. package/src/server/management/provider-capability-config.ts +35 -7
  148. package/src/server/management/provider-routes.ts +70 -18
  149. package/src/server/proxy-liveness.ts +97 -2
  150. package/src/server/relay.ts +17 -24
  151. package/src/server/request-log.ts +25 -1
  152. package/src/server/responses/adapter-continuation.ts +71 -27
  153. package/src/server/responses/adapter-delivery.ts +39 -8
  154. package/src/server/responses/adapter-dispatch.ts +52 -24
  155. package/src/server/responses/compact.ts +60 -11
  156. package/src/server/responses/core-codex-account.ts +83 -22
  157. package/src/server/responses/core-normalize.ts +12 -5
  158. package/src/server/responses/fetch-helpers.ts +68 -2
  159. package/src/server/responses/passthrough-delivery.ts +10 -1
  160. package/src/server/responses/passthrough-dispatch.ts +113 -48
  161. package/src/server/responses/passthrough-execution.ts +11 -1
  162. package/src/server/responses/request-prepare.ts +29 -0
  163. package/src/server/responses/request-send-budget.ts +84 -7
  164. package/src/server/responses/request-sidecar-auth.ts +16 -8
  165. package/src/server/responses/request-spend.ts +38 -9
  166. package/src/server/responses/request-transport.ts +13 -10
  167. package/src/server/responses/run-turn-execution.ts +20 -5
  168. package/src/server/responses/sidecar-execution.ts +2 -0
  169. package/src/server/responses/ws-upstream.ts +2 -1
  170. package/src/server/responses-custom-tool-repair.ts +2 -2
  171. package/src/server/sse-frame-buffer.ts +12 -10
  172. package/src/server/sse-payload-rewrite.ts +36 -9
  173. package/src/server/system-env-shell.ts +5 -1
  174. package/src/server/system-env.ts +7 -1
  175. package/src/server/workflow-refusal.ts +56 -2
  176. package/src/service/cli.ts +16 -6
  177. package/src/service/guards.ts +10 -0
  178. package/src/service/health.ts +43 -0
  179. package/src/service/state.ts +7 -2
  180. package/src/types/accounts.ts +4 -0
  181. package/src/types/config.ts +100 -3
  182. package/src/types/provider.ts +19 -0
  183. package/src/types/request.ts +7 -1
  184. package/src/types/wire.ts +9 -1
  185. package/src/usage/expected-prices.ts +28 -0
  186. package/src/usage/log.ts +87 -4
  187. package/src/web-search/passthrough-bridge.ts +39 -5
  188. package/gui/dist/assets/index-BbrHOIY0.js +0 -128
@@ -60,6 +60,9 @@ import { dirname, join } from "node:path";
60
60
  // Definition-site import, not the ../config barrel -- same reasoning as
61
61
  // src/quota/reset-seen-store.ts: the barrel pulls ~154 modules into a hot path.
62
62
  import { getConfigDir } from "../config/paths";
63
+ // Type-only, so it is erased before this module has a runtime import graph at all. The
64
+ // config SHAPE is what this file needs; the config loader is what the note above keeps out.
65
+ import type { OcxSpendConfig, OcxSpendScopeConfig } from "../types/config";
63
66
  import { assertNotRealHomeUnderTest } from "./test-home-guard";
64
67
  // Windows chmod does not remove inherited ACEs; this is the repository's icacls path.
65
68
  import { hardenSecretPath } from "./windows-secret-acl";
@@ -147,6 +150,24 @@ export interface SpendReservationRequest {
147
150
  /** Enforceable output ceiling -- max_output_tokens or the model's documented cap. */
148
151
  readonly outputCeilingTokens: number;
149
152
  readonly at?: number;
153
+ /**
154
+ * This send has ALREADY left for upstream and is being recorded rather than admitted.
155
+ *
156
+ * Some transports report their physical sends after the fact -- the passthrough ladder
157
+ * reports through `onSendsConsumed`, and an adapter's inner retries are counted when they
158
+ * finish. For those, a ceiling cannot refuse anything: the tokens are spent. Refusing to
159
+ * BOOK them is the worse answer, and it is not hypothetical -- it is a fixpoint. The send
160
+ * that would cross the ceiling gets dropped from the total, the total stays just under the
161
+ * limit forever, the scope never reads as exhausted, and the ceiling never fires again for
162
+ * any request. So a recorded send skips the limit check and takes the scope over its
163
+ * ceiling, which is what makes the NEXT request refusable.
164
+ *
165
+ * It skips the durability refusal for the same reason: a journal that could not be written
166
+ * is a reason to report degradation, never a reason to forget spend that really happened.
167
+ * Identity, capacity and journal-integrity denials still apply -- those say the ledger
168
+ * cannot account for the send at all, which no flag here can change.
169
+ */
170
+ readonly alreadySent?: boolean;
150
171
  }
151
172
 
152
173
  /**
@@ -468,6 +489,20 @@ export interface SpendReservationLedger {
468
489
  prune(now?: number): void;
469
490
  /** Whether this send id is already known, and therefore refused. */
470
491
  knows(sendId: string): boolean;
492
+ /**
493
+ * Replace the live policy.
494
+ *
495
+ * Every figure already accounted survives: raising, lowering or clearing a ceiling changes
496
+ * what is REFUSED from here on and never what was spent. Rebuilding the ledger instead
497
+ * would replay the journal into a second set of maps while the first still holds this
498
+ * process's open reservations, and the two would then disagree about what is in flight.
499
+ */
500
+ reconfigure(next: SpendReservationPolicy): void;
501
+ /**
502
+ * The policy in force. A live read, not a copy: a caller that formats a refusal has to name
503
+ * the ceiling this ledger would enforce on the NEXT request, not the one it was built with.
504
+ */
505
+ readonly policy: SpendReservationPolicy;
471
506
  /** Journal writes that failed; a nonzero count means durability is degraded. */
472
507
  readonly persistFailures: number;
473
508
  /**
@@ -495,13 +530,17 @@ export function createSpendReservationLedger(options: {
495
530
  */
496
531
  readonly salt?: string;
497
532
  } = {}): SpendReservationLedger {
498
- const policy = options.policy ?? DEFAULT_SPEND_RESERVATION_POLICY;
533
+ // Mutable because the ceilings are operator configuration, and configuration is reloadable.
534
+ // The three bounds below are read through functions for the same reason: a value captured
535
+ // at construction would answer for the policy this ledger was BUILT with, and an operator
536
+ // who raised a bound would keep the old one until the process restarted.
537
+ let policy = options.policy ?? DEFAULT_SPEND_RESERVATION_POLICY;
499
538
  const journal = options.journal;
500
539
  const now = options.now ?? (() => Date.now());
501
540
  const salt = options.salt ?? "";
502
- const maxTrackedScopes = policy.maxTrackedScopes ?? DEFAULT_MAX_TRACKED_SCOPES;
503
- const maxTrackedSends = policy.maxTrackedSends ?? DEFAULT_MAX_TRACKED_SENDS;
504
- const compactAfterRecords = policy.compactAfterRecords ?? DEFAULT_COMPACT_AFTER_RECORDS;
541
+ const maxTrackedScopes = (): number => policy.maxTrackedScopes ?? DEFAULT_MAX_TRACKED_SCOPES;
542
+ const maxTrackedSends = (): number => policy.maxTrackedSends ?? DEFAULT_MAX_TRACKED_SENDS;
543
+ const compactAfterRecords = (): number => policy.compactAfterRecords ?? DEFAULT_COMPACT_AFTER_RECORDS;
505
544
  const scopes = new Map<string, ScopeState>();
506
545
  const reservations = new Map<string, Reservation>();
507
546
  let persistFailures = 0;
@@ -753,7 +792,7 @@ export function createSpendReservationLedger(options: {
753
792
  */
754
793
  const compact = (at: number): void => {
755
794
  const rewrite = journal?.rewrite;
756
- if (!journal || !rewrite || recordsOnDisk < compactAfterRecords) return;
795
+ if (!journal || !rewrite || recordsOnDisk < compactAfterRecords()) return;
757
796
  const checkpoint: JournalRecord = {
758
797
  v: 1,
759
798
  kind: "checkpoint",
@@ -791,12 +830,12 @@ export function createSpendReservationLedger(options: {
791
830
  const makeRoom = (refs: readonly ScopeRef[], at: number): SpendDenial | undefined => {
792
831
  evictSends(at, false);
793
832
  evictScopes(at, false);
794
- while (reservations.size >= maxTrackedSends) {
833
+ while (reservations.size >= maxTrackedSends()) {
795
834
  if (evictSends(at, true) === 0) return { reason: "tracking-capacity-exhausted" };
796
835
  }
797
836
  let fresh = 0;
798
837
  for (const ref of refs) if (!scopes.has(scopeKey(ref.scope, ref.alias))) fresh += 1;
799
- while (scopes.size + fresh > maxTrackedScopes) {
838
+ while (scopes.size + fresh > maxTrackedScopes()) {
800
839
  if (evictScopes(at, true) === 0) {
801
840
  return { reason: "tracking-capacity-exhausted", scope: refs[0]?.scope };
802
841
  }
@@ -808,6 +847,7 @@ export function createSpendReservationLedger(options: {
808
847
  get persistFailures() { return persistFailures; },
809
848
  get corruptRecords() { return corruptRecords; },
810
849
  get degraded() { return persistFailures > 0 || corruptRecords > 0; },
850
+ get policy() { return policy; },
811
851
 
812
852
  reserve(request: SpendReservationRequest): SpendReservationDecision {
813
853
  const tokens = sanitizeTokens(request.inputTokens) + sanitizeTokens(request.outputCeilingTokens);
@@ -834,7 +874,9 @@ export function createSpendReservationLedger(options: {
834
874
  // reservation booked on the scopes that would have passed. Reading state without
835
875
  // creating it matters here -- a denied request must not leave a tracked scope behind.
836
876
  for (const ref of refs) {
837
- const limit = limitFor(ref.scope);
877
+ // A recorded send has no limit to fail: it already happened, and the point of booking
878
+ // it is to let the total go OVER the ceiling so the next request can be refused.
879
+ const limit = request.alreadySent === true ? undefined : limitFor(ref.scope);
838
880
  if (limit === undefined) continue;
839
881
  const state = scopes.get(scopeKey(ref.scope, ref.alias));
840
882
  const projected = (state ? state.settled + state.reserved + state.unresolved : 0) + tokens;
@@ -853,7 +895,7 @@ export function createSpendReservationLedger(options: {
853
895
  // limit a failed write refuses the request rather than admitting one that a restart
854
896
  // would forget -- which is exactly the disk-full and permission case durability is for.
855
897
  const durable = append({ v: 1, kind: "reserve", send, targets: refs, tokens, at });
856
- if (!durable && enforced) {
898
+ if (!durable && enforced && request.alreadySent !== true) {
857
899
  return { reserved: false, denial: { reason: "reserve-not-durable", sendId: request.sendId } };
858
900
  }
859
901
  applyReserve(send, refs, tokens, at);
@@ -931,10 +973,72 @@ export function createSpendReservationLedger(options: {
931
973
  evictSends(at, false);
932
974
  evictScopes(at, false);
933
975
  },
976
+
977
+ reconfigure(next: SpendReservationPolicy): void {
978
+ policy = next;
979
+ },
934
980
  };
935
981
  }
936
982
 
937
983
  let sharedLedger: SpendReservationLedger | undefined;
984
+ /**
985
+ * The operator policy in effect. Held beside the ledger rather than inside it because the
986
+ * ledger is built lazily: a configured ceiling has to be remembered from startup until the
987
+ * first request that actually reserves, and an install that configures nothing must still
988
+ * open no journal.
989
+ */
990
+ let sharedPolicy: SpendReservationPolicy = DEFAULT_SPEND_RESERVATION_POLICY;
991
+
992
+ /** Whether any scope carries a ceiling -- that is, whether anything at all can be refused. */
993
+ export function spendCeilingsConfigured(policy: SpendReservationPolicy = sharedPolicy): boolean {
994
+ return policy.root.maxTokens !== undefined
995
+ || policy.identity.maxTokens !== undefined
996
+ || policy.pool.maxTokens !== undefined;
997
+ }
998
+
999
+ /** The policy the process-wide ledger enforces right now. */
1000
+ export function sharedSpendPolicy(): SpendReservationPolicy {
1001
+ return sharedPolicy;
1002
+ }
1003
+
1004
+ const spendScopeLimitFromConfig = (scope: OcxSpendScopeConfig | undefined): SpendScopeLimit =>
1005
+ scope?.maxTokens !== undefined && Number.isFinite(scope.maxTokens) && scope.maxTokens > 0
1006
+ ? { maxTokens: Math.trunc(scope.maxTokens) }
1007
+ : {};
1008
+
1009
+ /**
1010
+ * The ledger policy an operator's `spend` section asks for.
1011
+ *
1012
+ * An absent section, an empty one, and one whose every ceiling is absent all produce the
1013
+ * unconfigured default: observe-only accounting that refuses nothing. That equivalence is the
1014
+ * load-bearing part. This ledger is on and journaling by default, so shipping a default
1015
+ * ceiling would start refusing real traffic on the first upgrade that ran this code, against
1016
+ * a number nobody chose. There is deliberately no default figure here at all.
1017
+ */
1018
+ export function spendPolicyFromConfig(spend: OcxSpendConfig | undefined): SpendReservationPolicy {
1019
+ const retentionDays = spend?.retentionDays;
1020
+ return {
1021
+ root: spendScopeLimitFromConfig(spend?.root),
1022
+ identity: spendScopeLimitFromConfig(spend?.identity),
1023
+ pool: spendScopeLimitFromConfig(spend?.pool),
1024
+ retentionMs: retentionDays !== undefined && Number.isFinite(retentionDays) && retentionDays > 0
1025
+ ? Math.trunc(retentionDays) * 24 * 60 * 60_000
1026
+ : DEFAULT_SPEND_RESERVATION_POLICY.retentionMs,
1027
+ };
1028
+ }
1029
+
1030
+ /**
1031
+ * Apply an operator policy to the process-wide ledger.
1032
+ *
1033
+ * Startup calls this with the loaded config, and a reload may call it again: the ledger keeps
1034
+ * every figure it has already accounted, so changing a ceiling changes what is refused from
1035
+ * here on and never what was spent. It does not CREATE the ledger -- an install that
1036
+ * configures no ceiling must not open a journal merely because the server started.
1037
+ */
1038
+ export function configureSharedSpendLedger(policy: SpendReservationPolicy): void {
1039
+ sharedPolicy = policy;
1040
+ sharedLedger?.reconfigure(policy);
1041
+ }
938
1042
 
939
1043
  /**
940
1044
  * Process-wide ledger backed by the journal under OPENCODEX_HOME. Created lazily so
@@ -947,6 +1051,7 @@ export function sharedSpendLedger(): SpendReservationLedger {
947
1051
  sharedLedger = createSpendReservationLedger({
948
1052
  journal: createFileSpendJournal(join(home, SPEND_LEDGER_JOURNAL_FILENAME)),
949
1053
  salt: loadOrCreateSpendLedgerSalt(join(home, SPEND_LEDGER_SALT_FILENAME)),
1054
+ policy: sharedPolicy,
950
1055
  });
951
1056
  }
952
1057
  return sharedLedger;
@@ -955,4 +1060,5 @@ export function sharedSpendLedger(): SpendReservationLedger {
955
1060
  /** Test seam. Production never discards the ledger: that would reset a spent budget. */
956
1061
  export function resetSharedSpendLedgerForTest(): void {
957
1062
  sharedLedger = undefined;
1063
+ sharedPolicy = DEFAULT_SPEND_RESERVATION_POLICY;
958
1064
  }
@@ -21,11 +21,47 @@
21
21
 
22
22
  import {
23
23
  sharedSpendLedger,
24
+ spendCeilingsConfigured,
24
25
  type SpendReservationLedger,
25
26
  type SpendScope,
26
27
  type SpendUsage,
27
28
  } from "./spend-reservation-ledger";
28
29
 
30
+ /**
31
+ * What a token-ceiling refusal has to be able to say.
32
+ *
33
+ * "Budget exhausted" on its own is the failure this repository keeps re-learning: a policy
34
+ * rejection wearing another error's clothing sends an operator to look at the provider. The
35
+ * scope says WHICH ceiling fired -- one task, one account, or the whole pool -- and the limit
36
+ * is the number they would otherwise have to read the journal to recover. The scope ID is
37
+ * deliberately not here: root ids are client thread headers and identity ids are credentials,
38
+ * and the ledger's rule is that neither is written down in the clear.
39
+ */
40
+ export interface WorkflowSpendDenialDetail {
41
+ readonly scope: SpendScope;
42
+ readonly limit: number;
43
+ /** Tokens the refused reservation would have taken the scope to, where that is known. */
44
+ readonly projected?: number;
45
+ }
46
+
47
+ /** Operator-facing name for each scope. What an operator calls it, not what the type calls it. */
48
+ const SPEND_SCOPE_LABEL: Record<SpendScope, string> = {
49
+ root: "task",
50
+ identity: "account",
51
+ pool: "provider pool",
52
+ };
53
+
54
+ /**
55
+ * Thousands separators, done here rather than by `toLocaleString`.
56
+ *
57
+ * A ceiling is an eight- or nine-digit number and an unseparated one is genuinely hard to read
58
+ * against the figure beside it. `toLocaleString` would do this too, but its output depends on
59
+ * the ICU data the runtime happens to carry, and a message a test pins must not differ between
60
+ * a developer's machine and a CI image.
61
+ */
62
+ const formatTokenCount = (tokens: number): string =>
63
+ Math.trunc(tokens).toString().replace(/\B(?=(\d{3})+(?!\d))/g, ",");
64
+
29
65
  export interface WorkflowBudgetPolicy {
30
66
  /** Children admitted concurrently under one root. */
31
67
  readonly maxConcurrentChildren: number;
@@ -172,7 +208,10 @@ export type WorkflowDenial =
172
208
  * therefore says which ceiling fired AND that no provider was contacted, because that is the
173
209
  * first thing an operator needs and the only place left to put it.
174
210
  */
175
- export function workflowDenialSummary(reason: WorkflowDenial): { code: string; message: string } {
211
+ export function workflowDenialSummary(
212
+ reason: WorkflowDenial,
213
+ spend?: WorkflowSpendDenialDetail,
214
+ ): { code: string; message: string } {
176
215
  switch (reason) {
177
216
  case "workflow-sends-exhausted":
178
217
  return {
@@ -197,8 +236,21 @@ export function workflowDenialSummary(reason: WorkflowDenial): { code: string; m
197
236
  case "workflow-spend-exhausted":
198
237
  return {
199
238
  code: "workflow_spend_exhausted",
200
- message: "This proxy refused the request locally: the task reached a configured token"
201
- + " ceiling, so no provider was contacted.",
239
+ // With the denial in hand the sentence names the ceiling that fired and its number,
240
+ // because the alternative is an operator who can see that something refused and has
241
+ // no way to find out what. Without one -- a caller that knows only the reason -- the
242
+ // original sentence is kept unchanged.
243
+ message: spend
244
+ ? "This proxy refused the request locally: the configured " + SPEND_SCOPE_LABEL[spend.scope]
245
+ + " token ceiling of " + formatTokenCount(spend.limit) + " is spent"
246
+ + (spend.projected !== undefined
247
+ ? " (this send would have taken it to " + formatTokenCount(spend.projected) + ")"
248
+ : "")
249
+ + ", so no provider was contacted. Spend is durable, so it does not roll forward"
250
+ + " with the send window: raise or remove spend." + spend.scope
251
+ + ".maxTokens in config.json to grant more."
252
+ : "This proxy refused the request locally: the task reached a configured token"
253
+ + " ceiling, so no provider was contacted.",
202
254
  };
203
255
  case "workflow-tracking-exhausted":
204
256
  return {
@@ -241,6 +293,10 @@ export interface WorkflowBudgetEvent {
241
293
  readonly rootId: string;
242
294
  /** The ceiling that fired. Present for `refused`, absent for `cleared`. */
243
295
  readonly reason?: WorkflowDenial;
296
+ /** Which token scope refused, on a spend denial. Absent on every count denial. */
297
+ readonly spendScope?: SpendScope;
298
+ /** That scope's ceiling, so the event is readable without the config open beside it. */
299
+ readonly spendLimit?: number;
244
300
  /** Windowed sends at the moment of the event. */
245
301
  readonly sends: number;
246
302
  /** Windowed distinct children at the moment of the event. */
@@ -289,6 +345,7 @@ export function recordWorkflowRefusalEvent(
289
345
  rootId: string | undefined,
290
346
  reason: WorkflowDenial,
291
347
  now: number = Date.now(),
348
+ spend?: WorkflowSpendDenialDetail,
292
349
  ): void {
293
350
  if (!rootId) return;
294
351
  const state = roots.get(rootId);
@@ -297,6 +354,7 @@ export function recordWorkflowRefusalEvent(
297
354
  kind: "refused",
298
355
  rootId,
299
356
  reason,
357
+ ...(spend ? { spendScope: spend.scope, spendLimit: spend.limit } : {}),
300
358
  sends: state ? windowedSends(state, now) : 0,
301
359
  children: state ? windowedChildren(state, now) : 0,
302
360
  });
@@ -323,6 +381,10 @@ export type WorkflowDecision =
323
381
  rootId: string;
324
382
  /** Which spend scope refused, when the denial came from the token ledger. */
325
383
  spendScope?: SpendScope;
384
+ /** That scope's configured ceiling, so a caller can say what it was. */
385
+ spendLimit?: number;
386
+ /** Tokens the refused reservation would have taken the scope to, where known. */
387
+ spendProjected?: number;
326
388
  };
327
389
 
328
390
  /**
@@ -424,24 +486,40 @@ export function admitWorkflowTurn(
424
486
  ): WorkflowDecision | undefined {
425
487
  if (!rootId) return undefined;
426
488
  // An explicit ledger is consulted even without a spend request, so root eviction can
427
- // still see spend-exhausted entries. With neither, no token tracking is in play.
428
- const ledger = spendLedger ?? (spend ? sharedSpendLedger() : undefined);
489
+ // still see spend-exhausted entries. The shared one is resolved whenever a ceiling is
490
+ // CONFIGURED, which is what lets admission refuse an already-spent scope before a body is
491
+ // parsed. An install that configured nothing resolves no ledger, opens no journal, and runs
492
+ // this function exactly as it did before -- the unconfigured path has to stay byte-identical
493
+ // because the ledger is on and journalling by default.
494
+ const ledger = spendLedger ?? (spend || spendCeilingsConfigured() ? sharedSpendLedger() : undefined);
429
495
  let state = roots.get(rootId);
430
496
  // Every refusal below goes on the record through this one seam. Recording at each return
431
497
  // site instead of at the HTTP caller is what makes the record complete: the spend denials
432
498
  // are decided inside the ledger branch and never surface as a distinct reason to the caller
433
499
  // that formats the response.
434
- const refuse = (reason: WorkflowDenial, spendScope?: SpendScope): WorkflowDecision => {
500
+ const refuse = (reason: WorkflowDenial, denial?: WorkflowSpendDenialDetail): WorkflowDecision => {
435
501
  const current = roots.get(rootId);
436
502
  recordBudgetEvent({
437
503
  at: now,
438
504
  kind: "refused",
439
505
  rootId,
440
506
  reason,
507
+ ...(denial ? { spendScope: denial.scope, spendLimit: denial.limit } : {}),
441
508
  sends: current ? windowedSends(current, now) : 0,
442
509
  children: current ? windowedChildren(current, now) : 0,
443
510
  });
444
- return { admitted: false, reason, rootId, ...(spendScope ? { spendScope } : {}) };
511
+ return {
512
+ admitted: false,
513
+ reason,
514
+ rootId,
515
+ ...(denial
516
+ ? {
517
+ spendScope: denial.scope,
518
+ spendLimit: denial.limit,
519
+ ...(denial.projected !== undefined ? { spendProjected: denial.projected } : {}),
520
+ }
521
+ : {}),
522
+ };
445
523
  };
446
524
  if (!state) {
447
525
  if (roots.size >= policy.maxTrackedRoots && !evictOneRoot(policy, ledger, now)) {
@@ -469,6 +547,26 @@ export function admitWorkflowTurn(
469
547
  return refuse("workflow-concurrency-exhausted");
470
548
  }
471
549
 
550
+ // The counts are checked first and the token ceiling second, and the order is deliberate
551
+ // rather than emergent. A count check reads two integers this process already holds; a token
552
+ // check may have to build the ledger and replay its journal. Checking the cheap bound first
553
+ // means the expensive one is never reached for a request the cheap one already refused.
554
+ //
555
+ // The two therefore CAN disagree, and the intersection is what is enforced: a request passes
556
+ // only when every count cap and every token ceiling admits it. A token denial happens before
557
+ // any count is charged, and a count denial happens before any reservation is booked, so
558
+ // neither leaves the other's accounting to unwind. Whichever refuses first is reported as
559
+ // itself -- one refusal is never relabelled as the other, because "sends exhausted" and
560
+ // "spend exhausted" send an operator to two different remedies.
561
+ //
562
+ // A scope whose ceiling is ALREADY spent is refused here rather than at the reservation. The
563
+ // reservation needs a token count, which is not known until the body is parsed and a route
564
+ // resolved; an exhausted scope needs neither and is the cheapest refusal available.
565
+ if (!spend && ledger) {
566
+ const reached = spentRootCeiling(rootId, ledger);
567
+ if (reached) return refuse("workflow-spend-exhausted", reached);
568
+ }
569
+
472
570
  if (spend && ledger) {
473
571
  const decision = ledger.reserve({
474
572
  sendId: spend.sendId,
@@ -491,7 +589,9 @@ export function admitWorkflowTurn(
491
589
  : "workflow-spend-exhausted";
492
590
  return refuse(
493
591
  reason,
494
- denial.reason === "spend-limit-exceeded" ? denial.scope : undefined,
592
+ denial.reason === "spend-limit-exceeded"
593
+ ? { scope: denial.scope, limit: denial.limit, projected: denial.projected }
594
+ : undefined,
495
595
  );
496
596
  }
497
597
  }
@@ -600,6 +700,43 @@ export function workflowSendCeilingReached(
600
700
  return state !== undefined && windowedSends(state, now) >= policy.maxPhysicalSends;
601
701
  }
602
702
 
703
+ /**
704
+ * The root scope's ceiling when that scope is already spent, or undefined.
705
+ *
706
+ * Root only: identity and pool are not known until routing has picked an account, so those two
707
+ * refuse at the reservation itself. `exhausted` sums settled spend, open reservations and
708
+ * unresolved spend, which is the same total the reservation compares, so this answers the same
709
+ * question the reservation would -- just without needing the request's token count.
710
+ */
711
+ function spentRootCeiling(
712
+ rootId: string,
713
+ ledger: SpendReservationLedger,
714
+ ): WorkflowSpendDenialDetail | undefined {
715
+ const limit = ledger.policy.root.maxTokens;
716
+ if (limit === undefined) return undefined;
717
+ return ledger.exhausted("root", rootId) ? { scope: "root", limit } : undefined;
718
+ }
719
+
720
+ /**
721
+ * The token ceiling a root has already spent, or undefined when it has room or has none.
722
+ *
723
+ * The count-side twin of {@link workflowSendCeilingReached}, and the responses path calls both
724
+ * at the same seam for the same reason: a refusal decided before dispatch can be reported as
725
+ * ITSELF -- a named ceiling, a synthetic log row, a machine-readable header -- instead of
726
+ * surfacing later as a generic send-budget error from whichever leg happened to run out first.
727
+ *
728
+ * Returns undefined when no ceiling is configured, without resolving a ledger, so an install
729
+ * that never opted in neither pays for this check nor opens a journal because of it.
730
+ */
731
+ export function workflowSpendCeilingReached(
732
+ rootId: string | undefined,
733
+ spendLedger?: SpendReservationLedger,
734
+ ): WorkflowSpendDenialDetail | undefined {
735
+ if (!rootId) return undefined;
736
+ const ledger = spendLedger ?? (spendCeilingsConfigured() ? sharedSpendLedger() : undefined);
737
+ return ledger ? spentRootCeiling(rootId, ledger) : undefined;
738
+ }
739
+
603
740
  export interface WorkflowBudgetSnapshot {
604
741
  active: number;
605
742
  /** Sends inside the window. This is the number the ceiling compares. */
@@ -13,6 +13,38 @@
13
13
  import { getCachedProviderAccountQuota, hasPassiveAccountQuota } from "../providers/quota";
14
14
  import { getKiroAccountExhaustion } from "../providers/kiro-usage";
15
15
 
16
+ /** Antigravity hosts Gemini and Claude windows on one account; ranking must not mix them. */
17
+ export type QuotaModelFamily = "gem" | "cla";
18
+
19
+ export function classifyModelFamilyForQuota(
20
+ provider: string,
21
+ modelId?: string | null,
22
+ ): QuotaModelFamily | undefined {
23
+ if (provider !== "google-antigravity" || typeof modelId !== "string" || !modelId.trim()) {
24
+ return undefined;
25
+ }
26
+ const id = modelId.toLowerCase();
27
+ // Gemma is not Gemini: a substring/prefix match would poison Gemini ranking.
28
+ if (/(?:^|[^a-z])gemma(?:[^a-z]|$)/.test(id)) return undefined;
29
+ // Catalog ids are gemini-*, never a bare gem- token. Window labels still match Gem via
30
+ // windowMatchesFamily; this classifier is only for request model ids.
31
+ if (/(?:^|[^a-z])gemini(?:[^a-z]|$)/.test(id)) return "gem";
32
+ if (
33
+ /(?:^|[^a-z])claude(?:[^a-z]|$)/.test(id)
34
+ || /(?:^|[^a-z])opus(?:[^a-z]|$)/.test(id)
35
+ || /(?:^|[^a-z])sonnet(?:[^a-z]|$)/.test(id)
36
+ || /(?:^|[^a-z])haiku(?:[^a-z]|$)/.test(id)
37
+ || /(?:^|[^a-z])gpt[-_]oss(?:[^a-z]|$)/.test(id)
38
+ ) return "cla";
39
+ return undefined;
40
+ }
41
+
42
+ function windowMatchesFamily(label: string, family: QuotaModelFamily): boolean {
43
+ const token = label.trim().split(/[\s(/]+/)[0] ?? "";
44
+ if (family === "gem") return /^gem(?:ini)?$/i.test(token);
45
+ return /^cla(?:ude)?$/i.test(token);
46
+ }
47
+
16
48
  /** Lower sorts earlier. Unknown sits between measured-healthy and measured-empty. */
17
49
  const RANK_HEALTHY = 0;
18
50
  const RANK_UNKNOWN = 1;
@@ -48,15 +80,24 @@ const PASSIVE_HEADROOM_MAX_AGE_MS = 60 * 60_000;
48
80
  /**
49
81
  * Remaining headroom across every window the provider reports.
50
82
  *
51
- * The minimum wins: an account at 5% of its five-hour window is unusable right now even if
52
- * its monthly allowance is barely touched.
53
- */
54
- function headroomOf(provider: string, accountId: string): number | null {
83
+ * The minimum wins: an account at 5% of its five-hour window is unusable right now even if
84
+ * its monthly allowance is barely touched.
85
+ */
86
+ function headroomOf(provider: string, accountId: string, requestedModelId?: string | null): number | null {
55
87
  const quota = getCachedProviderAccountQuota(provider, accountId);
56
88
  if (!quota) return null;
57
89
  // Null, not a low rank: this must reproduce "no evidence" so a stale roster degrades to
58
90
  // the unranked ring rather than to a differently wrong answer.
59
91
  if (hasPassiveAccountQuota(provider) && Date.now() - quota.updatedAt > PASSIVE_HEADROOM_MAX_AGE_MS) return null;
92
+ const family = classifyModelFamilyForQuota(provider, requestedModelId);
93
+ if (family) {
94
+ const percents = (quota.customWindows ?? [])
95
+ .filter(window => windowMatchesFamily(window.label, family))
96
+ .map(window => window.percent)
97
+ .filter((value): value is number => typeof value === "number");
98
+ if (percents.length === 0) return null;
99
+ return 100 - Math.max(...percents);
100
+ }
60
101
  const percents = [
61
102
  quota.fiveHourPercent,
62
103
  quota.weeklyPercent,
@@ -74,15 +115,23 @@ function headroomOf(provider: string, accountId: string): number | null {
74
115
  * than an ordering. Null stays null all the way out: a caller must decide what "unmeasured"
75
116
  * means for its own rule instead of being handed a fabricated 0 or 100.
76
117
  */
77
- export function accountHeadroomPercent(provider: string, accountId: string): number | null {
78
- return headroomOf(provider, accountId);
118
+ export function accountHeadroomPercent(
119
+ provider: string,
120
+ accountId: string,
121
+ requestedModelId?: string | null,
122
+ ): number | null {
123
+ return headroomOf(provider, accountId, requestedModelId);
79
124
  }
80
125
 
81
126
  /** Unknown usage is not exhaustion; Kiro's explicit overage verdict is authoritative. */
82
- export function isAccountQuotaExhausted(provider: string, accountId: string): boolean {
127
+ export function isAccountQuotaExhausted(
128
+ provider: string,
129
+ accountId: string,
130
+ requestedModelId?: string | null,
131
+ ): boolean {
83
132
  const exhaustion = provider === "kiro" ? getKiroAccountExhaustion(`${provider}\u0000${accountId}`) : null;
84
133
  if (exhaustion !== null) return exhaustion.exhausted;
85
- const headroom = headroomOf(provider, accountId);
134
+ const headroom = headroomOf(provider, accountId, requestedModelId);
86
135
  return headroom !== null && headroom <= 0;
87
136
  }
88
137
 
@@ -92,24 +141,28 @@ export function isAccountQuotaExhausted(provider: string, accountId: string): bo
92
141
  * Returns the input untouched when no candidate has quota evidence, which keeps every
93
142
  * provider without per-account quota on exactly the behaviour it has today.
94
143
  */
95
- export function rankAccountsByHeadroom(provider: string, ring: readonly string[]): string[] {
144
+ export function rankAccountsByHeadroom(
145
+ provider: string,
146
+ ring: readonly string[],
147
+ requestedModelId?: string | null,
148
+ ): string[] {
96
149
  if (ring.length < 2) return [...ring];
97
150
 
98
151
  let sawEvidence = false;
99
152
  // Same rule as hasHeadroomEvidence: a passive provider's partial roster must not rank
100
153
  // at all. The failover path calls this directly (selectFailoverAccount), so the guard
101
154
  // cannot live only in the pre-dispatch predicate.
102
- if (hasPassiveAccountQuota(provider) && !ring.every(id => headroomOf(provider, id) !== null)) {
155
+ if (hasPassiveAccountQuota(provider) && !ring.every(id => headroomOf(provider, id, requestedModelId) !== null)) {
103
156
  return [...ring];
104
157
  }
105
158
  const ranked: Ranked[] = ring.map((id, index) => {
106
159
  // A provider-declared exhaustion verdict outranks the percentage: an account may sit at
107
160
  // 100% and still be servable when overage is enabled, and the verdict knows that.
108
161
  const exhaustion = provider === "kiro" ? getKiroAccountExhaustion(`${provider}\u0000${id}`) : null;
109
- const headroom = headroomOf(provider, id);
162
+ const headroom = headroomOf(provider, id, requestedModelId);
110
163
  if (exhaustion !== null || headroom !== null) sawEvidence = true;
111
164
 
112
- if (isAccountQuotaExhausted(provider, id)) return { id, bucket: RANK_EXHAUSTED, headroom: 0, index };
165
+ if (isAccountQuotaExhausted(provider, id, requestedModelId)) return { id, bucket: RANK_EXHAUSTED, headroom: 0, index };
113
166
  if (headroom === null) return { id, bucket: RANK_UNKNOWN, headroom: 0, index };
114
167
  return { id, bucket: RANK_HEALTHY, headroom, index };
115
168
  });
@@ -129,7 +182,11 @@ export function rankAccountsByHeadroom(provider: string, ring: readonly string[]
129
182
  * told "ranked" when nothing was measured. Pre-dispatch selection asks this first so it
130
183
  * can decline to act on a roster it knows nothing about.
131
184
  */
132
- export function hasHeadroomEvidence(provider: string, ids: readonly string[]): boolean {
185
+ export function hasHeadroomEvidence(
186
+ provider: string,
187
+ ids: readonly string[],
188
+ requestedModelId?: string | null,
189
+ ): boolean {
133
190
  // A PASSIVE provider needs evidence for EVERY candidate, not any one of them.
134
191
  //
135
192
  // A probe fills the whole roster in one pass (fetchProviderAccountQuotas), so "any"
@@ -140,10 +197,10 @@ export function hasHeadroomEvidence(provider: string, ids: readonly string[]): b
140
197
  // AWAY from an unmeasured account and TOWARD the one account known to be spent, which
141
198
  // is the exact inversion of what ranking is for.
142
199
  if (hasPassiveAccountQuota(provider)) {
143
- return ids.length > 0 && ids.every(id => headroomOf(provider, id) !== null);
200
+ return ids.length > 0 && ids.every(id => headroomOf(provider, id, requestedModelId) !== null);
144
201
  }
145
202
  return ids.some(id =>
146
- headroomOf(provider, id) !== null
203
+ headroomOf(provider, id, requestedModelId) !== null
147
204
  || (provider === "kiro" && getKiroAccountExhaustion(`${provider}\u0000${id}`) !== null));
148
205
  }
149
206
  /**