@centerforagenticai/pi-multi-account 0.1.2 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,8 +6,10 @@ import {
6
6
  type MachineLeaseHandle,
7
7
  } from "./machine-lease.js";
8
8
  import {
9
+ MAX_USAGE_ATTEMPT_DELAY_MS,
9
10
  normalizeUsageEndpointPercent,
10
11
  type SharedUsageAttemptRecord,
12
+ type SharedUsageFailureReason,
11
13
  type SharedUsageStore,
12
14
  type UsageFailureDetail,
13
15
  } from "./shared-usage.js";
@@ -39,9 +41,8 @@ export const USAGE_FETCH_TIMEOUT_MS = 10_000;
39
41
  const MAX_RESPONSE_BYTES = 128 * 1024;
40
42
  const MAX_ERROR_CHARS = 512;
41
43
  const BASE_BACKOFF_MS = 30_000;
42
- const MAX_BACKOFF_MS = 15 * 60_000;
43
44
  const BACKOFF_LADDER_STEPS =
44
- Math.ceil(Math.log2(MAX_BACKOFF_MS / BASE_BACKOFF_MS)) + 1;
45
+ Math.ceil(Math.log2(MAX_USAGE_ATTEMPT_DELAY_MS / BASE_BACKOFF_MS)) + 1;
45
46
  /** Disable only after the capped rung has failed once more. */
46
47
  export const USAGE_FETCH_DISABLE_AFTER_FAILURES = BACKOFF_LADDER_STEPS + 1;
47
48
 
@@ -104,12 +105,7 @@ export interface UsageFetchStatus {
104
105
  readonly disabled: boolean;
105
106
  readonly failureCount: number;
106
107
  readonly nextAttemptAtMs?: number;
107
- readonly disabledReason?:
108
- | "rate-limit"
109
- | "server-error"
110
- | "credential-unavailable"
111
- | "malformed-response"
112
- | "network-error";
108
+ readonly disabledReason?: SharedUsageFailureReason;
113
109
  }
114
110
 
115
111
  export type UsageFetchResultStatus =
@@ -217,7 +213,10 @@ function retryAfterMs(response: UsageFetchResponse): number | undefined {
217
213
  }
218
214
  const seconds = Number(value);
219
215
  if (!Number.isFinite(seconds) || seconds < 0) return undefined;
220
- return Math.min(MAX_BACKOFF_MS, Math.max(0, Math.ceil(seconds * 1_000)));
216
+ return Math.min(
217
+ MAX_USAGE_ATTEMPT_DELAY_MS,
218
+ Math.max(0, Math.ceil(seconds * 1_000)),
219
+ );
221
220
  }
222
221
 
223
222
  /** Redacts bearer and access-token values before an error can escape this module. */
@@ -642,14 +641,28 @@ async function queryEndpoint(
642
641
  : normalizeAnthropicUsagePayload(payload);
643
642
  }
644
643
 
644
+ // Every attempt a reader here sees comes from `SharedUsageStore.latestAttempt`
645
+ // called with that reader's own clock, and that is what keeps a far-future
646
+ // deadline (internal issue #108) from suppressing polling:
647
+ //
648
+ // - the store refuses, on append and read, any attempt whose
649
+ // `nextAttemptAtMs - observedAtMs` exceeds MAX_USAGE_ATTEMPT_DELAY_MS; and
650
+ // - `latestAttempt` skips any attempt observed after `nowMs` (#133).
651
+ //
652
+ // Together these give `nextAttemptAtMs <= nowMs + MAX_USAGE_ATTEMPT_DELAY_MS`
653
+ // for every record a reader receives. A per-reader "plausible deadline" check
654
+ // used to restate that bound; once the origin filter moved into the store it
655
+ // could no longer fail on any reachable input and was removed (#133,
656
+ // CR-REFRESH-READER-BOUND-UNREACHABLE). Read attempts only through
657
+ // `latestAttempt(providerId, family, nowMs)` with the clock the decision uses.
658
+
645
659
  function statusFromAttempt(
646
660
  attempt: SharedUsageAttemptRecord | undefined,
647
661
  enabled: boolean,
648
662
  ): UsageFetchStatus {
649
663
  const nextAttemptAtMs =
650
664
  attempt?.failureCount === 0 ? undefined : attempt?.nextAttemptAtMs;
651
- const disabledReason =
652
- attempt?.failureReason as UsageFetchStatus["disabledReason"];
665
+ const disabledReason = attempt?.failureReason;
653
666
  return {
654
667
  enabled,
655
668
  disabled: attempt?.disabled ?? false,
@@ -831,7 +844,7 @@ export class UsageFetcher {
831
844
  ): UsageFetchStatus {
832
845
  const enabled = config.usageFetchEnabled?.[family] ?? true;
833
846
  return statusFromAttempt(
834
- this.#sharedStore.latestAttempt(providerId, family),
847
+ this.#sharedStore.latestAttempt(providerId, family, this.#now()),
835
848
  enabled,
836
849
  );
837
850
  }
@@ -994,18 +1007,16 @@ export class UsageFetcher {
994
1007
  // the debounce clause, so `already-answered` and `backoff` returned
995
1008
  // before it ran: an attempt with `observedAtMs` centuries ahead is
996
1009
  // trivially `>= failedAtMs`, and suppressed every failure refresh
997
- // forever. The bound belongs here, on the record, before any clause
998
- // reads it.
1010
+ // forever. `latestAttempt(..., nowMs)` now refuses such a record at the
1011
+ // source (#133), which also bounds `nextAttemptAtMs` to one capped rung
1012
+ // past `nowMs` (see the note above `statusFromAttempt`).
999
1013
  //
1000
- // `nextAttemptAtMs` is allowed one debounce interval of headroom because
1001
- // Only the ORIGIN timestamps are bounded, not `nextAttemptAtMs`: a real
1002
- // failure ladder legitimately schedules its next rung far ahead, up to
1003
- // MAX_BACKOFF_MS, and bounding that would break genuine backoff
1004
- // suppression. An honest record cannot have been OBSERVED in the future.
1014
+ // The failure trigger is not tied to `observedAtMs` by the store, so a
1015
+ // record observed in the past can still name a failure in the future;
1016
+ // such a marker cannot be honest and must not arm the debounce.
1005
1017
  if (
1006
- attempt.observedAtMs > nowMs ||
1007
- (attempt.failureTriggeredAtMs !== undefined &&
1008
- attempt.failureTriggeredAtMs > nowMs)
1018
+ attempt.failureTriggeredAtMs !== undefined &&
1019
+ attempt.failureTriggeredAtMs > nowMs
1009
1020
  ) {
1010
1021
  return undefined;
1011
1022
  }
@@ -1044,15 +1055,17 @@ export class UsageFetcher {
1044
1055
  }
1045
1056
 
1046
1057
  // Step 4/5: durable checks before the lease.
1058
+ const beforeLeaseNowMs = this.#now();
1047
1059
  const beforeLease = this.#sharedStore.latestAttempt(
1048
1060
  account.providerId,
1049
1061
  account.family,
1062
+ beforeLeaseNowMs,
1050
1063
  );
1051
1064
  if (
1052
1065
  this.#failureRefreshSuppressedBy(
1053
1066
  beforeLease,
1054
1067
  failedAtMs,
1055
- this.#now(),
1068
+ beforeLeaseNowMs,
1056
1069
  ) !== undefined
1057
1070
  ) {
1058
1071
  return { providerId: account.providerId, status: "not-due" };
@@ -1072,15 +1085,17 @@ export class UsageFetcher {
1072
1085
  const leaseGuard: AntigravityLeaseGuard = { lease, handedOff: false };
1073
1086
  try {
1074
1087
  // Step 7: the same checks again, now that nobody else can write.
1088
+ const underLeaseNowMs = this.#now();
1075
1089
  const underLease = this.#sharedStore.latestAttempt(
1076
1090
  account.providerId,
1077
1091
  account.family,
1092
+ underLeaseNowMs,
1078
1093
  );
1079
1094
  if (
1080
1095
  this.#failureRefreshSuppressedBy(
1081
1096
  underLease,
1082
1097
  failedAtMs,
1083
- this.#now(),
1098
+ underLeaseNowMs,
1084
1099
  ) !== undefined
1085
1100
  ) {
1086
1101
  return { providerId: account.providerId, status: "not-due" };
@@ -1169,6 +1184,7 @@ export class UsageFetcher {
1169
1184
  const prior = this.#sharedStore.latestAttempt(
1170
1185
  account.providerId,
1171
1186
  account.family,
1187
+ nowMs,
1172
1188
  );
1173
1189
  // Backoff is checked BEFORE `disabled`, deliberately.
1174
1190
  //
@@ -1190,7 +1206,7 @@ export class UsageFetcher {
1190
1206
  // Once the recorded backoff has elapsed, the account is retried whether or
1191
1207
  // not it was disabled. A retry that fails again simply re-arms the ladder at
1192
1208
  // its capped rung, so a genuinely dead endpoint is polled at most once per
1193
- // MAX_BACKOFF_MS rather than hammered.
1209
+ // MAX_USAGE_ATTEMPT_DELAY_MS rather than hammered.
1194
1210
  if (prior !== undefined && prior.nextAttemptAtMs > nowMs) {
1195
1211
  await this.#persistWindowGap(
1196
1212
  account,
@@ -1215,15 +1231,17 @@ export class UsageFetcher {
1215
1231
  }
1216
1232
  const leaseGuard: AntigravityLeaseGuard = { lease, handedOff: false };
1217
1233
  try {
1218
- const current = this.#sharedStore.latestAttempt(
1219
- account.providerId,
1220
- account.family,
1221
- );
1222
1234
  // Same ordering as the pre-lease check above: an elapsed backoff wins
1223
1235
  // over a stale `disabled` flag, so a recovered account can be observed
1224
1236
  // again. Re-read under the lease because a peer may have attempted in
1225
1237
  // between.
1226
- if (current !== undefined && current.nextAttemptAtMs > this.#now()) {
1238
+ const currentNowMs = this.#now();
1239
+ const current = this.#sharedStore.latestAttempt(
1240
+ account.providerId,
1241
+ account.family,
1242
+ currentNowMs,
1243
+ );
1244
+ if (current !== undefined && current.nextAttemptAtMs > currentNowMs) {
1227
1245
  await this.#persistWindowGap(
1228
1246
  account,
1229
1247
  this.#now(),
@@ -1555,7 +1573,7 @@ export class UsageFetcher {
1555
1573
  : new UsageEndpointError("usage endpoint failed", "network-error");
1556
1574
  const failureCount = (prior?.failureCount ?? 0) + 1;
1557
1575
  const backoff = Math.min(
1558
- MAX_BACKOFF_MS,
1576
+ MAX_USAGE_ATTEMPT_DELAY_MS,
1559
1577
  BASE_BACKOFF_MS * 2 ** Math.max(0, failureCount - 1),
1560
1578
  );
1561
1579
  const disabled = failureCount >= USAGE_FETCH_DISABLE_AFTER_FAILURES;
@@ -1617,13 +1635,15 @@ export class UsageFetcher {
1617
1635
  }
1618
1636
  // REQ-WINDOW-CADENCE: enforce cross-process 15-minute floor.
1619
1637
  // Check durable attempt record to prevent peer-process fetch within the window.
1638
+ const cadenceNowMs = this.#now();
1620
1639
  const prior = this.#sharedStore.latestAttempt(
1621
1640
  account.providerId,
1622
1641
  account.family,
1642
+ cadenceNowMs,
1623
1643
  );
1624
1644
  if (
1625
1645
  prior !== undefined &&
1626
- this.#now() - prior.observedAtMs < WINDOW_SAMPLE_INTERVAL_MS
1646
+ cadenceNowMs - prior.observedAtMs < WINDOW_SAMPLE_INTERVAL_MS
1627
1647
  ) {
1628
1648
  continue;
1629
1649
  }
package/src/usage.ts CHANGED
@@ -337,8 +337,8 @@ export class UsageLedger {
337
337
  * exhaustion hold is one. But a hold is machine-global: deleting the record
338
338
  * would revoke it for every other session on the machine, and those peers
339
339
  * may still be refusing work on that account. So the override is
340
- * PROCESS-LOCAL -- this session stops honouring holds installed up to the
341
- * moment of the clear, and peers keep theirs.
340
+ * PROCESS-LOCAL -- this session stops honouring holds whose failure happened
341
+ * at or before the clear, and peers keep theirs.
342
342
  *
343
343
  * Stored as the clear time rather than a boolean so a LATER failure still
344
344
  * installs a hold this session honours. A boolean would make one `clear`
@@ -346,6 +346,13 @@ export class UsageLedger {
346
346
  * which is a permanent effect from a transient instruction.
347
347
  */
348
348
  readonly #holdOverrides = new Map<string, number>();
349
+ /**
350
+ * When `clear()` with no account last ran. Clear-all is one point in time
351
+ * for EVERY account, including ones with no hold visible at that instant:
352
+ * enumerating the holds present then let a peer publish a pre-clear hold a
353
+ * moment later and still have it honoured (internal issue #110).
354
+ */
355
+ #clearAllAtMs: number | undefined;
349
356
  readonly #sharedStore: SharedUsageStore | undefined;
350
357
  readonly #now: () => number;
351
358
 
@@ -714,29 +721,29 @@ export class UsageLedger {
714
721
  family: AllowedFamily,
715
722
  nowMs = this.#now(),
716
723
  ): number | undefined {
717
- const holdUntilMs = this.#sharedStore?.activeExhaustionHoldUntilMs(
724
+ // An operator `clear` in THIS process stops us honouring holds whose
725
+ // failure happened at or before it. A hold from a LATER failure is
726
+ // honoured again: the override is a point in time, not a permanent
727
+ // exemption. Peers keep honouring the record either way, because it is
728
+ // not deleted.
729
+ //
730
+ // The store compares each hold's own validated `failedAtMs`. Deriving
731
+ // the failure time from the deadline (`holdUntilMs - EXHAUSTION_HOLD_MS`)
732
+ // is only right for an exact one-hour hold, and the store accepts every
733
+ // shorter span (internal issue #110).
734
+ const accountClearedAtMs = this.#holdOverrides.get(providerId);
735
+ const clearedAtMs =
736
+ accountClearedAtMs === undefined
737
+ ? this.#clearAllAtMs
738
+ : this.#clearAllAtMs === undefined
739
+ ? accountClearedAtMs
740
+ : Math.max(accountClearedAtMs, this.#clearAllAtMs);
741
+ return this.#sharedStore?.activeExhaustionHoldUntilMs(
718
742
  providerId,
719
743
  family,
720
744
  nowMs,
745
+ clearedAtMs,
721
746
  );
722
- if (holdUntilMs === undefined) return undefined;
723
- // An operator `clear` in THIS process stops us honouring holds that were
724
- // already installed when it ran. A hold from a LATER failure is honoured
725
- // again: the override is a point in time, not a permanent exemption.
726
- // Peers keep honouring the record either way, because it is not deleted.
727
- //
728
- // `holdUntilMs - EXHAUSTION_HOLD_MS` recovers the failure time from the
729
- // reported deadline without a second store read. The writer always sets
730
- // exactly that span (asserted by the hold-duration test), and the
731
- // store's own read bound refuses any record claiming more.
732
- const clearedAtMs = this.#holdOverrides.get(providerId);
733
- if (
734
- clearedAtMs !== undefined &&
735
- holdUntilMs - EXHAUSTION_HOLD_MS <= clearedAtMs
736
- ) {
737
- return undefined;
738
- }
739
- return holdUntilMs;
740
747
  }
741
748
 
742
749
  /**
@@ -811,11 +818,7 @@ export class UsageLedger {
811
818
  if (providerId === undefined) {
812
819
  this.#snapshots.clear();
813
820
  this.#quotaObservations.clear();
814
- for (const id of this.#sharedStore
815
- ?.readExhaustionHolds()
816
- .map((hold) => hold.providerId) ?? []) {
817
- this.#holdOverrides.set(id, nowMs);
818
- }
821
+ this.#clearAllAtMs = Math.max(this.#clearAllAtMs ?? nowMs, nowMs);
819
822
  } else {
820
823
  this.#snapshots.delete(providerId);
821
824
  this.#quotaObservations.delete(providerId);