@centerforagenticai/pi-multi-account 0.1.1 → 0.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -65,25 +65,22 @@ export const EXHAUSTION_HOLD_MS = 60 * 60_000;
65
65
  * policy.
66
66
  */
67
67
  const MAX_REFRESH_DEBOUNCE_MS = 60 * 60_000;
68
+ /** Longest persisted usage-attempt delay produced by the capped retry ladder. */
69
+ export const MAX_USAGE_ATTEMPT_DELAY_MS = 15 * 60_000;
68
70
 
69
71
  export const SHARED_USAGE_MAX_BYTES = 512 * 1024;
70
72
  const MAX_RECORD_BYTES = 4_096;
71
73
  const MAX_OBSERVER_ID_LENGTH = 256;
72
74
  /**
73
- * Shape a field name must have inside a record type this build does not
74
- * recognise, applied when deciding whether compaction may carry it forward
75
- * rather than delete it.
76
- *
77
- * A length bound alone was the first attempt and round 3 refuted it: a key
78
- * called `Bearer SECRET` is short, so the credential simply moved from the
79
- * value into the key. Field names this project writes are lower-camel
80
- * identifiers.
81
- *
82
- * There is no companion value-length bound: unknown string VALUES are refused
83
- * outright rather than length-capped, because a bounded short string is
84
- * exactly the shape of a leaked bearer token.
75
+ * Exactly the grammar `defaultObserverId()` can emit: a hostname mapped into
76
+ * `[A-Za-z0-9._-]`, then at most one `:` followed only by decimal pid digits,
77
+ * sliced to MAX_OBSERVER_ID_LENGTH. The rule is derived from the producer
78
+ * rather than from a guess about what a credential looks like
79
+ * (internal issue #108), and it deliberately does not require the
80
+ * colon or the pid: on a host whose name fills the slice, the delimiter and pid
81
+ * are cut off, and the store must not refuse or delete those records.
85
82
  */
86
- const CARRYABLE_FIELD_NAME = /^[a-z][A-Za-z0-9]{0,63}$/;
83
+ const OBSERVER_ID_PATTERN = /^(?=.{1,256}$)[A-Za-z0-9._-]*(?::[0-9]*)?$/;
87
84
 
88
85
  export type UsageObservationSource = "rate-limit-header" | "usage-endpoint";
89
86
 
@@ -123,6 +120,25 @@ const USAGE_FAILURE_DETAILS = new Set<UsageFailureDetail>([
123
120
  "quota-summary-error",
124
121
  ]);
125
122
 
123
+ /**
124
+ * The closed set `failureReason()` in usage-fetch.ts can produce. A persisted
125
+ * attempt may carry only one of these, so the field cannot hold caller text.
126
+ */
127
+ export type SharedUsageFailureReason =
128
+ | "rate-limit"
129
+ | "server-error"
130
+ | "credential-unavailable"
131
+ | "malformed-response"
132
+ | "network-error";
133
+
134
+ const USAGE_FAILURE_REASONS = new Set<SharedUsageFailureReason>([
135
+ "rate-limit",
136
+ "server-error",
137
+ "credential-unavailable",
138
+ "malformed-response",
139
+ "network-error",
140
+ ]);
141
+
126
142
  /** Machine-global fetch-attempt state; deliberately ignored by usage aggregation. */
127
143
  export interface SharedUsageAttemptRecord {
128
144
  readonly recordType: "usage-attempt";
@@ -133,9 +149,10 @@ export interface SharedUsageAttemptRecord {
133
149
  readonly observedAtMs: number;
134
150
  readonly observerId: string;
135
151
  readonly failureCount: number;
152
+ /** Bounded relative to `observedAtMs` by the capped retry ladder. */
136
153
  readonly nextAttemptAtMs: number;
137
154
  readonly disabled: boolean;
138
- readonly failureReason?: string;
155
+ readonly failureReason?: SharedUsageFailureReason;
139
156
  /** Fixed, sanitized classification detail; never an upstream error body. */
140
157
  readonly failureDetail?: UsageFailureDetail;
141
158
  /**
@@ -238,141 +255,9 @@ function validTimestamp(value: unknown): value is number {
238
255
  return typeof value === "number" && Number.isFinite(value) && value >= 0;
239
256
  }
240
257
 
241
- /**
242
- * Whether a record this build cannot interpret may survive compaction.
243
- *
244
- * Compaction rebuilds the file from recognised records, so anything not carried
245
- * here is deleted. That is how an older build destroys a record type added
246
- * after it. Carrying unknown records forward keeps a newer build's state alive
247
- * across a mixed-version fleet.
248
- *
249
- * The filter exists because "unrecognised" also covers records this build
250
- * rejected as malformed, including the credential-bearing ones that
251
- * `append` refuses and compaction currently scrubs. Carrying those would turn a
252
- * durability fix into a privacy regression, so a carried record must still look
253
- * like a usage record for a known account: a `recordType` string this build
254
- * does not know, the same account identity fields every record carries, and no
255
- * field outside that shape. A future record type satisfies this; a leaked
256
- * authorization header does not.
257
- */
258
- function carryableUnknownRecord(value: unknown): boolean {
259
- if (typeof value !== "object" || value === null || Array.isArray(value))
260
- return false;
261
- const record = value as Record<string, unknown>;
262
- // `recordType` must look like a record type, not merely be non-empty.
263
- //
264
- // Round 2 proved "non-empty string" is not a constraint: `Bearer <token>`
265
- // is a non-empty string, so a credential placed here survived compaction.
266
- // A real record type is a lower-kebab identifier, and nothing that fails
267
- // this shape is a record type this build should carry blind.
268
- if (
269
- typeof record.recordType !== "string" ||
270
- !CARRYABLE_RECORD_TYPE.test(record.recordType)
271
- )
272
- return false;
273
- if (
274
- typeof record.providerId !== "string" ||
275
- typeof record.family !== "string" ||
276
- !isAllowedFamily(record.family) ||
277
- !isCanonicalManagedProviderId(record.providerId, record.family) ||
278
- !validTimestamp(record.observedAtMs) ||
279
- !validObserverId(record.observerId) ||
280
- // Stricter than `validObserverId` on purpose: that only bounds length,
281
- // and a bearer token is a bounded string. See CARRYABLE_OBSERVER_ID.
282
- !CARRYABLE_OBSERVER_ID.test(record.observerId)
283
- )
284
- return false;
285
- // Beyond the identity fields above, a carried record may hold only
286
- // NON-STRING values.
287
- //
288
- // The first version of this filter bounded value shape -- primitives only,
289
- // length-capped -- and review proved it unsound: `{ authorization: "Bearer
290
- // SHORT" }` is a bounded primitive and survived compaction, which is exactly
291
- // the privacy regression the carry-through was not allowed to create.
292
- //
293
- // Refusing unknown strings outright is the only defensible rule here. A
294
- // denylist of sensitive-looking field names would be a guess about what a
295
- // future record type calls its fields, and every credential this project
296
- // handles is a string. Timestamps, counts, fractions and flags -- what a
297
- // forward-compatible usage record actually needs -- are unaffected. A future
298
- // type that genuinely needs a string field must teach this build about
299
- // itself rather than rely on being carried blind.
300
- for (const [key, entry] of Object.entries(record)) {
301
- // Field NAMES are constrained by shape, not only length. Round 3 found
302
- // a length bound alone lets a key called `Bearer SECRET` through: the
303
- // credential rides in the key rather than the value. A field name in a
304
- // JSON record written by this project is a lower-camel identifier.
305
- if (!CARRYABLE_FIELD_NAME.test(key)) return false;
306
- if (CARRYABLE_IDENTITY_FIELDS.has(key)) continue;
307
- if (!carryableUnknownValue(entry)) return false;
308
- }
309
- return true;
310
- }
311
-
312
- /**
313
- * Record types compaction may carry forward: the project's own namespace.
314
- *
315
- * Two weaker rules were tried and both refuted. "Non-empty" fell to
316
- * `Bearer <token>` in round 2. A lower-kebab shape fell in round 3 to
317
- * `sk-ant-api03-deadbeef`, which IS lower-kebab -- an API key and a record
318
- * type are not distinguishable by shape, so no amount of character-class
319
- * tightening can separate them.
320
- *
321
- * A namespace can. Every record type this file writes is `usage-`-prefixed
322
- * (`usage-attempt`, `usage-exhaustion-hold`), so a future type from a newer
323
- * build will be too. That is a property of the writer rather than a guess
324
- * about what a credential looks like, which is why it holds where the shape
325
- * checks did not.
326
- */
327
- const CARRYABLE_RECORD_TYPE = /^usage-[a-z][a-z0-9-]{0,56}$/;
328
-
329
- /**
330
- * Shape a carried record's `observerId` must have.
331
- *
332
- * `validObserverId` only bounds length, and round 2 proved that insufficient:
333
- * a bearer token is a bounded string. This restricts the CHARACTER SET instead,
334
- * which is what makes the field unusable for smuggling while still accepting
335
- * everything the producer can emit.
336
- *
337
- * The colon is deliberately NOT required. `defaultObserverId` builds
338
- * `${hostname()}:${process.pid}` and then truncates to MAX_OBSERVER_ID_LENGTH,
339
- * so on a host with a very long name the pid -- and the colon with it -- is cut
340
- * off entirely. Round 3 found an earlier version of this expression required
341
- * the colon, which would have made compaction DELETE legitimate records on such
342
- * a machine: the precise data loss this carry-through exists to prevent, caused
343
- * by the fix for it. A verified probe produced a 256-character id with no colon
344
- * at all.
345
- *
346
- * The length bound matches MAX_OBSERVER_ID_LENGTH rather than guessing a
347
- * narrower one, so the accepted domain covers every value the producer can
348
- * actually return.
349
- */
350
- const CARRYABLE_OBSERVER_ID = /^[A-Za-z0-9._:-]{1,256}$/;
351
-
352
- /**
353
- * Identity fields every record carries. They are the only strings a carried
354
- * unknown record may contain, and each is format-checked above rather than
355
- * merely bounded.
356
- */
357
- const CARRYABLE_IDENTITY_FIELDS = new Set([
358
- "recordType",
359
- "providerId",
360
- "family",
361
- "observerId",
362
- ]);
363
-
364
- function carryableUnknownValue(value: unknown): boolean {
365
- return (
366
- value === null || typeof value === "boolean" || typeof value === "number"
367
- );
368
- }
369
258
 
370
259
  function validObserverId(value: unknown): value is string {
371
- return (
372
- typeof value === "string" &&
373
- value.length > 0 &&
374
- value.length <= MAX_OBSERVER_ID_LENGTH
375
- );
260
+ return typeof value === "string" && OBSERVER_ID_PATTERN.test(value);
376
261
  }
377
262
 
378
263
  const TOKEN_FIELDS = [
@@ -474,10 +359,15 @@ function validAttemptRecord(value: unknown): value is SharedUsageAttemptRecord {
474
359
  validObserverId(record.observerId) &&
475
360
  nonNegativeInteger(record.failureCount) &&
476
361
  validTimestamp(record.nextAttemptAtMs) &&
362
+ record.nextAttemptAtMs >= record.observedAtMs &&
363
+ record.nextAttemptAtMs - record.observedAtMs <=
364
+ MAX_USAGE_ATTEMPT_DELAY_MS &&
477
365
  typeof record.disabled === "boolean" &&
478
366
  (record.failureReason === undefined ||
479
367
  (typeof record.failureReason === "string" &&
480
- record.failureReason.length <= 128)) &&
368
+ USAGE_FAILURE_REASONS.has(
369
+ record.failureReason as SharedUsageFailureReason,
370
+ ))) &&
481
371
  (record.failureDetail === undefined ||
482
372
  (typeof record.failureDetail === "string" &&
483
373
  USAGE_FAILURE_DETAILS.has(record.failureDetail as UsageFailureDetail))) &&
@@ -551,9 +441,19 @@ function defaultStorePath(): string {
551
441
  return join(agentDir, "pi-multi-account", "usage.ndjson");
552
442
  }
553
443
 
554
- /** Identity is bounded metadata only; it contains no credential-derived value. */
555
- export function defaultObserverId(): string {
556
- return `${hostname()}:${process.pid}`.slice(0, MAX_OBSERVER_ID_LENGTH);
444
+ /**
445
+ * Identity is bounded metadata only; it contains no credential-derived value.
446
+ *
447
+ * Hostname characters outside the observer-id alphabet are mapped to `-`, so
448
+ * every value this producer returns is one `validObserverId` accepts. Without
449
+ * that, an unusual hostname would make the default store constructor throw.
450
+ */
451
+ export function defaultObserverId(
452
+ host: string = hostname(),
453
+ pid: number = process.pid,
454
+ ): string {
455
+ const safeHost = host.replace(/[^A-Za-z0-9._-]/g, "-");
456
+ return `${safeHost}:${pid}`.slice(0, MAX_OBSERVER_ID_LENGTH);
557
457
  }
558
458
 
559
459
  function projectRecord(record: SharedUsageRecord): SharedUsageRecord {
@@ -829,24 +729,26 @@ function compactUsageFile(options: {
829
729
  const latestRateLimits = new Map<string, SharedUsageRecord>();
830
730
  const latestAttempts = new Map<string, SharedUsageAttemptRecord>();
831
731
  const latestHolds = new Map<string, SharedUsageExhaustionHoldRecord>();
832
- // Records this build has no reader for are carried through verbatim.
833
- //
834
- // Compaction rebuilds the file from what it recognises, so without this a
835
- // process running an older build silently deletes every record type added
836
- // after it -- including the exhaustion holds that keep a spent account out
837
- // of rotation. The account then looks healthy
838
- // to the next process and gets routed to again.
732
+ // Compaction is a CLOSED REGISTRY: only the three record types this
733
+ // build validates survive, and each survives as the JSON of its own
734
+ // validated projection, never as the source bytes.
839
735
  //
840
- // Carrying them verbatim rather than re-serialising keeps this build from
841
- // imposing a shape on data it does not understand, and matches how the
842
- // writer tail below is copied byte-for-byte.
736
+ // internal MR !83 carried unrecognised records verbatim so an
737
+ // older build would not delete a newer build's record types. Four rounds
738
+ // of shape rules (non-empty, lower-kebab, `usage-` namespace, lower-camel
739
+ // field names) tried to tell a future record from credential text and
740
+ // all were refuted (internal issue #110): `sk-ant-api03-deadbeef`
741
+ // is indistinguishable by shape from an identifier, a `usage-` prefix is
742
+ // free to any writer, and a retained source line keeps every field name
743
+ // and raw numeric text too. A record this build cannot interpret cannot
744
+ // be proved credential-free, so it is dropped.
843
745
  //
844
- // This is deliberately limited to whole unrecognised *records*.
845
- // Unrecognised *fields* on a recognised record are still stripped by
846
- // projectRecord/projectAttempt: that projection is the credential scrub
847
- // required by AGENTS.md, and widening it here would trade a privacy
848
- // guarantee for a durability one.
849
- const unrecognisedLines: string[] = [];
746
+ // Mixed-version safety is kept for every type this build knows, the
747
+ // exhaustion hold included: each is re-emitted from its validated fields,
748
+ // so a peer on this build never deletes a live hold. The accepted cost
749
+ // is that a record type added after this build is dropped when this
750
+ // build compacts; a new type must stay advisory until every build in the
751
+ // fleet recognises it.
850
752
 
851
753
  for (const line of completeUsageLines(options.path, completePrefixBytes)) {
852
754
  let parsed: unknown;
@@ -893,10 +795,7 @@ function compactUsageFile(options: {
893
795
  if (!previous || hold.observedAtMs >= previous.observedAtMs) {
894
796
  latestHolds.set(hold.providerId, hold);
895
797
  }
896
- } else if (carryableUnknownRecord(parsed)) {
897
- unrecognisedLines.push(line);
898
798
  }
899
-
900
799
  }
901
800
  const compactedRecords = new Set([
902
801
  ...latest.values(),
@@ -905,12 +804,9 @@ function compactUsageFile(options: {
905
804
  ...latestAttempts.values(),
906
805
  ...latestHolds.values(),
907
806
  ]);
908
- // Carried records go first and verbatim, so a build that cannot interpret
909
- // them neither reorders them relative to each other nor reformats them.
910
- const compacted = [
911
- ...unrecognisedLines.map((line) => `${line}\n`),
912
- ...[...compactedRecords].map((record) => `${JSON.stringify(record)}\n`),
913
- ].join("");
807
+ const compacted = [...compactedRecords]
808
+ .map((record) => `${JSON.stringify(record)}\n`)
809
+ .join("");
914
810
  const compactedBytes = Buffer.byteLength(compacted, "utf8");
915
811
  if (compactedBytes > options.maxBytes) return false;
916
812
 
@@ -1009,6 +905,11 @@ export class SharedUsageStore {
1009
905
 
1010
906
  append(record: SharedUsageLogRecord): boolean {
1011
907
  try {
908
+ // Judge the caller's own value, not the projection: projecting slices
909
+ // the observer id, and a value this store would have to rewrite is
910
+ // not one the producer emitted.
911
+ if (!validObserverId((record as { observerId?: unknown }).observerId))
912
+ return false;
1012
913
  const projected = validExhaustionHoldRecord(record)
1013
914
  ? projectExhaustionHold(record)
1014
915
  : validAttemptRecord(record)
@@ -1134,16 +1035,22 @@ export class SharedUsageStore {
1134
1035
  * records, so a hold that has lapsed simply stops being reported. Callers get
1135
1036
  * a time to compare, not a boolean, because routing has to combine it with an
1136
1037
  * authoritative recovery time and take whichever is later.
1038
+ *
1039
+ * `clearedAtMs` excludes holds whose own validated `failedAtMs` is at or
1040
+ * before it, so a caller's operator clear is judged against the failure
1041
+ * the record states rather than one reconstructed from its deadline.
1137
1042
  */
1138
1043
  activeExhaustionHoldUntilMs(
1139
1044
  providerId: string,
1140
1045
  family: AllowedFamily,
1141
1046
  nowMs: number,
1047
+ clearedAtMs?: number,
1142
1048
  ): number | undefined {
1143
1049
  let latest: number | undefined;
1144
1050
  for (const hold of this.readExhaustionHolds()) {
1145
1051
  if (hold.providerId !== providerId || hold.family !== family) continue;
1146
1052
  if (hold.holdUntilMs <= nowMs) continue;
1053
+ if (clearedAtMs !== undefined && hold.failedAtMs <= clearedAtMs) continue;
1147
1054
  // A relative bound alone is not enough. `holdUntilMs - failedAtMs` can
1148
1055
  // be a legitimate 60 minutes while `failedAtMs` itself sits in the year
1149
1056
  // 3138, which would exclude the account for centuries. No honest hold
@@ -6,8 +6,10 @@ import {
6
6
  type MachineLeaseHandle,
7
7
  } from "./machine-lease.js";
8
8
  import {
9
+ MAX_USAGE_ATTEMPT_DELAY_MS,
9
10
  normalizeUsageEndpointPercent,
10
11
  type SharedUsageAttemptRecord,
12
+ type SharedUsageFailureReason,
11
13
  type SharedUsageStore,
12
14
  type UsageFailureDetail,
13
15
  } from "./shared-usage.js";
@@ -39,9 +41,8 @@ export const USAGE_FETCH_TIMEOUT_MS = 10_000;
39
41
  const MAX_RESPONSE_BYTES = 128 * 1024;
40
42
  const MAX_ERROR_CHARS = 512;
41
43
  const BASE_BACKOFF_MS = 30_000;
42
- const MAX_BACKOFF_MS = 15 * 60_000;
43
44
  const BACKOFF_LADDER_STEPS =
44
- Math.ceil(Math.log2(MAX_BACKOFF_MS / BASE_BACKOFF_MS)) + 1;
45
+ Math.ceil(Math.log2(MAX_USAGE_ATTEMPT_DELAY_MS / BASE_BACKOFF_MS)) + 1;
45
46
  /** Disable only after the capped rung has failed once more. */
46
47
  export const USAGE_FETCH_DISABLE_AFTER_FAILURES = BACKOFF_LADDER_STEPS + 1;
47
48
 
@@ -104,12 +105,7 @@ export interface UsageFetchStatus {
104
105
  readonly disabled: boolean;
105
106
  readonly failureCount: number;
106
107
  readonly nextAttemptAtMs?: number;
107
- readonly disabledReason?:
108
- | "rate-limit"
109
- | "server-error"
110
- | "credential-unavailable"
111
- | "malformed-response"
112
- | "network-error";
108
+ readonly disabledReason?: SharedUsageFailureReason;
113
109
  }
114
110
 
115
111
  export type UsageFetchResultStatus =
@@ -217,7 +213,10 @@ function retryAfterMs(response: UsageFetchResponse): number | undefined {
217
213
  }
218
214
  const seconds = Number(value);
219
215
  if (!Number.isFinite(seconds) || seconds < 0) return undefined;
220
- return Math.min(MAX_BACKOFF_MS, Math.max(0, Math.ceil(seconds * 1_000)));
216
+ return Math.min(
217
+ MAX_USAGE_ATTEMPT_DELAY_MS,
218
+ Math.max(0, Math.ceil(seconds * 1_000)),
219
+ );
221
220
  }
222
221
 
223
222
  /** Redacts bearer and access-token values before an error can escape this module. */
@@ -642,14 +641,37 @@ async function queryEndpoint(
642
641
  : normalizeAnthropicUsagePayload(payload);
643
642
  }
644
643
 
644
+ /**
645
+ * The attempt's deadline, or undefined when no honest ladder could have set it.
646
+ *
647
+ * The store already bounds `nextAttemptAtMs` against the record's own
648
+ * `observedAtMs` on append and read (internal issue #108). That span
649
+ * check alone is not enough for a reader: a record whose ORIGIN sits centuries
650
+ * ahead carries a legitimate-looking span and would still suppress polling for
651
+ * that account forever. No honest deadline lies further than one capped rung
652
+ * from now, so anything beyond that is ignored rather than trusted.
653
+ */
654
+ function plausibleAttemptDeadline(
655
+ attempt: SharedUsageAttemptRecord | undefined,
656
+ nowMs: number,
657
+ ): number | undefined {
658
+ if (attempt === undefined) return undefined;
659
+ if (attempt.nextAttemptAtMs > nowMs + MAX_USAGE_ATTEMPT_DELAY_MS) {
660
+ return undefined;
661
+ }
662
+ return attempt.nextAttemptAtMs;
663
+ }
664
+
645
665
  function statusFromAttempt(
646
666
  attempt: SharedUsageAttemptRecord | undefined,
647
667
  enabled: boolean,
668
+ nowMs: number,
648
669
  ): UsageFetchStatus {
649
670
  const nextAttemptAtMs =
650
- attempt?.failureCount === 0 ? undefined : attempt?.nextAttemptAtMs;
651
- const disabledReason =
652
- attempt?.failureReason as UsageFetchStatus["disabledReason"];
671
+ attempt?.failureCount === 0
672
+ ? undefined
673
+ : plausibleAttemptDeadline(attempt, nowMs);
674
+ const disabledReason = attempt?.failureReason;
653
675
  return {
654
676
  enabled,
655
677
  disabled: attempt?.disabled ?? false,
@@ -833,6 +855,7 @@ export class UsageFetcher {
833
855
  return statusFromAttempt(
834
856
  this.#sharedStore.latestAttempt(providerId, family),
835
857
  enabled,
858
+ this.#now(),
836
859
  );
837
860
  }
838
861
 
@@ -997,11 +1020,9 @@ export class UsageFetcher {
997
1020
  // forever. The bound belongs here, on the record, before any clause
998
1021
  // reads it.
999
1022
  //
1000
- // `nextAttemptAtMs` is allowed one debounce interval of headroom because
1001
- // Only the ORIGIN timestamps are bounded, not `nextAttemptAtMs`: a real
1002
- // failure ladder legitimately schedules its next rung far ahead, up to
1003
- // MAX_BACKOFF_MS, and bounding that would break genuine backoff
1004
- // suppression. An honest record cannot have been OBSERVED in the future.
1023
+ // `nextAttemptAtMs` is bounded relative to `observedAtMs` by the store and
1024
+ // checked again here before it may suppress a send. An honest record cannot
1025
+ // have been OBSERVED in the future either.
1005
1026
  if (
1006
1027
  attempt.observedAtMs > nowMs ||
1007
1028
  (attempt.failureTriggeredAtMs !== undefined &&
@@ -1009,7 +1030,12 @@ export class UsageFetcher {
1009
1030
  ) {
1010
1031
  return undefined;
1011
1032
  }
1012
- if (attempt.failureCount > 0 && attempt.nextAttemptAtMs > nowMs) {
1033
+ const nextAttemptAtMs = plausibleAttemptDeadline(attempt, nowMs);
1034
+ if (
1035
+ attempt.failureCount > 0 &&
1036
+ nextAttemptAtMs !== undefined &&
1037
+ nextAttemptAtMs > nowMs
1038
+ ) {
1013
1039
  return "backoff";
1014
1040
  }
1015
1041
  if (attempt.observedAtMs >= failedAtMs) return "already-answered";
@@ -1190,8 +1216,13 @@ export class UsageFetcher {
1190
1216
  // Once the recorded backoff has elapsed, the account is retried whether or
1191
1217
  // not it was disabled. A retry that fails again simply re-arms the ladder at
1192
1218
  // its capped rung, so a genuinely dead endpoint is polled at most once per
1193
- // MAX_BACKOFF_MS rather than hammered.
1194
- if (prior !== undefined && prior.nextAttemptAtMs > nowMs) {
1219
+ // MAX_USAGE_ATTEMPT_DELAY_MS rather than hammered.
1220
+ const priorDeadline = plausibleAttemptDeadline(prior, nowMs);
1221
+ if (
1222
+ prior !== undefined &&
1223
+ priorDeadline !== undefined &&
1224
+ priorDeadline > nowMs
1225
+ ) {
1195
1226
  await this.#persistWindowGap(
1196
1227
  account,
1197
1228
  nowMs,
@@ -1223,7 +1254,13 @@ export class UsageFetcher {
1223
1254
  // over a stale `disabled` flag, so a recovered account can be observed
1224
1255
  // again. Re-read under the lease because a peer may have attempted in
1225
1256
  // between.
1226
- if (current !== undefined && current.nextAttemptAtMs > this.#now()) {
1257
+ const currentNowMs = this.#now();
1258
+ const currentDeadline = plausibleAttemptDeadline(current, currentNowMs);
1259
+ if (
1260
+ current !== undefined &&
1261
+ currentDeadline !== undefined &&
1262
+ currentDeadline > currentNowMs
1263
+ ) {
1227
1264
  await this.#persistWindowGap(
1228
1265
  account,
1229
1266
  this.#now(),
@@ -1555,7 +1592,7 @@ export class UsageFetcher {
1555
1592
  : new UsageEndpointError("usage endpoint failed", "network-error");
1556
1593
  const failureCount = (prior?.failureCount ?? 0) + 1;
1557
1594
  const backoff = Math.min(
1558
- MAX_BACKOFF_MS,
1595
+ MAX_USAGE_ATTEMPT_DELAY_MS,
1559
1596
  BASE_BACKOFF_MS * 2 ** Math.max(0, failureCount - 1),
1560
1597
  );
1561
1598
  const disabled = failureCount >= USAGE_FETCH_DISABLE_AFTER_FAILURES;
package/src/usage.ts CHANGED
@@ -337,8 +337,8 @@ export class UsageLedger {
337
337
  * exhaustion hold is one. But a hold is machine-global: deleting the record
338
338
  * would revoke it for every other session on the machine, and those peers
339
339
  * may still be refusing work on that account. So the override is
340
- * PROCESS-LOCAL -- this session stops honouring holds installed up to the
341
- * moment of the clear, and peers keep theirs.
340
+ * PROCESS-LOCAL -- this session stops honouring holds whose failure happened
341
+ * at or before the clear, and peers keep theirs.
342
342
  *
343
343
  * Stored as the clear time rather than a boolean so a LATER failure still
344
344
  * installs a hold this session honours. A boolean would make one `clear`
@@ -346,6 +346,13 @@ export class UsageLedger {
346
346
  * which is a permanent effect from a transient instruction.
347
347
  */
348
348
  readonly #holdOverrides = new Map<string, number>();
349
+ /**
350
+ * When `clear()` with no account last ran. Clear-all is one point in time
351
+ * for EVERY account, including ones with no hold visible at that instant:
352
+ * enumerating the holds present then let a peer publish a pre-clear hold a
353
+ * moment later and still have it honoured (internal issue #110).
354
+ */
355
+ #clearAllAtMs: number | undefined;
349
356
  readonly #sharedStore: SharedUsageStore | undefined;
350
357
  readonly #now: () => number;
351
358
 
@@ -714,29 +721,29 @@ export class UsageLedger {
714
721
  family: AllowedFamily,
715
722
  nowMs = this.#now(),
716
723
  ): number | undefined {
717
- const holdUntilMs = this.#sharedStore?.activeExhaustionHoldUntilMs(
724
+ // An operator `clear` in THIS process stops us honouring holds whose
725
+ // failure happened at or before it. A hold from a LATER failure is
726
+ // honoured again: the override is a point in time, not a permanent
727
+ // exemption. Peers keep honouring the record either way, because it is
728
+ // not deleted.
729
+ //
730
+ // The store compares each hold's own validated `failedAtMs`. Deriving
731
+ // the failure time from the deadline (`holdUntilMs - EXHAUSTION_HOLD_MS`)
732
+ // is only right for an exact one-hour hold, and the store accepts every
733
+ // shorter span (internal issue #110).
734
+ const accountClearedAtMs = this.#holdOverrides.get(providerId);
735
+ const clearedAtMs =
736
+ accountClearedAtMs === undefined
737
+ ? this.#clearAllAtMs
738
+ : this.#clearAllAtMs === undefined
739
+ ? accountClearedAtMs
740
+ : Math.max(accountClearedAtMs, this.#clearAllAtMs);
741
+ return this.#sharedStore?.activeExhaustionHoldUntilMs(
718
742
  providerId,
719
743
  family,
720
744
  nowMs,
745
+ clearedAtMs,
721
746
  );
722
- if (holdUntilMs === undefined) return undefined;
723
- // An operator `clear` in THIS process stops us honouring holds that were
724
- // already installed when it ran. A hold from a LATER failure is honoured
725
- // again: the override is a point in time, not a permanent exemption.
726
- // Peers keep honouring the record either way, because it is not deleted.
727
- //
728
- // `holdUntilMs - EXHAUSTION_HOLD_MS` recovers the failure time from the
729
- // reported deadline without a second store read. The writer always sets
730
- // exactly that span (asserted by the hold-duration test), and the
731
- // store's own read bound refuses any record claiming more.
732
- const clearedAtMs = this.#holdOverrides.get(providerId);
733
- if (
734
- clearedAtMs !== undefined &&
735
- holdUntilMs - EXHAUSTION_HOLD_MS <= clearedAtMs
736
- ) {
737
- return undefined;
738
- }
739
- return holdUntilMs;
740
747
  }
741
748
 
742
749
  /**
@@ -811,11 +818,7 @@ export class UsageLedger {
811
818
  if (providerId === undefined) {
812
819
  this.#snapshots.clear();
813
820
  this.#quotaObservations.clear();
814
- for (const id of this.#sharedStore
815
- ?.readExhaustionHolds()
816
- .map((hold) => hold.providerId) ?? []) {
817
- this.#holdOverrides.set(id, nowMs);
818
- }
821
+ this.#clearAllAtMs = Math.max(this.#clearAllAtMs ?? nowMs, nowMs);
819
822
  } else {
820
823
  this.#snapshots.delete(providerId);
821
824
  this.#quotaObservations.delete(providerId);