@centerforagenticai/pi-multi-account 0.1.2 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/packages/pi-anthropic-oauth/src/stream.ts +8 -0
- package/src/anthropic-adaptive-stream.ts +81 -15
- package/src/anthropic-alias-stream.ts +9 -1
- package/src/codex-adapter.ts +109 -11
- package/src/config.ts +151 -0
- package/src/error-classification.ts +13 -1
- package/src/host-final-stop-message.ts +136 -0
- package/src/index.ts +41 -4
- package/src/logical-provider.ts +157 -2
- package/src/model-fallback-policy.ts +384 -0
- package/src/recovery-engine.ts +354 -68
- package/src/recovery-plan.ts +7 -1
- package/src/recovery-send-evidence.ts +29 -0
- package/src/refusal-advice.ts +139 -0
- package/src/routing.ts +17 -1
- package/src/shared-usage.ts +100 -176
- package/src/usage-fetch.ts +52 -32
- package/src/usage.ts +29 -26
package/src/usage-fetch.ts
CHANGED
|
@@ -6,8 +6,10 @@ import {
|
|
|
6
6
|
type MachineLeaseHandle,
|
|
7
7
|
} from "./machine-lease.js";
|
|
8
8
|
import {
|
|
9
|
+
MAX_USAGE_ATTEMPT_DELAY_MS,
|
|
9
10
|
normalizeUsageEndpointPercent,
|
|
10
11
|
type SharedUsageAttemptRecord,
|
|
12
|
+
type SharedUsageFailureReason,
|
|
11
13
|
type SharedUsageStore,
|
|
12
14
|
type UsageFailureDetail,
|
|
13
15
|
} from "./shared-usage.js";
|
|
@@ -39,9 +41,8 @@ export const USAGE_FETCH_TIMEOUT_MS = 10_000;
|
|
|
39
41
|
const MAX_RESPONSE_BYTES = 128 * 1024;
|
|
40
42
|
const MAX_ERROR_CHARS = 512;
|
|
41
43
|
const BASE_BACKOFF_MS = 30_000;
|
|
42
|
-
const MAX_BACKOFF_MS = 15 * 60_000;
|
|
43
44
|
const BACKOFF_LADDER_STEPS =
|
|
44
|
-
Math.ceil(Math.log2(
|
|
45
|
+
Math.ceil(Math.log2(MAX_USAGE_ATTEMPT_DELAY_MS / BASE_BACKOFF_MS)) + 1;
|
|
45
46
|
/** Disable only after the capped rung has failed once more. */
|
|
46
47
|
export const USAGE_FETCH_DISABLE_AFTER_FAILURES = BACKOFF_LADDER_STEPS + 1;
|
|
47
48
|
|
|
@@ -104,12 +105,7 @@ export interface UsageFetchStatus {
|
|
|
104
105
|
readonly disabled: boolean;
|
|
105
106
|
readonly failureCount: number;
|
|
106
107
|
readonly nextAttemptAtMs?: number;
|
|
107
|
-
readonly disabledReason?:
|
|
108
|
-
| "rate-limit"
|
|
109
|
-
| "server-error"
|
|
110
|
-
| "credential-unavailable"
|
|
111
|
-
| "malformed-response"
|
|
112
|
-
| "network-error";
|
|
108
|
+
readonly disabledReason?: SharedUsageFailureReason;
|
|
113
109
|
}
|
|
114
110
|
|
|
115
111
|
export type UsageFetchResultStatus =
|
|
@@ -217,7 +213,10 @@ function retryAfterMs(response: UsageFetchResponse): number | undefined {
|
|
|
217
213
|
}
|
|
218
214
|
const seconds = Number(value);
|
|
219
215
|
if (!Number.isFinite(seconds) || seconds < 0) return undefined;
|
|
220
|
-
return Math.min(
|
|
216
|
+
return Math.min(
|
|
217
|
+
MAX_USAGE_ATTEMPT_DELAY_MS,
|
|
218
|
+
Math.max(0, Math.ceil(seconds * 1_000)),
|
|
219
|
+
);
|
|
221
220
|
}
|
|
222
221
|
|
|
223
222
|
/** Redacts bearer and access-token values before an error can escape this module. */
|
|
@@ -642,14 +641,28 @@ async function queryEndpoint(
|
|
|
642
641
|
: normalizeAnthropicUsagePayload(payload);
|
|
643
642
|
}
|
|
644
643
|
|
|
644
|
+
// Every attempt a reader here sees comes from `SharedUsageStore.latestAttempt`
|
|
645
|
+
// called with that reader's own clock, and that is what keeps a far-future
|
|
646
|
+
// deadline (internal issue #108) from suppressing polling:
|
|
647
|
+
//
|
|
648
|
+
// - the store refuses, on append and read, any attempt whose
|
|
649
|
+
// `nextAttemptAtMs - observedAtMs` exceeds MAX_USAGE_ATTEMPT_DELAY_MS; and
|
|
650
|
+
// - `latestAttempt` skips any attempt observed after `nowMs` (#133).
|
|
651
|
+
//
|
|
652
|
+
// Together these give `nextAttemptAtMs <= nowMs + MAX_USAGE_ATTEMPT_DELAY_MS`
|
|
653
|
+
// for every record a reader receives. A per-reader "plausible deadline" check
|
|
654
|
+
// used to restate that bound; once the origin filter moved into the store it
|
|
655
|
+
// could no longer fail on any reachable input and was removed (#133,
|
|
656
|
+
// CR-REFRESH-READER-BOUND-UNREACHABLE). Read attempts only through
|
|
657
|
+
// `latestAttempt(providerId, family, nowMs)` with the clock the decision uses.
|
|
658
|
+
|
|
645
659
|
function statusFromAttempt(
|
|
646
660
|
attempt: SharedUsageAttemptRecord | undefined,
|
|
647
661
|
enabled: boolean,
|
|
648
662
|
): UsageFetchStatus {
|
|
649
663
|
const nextAttemptAtMs =
|
|
650
664
|
attempt?.failureCount === 0 ? undefined : attempt?.nextAttemptAtMs;
|
|
651
|
-
const disabledReason =
|
|
652
|
-
attempt?.failureReason as UsageFetchStatus["disabledReason"];
|
|
665
|
+
const disabledReason = attempt?.failureReason;
|
|
653
666
|
return {
|
|
654
667
|
enabled,
|
|
655
668
|
disabled: attempt?.disabled ?? false,
|
|
@@ -831,7 +844,7 @@ export class UsageFetcher {
|
|
|
831
844
|
): UsageFetchStatus {
|
|
832
845
|
const enabled = config.usageFetchEnabled?.[family] ?? true;
|
|
833
846
|
return statusFromAttempt(
|
|
834
|
-
this.#sharedStore.latestAttempt(providerId, family),
|
|
847
|
+
this.#sharedStore.latestAttempt(providerId, family, this.#now()),
|
|
835
848
|
enabled,
|
|
836
849
|
);
|
|
837
850
|
}
|
|
@@ -994,18 +1007,16 @@ export class UsageFetcher {
|
|
|
994
1007
|
// the debounce clause, so `already-answered` and `backoff` returned
|
|
995
1008
|
// before it ran: an attempt with `observedAtMs` centuries ahead is
|
|
996
1009
|
// trivially `>= failedAtMs`, and suppressed every failure refresh
|
|
997
|
-
// forever.
|
|
998
|
-
//
|
|
1010
|
+
// forever. `latestAttempt(..., nowMs)` now refuses such a record at the
|
|
1011
|
+
// source (#133), which also bounds `nextAttemptAtMs` to one capped rung
|
|
1012
|
+
// past `nowMs` (see the note above `statusFromAttempt`).
|
|
999
1013
|
//
|
|
1000
|
-
//
|
|
1001
|
-
//
|
|
1002
|
-
//
|
|
1003
|
-
// MAX_BACKOFF_MS, and bounding that would break genuine backoff
|
|
1004
|
-
// suppression. An honest record cannot have been OBSERVED in the future.
|
|
1014
|
+
// The failure trigger is not tied to `observedAtMs` by the store, so a
|
|
1015
|
+
// record observed in the past can still name a failure in the future;
|
|
1016
|
+
// such a marker cannot be honest and must not arm the debounce.
|
|
1005
1017
|
if (
|
|
1006
|
-
attempt.
|
|
1007
|
-
|
|
1008
|
-
attempt.failureTriggeredAtMs > nowMs)
|
|
1018
|
+
attempt.failureTriggeredAtMs !== undefined &&
|
|
1019
|
+
attempt.failureTriggeredAtMs > nowMs
|
|
1009
1020
|
) {
|
|
1010
1021
|
return undefined;
|
|
1011
1022
|
}
|
|
@@ -1044,15 +1055,17 @@ export class UsageFetcher {
|
|
|
1044
1055
|
}
|
|
1045
1056
|
|
|
1046
1057
|
// Step 4/5: durable checks before the lease.
|
|
1058
|
+
const beforeLeaseNowMs = this.#now();
|
|
1047
1059
|
const beforeLease = this.#sharedStore.latestAttempt(
|
|
1048
1060
|
account.providerId,
|
|
1049
1061
|
account.family,
|
|
1062
|
+
beforeLeaseNowMs,
|
|
1050
1063
|
);
|
|
1051
1064
|
if (
|
|
1052
1065
|
this.#failureRefreshSuppressedBy(
|
|
1053
1066
|
beforeLease,
|
|
1054
1067
|
failedAtMs,
|
|
1055
|
-
|
|
1068
|
+
beforeLeaseNowMs,
|
|
1056
1069
|
) !== undefined
|
|
1057
1070
|
) {
|
|
1058
1071
|
return { providerId: account.providerId, status: "not-due" };
|
|
@@ -1072,15 +1085,17 @@ export class UsageFetcher {
|
|
|
1072
1085
|
const leaseGuard: AntigravityLeaseGuard = { lease, handedOff: false };
|
|
1073
1086
|
try {
|
|
1074
1087
|
// Step 7: the same checks again, now that nobody else can write.
|
|
1088
|
+
const underLeaseNowMs = this.#now();
|
|
1075
1089
|
const underLease = this.#sharedStore.latestAttempt(
|
|
1076
1090
|
account.providerId,
|
|
1077
1091
|
account.family,
|
|
1092
|
+
underLeaseNowMs,
|
|
1078
1093
|
);
|
|
1079
1094
|
if (
|
|
1080
1095
|
this.#failureRefreshSuppressedBy(
|
|
1081
1096
|
underLease,
|
|
1082
1097
|
failedAtMs,
|
|
1083
|
-
|
|
1098
|
+
underLeaseNowMs,
|
|
1084
1099
|
) !== undefined
|
|
1085
1100
|
) {
|
|
1086
1101
|
return { providerId: account.providerId, status: "not-due" };
|
|
@@ -1169,6 +1184,7 @@ export class UsageFetcher {
|
|
|
1169
1184
|
const prior = this.#sharedStore.latestAttempt(
|
|
1170
1185
|
account.providerId,
|
|
1171
1186
|
account.family,
|
|
1187
|
+
nowMs,
|
|
1172
1188
|
);
|
|
1173
1189
|
// Backoff is checked BEFORE `disabled`, deliberately.
|
|
1174
1190
|
//
|
|
@@ -1190,7 +1206,7 @@ export class UsageFetcher {
|
|
|
1190
1206
|
// Once the recorded backoff has elapsed, the account is retried whether or
|
|
1191
1207
|
// not it was disabled. A retry that fails again simply re-arms the ladder at
|
|
1192
1208
|
// its capped rung, so a genuinely dead endpoint is polled at most once per
|
|
1193
|
-
//
|
|
1209
|
+
// MAX_USAGE_ATTEMPT_DELAY_MS rather than hammered.
|
|
1194
1210
|
if (prior !== undefined && prior.nextAttemptAtMs > nowMs) {
|
|
1195
1211
|
await this.#persistWindowGap(
|
|
1196
1212
|
account,
|
|
@@ -1215,15 +1231,17 @@ export class UsageFetcher {
|
|
|
1215
1231
|
}
|
|
1216
1232
|
const leaseGuard: AntigravityLeaseGuard = { lease, handedOff: false };
|
|
1217
1233
|
try {
|
|
1218
|
-
const current = this.#sharedStore.latestAttempt(
|
|
1219
|
-
account.providerId,
|
|
1220
|
-
account.family,
|
|
1221
|
-
);
|
|
1222
1234
|
// Same ordering as the pre-lease check above: an elapsed backoff wins
|
|
1223
1235
|
// over a stale `disabled` flag, so a recovered account can be observed
|
|
1224
1236
|
// again. Re-read under the lease because a peer may have attempted in
|
|
1225
1237
|
// between.
|
|
1226
|
-
|
|
1238
|
+
const currentNowMs = this.#now();
|
|
1239
|
+
const current = this.#sharedStore.latestAttempt(
|
|
1240
|
+
account.providerId,
|
|
1241
|
+
account.family,
|
|
1242
|
+
currentNowMs,
|
|
1243
|
+
);
|
|
1244
|
+
if (current !== undefined && current.nextAttemptAtMs > currentNowMs) {
|
|
1227
1245
|
await this.#persistWindowGap(
|
|
1228
1246
|
account,
|
|
1229
1247
|
this.#now(),
|
|
@@ -1555,7 +1573,7 @@ export class UsageFetcher {
|
|
|
1555
1573
|
: new UsageEndpointError("usage endpoint failed", "network-error");
|
|
1556
1574
|
const failureCount = (prior?.failureCount ?? 0) + 1;
|
|
1557
1575
|
const backoff = Math.min(
|
|
1558
|
-
|
|
1576
|
+
MAX_USAGE_ATTEMPT_DELAY_MS,
|
|
1559
1577
|
BASE_BACKOFF_MS * 2 ** Math.max(0, failureCount - 1),
|
|
1560
1578
|
);
|
|
1561
1579
|
const disabled = failureCount >= USAGE_FETCH_DISABLE_AFTER_FAILURES;
|
|
@@ -1617,13 +1635,15 @@ export class UsageFetcher {
|
|
|
1617
1635
|
}
|
|
1618
1636
|
// REQ-WINDOW-CADENCE: enforce cross-process 15-minute floor.
|
|
1619
1637
|
// Check durable attempt record to prevent peer-process fetch within the window.
|
|
1638
|
+
const cadenceNowMs = this.#now();
|
|
1620
1639
|
const prior = this.#sharedStore.latestAttempt(
|
|
1621
1640
|
account.providerId,
|
|
1622
1641
|
account.family,
|
|
1642
|
+
cadenceNowMs,
|
|
1623
1643
|
);
|
|
1624
1644
|
if (
|
|
1625
1645
|
prior !== undefined &&
|
|
1626
|
-
|
|
1646
|
+
cadenceNowMs - prior.observedAtMs < WINDOW_SAMPLE_INTERVAL_MS
|
|
1627
1647
|
) {
|
|
1628
1648
|
continue;
|
|
1629
1649
|
}
|
package/src/usage.ts
CHANGED
|
@@ -337,8 +337,8 @@ export class UsageLedger {
|
|
|
337
337
|
* exhaustion hold is one. But a hold is machine-global: deleting the record
|
|
338
338
|
* would revoke it for every other session on the machine, and those peers
|
|
339
339
|
* may still be refusing work on that account. So the override is
|
|
340
|
-
* PROCESS-LOCAL -- this session stops honouring holds
|
|
341
|
-
*
|
|
340
|
+
* PROCESS-LOCAL -- this session stops honouring holds whose failure happened
|
|
341
|
+
* at or before the clear, and peers keep theirs.
|
|
342
342
|
*
|
|
343
343
|
* Stored as the clear time rather than a boolean so a LATER failure still
|
|
344
344
|
* installs a hold this session honours. A boolean would make one `clear`
|
|
@@ -346,6 +346,13 @@ export class UsageLedger {
|
|
|
346
346
|
* which is a permanent effect from a transient instruction.
|
|
347
347
|
*/
|
|
348
348
|
readonly #holdOverrides = new Map<string, number>();
|
|
349
|
+
/**
|
|
350
|
+
* When `clear()` with no account last ran. Clear-all is one point in time
|
|
351
|
+
* for EVERY account, including ones with no hold visible at that instant:
|
|
352
|
+
* enumerating the holds present then let a peer publish a pre-clear hold a
|
|
353
|
+
* moment later and still have it honoured (internal issue #110).
|
|
354
|
+
*/
|
|
355
|
+
#clearAllAtMs: number | undefined;
|
|
349
356
|
readonly #sharedStore: SharedUsageStore | undefined;
|
|
350
357
|
readonly #now: () => number;
|
|
351
358
|
|
|
@@ -714,29 +721,29 @@ export class UsageLedger {
|
|
|
714
721
|
family: AllowedFamily,
|
|
715
722
|
nowMs = this.#now(),
|
|
716
723
|
): number | undefined {
|
|
717
|
-
|
|
724
|
+
// An operator `clear` in THIS process stops us honouring holds whose
|
|
725
|
+
// failure happened at or before it. A hold from a LATER failure is
|
|
726
|
+
// honoured again: the override is a point in time, not a permanent
|
|
727
|
+
// exemption. Peers keep honouring the record either way, because it is
|
|
728
|
+
// not deleted.
|
|
729
|
+
//
|
|
730
|
+
// The store compares each hold's own validated `failedAtMs`. Deriving
|
|
731
|
+
// the failure time from the deadline (`holdUntilMs - EXHAUSTION_HOLD_MS`)
|
|
732
|
+
// is only right for an exact one-hour hold, and the store accepts every
|
|
733
|
+
// shorter span (internal issue #110).
|
|
734
|
+
const accountClearedAtMs = this.#holdOverrides.get(providerId);
|
|
735
|
+
const clearedAtMs =
|
|
736
|
+
accountClearedAtMs === undefined
|
|
737
|
+
? this.#clearAllAtMs
|
|
738
|
+
: this.#clearAllAtMs === undefined
|
|
739
|
+
? accountClearedAtMs
|
|
740
|
+
: Math.max(accountClearedAtMs, this.#clearAllAtMs);
|
|
741
|
+
return this.#sharedStore?.activeExhaustionHoldUntilMs(
|
|
718
742
|
providerId,
|
|
719
743
|
family,
|
|
720
744
|
nowMs,
|
|
745
|
+
clearedAtMs,
|
|
721
746
|
);
|
|
722
|
-
if (holdUntilMs === undefined) return undefined;
|
|
723
|
-
// An operator `clear` in THIS process stops us honouring holds that were
|
|
724
|
-
// already installed when it ran. A hold from a LATER failure is honoured
|
|
725
|
-
// again: the override is a point in time, not a permanent exemption.
|
|
726
|
-
// Peers keep honouring the record either way, because it is not deleted.
|
|
727
|
-
//
|
|
728
|
-
// `holdUntilMs - EXHAUSTION_HOLD_MS` recovers the failure time from the
|
|
729
|
-
// reported deadline without a second store read. The writer always sets
|
|
730
|
-
// exactly that span (asserted by the hold-duration test), and the
|
|
731
|
-
// store's own read bound refuses any record claiming more.
|
|
732
|
-
const clearedAtMs = this.#holdOverrides.get(providerId);
|
|
733
|
-
if (
|
|
734
|
-
clearedAtMs !== undefined &&
|
|
735
|
-
holdUntilMs - EXHAUSTION_HOLD_MS <= clearedAtMs
|
|
736
|
-
) {
|
|
737
|
-
return undefined;
|
|
738
|
-
}
|
|
739
|
-
return holdUntilMs;
|
|
740
747
|
}
|
|
741
748
|
|
|
742
749
|
/**
|
|
@@ -811,11 +818,7 @@ export class UsageLedger {
|
|
|
811
818
|
if (providerId === undefined) {
|
|
812
819
|
this.#snapshots.clear();
|
|
813
820
|
this.#quotaObservations.clear();
|
|
814
|
-
|
|
815
|
-
?.readExhaustionHolds()
|
|
816
|
-
.map((hold) => hold.providerId) ?? []) {
|
|
817
|
-
this.#holdOverrides.set(id, nowMs);
|
|
818
|
-
}
|
|
821
|
+
this.#clearAllAtMs = Math.max(this.#clearAllAtMs ?? nowMs, nowMs);
|
|
819
822
|
} else {
|
|
820
823
|
this.#snapshots.delete(providerId);
|
|
821
824
|
this.#quotaObservations.delete(providerId);
|