maxpool 1.5.16 → 1.5.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/account-manager.js +160 -17
- package/src/index.js +3 -0
- package/src/oauth.js +55 -0
- package/src/prober.js +23 -6
- package/src/tui.js +39 -11
package/package.json
CHANGED
package/src/account-manager.js
CHANGED
|
@@ -42,6 +42,19 @@ function emptyQuota() {
|
|
|
42
42
|
scopedWeekly: {},
|
|
43
43
|
unifiedStatus: null, // allowed | allowed_warning | rejected
|
|
44
44
|
resetsAt: null,
|
|
45
|
+
// Provider (z.ai / Kimi) quota — kept SEPARATE from unified* so a provider
|
|
46
|
+
// reading never leaks into the OAuth quota gates (_isAvailable / _weeklyRawState
|
|
47
|
+
// / _accountScarcity read unified* only). z.ai is pollable; Kimi is not.
|
|
48
|
+
providerSes: null, // utilization 0-1 (z.ai 5h token window)
|
|
49
|
+
providerSesReset: null, // ms
|
|
50
|
+
providerWk: null, // utilization 0-1 (z.ai weekly), null if plan has none
|
|
51
|
+
providerWkReset: null, // ms
|
|
52
|
+
providerQuotaSource: null, // 'zai' (pollable) | 'console-only' (kimi) | null
|
|
53
|
+
// Freshness: last time a background usage PROBE succeeded for this account
|
|
54
|
+
// (oauth fetchUsage OR provider fetchProviderUsage). Drives the TUI staleness
|
|
55
|
+
// marker — a swallowed failing probe no longer silently freezes a stale tag.
|
|
56
|
+
// Header-driven updates do NOT stamp this (headers can't refresh scoped/provider).
|
|
57
|
+
lastProbeOkAt: null, // ms
|
|
45
58
|
};
|
|
46
59
|
}
|
|
47
60
|
|
|
@@ -435,6 +448,10 @@ export class AccountManager {
|
|
|
435
448
|
hasAvailableRoute(requestInfo = {}, excludedIndexes = new Set()) {
|
|
436
449
|
this.refreshExpiredQuotas();
|
|
437
450
|
const profile = requestInfo.profile || 'claude';
|
|
451
|
+
// Route-EXISTENCE check only (order-independent `.some`): unlike the acquire
|
|
452
|
+
// path's re-home loop, pass ORDER doesn't matter here — the bound 2-entry set
|
|
453
|
+
// covers the same accounts (reserve+critical) — so it intentionally is NOT
|
|
454
|
+
// unified to the healthy-first ladder. Do not "sync" these.
|
|
438
455
|
const hasBinding = Boolean(requestInfo.sessionKey && this.sessionBindings.has(requestInfo.sessionKey));
|
|
439
456
|
const weeklyPasses = hasBinding
|
|
440
457
|
? [
|
|
@@ -589,8 +606,14 @@ export class AccountManager {
|
|
|
589
606
|
if (!fam) return null;
|
|
590
607
|
const e = account.quota?.scopedWeekly?.[fam];
|
|
591
608
|
if (!e || e.isActive === false) return null;
|
|
592
|
-
|
|
593
|
-
|
|
609
|
+
// Bench ONLY at genuine exhaustion (>= weeklyExhaustedThreshold). Anthropic
|
|
610
|
+
// labels a scoped weekly `severity:'critical'` well before it's actually
|
|
611
|
+
// capped (~90%), where the model still has headroom and is still served — a
|
|
612
|
+
// hard bench there strands the remainder and mislabels the account "maxed".
|
|
613
|
+
// A real scoped 429 writes utilization:1 (markRateLimited), so genuine
|
|
614
|
+
// exhaustion is still caught by the threshold. Same predicate drives the TUI
|
|
615
|
+
// `maxed` tag, so "maxed" renders iff the model is actually benched.
|
|
616
|
+
const exhausted = e.utilization != null && e.utilization >= this.scheduler.weeklyExhaustedThreshold;
|
|
594
617
|
return exhausted ? { resetAt: e.resetAt || null, family: fam } : null;
|
|
595
618
|
}
|
|
596
619
|
|
|
@@ -1087,6 +1110,11 @@ export class AccountManager {
|
|
|
1087
1110
|
// Keep a signed-thinking session's live migration on Claude accounts only —
|
|
1088
1111
|
// don't shuttle an Anthropic-signed block onto a provider mid-session even
|
|
1089
1112
|
// though _isRequestCompatible now allows providers for thinking under policy.
|
|
1113
|
+
// LOAD-BEARING for the absolute-near-cap `return true` below: it's what
|
|
1114
|
+
// guarantees that path only fires with a CLAUDE target present (a signed
|
|
1115
|
+
// session with no healthy Claude alt → bestScore stays Infinity → return
|
|
1116
|
+
// false → stays on its Claude account, never re-homed to a provider). Do not
|
|
1117
|
+
// remove assuming the fall-through protects signed thinking — it does not.
|
|
1090
1118
|
if (requestInfo.requiresAnthropicThinkingIntegrity === true && account.type === 'provider') continue;
|
|
1091
1119
|
if (!this._matchesRequest(account, profile, requestInfo)) continue;
|
|
1092
1120
|
// Genuinely-healthy alternatives only (normal/soft/unknown weekly + model headroom).
|
|
@@ -1097,8 +1125,26 @@ export class AccountManager {
|
|
|
1097
1125
|
bestTier = WEEKLY_TIER[this._weeklyPaceState(account)] ?? 0;
|
|
1098
1126
|
}
|
|
1099
1127
|
}
|
|
1100
|
-
if (!Number.isFinite(bestScore)) return false; // no healthy alternative
|
|
1101
|
-
|
|
1128
|
+
if (!Number.isFinite(bestScore)) return false; // no RAW-healthy alternative → don't move (no stranding)
|
|
1129
|
+
|
|
1130
|
+
// Absolute near-cap: the bound account is genuinely low on weekly headroom
|
|
1131
|
+
// (RAW reserve+, not merely burning fast). Every candidate the loop kept is RAW
|
|
1132
|
+
// normal/soft, so this is a STRICT absolute-headroom improvement and flap-stable
|
|
1133
|
+
// (a reserve account can never be a target → no bounce-back). Skip the
|
|
1134
|
+
// concurrency-relief margin: it's calibrated for moving off a LOADED account and
|
|
1135
|
+
// is UNSATISFIABLE for an idle near-cap one — an idle boundScore ≈ the ~2
|
|
1136
|
+
// concurrency floor, so boundScore*0.5 ≈ 1 sits below every account's minimum
|
|
1137
|
+
// score, pinning the session to the near-exhausted account. Preserving the thin
|
|
1138
|
+
// remaining weekly headroom dominates concurrency spread; the per-request
|
|
1139
|
+
// candidate loop lands on the least-loaded healthy account, spreading the move.
|
|
1140
|
+
if ((WEEKLY_TIER[this._weeklyRawState(bound)] ?? 0) >= WEEKLY_TIER.reserve) return true;
|
|
1141
|
+
|
|
1142
|
+
// Otherwise the trigger was PACE-only on a RAW-healthy account — a fast-burner
|
|
1143
|
+
// that still has real absolute headroom (RAW soft but pace reserve/critical,
|
|
1144
|
+
// e.g. 79% used resetting in ~3.5d). Keep the conservative gate so it isn't
|
|
1145
|
+
// churned off an account that's genuinely fine: move only for a clearly-cheaper,
|
|
1146
|
+
// strictly-healthier-pace-tier target. (A RAW-soft/pace-normal fast-burner with
|
|
1147
|
+
// lots of headroom never even reaches here — _isBoundAccountHot gates it out.)
|
|
1102
1148
|
return bestScore <= boundScore * REBALANCE_SCORE_MARGIN
|
|
1103
1149
|
&& (boundScore - bestScore) >= REBALANCE_MIN_ABS_GAP
|
|
1104
1150
|
&& bestTier < boundTier;
|
|
@@ -1244,7 +1290,6 @@ export class AccountManager {
|
|
|
1244
1290
|
}
|
|
1245
1291
|
}
|
|
1246
1292
|
|
|
1247
|
-
const hasBinding = Boolean(requestInfo.sessionKey && this.sessionBindings.has(requestInfo.sessionKey));
|
|
1248
1293
|
const preferred = this._preferredAccount(profile, excludedIndexes, requestInfo);
|
|
1249
1294
|
if (preferred) {
|
|
1250
1295
|
const preferredPasses = [
|
|
@@ -1265,16 +1310,17 @@ export class AccountManager {
|
|
|
1265
1310
|
// Else fall through to the candidate score loop, which re-homes the session
|
|
1266
1311
|
// onto the best healthy account via _bindSession on acquire.
|
|
1267
1312
|
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1313
|
+
// Healthy-first ladder for BOTH bound and unbound sessions. A bound session
|
|
1314
|
+
// only reaches this loop once we've decided to LEAVE its account (rebalance
|
|
1315
|
+
// fired / a higher-priority account is available / the bound account is down) —
|
|
1316
|
+
// the sticky "stay put" path returns above without reaching here — so re-homing
|
|
1317
|
+
// must prefer a genuinely-healthy account and fall back to reserve/critical only
|
|
1318
|
+
// if none exists, never re-pick the idle reserve account it's leaving.
|
|
1319
|
+
const weeklyPasses = [
|
|
1320
|
+
{ allowWeeklyReserve: false, allowWeeklyCritical: false },
|
|
1321
|
+
{ allowWeeklyReserve: true, allowWeeklyCritical: false },
|
|
1322
|
+
{ allowWeeklyReserve: true, allowWeeklyCritical: true },
|
|
1323
|
+
];
|
|
1278
1324
|
|
|
1279
1325
|
for (const weeklyOptions of weeklyPasses) {
|
|
1280
1326
|
best = null;
|
|
@@ -1648,6 +1694,16 @@ export class AccountManager {
|
|
|
1648
1694
|
// soft de-preference of accounts burning ahead of an even pace. Never a bench.
|
|
1649
1695
|
const paceCost = this._accountScarcity(account, now) * this.scheduler.paceCostWeight;
|
|
1650
1696
|
|
|
1697
|
+
// Per-model weekly de-preference: an account whose scoped weekly for THIS
|
|
1698
|
+
// request's model (e.g. Fable) is high-but-not-exhausted is a poor pick for
|
|
1699
|
+
// that model — shed its load toward healthier accounts BEFORE the hard bench
|
|
1700
|
+
// at weeklyExhaustedThreshold, so it's chosen only as overflow (rarely
|
|
1701
|
+
// re-429ing). Scoped weekly is otherwise absent from scoring. Soft, never a
|
|
1702
|
+
// bench; 0 below reserve and for models with no scoped cap.
|
|
1703
|
+
const scopedPace = requestInfo.model
|
|
1704
|
+
? this._scopedScarcity(account, requestInfo.model, now) * this.scheduler.paceCostWeight
|
|
1705
|
+
: 0;
|
|
1706
|
+
|
|
1651
1707
|
const fleetRecentWeight = ctx?.fleetRecentWeight ?? 0;
|
|
1652
1708
|
const recentWeight = this._loadSummary(account, this.scheduler.spreadWindowMs, now).weight;
|
|
1653
1709
|
const share = fleetRecentWeight > 0 ? recentWeight / fleetRecentWeight : 0;
|
|
@@ -1663,7 +1719,22 @@ export class AccountManager {
|
|
|
1663
1719
|
// default) learns the real number within a cycle. `probing`/requalify still
|
|
1664
1720
|
// flags a never-seen account for learning — that path is unchanged.
|
|
1665
1721
|
|
|
1666
|
-
return concurrency + capPenalty + paceCost + spread + ramp + failurePenalty;
|
|
1722
|
+
return concurrency + capPenalty + paceCost + scopedPace + spread + ramp + failurePenalty;
|
|
1723
|
+
}
|
|
1724
|
+
|
|
1725
|
+
/**
|
|
1726
|
+
* Per-model weekly pace-overage for `model`'s family, or 0 when the account has
|
|
1727
|
+
* no scoped cap for it, the cap is inactive, or it's below the reserve tier
|
|
1728
|
+
* (plenty of headroom → no steering). Same pace discount as _windowScarcity, so
|
|
1729
|
+
* a scoped window about to reset is cheap to spend.
|
|
1730
|
+
*/
|
|
1731
|
+
_scopedScarcity(account, model, now = Date.now()) {
|
|
1732
|
+
const fam = modelFamily(model);
|
|
1733
|
+
if (!fam) return 0;
|
|
1734
|
+
const e = account.quota?.scopedWeekly?.[fam];
|
|
1735
|
+
if (!e || e.isActive === false || e.utilization == null) return 0;
|
|
1736
|
+
if (e.utilization < this.scheduler.weeklyReserveThreshold) return 0;
|
|
1737
|
+
return this._windowScarcity(e.utilization, e.resetAt, WEEK_MS, now);
|
|
1667
1738
|
}
|
|
1668
1739
|
|
|
1669
1740
|
/**
|
|
@@ -1795,8 +1866,15 @@ export class AccountManager {
|
|
|
1795
1866
|
}
|
|
1796
1867
|
// Per-model weekly sub-limits (Fable, Opus, ...). Replace wholesale with the
|
|
1797
1868
|
// fresh probe set so a family that dropped out of the response doesn't linger
|
|
1798
|
-
// stale; expiry on reset is a backstop for the between-probe window.
|
|
1869
|
+
// stale; expiry on reset is a backstop for the between-probe window. EXCEPTION:
|
|
1870
|
+
// a `reactive` scoped-429 bench is authoritative-high — while its resetAt is
|
|
1871
|
+
// still future, a lagging probe may neither lower it nor drop it (a probe that
|
|
1872
|
+
// omits the family, or reports the pre-429 level, would otherwise un-bench it
|
|
1873
|
+
// and trigger an immediate re-429 flap). _clearExpiredQuotas self-clears it at
|
|
1874
|
+
// resetAt even if probes die.
|
|
1799
1875
|
if (usage.scopedWeekly && typeof usage.scopedWeekly === 'object') {
|
|
1876
|
+
const now = Date.now();
|
|
1877
|
+
const prev = (q.scopedWeekly && typeof q.scopedWeekly === 'object') ? q.scopedWeekly : {};
|
|
1800
1878
|
const fresh = {};
|
|
1801
1879
|
for (const [fam, e] of Object.entries(usage.scopedWeekly)) {
|
|
1802
1880
|
if (!e) continue;
|
|
@@ -1807,9 +1885,21 @@ export class AccountManager {
|
|
|
1807
1885
|
isActive: e.isActive !== false,
|
|
1808
1886
|
};
|
|
1809
1887
|
}
|
|
1888
|
+
for (const [fam, pe] of Object.entries(prev)) {
|
|
1889
|
+
if (!pe || !pe.reactive || pe.resetAt == null || pe.resetAt <= now) continue;
|
|
1890
|
+
const f = fresh[fam];
|
|
1891
|
+
// Probe absent, null, or LOWER than the reactive bench → keep the bench.
|
|
1892
|
+
// Probe CONFIRMS >= the reactive level → take the fresh reading (still >=
|
|
1893
|
+
// threshold, so still exhausted; no stickiness needed).
|
|
1894
|
+
if (!f || f.utilization == null || f.utilization < (pe.utilization ?? 0)) {
|
|
1895
|
+
fresh[fam] = { ...pe };
|
|
1896
|
+
}
|
|
1897
|
+
}
|
|
1810
1898
|
q.scopedWeekly = fresh;
|
|
1811
1899
|
}
|
|
1812
1900
|
|
|
1901
|
+
q.lastProbeOkAt = Date.now();
|
|
1902
|
+
|
|
1813
1903
|
// If we just learned this account's weekly window while probing, re-evaluate
|
|
1814
1904
|
// selection (same path as learning it from a live response).
|
|
1815
1905
|
if (account.probing && q.unified7dReset != null) {
|
|
@@ -1818,6 +1908,55 @@ export class AccountManager {
|
|
|
1818
1908
|
}
|
|
1819
1909
|
}
|
|
1820
1910
|
|
|
1911
|
+
/**
|
|
1912
|
+
* Update a PROVIDER account's quota from a provider usage probe
|
|
1913
|
+
* (fetchProviderUsage). z.ai maps to Ses/Wk token windows; Kimi has no pollable
|
|
1914
|
+
* source and only sets a `console-only` marker. Writes ONLY the provider* fields
|
|
1915
|
+
* (never the unified or scopedWeekly fields) so a provider reading can't reach
|
|
1916
|
+
* the OAuth quota gates.
|
|
1917
|
+
*/
|
|
1918
|
+
applyProviderUsage(accountIndex, usage) {
|
|
1919
|
+
const account = this.accounts[accountIndex];
|
|
1920
|
+
if (!account || !usage) return;
|
|
1921
|
+
const q = account.quota;
|
|
1922
|
+
if (usage.error) {
|
|
1923
|
+
// Distinguish "no pollable quota" (Kimi) from a transient probe failure.
|
|
1924
|
+
// Never clear existing values on a transient error — let them age into the
|
|
1925
|
+
// staleness marker instead of blanking the bars.
|
|
1926
|
+
if (usage.source === 'console-only') q.providerQuotaSource = 'console-only';
|
|
1927
|
+
return;
|
|
1928
|
+
}
|
|
1929
|
+
q.providerQuotaSource = usage.source || 'zai';
|
|
1930
|
+
if (usage.ses) {
|
|
1931
|
+
if (usage.ses.utilization != null) q.providerSes = clamp01(usage.ses.utilization);
|
|
1932
|
+
if (usage.ses.resetAt != null) q.providerSesReset = usage.ses.resetAt;
|
|
1933
|
+
}
|
|
1934
|
+
if (usage.wk) {
|
|
1935
|
+
if (usage.wk.utilization != null) q.providerWk = clamp01(usage.wk.utilization);
|
|
1936
|
+
if (usage.wk.resetAt != null) q.providerWkReset = usage.wk.resetAt;
|
|
1937
|
+
} else {
|
|
1938
|
+
// Weekly window absent from this plan/response — clear so a stale weekly
|
|
1939
|
+
// reading doesn't linger after a plan/window change.
|
|
1940
|
+
q.providerWk = null;
|
|
1941
|
+
q.providerWkReset = null;
|
|
1942
|
+
}
|
|
1943
|
+
q.lastProbeOkAt = Date.now();
|
|
1944
|
+
}
|
|
1945
|
+
|
|
1946
|
+
/**
|
|
1947
|
+
* True when the background quota probe hasn't succeeded in > 2× its interval —
|
|
1948
|
+
* the last-known scoped/provider values are aging with no confirmation. Returns
|
|
1949
|
+
* false when the probe is off (nothing to be stale against) or has never yet
|
|
1950
|
+
* succeeded (startup — shown as "no data", not "stale").
|
|
1951
|
+
*/
|
|
1952
|
+
_quotaProbeStale(account, now = Date.now()) {
|
|
1953
|
+
const interval = this.quotaProbeIntervalMs;
|
|
1954
|
+
if (!interval || interval <= 0) return false;
|
|
1955
|
+
const last = account?.quota?.lastProbeOkAt;
|
|
1956
|
+
if (last == null) return false;
|
|
1957
|
+
return (now - last) > Math.max(2 * interval, 120_000);
|
|
1958
|
+
}
|
|
1959
|
+
|
|
1821
1960
|
/**
|
|
1822
1961
|
* Update an account's quota tracking from upstream response headers.
|
|
1823
1962
|
*/
|
|
@@ -1956,6 +2095,10 @@ export class AccountManager {
|
|
|
1956
2095
|
resetAt: Date.now() + (retryAfter * 1000),
|
|
1957
2096
|
severity: 'critical',
|
|
1958
2097
|
isActive: true,
|
|
2098
|
+
// Authoritative-high: a real reject. A lagging 60s probe reporting the
|
|
2099
|
+
// pre-429 level (e.g. 0.96) must NOT lower/drop this before resetAt, else
|
|
2100
|
+
// the account un-benches and immediately re-429s — a per-probe flap.
|
|
2101
|
+
reactive: true,
|
|
1959
2102
|
};
|
|
1960
2103
|
account.lastStatus = options.status || 429;
|
|
1961
2104
|
account.lastErrorAt = Date.now();
|
package/src/index.js
CHANGED
|
@@ -564,6 +564,9 @@ async function serverWorkerCommand() {
|
|
|
564
564
|
// assertions (mirrors MAXPOOL_DISABLE_SLEEP_GUARD).
|
|
565
565
|
const probeSeconds = process.env.MAXPOOL_DISABLE_QUOTA_PROBE === '1' ? 0 : (config.quotaProbeSeconds || 0);
|
|
566
566
|
const prober = new Prober(accountManager, { intervalMs: probeSeconds * 1000 });
|
|
567
|
+
// Tell the AM the probe cadence so the TUI can flag a scoped/provider tag whose
|
|
568
|
+
// background probe has gone stale (> 2× interval since last success).
|
|
569
|
+
accountManager.quotaProbeIntervalMs = probeSeconds * 1000;
|
|
567
570
|
|
|
568
571
|
// Persist refreshed tokens back to config. Defense-in-depth: the updater reads
|
|
569
572
|
// the on-disk refresh token and SKIPS the rotation if a fresher writer already
|
package/src/oauth.js
CHANGED
|
@@ -245,6 +245,61 @@ export async function fetchUsage(accessToken) {
|
|
|
245
245
|
}
|
|
246
246
|
}
|
|
247
247
|
|
|
248
|
+
// z.ai quota monitor — a DIFFERENT host/path than the message upstream
|
|
249
|
+
// (account.upstream is the Anthropic-compat endpoint). Zero-spend read.
|
|
250
|
+
const ZAI_QUOTA_URL = 'https://api.z.ai/api/monitor/usage/quota/limit';
|
|
251
|
+
|
|
252
|
+
/** Classify one z.ai `limits[]` entry into a Ses (5h) or Wk (weekly) TOKEN window.
|
|
253
|
+
* Only `TOKENS_LIMIT` maps to the quota bars; `TIME_LIMIT` is a tool-call cap
|
|
254
|
+
* (web-search/reader counts) and is intentionally ignored. `unit` is z.ai's
|
|
255
|
+
* window enum (3 = 5-hour session, 6 = weekly); we fall back to reset-distance
|
|
256
|
+
* when the code is unfamiliar so a new plan tier still classifies sanely. */
|
|
257
|
+
export function classifyZaiLimit(l, now = Date.now()) {
|
|
258
|
+
if (!l || l.type !== 'TOKENS_LIMIT') return null;
|
|
259
|
+
const reset = Number(l.nextResetTime);
|
|
260
|
+
const resetAt = Number.isFinite(reset) && reset > 0 ? reset : null;
|
|
261
|
+
const pct = typeof l.percentage === 'number' ? l.percentage : parseFloat(l.percentage);
|
|
262
|
+
const utilization = Number.isFinite(pct) ? Math.max(0, Math.min(1, pct / 100)) : null;
|
|
263
|
+
let bucket;
|
|
264
|
+
if (l.unit === 3) bucket = 'ses';
|
|
265
|
+
else if (l.unit === 6) bucket = 'wk';
|
|
266
|
+
else if (resetAt) bucket = (resetAt - now) <= 12 * 60 * 60 * 1000 ? 'ses' : 'wk';
|
|
267
|
+
else bucket = 'ses';
|
|
268
|
+
return { bucket, utilization, resetAt };
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
/** Read a provider account's quota. z.ai has a pollable monitor endpoint mapped
|
|
272
|
+
* to Ses/Wk token windows; Kimi (Moonshot coding key) has NO pollable quota
|
|
273
|
+
* (web console only), so it returns a `console-only` marker instead of fake
|
|
274
|
+
* bars. Returns { ses, wk, level } | { error, status?, source? }. */
|
|
275
|
+
export async function fetchProviderUsage(account) {
|
|
276
|
+
const provider = account?.provider;
|
|
277
|
+
const token = account?.credential;
|
|
278
|
+
if (provider === 'kimi') return { error: 'unsupported', source: 'console-only' };
|
|
279
|
+
if (provider !== 'zai' || !token) return { error: 'unsupported', source: null };
|
|
280
|
+
try {
|
|
281
|
+
const res = await fetch(ZAI_QUOTA_URL, {
|
|
282
|
+
headers: { 'Authorization': `Bearer ${token}`, 'Accept': 'application/json' },
|
|
283
|
+
});
|
|
284
|
+
if (!res.ok) return { error: `HTTP ${res.status}`, status: res.status };
|
|
285
|
+
const data = await res.json();
|
|
286
|
+
if (data?.code !== 200 || !data?.data) {
|
|
287
|
+
return { error: `bad_envelope${data?.code != null ? ' code=' + data.code : ''}` };
|
|
288
|
+
}
|
|
289
|
+
const now = Date.now();
|
|
290
|
+
let ses = null, wk = null;
|
|
291
|
+
for (const l of (Array.isArray(data.data.limits) ? data.data.limits : [])) {
|
|
292
|
+
const c = classifyZaiLimit(l, now);
|
|
293
|
+
if (!c) continue;
|
|
294
|
+
if (c.bucket === 'ses') ses = { utilization: c.utilization, resetAt: c.resetAt };
|
|
295
|
+
else if (c.bucket === 'wk') wk = { utilization: c.utilization, resetAt: c.resetAt };
|
|
296
|
+
}
|
|
297
|
+
return { ses, wk, level: data.data.level || null, source: 'zai' };
|
|
298
|
+
} catch (err) {
|
|
299
|
+
return { error: err.message || String(err), status: null };
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
|
|
248
303
|
// OAuth config (extracted from Claude Code)
|
|
249
304
|
const OAUTH_CLIENT_ID = '9d1c250a-e61b-44d9-88ed-5944d1962f5e';
|
|
250
305
|
const OAUTH_AUTHORIZE = 'https://claude.ai/oauth/authorize';
|
package/src/prober.js
CHANGED
|
@@ -7,13 +7,14 @@
|
|
|
7
7
|
// is blind to an account it isn't actively routing to and will pile traffic onto it.
|
|
8
8
|
// This is the one sanctioned active-upstream feature; the proxy is otherwise passive.
|
|
9
9
|
|
|
10
|
-
import { fetchUsage } from './oauth.js';
|
|
10
|
+
import { fetchUsage, fetchProviderUsage } from './oauth.js';
|
|
11
11
|
|
|
12
12
|
export class Prober {
|
|
13
|
-
constructor(accountManager, { intervalMs = 0, probeFn = fetchUsage, timeoutMs = 10_000, log = console.log } = {}) {
|
|
13
|
+
constructor(accountManager, { intervalMs = 0, probeFn = fetchUsage, providerProbeFn = fetchProviderUsage, timeoutMs = 10_000, log = console.log } = {}) {
|
|
14
14
|
this.am = accountManager;
|
|
15
15
|
this.intervalMs = intervalMs;
|
|
16
16
|
this.probeFn = probeFn;
|
|
17
|
+
this.providerProbeFn = providerProbeFn;
|
|
17
18
|
this.timeoutMs = timeoutMs;
|
|
18
19
|
this.log = log;
|
|
19
20
|
this.timer = null;
|
|
@@ -49,15 +50,21 @@ export class Prober {
|
|
|
49
50
|
if (this._inflight) { try { await this._inflight; } catch { /* swallow */ } }
|
|
50
51
|
}
|
|
51
52
|
|
|
52
|
-
/** Probe every OAuth account
|
|
53
|
-
*
|
|
53
|
+
/** Probe every OAuth account (usage endpoint) and every provider account with a
|
|
54
|
+
* pollable/known quota source (z.ai monitor; Kimi → console-only marker) once.
|
|
55
|
+
* Overlapping cycles are skipped. The active cycle is tracked on `_inflight` so
|
|
56
|
+
* stop() can await it. */
|
|
54
57
|
probeAll() {
|
|
55
58
|
if (this._running) return this._inflight || Promise.resolve();
|
|
56
59
|
this._running = true;
|
|
57
60
|
this._inflight = (async () => {
|
|
58
61
|
try {
|
|
59
|
-
const
|
|
60
|
-
|
|
62
|
+
const oauth = this.am.accounts.filter(a => a.type === 'oauth' && a.credential);
|
|
63
|
+
const providers = this.am.accounts.filter(a => a.type === 'provider' && a.credential);
|
|
64
|
+
await Promise.all([
|
|
65
|
+
...oauth.map(a => this.probeOne(a)),
|
|
66
|
+
...providers.map(a => this.probeProvider(a)),
|
|
67
|
+
]);
|
|
61
68
|
} finally {
|
|
62
69
|
this._running = false;
|
|
63
70
|
this._inflight = null;
|
|
@@ -66,6 +73,16 @@ export class Prober {
|
|
|
66
73
|
return this._inflight;
|
|
67
74
|
}
|
|
68
75
|
|
|
76
|
+
/** Probe one PROVIDER account. Provider tokens are static API keys (no OAuth
|
|
77
|
+
* refresh). Best-effort; never throws. */
|
|
78
|
+
async probeProvider(account) {
|
|
79
|
+
try {
|
|
80
|
+
const usage = await this._withTimeout(this.providerProbeFn(account));
|
|
81
|
+
if (!usage) return; // timed out — try again next cycle
|
|
82
|
+
this.am.applyProviderUsage(account.index, usage);
|
|
83
|
+
} catch { /* best-effort; never let a probe throw */ }
|
|
84
|
+
}
|
|
85
|
+
|
|
69
86
|
async probeOne(account) {
|
|
70
87
|
try {
|
|
71
88
|
await this.am.ensureTokenFresh(account.index);
|
package/src/tui.js
CHANGED
|
@@ -984,7 +984,7 @@ export class TUI {
|
|
|
984
984
|
status = rpad(status, 13);
|
|
985
985
|
|
|
986
986
|
if (a.type === 'provider') {
|
|
987
|
-
return this._renderProviderAcct(sel, cur, name, type, status, a);
|
|
987
|
+
return this._renderProviderAcct(sel, cur, name, type, status, a, bw, showBoth);
|
|
988
988
|
}
|
|
989
989
|
|
|
990
990
|
// Quota ratios — prefer unified (Claude Max), fall back to standard (API key)
|
|
@@ -1013,25 +1013,53 @@ export class TUI {
|
|
|
1013
1013
|
}
|
|
1014
1014
|
const weekly = weeklyPolicyText(this.am, a);
|
|
1015
1015
|
if (weekly) line += ` ${weekly}`;
|
|
1016
|
-
// Per-model weekly caps (e.g. Fable
|
|
1017
|
-
// headroom)
|
|
1018
|
-
//
|
|
1016
|
+
// Per-model weekly caps (e.g. Fable, while the unified weekly still has
|
|
1017
|
+
// headroom). Show the ACTUAL utilization — "Fable 90%" (yellow) while high but
|
|
1018
|
+
// still usable, "Fable maxed" (red) ONLY at genuine exhaustion. This is the
|
|
1019
|
+
// SAME predicate the router benches on (_scopedExhausted), so "maxed" renders
|
|
1020
|
+
// iff the model is actually benched — 90%/critical is no longer mislabelled.
|
|
1019
1021
|
const exhaustedFloor = this.am.scheduler?.weeklyExhaustedThreshold ?? 0.985;
|
|
1020
|
-
const
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1022
|
+
const reserveFloor = this.am.scheduler?.weeklyReserveThreshold ?? 0.85;
|
|
1023
|
+
const scopedTags = [];
|
|
1024
|
+
for (const [fam, e] of Object.entries(q.scopedWeekly || {})) {
|
|
1025
|
+
if (!e || e.isActive === false || e.utilization == null) continue;
|
|
1026
|
+
if (e.utilization < reserveFloor) continue;
|
|
1027
|
+
const Fam = fam.charAt(0).toUpperCase() + fam.slice(1);
|
|
1028
|
+
scopedTags.push(e.utilization >= exhaustedFloor
|
|
1029
|
+
? red(`${Fam} maxed`)
|
|
1030
|
+
: yellow(`${Fam} ${Math.round(e.utilization * 100)}%`));
|
|
1031
|
+
}
|
|
1032
|
+
if (scopedTags.length) line += ` ${scopedTags.join(' ')}`;
|
|
1033
|
+
// Freshness: scoped caps are refreshed ONLY by the background probe (response
|
|
1034
|
+
// headers don't carry them). If the probe has gone stale (> 2× interval since
|
|
1035
|
+
// last success), say so rather than imply the last-known value is current.
|
|
1036
|
+
if (this.am._quotaProbeStale?.(a)) line += ` ${dim('stale')}`;
|
|
1025
1037
|
line += ` ${dim(loadText(this._accountLoad(a)))}`;
|
|
1026
1038
|
return line;
|
|
1027
1039
|
}
|
|
1028
1040
|
|
|
1029
|
-
_renderProviderAcct(sel, cur, name, type, status, a) {
|
|
1041
|
+
_renderProviderAcct(sel, cur, name, type, status, a, bw = 11, showBoth = true) {
|
|
1030
1042
|
const completed = a.completedRequests || 0;
|
|
1031
1043
|
const failed = a.failedRequests || 0;
|
|
1032
1044
|
const active = a.inFlight || 0;
|
|
1033
1045
|
const last = a.lastStatus ? `${statusColor(a.lastStatus)} ${formatMs(a.lastResponseMs)}` : '-';
|
|
1034
1046
|
const q = a.quota || {};
|
|
1047
|
+
|
|
1048
|
+
// Quota segment. z.ai has a pollable monitor endpoint → real Ses/Wk token bars,
|
|
1049
|
+
// same rendering as OAuth accounts. Kimi (Moonshot coding key) has NO pollable
|
|
1050
|
+
// quota (web console only) → an honest label, never a fake bar. Fields are the
|
|
1051
|
+
// SEPARATE provider* set, so a provider reading never leaks into an OAuth bar.
|
|
1052
|
+
let quotaSeg = '';
|
|
1053
|
+
if (q.providerSes != null || q.providerWk != null) {
|
|
1054
|
+
quotaSeg = ` Ses ${bar(q.providerSes, bw, q.providerSesReset)}`;
|
|
1055
|
+
if (showBoth && q.providerWk != null) quotaSeg += ` Wk ${bar(q.providerWk, bw, q.providerWkReset)}`;
|
|
1056
|
+
if (this.am._quotaProbeStale?.(a)) quotaSeg += ` ${dim('stale')}`;
|
|
1057
|
+
} else if (q.providerQuotaSource === 'console-only') {
|
|
1058
|
+
quotaSeg = ` ${dim('Quota console-only')}`;
|
|
1059
|
+
} else if (q.providerQuotaSource === 'zai') {
|
|
1060
|
+
quotaSeg = ` ${dim('Ses/Wk probing')}`;
|
|
1061
|
+
}
|
|
1062
|
+
|
|
1035
1063
|
let limit = '';
|
|
1036
1064
|
if (q.genericLimit != null && q.genericRemaining != null) {
|
|
1037
1065
|
const used = q.genericLimit - q.genericRemaining;
|
|
@@ -1039,7 +1067,7 @@ export class TUI {
|
|
|
1039
1067
|
limit = ` Lim ${used}/${q.genericLimit}${reset ? ` ${reset}` : ''}`;
|
|
1040
1068
|
}
|
|
1041
1069
|
const err = a.lastError ? ` Err ${String(a.lastError).slice(0, 18)}` : '';
|
|
1042
|
-
return ` ${sel}${cur} ${name} ${type} ${status} Act ${String(active).padStart(2)} OK ${String(completed).padStart(3)} Fail ${String(failed).padStart(2)} Last ${last} ${dim(loadText(this._accountLoad(a)))}${limit}${err}`;
|
|
1070
|
+
return ` ${sel}${cur} ${name} ${type} ${status} Act ${String(active).padStart(2)} OK ${String(completed).padStart(3)} Fail ${String(failed).padStart(2)} Last ${last}${quotaSeg} ${dim(loadText(this._accountLoad(a)))}${limit}${err}`;
|
|
1043
1071
|
}
|
|
1044
1072
|
|
|
1045
1073
|
_accountLoad(account) {
|