maxpool 1.5.16 → 1.5.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.16",
3
+ "version": "1.5.18",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -42,6 +42,19 @@ function emptyQuota() {
42
42
  scopedWeekly: {},
43
43
  unifiedStatus: null, // allowed | allowed_warning | rejected
44
44
  resetsAt: null,
45
+ // Provider (z.ai / Kimi) quota — kept SEPARATE from unified* so a provider
46
+ // reading never leaks into the OAuth quota gates (_isAvailable / _weeklyRawState
47
+ // / _accountScarcity read unified* only). z.ai is pollable; Kimi is not.
48
+ providerSes: null, // utilization 0-1 (z.ai 5h token window)
49
+ providerSesReset: null, // ms
50
+ providerWk: null, // utilization 0-1 (z.ai weekly), null if plan has none
51
+ providerWkReset: null, // ms
52
+ providerQuotaSource: null, // 'zai' (pollable) | 'console-only' (kimi) | null
53
+ // Freshness: last time a background usage PROBE succeeded for this account
54
+ // (oauth fetchUsage OR provider fetchProviderUsage). Drives the TUI staleness
55
+ // marker — a swallowed failing probe no longer silently freezes a stale tag.
56
+ // Header-driven updates do NOT stamp this (headers can't refresh scoped/provider).
57
+ lastProbeOkAt: null, // ms
45
58
  };
46
59
  }
47
60
 
@@ -435,6 +448,10 @@ export class AccountManager {
435
448
  hasAvailableRoute(requestInfo = {}, excludedIndexes = new Set()) {
436
449
  this.refreshExpiredQuotas();
437
450
  const profile = requestInfo.profile || 'claude';
451
+ // Route-EXISTENCE check only (order-independent `.some`): unlike the acquire
452
+ // path's re-home loop, pass ORDER doesn't matter here — the bound 2-entry set
453
+ // covers the same accounts (reserve+critical) — so it intentionally is NOT
454
+ // unified to the healthy-first ladder. Do not "sync" these.
438
455
  const hasBinding = Boolean(requestInfo.sessionKey && this.sessionBindings.has(requestInfo.sessionKey));
439
456
  const weeklyPasses = hasBinding
440
457
  ? [
@@ -589,8 +606,14 @@ export class AccountManager {
589
606
  if (!fam) return null;
590
607
  const e = account.quota?.scopedWeekly?.[fam];
591
608
  if (!e || e.isActive === false) return null;
592
- const exhausted = e.severity === 'critical'
593
- || (e.utilization != null && e.utilization >= this.scheduler.weeklyExhaustedThreshold);
609
+ // Bench ONLY at genuine exhaustion (>= weeklyExhaustedThreshold). Anthropic
610
+ // labels a scoped weekly `severity:'critical'` well before it's actually
611
+ // capped (~90%), where the model still has headroom and is still served — a
612
+ // hard bench there strands the remainder and mislabels the account "maxed".
613
+ // A real scoped 429 writes utilization:1 (markRateLimited), so genuine
614
+ // exhaustion is still caught by the threshold. Same predicate drives the TUI
615
+ // `maxed` tag, so "maxed" renders iff the model is actually benched.
616
+ const exhausted = e.utilization != null && e.utilization >= this.scheduler.weeklyExhaustedThreshold;
594
617
  return exhausted ? { resetAt: e.resetAt || null, family: fam } : null;
595
618
  }
596
619
 
@@ -1087,6 +1110,11 @@ export class AccountManager {
1087
1110
  // Keep a signed-thinking session's live migration on Claude accounts only —
1088
1111
  // don't shuttle an Anthropic-signed block onto a provider mid-session even
1089
1112
  // though _isRequestCompatible now allows providers for thinking under policy.
1113
+ // LOAD-BEARING for the absolute-near-cap `return true` below: it's what
1114
+ // guarantees that path only fires with a CLAUDE target present (a signed
1115
+ // session with no healthy Claude alt → bestScore stays Infinity → return
1116
+ // false → stays on its Claude account, never re-homed to a provider). Do not
1117
+ // remove assuming the fall-through protects signed thinking — it does not.
1090
1118
  if (requestInfo.requiresAnthropicThinkingIntegrity === true && account.type === 'provider') continue;
1091
1119
  if (!this._matchesRequest(account, profile, requestInfo)) continue;
1092
1120
  // Genuinely-healthy alternatives only (normal/soft/unknown weekly + model headroom).
@@ -1097,8 +1125,26 @@ export class AccountManager {
1097
1125
  bestTier = WEEKLY_TIER[this._weeklyPaceState(account)] ?? 0;
1098
1126
  }
1099
1127
  }
1100
- if (!Number.isFinite(bestScore)) return false; // no healthy alternative
1101
-
1128
+ if (!Number.isFinite(bestScore)) return false; // no RAW-healthy alternative → don't move (no stranding)
1129
+
1130
+ // Absolute near-cap: the bound account is genuinely low on weekly headroom
1131
+ // (RAW reserve+, not merely burning fast). Every candidate the loop kept is RAW
1132
+ // normal/soft, so this is a STRICT absolute-headroom improvement and flap-stable
1133
+ // (a reserve account can never be a target → no bounce-back). Skip the
1134
+ // concurrency-relief margin: it's calibrated for moving off a LOADED account and
1135
+ // is UNSATISFIABLE for an idle near-cap one — an idle boundScore ≈ the ~2
1136
+ // concurrency floor, so boundScore*0.5 ≈ 1 sits below every account's minimum
1137
+ // score, pinning the session to the near-exhausted account. Preserving the thin
1138
+ // remaining weekly headroom dominates concurrency spread; the per-request
1139
+ // candidate loop lands on the least-loaded healthy account, spreading the move.
1140
+ if ((WEEKLY_TIER[this._weeklyRawState(bound)] ?? 0) >= WEEKLY_TIER.reserve) return true;
1141
+
1142
+ // Otherwise the trigger was PACE-only on a RAW-healthy account — a fast-burner
1143
+ // that still has real absolute headroom (RAW soft but pace reserve/critical,
1144
+ // e.g. 79% used resetting in ~3.5d). Keep the conservative gate so it isn't
1145
+ // churned off an account that's genuinely fine: move only for a clearly-cheaper,
1146
+ // strictly-healthier-pace-tier target. (A RAW-soft/pace-normal fast-burner with
1147
+ // lots of headroom never even reaches here — _isBoundAccountHot gates it out.)
1102
1148
  return bestScore <= boundScore * REBALANCE_SCORE_MARGIN
1103
1149
  && (boundScore - bestScore) >= REBALANCE_MIN_ABS_GAP
1104
1150
  && bestTier < boundTier;
@@ -1244,7 +1290,6 @@ export class AccountManager {
1244
1290
  }
1245
1291
  }
1246
1292
 
1247
- const hasBinding = Boolean(requestInfo.sessionKey && this.sessionBindings.has(requestInfo.sessionKey));
1248
1293
  const preferred = this._preferredAccount(profile, excludedIndexes, requestInfo);
1249
1294
  if (preferred) {
1250
1295
  const preferredPasses = [
@@ -1265,16 +1310,17 @@ export class AccountManager {
1265
1310
  // Else fall through to the candidate score loop, which re-homes the session
1266
1311
  // onto the best healthy account via _bindSession on acquire.
1267
1312
 
1268
- const weeklyPasses = hasBinding
1269
- ? [
1270
- { allowWeeklyReserve: true, allowWeeklyCritical: false },
1271
- { allowWeeklyReserve: true, allowWeeklyCritical: true },
1272
- ]
1273
- : [
1274
- { allowWeeklyReserve: false, allowWeeklyCritical: false },
1275
- { allowWeeklyReserve: true, allowWeeklyCritical: false },
1276
- { allowWeeklyReserve: true, allowWeeklyCritical: true },
1277
- ];
1313
+ // Healthy-first ladder for BOTH bound and unbound sessions. A bound session
1314
+ // only reaches this loop once we've decided to LEAVE its account (rebalance
1315
+ // fired / a higher-priority account is available / the bound account is down) —
1316
+ // the sticky "stay put" path returns above without reaching here — so re-homing
1317
+ // must prefer a genuinely-healthy account and fall back to reserve/critical only
1318
+ // if none exists, never re-pick the idle reserve account it's leaving.
1319
+ const weeklyPasses = [
1320
+ { allowWeeklyReserve: false, allowWeeklyCritical: false },
1321
+ { allowWeeklyReserve: true, allowWeeklyCritical: false },
1322
+ { allowWeeklyReserve: true, allowWeeklyCritical: true },
1323
+ ];
1278
1324
 
1279
1325
  for (const weeklyOptions of weeklyPasses) {
1280
1326
  best = null;
@@ -1648,6 +1694,16 @@ export class AccountManager {
1648
1694
  // soft de-preference of accounts burning ahead of an even pace. Never a bench.
1649
1695
  const paceCost = this._accountScarcity(account, now) * this.scheduler.paceCostWeight;
1650
1696
 
1697
+ // Per-model weekly de-preference: an account whose scoped weekly for THIS
1698
+ // request's model (e.g. Fable) is high-but-not-exhausted is a poor pick for
1699
+ // that model — shed its load toward healthier accounts BEFORE the hard bench
1700
+ // at weeklyExhaustedThreshold, so it's chosen only as overflow (rarely
1701
+ // re-429ing). Scoped weekly is otherwise absent from scoring. Soft, never a
1702
+ // bench; 0 below reserve and for models with no scoped cap.
1703
+ const scopedPace = requestInfo.model
1704
+ ? this._scopedScarcity(account, requestInfo.model, now) * this.scheduler.paceCostWeight
1705
+ : 0;
1706
+
1651
1707
  const fleetRecentWeight = ctx?.fleetRecentWeight ?? 0;
1652
1708
  const recentWeight = this._loadSummary(account, this.scheduler.spreadWindowMs, now).weight;
1653
1709
  const share = fleetRecentWeight > 0 ? recentWeight / fleetRecentWeight : 0;
@@ -1663,7 +1719,22 @@ export class AccountManager {
1663
1719
  // default) learns the real number within a cycle. `probing`/requalify still
1664
1720
  // flags a never-seen account for learning — that path is unchanged.
1665
1721
 
1666
- return concurrency + capPenalty + paceCost + spread + ramp + failurePenalty;
1722
+ return concurrency + capPenalty + paceCost + scopedPace + spread + ramp + failurePenalty;
1723
+ }
1724
+
1725
+ /**
1726
+ * Per-model weekly pace-overage for `model`'s family, or 0 when the account has
1727
+ * no scoped cap for it, the cap is inactive, or it's below the reserve tier
1728
+ * (plenty of headroom → no steering). Same pace discount as _windowScarcity, so
1729
+ * a scoped window about to reset is cheap to spend.
1730
+ */
1731
+ _scopedScarcity(account, model, now = Date.now()) {
1732
+ const fam = modelFamily(model);
1733
+ if (!fam) return 0;
1734
+ const e = account.quota?.scopedWeekly?.[fam];
1735
+ if (!e || e.isActive === false || e.utilization == null) return 0;
1736
+ if (e.utilization < this.scheduler.weeklyReserveThreshold) return 0;
1737
+ return this._windowScarcity(e.utilization, e.resetAt, WEEK_MS, now);
1667
1738
  }
1668
1739
 
1669
1740
  /**
@@ -1795,8 +1866,15 @@ export class AccountManager {
1795
1866
  }
1796
1867
  // Per-model weekly sub-limits (Fable, Opus, ...). Replace wholesale with the
1797
1868
  // fresh probe set so a family that dropped out of the response doesn't linger
1798
- // stale; expiry on reset is a backstop for the between-probe window.
1869
+ // stale; expiry on reset is a backstop for the between-probe window. EXCEPTION:
1870
+ // a `reactive` scoped-429 bench is authoritative-high — while its resetAt is
1871
+ // still future, a lagging probe may neither lower it nor drop it (a probe that
1872
+ // omits the family, or reports the pre-429 level, would otherwise un-bench it
1873
+ // and trigger an immediate re-429 flap). _clearExpiredQuotas self-clears it at
1874
+ // resetAt even if probes die.
1799
1875
  if (usage.scopedWeekly && typeof usage.scopedWeekly === 'object') {
1876
+ const now = Date.now();
1877
+ const prev = (q.scopedWeekly && typeof q.scopedWeekly === 'object') ? q.scopedWeekly : {};
1800
1878
  const fresh = {};
1801
1879
  for (const [fam, e] of Object.entries(usage.scopedWeekly)) {
1802
1880
  if (!e) continue;
@@ -1807,9 +1885,21 @@ export class AccountManager {
1807
1885
  isActive: e.isActive !== false,
1808
1886
  };
1809
1887
  }
1888
+ for (const [fam, pe] of Object.entries(prev)) {
1889
+ if (!pe || !pe.reactive || pe.resetAt == null || pe.resetAt <= now) continue;
1890
+ const f = fresh[fam];
1891
+ // Probe absent, null, or LOWER than the reactive bench → keep the bench.
1892
+ // Probe CONFIRMS >= the reactive level → take the fresh reading (still >=
1893
+ // threshold, so still exhausted; no stickiness needed).
1894
+ if (!f || f.utilization == null || f.utilization < (pe.utilization ?? 0)) {
1895
+ fresh[fam] = { ...pe };
1896
+ }
1897
+ }
1810
1898
  q.scopedWeekly = fresh;
1811
1899
  }
1812
1900
 
1901
+ q.lastProbeOkAt = Date.now();
1902
+
1813
1903
  // If we just learned this account's weekly window while probing, re-evaluate
1814
1904
  // selection (same path as learning it from a live response).
1815
1905
  if (account.probing && q.unified7dReset != null) {
@@ -1818,6 +1908,55 @@ export class AccountManager {
1818
1908
  }
1819
1909
  }
1820
1910
 
1911
+ /**
1912
+ * Update a PROVIDER account's quota from a provider usage probe
1913
+ * (fetchProviderUsage). z.ai maps to Ses/Wk token windows; Kimi has no pollable
1914
+ * source and only sets a `console-only` marker. Writes ONLY the provider* fields
1915
+ * (never the unified or scopedWeekly fields) so a provider reading can't reach
1916
+ * the OAuth quota gates.
1917
+ */
1918
+ applyProviderUsage(accountIndex, usage) {
1919
+ const account = this.accounts[accountIndex];
1920
+ if (!account || !usage) return;
1921
+ const q = account.quota;
1922
+ if (usage.error) {
1923
+ // Distinguish "no pollable quota" (Kimi) from a transient probe failure.
1924
+ // Never clear existing values on a transient error — let them age into the
1925
+ // staleness marker instead of blanking the bars.
1926
+ if (usage.source === 'console-only') q.providerQuotaSource = 'console-only';
1927
+ return;
1928
+ }
1929
+ q.providerQuotaSource = usage.source || 'zai';
1930
+ if (usage.ses) {
1931
+ if (usage.ses.utilization != null) q.providerSes = clamp01(usage.ses.utilization);
1932
+ if (usage.ses.resetAt != null) q.providerSesReset = usage.ses.resetAt;
1933
+ }
1934
+ if (usage.wk) {
1935
+ if (usage.wk.utilization != null) q.providerWk = clamp01(usage.wk.utilization);
1936
+ if (usage.wk.resetAt != null) q.providerWkReset = usage.wk.resetAt;
1937
+ } else {
1938
+ // Weekly window absent from this plan/response — clear so a stale weekly
1939
+ // reading doesn't linger after a plan/window change.
1940
+ q.providerWk = null;
1941
+ q.providerWkReset = null;
1942
+ }
1943
+ q.lastProbeOkAt = Date.now();
1944
+ }
1945
+
1946
+ /**
1947
+ * True when the background quota probe hasn't succeeded in > 2× its interval —
1948
+ * the last-known scoped/provider values are aging with no confirmation. Returns
1949
+ * false when the probe is off (nothing to be stale against) or has never yet
1950
+ * succeeded (startup — shown as "no data", not "stale").
1951
+ */
1952
+ _quotaProbeStale(account, now = Date.now()) {
1953
+ const interval = this.quotaProbeIntervalMs;
1954
+ if (!interval || interval <= 0) return false;
1955
+ const last = account?.quota?.lastProbeOkAt;
1956
+ if (last == null) return false;
1957
+ return (now - last) > Math.max(2 * interval, 120_000);
1958
+ }
1959
+
1821
1960
  /**
1822
1961
  * Update an account's quota tracking from upstream response headers.
1823
1962
  */
@@ -1956,6 +2095,10 @@ export class AccountManager {
1956
2095
  resetAt: Date.now() + (retryAfter * 1000),
1957
2096
  severity: 'critical',
1958
2097
  isActive: true,
2098
+ // Authoritative-high: a real reject. A lagging 60s probe reporting the
2099
+ // pre-429 level (e.g. 0.96) must NOT lower/drop this before resetAt, else
2100
+ // the account un-benches and immediately re-429s — a per-probe flap.
2101
+ reactive: true,
1959
2102
  };
1960
2103
  account.lastStatus = options.status || 429;
1961
2104
  account.lastErrorAt = Date.now();
package/src/index.js CHANGED
@@ -564,6 +564,9 @@ async function serverWorkerCommand() {
564
564
  // assertions (mirrors MAXPOOL_DISABLE_SLEEP_GUARD).
565
565
  const probeSeconds = process.env.MAXPOOL_DISABLE_QUOTA_PROBE === '1' ? 0 : (config.quotaProbeSeconds || 0);
566
566
  const prober = new Prober(accountManager, { intervalMs: probeSeconds * 1000 });
567
+ // Tell the AM the probe cadence so the TUI can flag a scoped/provider tag whose
568
+ // background probe has gone stale (> 2× interval since last success).
569
+ accountManager.quotaProbeIntervalMs = probeSeconds * 1000;
567
570
 
568
571
  // Persist refreshed tokens back to config. Defense-in-depth: the updater reads
569
572
  // the on-disk refresh token and SKIPS the rotation if a fresher writer already
package/src/oauth.js CHANGED
@@ -245,6 +245,61 @@ export async function fetchUsage(accessToken) {
245
245
  }
246
246
  }
247
247
 
248
+ // z.ai quota monitor — a DIFFERENT host/path than the message upstream
249
+ // (account.upstream is the Anthropic-compat endpoint). Zero-spend read.
250
+ const ZAI_QUOTA_URL = 'https://api.z.ai/api/monitor/usage/quota/limit';
251
+
252
+ /** Classify one z.ai `limits[]` entry into a Ses (5h) or Wk (weekly) TOKEN window.
253
+ * Only `TOKENS_LIMIT` maps to the quota bars; `TIME_LIMIT` is a tool-call cap
254
+ * (web-search/reader counts) and is intentionally ignored. `unit` is z.ai's
255
+ * window enum (3 = 5-hour session, 6 = weekly); we fall back to reset-distance
256
+ * when the code is unfamiliar so a new plan tier still classifies sanely. */
257
+ export function classifyZaiLimit(l, now = Date.now()) {
258
+ if (!l || l.type !== 'TOKENS_LIMIT') return null;
259
+ const reset = Number(l.nextResetTime);
260
+ const resetAt = Number.isFinite(reset) && reset > 0 ? reset : null;
261
+ const pct = typeof l.percentage === 'number' ? l.percentage : parseFloat(l.percentage);
262
+ const utilization = Number.isFinite(pct) ? Math.max(0, Math.min(1, pct / 100)) : null;
263
+ let bucket;
264
+ if (l.unit === 3) bucket = 'ses';
265
+ else if (l.unit === 6) bucket = 'wk';
266
+ else if (resetAt) bucket = (resetAt - now) <= 12 * 60 * 60 * 1000 ? 'ses' : 'wk';
267
+ else bucket = 'ses';
268
+ return { bucket, utilization, resetAt };
269
+ }
270
+
271
+ /** Read a provider account's quota. z.ai has a pollable monitor endpoint mapped
272
+ * to Ses/Wk token windows; Kimi (Moonshot coding key) has NO pollable quota
273
+ * (web console only), so it returns a `console-only` marker instead of fake
274
+ * bars. Returns { ses, wk, level } | { error, status?, source? }. */
275
+ export async function fetchProviderUsage(account) {
276
+ const provider = account?.provider;
277
+ const token = account?.credential;
278
+ if (provider === 'kimi') return { error: 'unsupported', source: 'console-only' };
279
+ if (provider !== 'zai' || !token) return { error: 'unsupported', source: null };
280
+ try {
281
+ const res = await fetch(ZAI_QUOTA_URL, {
282
+ headers: { 'Authorization': `Bearer ${token}`, 'Accept': 'application/json' },
283
+ });
284
+ if (!res.ok) return { error: `HTTP ${res.status}`, status: res.status };
285
+ const data = await res.json();
286
+ if (data?.code !== 200 || !data?.data) {
287
+ return { error: `bad_envelope${data?.code != null ? ' code=' + data.code : ''}` };
288
+ }
289
+ const now = Date.now();
290
+ let ses = null, wk = null;
291
+ for (const l of (Array.isArray(data.data.limits) ? data.data.limits : [])) {
292
+ const c = classifyZaiLimit(l, now);
293
+ if (!c) continue;
294
+ if (c.bucket === 'ses') ses = { utilization: c.utilization, resetAt: c.resetAt };
295
+ else if (c.bucket === 'wk') wk = { utilization: c.utilization, resetAt: c.resetAt };
296
+ }
297
+ return { ses, wk, level: data.data.level || null, source: 'zai' };
298
+ } catch (err) {
299
+ return { error: err.message || String(err), status: null };
300
+ }
301
+ }
302
+
248
303
  // OAuth config (extracted from Claude Code)
249
304
  const OAUTH_CLIENT_ID = '9d1c250a-e61b-44d9-88ed-5944d1962f5e';
250
305
  const OAUTH_AUTHORIZE = 'https://claude.ai/oauth/authorize';
package/src/prober.js CHANGED
@@ -7,13 +7,14 @@
7
7
  // is blind to an account it isn't actively routing to and will pile traffic onto it.
8
8
  // This is the one sanctioned active-upstream feature; the proxy is otherwise passive.
9
9
 
10
- import { fetchUsage } from './oauth.js';
10
+ import { fetchUsage, fetchProviderUsage } from './oauth.js';
11
11
 
12
12
  export class Prober {
13
- constructor(accountManager, { intervalMs = 0, probeFn = fetchUsage, timeoutMs = 10_000, log = console.log } = {}) {
13
+ constructor(accountManager, { intervalMs = 0, probeFn = fetchUsage, providerProbeFn = fetchProviderUsage, timeoutMs = 10_000, log = console.log } = {}) {
14
14
  this.am = accountManager;
15
15
  this.intervalMs = intervalMs;
16
16
  this.probeFn = probeFn;
17
+ this.providerProbeFn = providerProbeFn;
17
18
  this.timeoutMs = timeoutMs;
18
19
  this.log = log;
19
20
  this.timer = null;
@@ -49,15 +50,21 @@ export class Prober {
49
50
  if (this._inflight) { try { await this._inflight; } catch { /* swallow */ } }
50
51
  }
51
52
 
52
- /** Probe every OAuth account once. Overlapping cycles are skipped. The active
53
- * cycle is tracked on `_inflight` so stop() can await it. */
53
+ /** Probe every OAuth account (usage endpoint) and every provider account with a
54
+ * pollable/known quota source (z.ai monitor; Kimi → console-only marker) once.
55
+ * Overlapping cycles are skipped. The active cycle is tracked on `_inflight` so
56
+ * stop() can await it. */
54
57
  probeAll() {
55
58
  if (this._running) return this._inflight || Promise.resolve();
56
59
  this._running = true;
57
60
  this._inflight = (async () => {
58
61
  try {
59
- const accounts = this.am.accounts.filter(a => a.type === 'oauth' && a.credential);
60
- await Promise.all(accounts.map(a => this.probeOne(a)));
62
+ const oauth = this.am.accounts.filter(a => a.type === 'oauth' && a.credential);
63
+ const providers = this.am.accounts.filter(a => a.type === 'provider' && a.credential);
64
+ await Promise.all([
65
+ ...oauth.map(a => this.probeOne(a)),
66
+ ...providers.map(a => this.probeProvider(a)),
67
+ ]);
61
68
  } finally {
62
69
  this._running = false;
63
70
  this._inflight = null;
@@ -66,6 +73,16 @@ export class Prober {
66
73
  return this._inflight;
67
74
  }
68
75
 
76
+ /** Probe one PROVIDER account. Provider tokens are static API keys (no OAuth
77
+ * refresh). Best-effort; never throws. */
78
+ async probeProvider(account) {
79
+ try {
80
+ const usage = await this._withTimeout(this.providerProbeFn(account));
81
+ if (!usage) return; // timed out — try again next cycle
82
+ this.am.applyProviderUsage(account.index, usage);
83
+ } catch { /* best-effort; never let a probe throw */ }
84
+ }
85
+
69
86
  async probeOne(account) {
70
87
  try {
71
88
  await this.am.ensureTokenFresh(account.index);
package/src/tui.js CHANGED
@@ -984,7 +984,7 @@ export class TUI {
984
984
  status = rpad(status, 13);
985
985
 
986
986
  if (a.type === 'provider') {
987
- return this._renderProviderAcct(sel, cur, name, type, status, a);
987
+ return this._renderProviderAcct(sel, cur, name, type, status, a, bw, showBoth);
988
988
  }
989
989
 
990
990
  // Quota ratios — prefer unified (Claude Max), fall back to standard (API key)
@@ -1013,25 +1013,53 @@ export class TUI {
1013
1013
  }
1014
1014
  const weekly = weeklyPolicyText(this.am, a);
1015
1015
  if (weekly) line += ` ${weekly}`;
1016
- // Per-model weekly caps (e.g. Fable maxed while the unified weekly still has
1017
- // headroom) — surfaced separately so a capped model on an otherwise-healthy
1018
- // account is visible, and routing away from it is explained.
1016
+ // Per-model weekly caps (e.g. Fable, while the unified weekly still has
1017
+ // headroom). Show the ACTUAL utilization — "Fable 90%" (yellow) while high but
1018
+ // still usable, "Fable maxed" (red) ONLY at genuine exhaustion. This is the
1019
+ // SAME predicate the router benches on (_scopedExhausted), so "maxed" renders
1020
+ // iff the model is actually benched — 90%/critical is no longer mislabelled.
1019
1021
  const exhaustedFloor = this.am.scheduler?.weeklyExhaustedThreshold ?? 0.985;
1020
- const capped = Object.entries(q.scopedWeekly || {})
1021
- .filter(([, e]) => e && e.isActive !== false
1022
- && (e.severity === 'critical' || (e.utilization != null && e.utilization >= exhaustedFloor)))
1023
- .map(([fam]) => fam.charAt(0).toUpperCase() + fam.slice(1));
1024
- if (capped.length) line += ` ${red(`${capped.join(',')} maxed`)}`;
1022
+ const reserveFloor = this.am.scheduler?.weeklyReserveThreshold ?? 0.85;
1023
+ const scopedTags = [];
1024
+ for (const [fam, e] of Object.entries(q.scopedWeekly || {})) {
1025
+ if (!e || e.isActive === false || e.utilization == null) continue;
1026
+ if (e.utilization < reserveFloor) continue;
1027
+ const Fam = fam.charAt(0).toUpperCase() + fam.slice(1);
1028
+ scopedTags.push(e.utilization >= exhaustedFloor
1029
+ ? red(`${Fam} maxed`)
1030
+ : yellow(`${Fam} ${Math.round(e.utilization * 100)}%`));
1031
+ }
1032
+ if (scopedTags.length) line += ` ${scopedTags.join(' ')}`;
1033
+ // Freshness: scoped caps are refreshed ONLY by the background probe (response
1034
+ // headers don't carry them). If the probe has gone stale (> 2× interval since
1035
+ // last success), say so rather than imply the last-known value is current.
1036
+ if (this.am._quotaProbeStale?.(a)) line += ` ${dim('stale')}`;
1025
1037
  line += ` ${dim(loadText(this._accountLoad(a)))}`;
1026
1038
  return line;
1027
1039
  }
1028
1040
 
1029
- _renderProviderAcct(sel, cur, name, type, status, a) {
1041
+ _renderProviderAcct(sel, cur, name, type, status, a, bw = 11, showBoth = true) {
1030
1042
  const completed = a.completedRequests || 0;
1031
1043
  const failed = a.failedRequests || 0;
1032
1044
  const active = a.inFlight || 0;
1033
1045
  const last = a.lastStatus ? `${statusColor(a.lastStatus)} ${formatMs(a.lastResponseMs)}` : '-';
1034
1046
  const q = a.quota || {};
1047
+
1048
+ // Quota segment. z.ai has a pollable monitor endpoint → real Ses/Wk token bars,
1049
+ // same rendering as OAuth accounts. Kimi (Moonshot coding key) has NO pollable
1050
+ // quota (web console only) → an honest label, never a fake bar. Fields are the
1051
+ // SEPARATE provider* set, so a provider reading never leaks into an OAuth bar.
1052
+ let quotaSeg = '';
1053
+ if (q.providerSes != null || q.providerWk != null) {
1054
+ quotaSeg = ` Ses ${bar(q.providerSes, bw, q.providerSesReset)}`;
1055
+ if (showBoth && q.providerWk != null) quotaSeg += ` Wk ${bar(q.providerWk, bw, q.providerWkReset)}`;
1056
+ if (this.am._quotaProbeStale?.(a)) quotaSeg += ` ${dim('stale')}`;
1057
+ } else if (q.providerQuotaSource === 'console-only') {
1058
+ quotaSeg = ` ${dim('Quota console-only')}`;
1059
+ } else if (q.providerQuotaSource === 'zai') {
1060
+ quotaSeg = ` ${dim('Ses/Wk probing')}`;
1061
+ }
1062
+
1035
1063
  let limit = '';
1036
1064
  if (q.genericLimit != null && q.genericRemaining != null) {
1037
1065
  const used = q.genericLimit - q.genericRemaining;
@@ -1039,7 +1067,7 @@ export class TUI {
1039
1067
  limit = ` Lim ${used}/${q.genericLimit}${reset ? ` ${reset}` : ''}`;
1040
1068
  }
1041
1069
  const err = a.lastError ? ` Err ${String(a.lastError).slice(0, 18)}` : '';
1042
- return ` ${sel}${cur} ${name} ${type} ${status} Act ${String(active).padStart(2)} OK ${String(completed).padStart(3)} Fail ${String(failed).padStart(2)} Last ${last} ${dim(loadText(this._accountLoad(a)))}${limit}${err}`;
1070
+ return ` ${sel}${cur} ${name} ${type} ${status} Act ${String(active).padStart(2)} OK ${String(completed).padStart(3)} Fail ${String(failed).padStart(2)} Last ${last}${quotaSeg} ${dim(loadText(this._accountLoad(a)))}${limit}${err}`;
1043
1071
  }
1044
1072
 
1045
1073
  _accountLoad(account) {