maxpool 1.5.15 → 1.5.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/package.json +1 -1
- package/src/account-manager.js +175 -62
- package/src/index.js +3 -0
- package/src/oauth.js +55 -0
- package/src/prober.js +23 -6
- package/src/server.js +80 -40
- package/src/tui.js +44 -11
package/README.md
CHANGED
|
@@ -194,7 +194,9 @@ Provider fallback credentials can be supplied per Claude Code process with `ANTH
|
|
|
194
194
|
|
|
195
195
|
`x-maxpool-session: <id>` enables session affinity. With this header, the first request for a Claude Code process is routed by the adaptive load balancer, then later requests from the same process keep using that home account while it remains available. If the home account is rate-limited, exhausted, in cooldown, or removed, the session temporarily uses another eligible route. When the home account becomes available again, the session returns to it. For the `all` profile, a Claude session that had to spill onto GLM/Kimi moves back to Claude once a Claude account frees up — **as long as it produced no provider-format tool calls while there**.
|
|
196
196
|
|
|
197
|
-
**Cross-provider
|
|
197
|
+
**Cross-provider fallback + compatibility (`all` profile).** Claude, GLM (z.ai), and Kimi (Moonshot) interoperate for ordinary sessions — a regular tool-use id (`call_`/`tool_`) passes Anthropic's loose validation, so a Kimi or GLM session runs fine on Claude and a Claude session spills onto GLM/Kimi when Claude is unavailable. `scheduler.crossProviderFallbackPolicy` controls it, cyclable live in the TUI Routing menu (`m` → `f`): `never` = strict same-family (Claude→Claude, GLM→GLM, Kimi→Kimi), `when-exhausted` (default) = cross only once the home family is exhausted, `always` = providers peer with Claude.
|
|
198
|
+
|
|
199
|
+
The one hard incompatibility Maxpool guards is a **`server_tool_use` id** that isn't `srvtoolu_` — Anthropic 400s that on replay (e.g. a `cc glm` session that used a server tool, resumed under `cc all`). Maxpool predicts that case ahead (pins the session to GLM/Kimi) and, for anything it can't predict (a rejected thinking signature), **self-heals**: the 400 is pre-stream, so Maxpool latches the session provider-only and transparently retries on GLM/Kimi — the client never sees the error. (Note: Claude Code may warn a resumed `glm-*` model "could not be restored" and fall back to `claude-opus-4-8`; that's a client-side message — routing is unaffected.)
|
|
198
200
|
|
|
199
201
|
Provider rows do not use Claude Max session/week bars unless the provider returns compatible quota headers. For GLM/Kimi, Maxpool always tracks operational telemetry (`Act`, `OK`, `Fail`, `Last`) and also parses common `x-ratelimit-*` / `ratelimit-*` headers if present.
|
|
200
202
|
|
package/package.json
CHANGED
package/src/account-manager.js
CHANGED
|
@@ -42,6 +42,19 @@ function emptyQuota() {
|
|
|
42
42
|
scopedWeekly: {},
|
|
43
43
|
unifiedStatus: null, // allowed | allowed_warning | rejected
|
|
44
44
|
resetsAt: null,
|
|
45
|
+
// Provider (z.ai / Kimi) quota — kept SEPARATE from unified* so a provider
|
|
46
|
+
// reading never leaks into the OAuth quota gates (_isAvailable / _weeklyRawState
|
|
47
|
+
// / _accountScarcity read unified* only). z.ai is pollable; Kimi is not.
|
|
48
|
+
providerSes: null, // utilization 0-1 (z.ai 5h token window)
|
|
49
|
+
providerSesReset: null, // ms
|
|
50
|
+
providerWk: null, // utilization 0-1 (z.ai weekly), null if plan has none
|
|
51
|
+
providerWkReset: null, // ms
|
|
52
|
+
providerQuotaSource: null, // 'zai' (pollable) | 'console-only' (kimi) | null
|
|
53
|
+
// Freshness: last time a background usage PROBE succeeded for this account
|
|
54
|
+
// (oauth fetchUsage OR provider fetchProviderUsage). Drives the TUI staleness
|
|
55
|
+
// marker — a swallowed failing probe no longer silently freezes a stale tag.
|
|
56
|
+
// Header-driven updates do NOT stamp this (headers can't refresh scoped/provider).
|
|
57
|
+
lastProbeOkAt: null, // ms
|
|
45
58
|
};
|
|
46
59
|
}
|
|
47
60
|
|
|
@@ -589,8 +602,14 @@ export class AccountManager {
|
|
|
589
602
|
if (!fam) return null;
|
|
590
603
|
const e = account.quota?.scopedWeekly?.[fam];
|
|
591
604
|
if (!e || e.isActive === false) return null;
|
|
592
|
-
|
|
593
|
-
|
|
605
|
+
// Bench ONLY at genuine exhaustion (>= weeklyExhaustedThreshold). Anthropic
|
|
606
|
+
// labels a scoped weekly `severity:'critical'` well before it's actually
|
|
607
|
+
// capped (~90%), where the model still has headroom and is still served — a
|
|
608
|
+
// hard bench there strands the remainder and mislabels the account "maxed".
|
|
609
|
+
// A real scoped 429 writes utilization:1 (markRateLimited), so genuine
|
|
610
|
+
// exhaustion is still caught by the threshold. Same predicate drives the TUI
|
|
611
|
+
// `maxed` tag, so "maxed" renders iff the model is actually benched.
|
|
612
|
+
const exhausted = e.utilization != null && e.utilization >= this.scheduler.weeklyExhaustedThreshold;
|
|
594
613
|
return exhausted ? { resetAt: e.resetAt || null, family: fam } : null;
|
|
595
614
|
}
|
|
596
615
|
|
|
@@ -1034,16 +1053,16 @@ export class AccountManager {
|
|
|
1034
1053
|
_migrationSafeForRequest(requestInfo = {}) {
|
|
1035
1054
|
// Fail closed on any body we couldn't fully scan (non-JSON / parse error).
|
|
1036
1055
|
if (requestInfo.bodyThinkingScanned !== true) return false;
|
|
1037
|
-
//
|
|
1038
|
-
//
|
|
1039
|
-
//
|
|
1040
|
-
if (this.
|
|
1056
|
+
// An Anthropic-incompatible (provider-pinned) session's only possible cross is
|
|
1057
|
+
// GLM↔Kimi, whose mismatched thinking formats risk a reasoning-loop — keep it on
|
|
1058
|
+
// its bound provider rather than rebalancing.
|
|
1059
|
+
if (this._effectiveIncompatible(requestInfo).incompatible) return false;
|
|
1041
1060
|
// A signed-thinking request is migration-safe when cross-account thinking
|
|
1042
1061
|
// migration is enabled: the signature is content/model integrity, not account-
|
|
1043
|
-
// bound
|
|
1044
|
-
//
|
|
1045
|
-
//
|
|
1046
|
-
//
|
|
1062
|
+
// bound. The rebalance candidate loop (_shouldRebalanceBoundSession) additionally
|
|
1063
|
+
// skips PROVIDER targets for a signed-thinking request, so every migration target
|
|
1064
|
+
// stays a Claude account (a signed block isn't shuttled to a provider mid-session).
|
|
1065
|
+
// When the flag is off, keep the conservative bar (never migrate signed thinking).
|
|
1047
1066
|
if (requestInfo.requiresAnthropicThinkingIntegrity === true) {
|
|
1048
1067
|
return this.scheduler.crossAccountThinkingMigration === true;
|
|
1049
1068
|
}
|
|
@@ -1084,6 +1103,10 @@ export class AccountManager {
|
|
|
1084
1103
|
for (const account of this.accounts) {
|
|
1085
1104
|
if (account.index === bound.index) continue;
|
|
1086
1105
|
if (excludedIndexes.has(account.index)) continue;
|
|
1106
|
+
// Keep a signed-thinking session's live migration on Claude accounts only —
|
|
1107
|
+
// don't shuttle an Anthropic-signed block onto a provider mid-session even
|
|
1108
|
+
// though _isRequestCompatible now allows providers for thinking under policy.
|
|
1109
|
+
if (requestInfo.requiresAnthropicThinkingIntegrity === true && account.type === 'provider') continue;
|
|
1087
1110
|
if (!this._matchesRequest(account, profile, requestInfo)) continue;
|
|
1088
1111
|
// Genuinely-healthy alternatives only (normal/soft/unknown weekly + model headroom).
|
|
1089
1112
|
if (!this._isAvailable(account, { allowWeeklyReserve: false, allowWeeklyCritical: false, model: requestInfo.model })) continue;
|
|
@@ -1493,7 +1516,7 @@ export class AccountManager {
|
|
|
1493
1516
|
const base = Number.isFinite(account.priority) ? account.priority : 0;
|
|
1494
1517
|
if (account.type === 'provider'
|
|
1495
1518
|
&& this._crossProviderFallbackPolicy() === 'always'
|
|
1496
|
-
&& this.
|
|
1519
|
+
&& !this._effectiveIncompatible(requestInfo).incompatible) {
|
|
1497
1520
|
return 0;
|
|
1498
1521
|
}
|
|
1499
1522
|
return base;
|
|
@@ -1506,53 +1529,45 @@ export class AccountManager {
|
|
|
1506
1529
|
return true;
|
|
1507
1530
|
}
|
|
1508
1531
|
|
|
1509
|
-
// Effective
|
|
1510
|
-
//
|
|
1511
|
-
//
|
|
1512
|
-
//
|
|
1513
|
-
//
|
|
1514
|
-
|
|
1532
|
+
// Effective Anthropic-incompatibility for a request: the request's own transcript
|
|
1533
|
+
// verdict, OR a sticky latch on the session — set once a foreign server_tool_use
|
|
1534
|
+
// id is seen, or once Anthropic REJECTED the transcript on replay (react-and-heal
|
|
1535
|
+
// in server.js). Never downgrades, so a later no-tool follow-up turn stays
|
|
1536
|
+
// provider-pinned. Both the selector AND the retry oracle read this so they never
|
|
1537
|
+
// disagree. homeProvider is a SOFT hint (first foreign id shape) for 'never' only.
|
|
1538
|
+
_effectiveIncompatible(requestInfo = {}) {
|
|
1515
1539
|
const sticky = requestInfo.sessionKey ? this.sessionPolicies.get(requestInfo.sessionKey) : null;
|
|
1516
|
-
if (sticky?.originClass === 'foreign') {
|
|
1517
|
-
return { class: 'foreign', provider: sticky.originProvider || requestInfo.originProvider || null };
|
|
1518
|
-
}
|
|
1519
|
-
if (requestInfo.originClass === 'foreign') {
|
|
1520
|
-
return { class: 'foreign', provider: requestInfo.originProvider || null };
|
|
1521
|
-
}
|
|
1522
1540
|
return {
|
|
1523
|
-
|
|
1524
|
-
|
|
1541
|
+
incompatible: Boolean(requestInfo.anthropicIncompatible || sticky?.anthropicIncompatible),
|
|
1542
|
+
homeProvider: requestInfo.homeProvider || sticky?.homeProvider || null,
|
|
1525
1543
|
};
|
|
1526
1544
|
}
|
|
1527
1545
|
|
|
1528
1546
|
_isRequestCompatible(account, profile, requestInfo = {}) {
|
|
1529
1547
|
if (!this._matchesProfile(account, profile)) return false;
|
|
1530
1548
|
|
|
1531
|
-
const
|
|
1549
|
+
const { incompatible, homeProvider } = this._effectiveIncompatible(requestInfo);
|
|
1550
|
+
const policy = this._crossProviderFallbackPolicy();
|
|
1532
1551
|
|
|
1533
|
-
if (
|
|
1534
|
-
//
|
|
1535
|
-
// Anthropic
|
|
1536
|
-
//
|
|
1537
|
-
//
|
|
1538
|
-
// force-pinned to Claude (the misroute that caused the reported 400).
|
|
1552
|
+
if (incompatible) {
|
|
1553
|
+
// The transcript can't replay to Claude (a foreign server_tool_use id, or
|
|
1554
|
+
// content Anthropic rejected on replay) — provider accounts ONLY, regardless
|
|
1555
|
+
// of policy. Providers are lenient and accept each other's ids (GLM↔Kimi is
|
|
1556
|
+
// fine); 'never' pins to the detected home provider when known.
|
|
1539
1557
|
if (account.type !== 'provider') return false;
|
|
1540
|
-
|
|
1541
|
-
// 'when-exhausted'/'always' let the sibling provider serve too (priority 10<20
|
|
1542
|
-
// keeps it a fallback). Unknown origin provider ⇒ any provider allowed.
|
|
1543
|
-
if (this._crossProviderFallbackPolicy() === 'never'
|
|
1544
|
-
&& origin.provider && account.provider !== origin.provider) return false;
|
|
1558
|
+
if (policy === 'never' && homeProvider && account.provider !== homeProvider) return false;
|
|
1545
1559
|
return true;
|
|
1546
1560
|
}
|
|
1547
1561
|
|
|
1548
|
-
//
|
|
1549
|
-
//
|
|
1550
|
-
//
|
|
1551
|
-
//
|
|
1552
|
-
|
|
1553
|
-
|
|
1554
|
-
|
|
1555
|
-
|
|
1562
|
+
// Compatible session — includes Kimi and GLM-without-server-tools, whose regular
|
|
1563
|
+
// tool_use ids pass Anthropic's loose validation, AND ordinary Claude sessions.
|
|
1564
|
+
// Claude is eligible + preferred (priority 0). Provider fallback is the SAFE
|
|
1565
|
+
// direction (lenient providers accept Anthropic ids/signatures) and is
|
|
1566
|
+
// policy-gated ONLY: 'never' keeps a Claude session on Claude; 'when-exhausted'
|
|
1567
|
+
// lets providers serve as a priority-fallback; 'always' peers them
|
|
1568
|
+
// (_effectivePriority). Signed thinking no longer bars providers here — but its
|
|
1569
|
+
// live MIGRATION stays Claude-only (see the rebalance guard).
|
|
1570
|
+
if (account.type === 'provider' && policy === 'never') return false;
|
|
1556
1571
|
return true;
|
|
1557
1572
|
}
|
|
1558
1573
|
|
|
@@ -1561,33 +1576,34 @@ export class AccountManager {
|
|
|
1561
1576
|
if (requestInfo.requiresAnthropicThinkingIntegrity) {
|
|
1562
1577
|
this.markSessionThinkingProtected(requestInfo.sessionKey, requestInfo.model);
|
|
1563
1578
|
}
|
|
1564
|
-
if (requestInfo.
|
|
1565
|
-
this.
|
|
1579
|
+
if (requestInfo.anthropicIncompatible) {
|
|
1580
|
+
this.markSessionIncompatible(requestInfo.sessionKey, requestInfo.homeProvider);
|
|
1566
1581
|
}
|
|
1567
1582
|
}
|
|
1568
1583
|
|
|
1569
|
-
// Latch a session
|
|
1570
|
-
// Anthropic
|
|
1571
|
-
// provider-pinned
|
|
1572
|
-
|
|
1573
|
-
if (!sessionKey
|
|
1584
|
+
// Latch a session as Anthropic-incompatible (a foreign server_tool_use id, or a
|
|
1585
|
+
// transcript Anthropic rejected on replay). Sticky + never-downgrades so the
|
|
1586
|
+
// session stays provider-pinned across later follow-up turns.
|
|
1587
|
+
markSessionIncompatible(sessionKey, homeProvider = null) {
|
|
1588
|
+
if (!sessionKey) return;
|
|
1574
1589
|
const existing = this.sessionPolicies.get(sessionKey) || {};
|
|
1575
|
-
if (existing.
|
|
1576
|
-
console.log(`[Maxpool] Session "${sessionKey}" is ${
|
|
1590
|
+
if (!existing.anthropicIncompatible) {
|
|
1591
|
+
console.log(`[Maxpool] Session "${sessionKey}" is Anthropic-incompatible (${homeProvider || 'provider'} transcript) — pinned to GLM/Kimi`);
|
|
1577
1592
|
}
|
|
1578
1593
|
this.sessionPolicies.set(sessionKey, {
|
|
1579
1594
|
...existing,
|
|
1580
|
-
|
|
1581
|
-
|
|
1595
|
+
anthropicIncompatible: true,
|
|
1596
|
+
homeProvider: existing.homeProvider || homeProvider || null,
|
|
1582
1597
|
});
|
|
1583
1598
|
}
|
|
1584
1599
|
|
|
1600
|
+
// Marks a session as containing Anthropic signed thinking. This no longer bars
|
|
1601
|
+
// provider fallback (a lenient provider accepts an Anthropic signature) — it only
|
|
1602
|
+
// keeps the session's live cross-account MIGRATION on Claude (the rebalance guard),
|
|
1603
|
+
// so a signed block isn't needlessly shuttled to a provider mid-session.
|
|
1585
1604
|
markSessionThinkingProtected(sessionKey, model = null) {
|
|
1586
1605
|
if (!sessionKey) return;
|
|
1587
1606
|
const existing = this.sessionPolicies.get(sessionKey) || {};
|
|
1588
|
-
if (!existing.requiresAnthropicThinkingIntegrity) {
|
|
1589
|
-
console.log(`[Maxpool] Session "${sessionKey}" contains Anthropic signed thinking; provider fallback disabled`);
|
|
1590
|
-
}
|
|
1591
1607
|
this.sessionPolicies.set(sessionKey, {
|
|
1592
1608
|
...existing,
|
|
1593
1609
|
requiresAnthropicThinkingIntegrity: true,
|
|
@@ -1651,6 +1667,16 @@ export class AccountManager {
|
|
|
1651
1667
|
// soft de-preference of accounts burning ahead of an even pace. Never a bench.
|
|
1652
1668
|
const paceCost = this._accountScarcity(account, now) * this.scheduler.paceCostWeight;
|
|
1653
1669
|
|
|
1670
|
+
// Per-model weekly de-preference: an account whose scoped weekly for THIS
|
|
1671
|
+
// request's model (e.g. Fable) is high-but-not-exhausted is a poor pick for
|
|
1672
|
+
// that model — shed its load toward healthier accounts BEFORE the hard bench
|
|
1673
|
+
// at weeklyExhaustedThreshold, so it's chosen only as overflow (rarely
|
|
1674
|
+
// re-429ing). Scoped weekly is otherwise absent from scoring. Soft, never a
|
|
1675
|
+
// bench; 0 below reserve and for models with no scoped cap.
|
|
1676
|
+
const scopedPace = requestInfo.model
|
|
1677
|
+
? this._scopedScarcity(account, requestInfo.model, now) * this.scheduler.paceCostWeight
|
|
1678
|
+
: 0;
|
|
1679
|
+
|
|
1654
1680
|
const fleetRecentWeight = ctx?.fleetRecentWeight ?? 0;
|
|
1655
1681
|
const recentWeight = this._loadSummary(account, this.scheduler.spreadWindowMs, now).weight;
|
|
1656
1682
|
const share = fleetRecentWeight > 0 ? recentWeight / fleetRecentWeight : 0;
|
|
@@ -1666,7 +1692,22 @@ export class AccountManager {
|
|
|
1666
1692
|
// default) learns the real number within a cycle. `probing`/requalify still
|
|
1667
1693
|
// flags a never-seen account for learning — that path is unchanged.
|
|
1668
1694
|
|
|
1669
|
-
return concurrency + capPenalty + paceCost + spread + ramp + failurePenalty;
|
|
1695
|
+
return concurrency + capPenalty + paceCost + scopedPace + spread + ramp + failurePenalty;
|
|
1696
|
+
}
|
|
1697
|
+
|
|
1698
|
+
/**
|
|
1699
|
+
* Per-model weekly pace-overage for `model`'s family, or 0 when the account has
|
|
1700
|
+
* no scoped cap for it, the cap is inactive, or it's below the reserve tier
|
|
1701
|
+
* (plenty of headroom → no steering). Same pace discount as _windowScarcity, so
|
|
1702
|
+
* a scoped window about to reset is cheap to spend.
|
|
1703
|
+
*/
|
|
1704
|
+
_scopedScarcity(account, model, now = Date.now()) {
|
|
1705
|
+
const fam = modelFamily(model);
|
|
1706
|
+
if (!fam) return 0;
|
|
1707
|
+
const e = account.quota?.scopedWeekly?.[fam];
|
|
1708
|
+
if (!e || e.isActive === false || e.utilization == null) return 0;
|
|
1709
|
+
if (e.utilization < this.scheduler.weeklyReserveThreshold) return 0;
|
|
1710
|
+
return this._windowScarcity(e.utilization, e.resetAt, WEEK_MS, now);
|
|
1670
1711
|
}
|
|
1671
1712
|
|
|
1672
1713
|
/**
|
|
@@ -1798,8 +1839,15 @@ export class AccountManager {
|
|
|
1798
1839
|
}
|
|
1799
1840
|
// Per-model weekly sub-limits (Fable, Opus, ...). Replace wholesale with the
|
|
1800
1841
|
// fresh probe set so a family that dropped out of the response doesn't linger
|
|
1801
|
-
// stale; expiry on reset is a backstop for the between-probe window.
|
|
1842
|
+
// stale; expiry on reset is a backstop for the between-probe window. EXCEPTION:
|
|
1843
|
+
// a `reactive` scoped-429 bench is authoritative-high — while its resetAt is
|
|
1844
|
+
// still future, a lagging probe may neither lower it nor drop it (a probe that
|
|
1845
|
+
// omits the family, or reports the pre-429 level, would otherwise un-bench it
|
|
1846
|
+
// and trigger an immediate re-429 flap). _clearExpiredQuotas self-clears it at
|
|
1847
|
+
// resetAt even if probes die.
|
|
1802
1848
|
if (usage.scopedWeekly && typeof usage.scopedWeekly === 'object') {
|
|
1849
|
+
const now = Date.now();
|
|
1850
|
+
const prev = (q.scopedWeekly && typeof q.scopedWeekly === 'object') ? q.scopedWeekly : {};
|
|
1803
1851
|
const fresh = {};
|
|
1804
1852
|
for (const [fam, e] of Object.entries(usage.scopedWeekly)) {
|
|
1805
1853
|
if (!e) continue;
|
|
@@ -1810,9 +1858,21 @@ export class AccountManager {
|
|
|
1810
1858
|
isActive: e.isActive !== false,
|
|
1811
1859
|
};
|
|
1812
1860
|
}
|
|
1861
|
+
for (const [fam, pe] of Object.entries(prev)) {
|
|
1862
|
+
if (!pe || !pe.reactive || pe.resetAt == null || pe.resetAt <= now) continue;
|
|
1863
|
+
const f = fresh[fam];
|
|
1864
|
+
// Probe absent, null, or LOWER than the reactive bench → keep the bench.
|
|
1865
|
+
// Probe CONFIRMS >= the reactive level → take the fresh reading (still >=
|
|
1866
|
+
// threshold, so still exhausted; no stickiness needed).
|
|
1867
|
+
if (!f || f.utilization == null || f.utilization < (pe.utilization ?? 0)) {
|
|
1868
|
+
fresh[fam] = { ...pe };
|
|
1869
|
+
}
|
|
1870
|
+
}
|
|
1813
1871
|
q.scopedWeekly = fresh;
|
|
1814
1872
|
}
|
|
1815
1873
|
|
|
1874
|
+
q.lastProbeOkAt = Date.now();
|
|
1875
|
+
|
|
1816
1876
|
// If we just learned this account's weekly window while probing, re-evaluate
|
|
1817
1877
|
// selection (same path as learning it from a live response).
|
|
1818
1878
|
if (account.probing && q.unified7dReset != null) {
|
|
@@ -1821,6 +1881,55 @@ export class AccountManager {
|
|
|
1821
1881
|
}
|
|
1822
1882
|
}
|
|
1823
1883
|
|
|
1884
|
+
/**
|
|
1885
|
+
* Update a PROVIDER account's quota from a provider usage probe
|
|
1886
|
+
* (fetchProviderUsage). z.ai maps to Ses/Wk token windows; Kimi has no pollable
|
|
1887
|
+
* source and only sets a `console-only` marker. Writes ONLY the provider* fields
|
|
1888
|
+
* (never the unified or scopedWeekly fields) so a provider reading can't reach
|
|
1889
|
+
* the OAuth quota gates.
|
|
1890
|
+
*/
|
|
1891
|
+
applyProviderUsage(accountIndex, usage) {
|
|
1892
|
+
const account = this.accounts[accountIndex];
|
|
1893
|
+
if (!account || !usage) return;
|
|
1894
|
+
const q = account.quota;
|
|
1895
|
+
if (usage.error) {
|
|
1896
|
+
// Distinguish "no pollable quota" (Kimi) from a transient probe failure.
|
|
1897
|
+
// Never clear existing values on a transient error — let them age into the
|
|
1898
|
+
// staleness marker instead of blanking the bars.
|
|
1899
|
+
if (usage.source === 'console-only') q.providerQuotaSource = 'console-only';
|
|
1900
|
+
return;
|
|
1901
|
+
}
|
|
1902
|
+
q.providerQuotaSource = usage.source || 'zai';
|
|
1903
|
+
if (usage.ses) {
|
|
1904
|
+
if (usage.ses.utilization != null) q.providerSes = clamp01(usage.ses.utilization);
|
|
1905
|
+
if (usage.ses.resetAt != null) q.providerSesReset = usage.ses.resetAt;
|
|
1906
|
+
}
|
|
1907
|
+
if (usage.wk) {
|
|
1908
|
+
if (usage.wk.utilization != null) q.providerWk = clamp01(usage.wk.utilization);
|
|
1909
|
+
if (usage.wk.resetAt != null) q.providerWkReset = usage.wk.resetAt;
|
|
1910
|
+
} else {
|
|
1911
|
+
// Weekly window absent from this plan/response — clear so a stale weekly
|
|
1912
|
+
// reading doesn't linger after a plan/window change.
|
|
1913
|
+
q.providerWk = null;
|
|
1914
|
+
q.providerWkReset = null;
|
|
1915
|
+
}
|
|
1916
|
+
q.lastProbeOkAt = Date.now();
|
|
1917
|
+
}
|
|
1918
|
+
|
|
1919
|
+
/**
|
|
1920
|
+
* True when the background quota probe hasn't succeeded in > 2× its interval —
|
|
1921
|
+
* the last-known scoped/provider values are aging with no confirmation. Returns
|
|
1922
|
+
* false when the probe is off (nothing to be stale against) or has never yet
|
|
1923
|
+
* succeeded (startup — shown as "no data", not "stale").
|
|
1924
|
+
*/
|
|
1925
|
+
_quotaProbeStale(account, now = Date.now()) {
|
|
1926
|
+
const interval = this.quotaProbeIntervalMs;
|
|
1927
|
+
if (!interval || interval <= 0) return false;
|
|
1928
|
+
const last = account?.quota?.lastProbeOkAt;
|
|
1929
|
+
if (last == null) return false;
|
|
1930
|
+
return (now - last) > Math.max(2 * interval, 120_000);
|
|
1931
|
+
}
|
|
1932
|
+
|
|
1824
1933
|
/**
|
|
1825
1934
|
* Update an account's quota tracking from upstream response headers.
|
|
1826
1935
|
*/
|
|
@@ -1959,6 +2068,10 @@ export class AccountManager {
|
|
|
1959
2068
|
resetAt: Date.now() + (retryAfter * 1000),
|
|
1960
2069
|
severity: 'critical',
|
|
1961
2070
|
isActive: true,
|
|
2071
|
+
// Authoritative-high: a real reject. A lagging 60s probe reporting the
|
|
2072
|
+
// pre-429 level (e.g. 0.96) must NOT lower/drop this before resetAt, else
|
|
2073
|
+
// the account un-benches and immediately re-429s — a per-probe flap.
|
|
2074
|
+
reactive: true,
|
|
1962
2075
|
};
|
|
1963
2076
|
account.lastStatus = options.status || 429;
|
|
1964
2077
|
account.lastErrorAt = Date.now();
|
|
@@ -2416,7 +2529,7 @@ export class AccountManager {
|
|
|
2416
2529
|
sessions: {
|
|
2417
2530
|
stickyBindings: this.sessionBindings.size,
|
|
2418
2531
|
thinkingProtected: [...this.sessionPolicies.values()].filter(p => p.requiresAnthropicThinkingIntegrity).length,
|
|
2419
|
-
|
|
2532
|
+
providerPinned: [...this.sessionPolicies.values()].filter(p => p.anthropicIncompatible).length,
|
|
2420
2533
|
},
|
|
2421
2534
|
};
|
|
2422
2535
|
}
|
package/src/index.js
CHANGED
|
@@ -564,6 +564,9 @@ async function serverWorkerCommand() {
|
|
|
564
564
|
// assertions (mirrors MAXPOOL_DISABLE_SLEEP_GUARD).
|
|
565
565
|
const probeSeconds = process.env.MAXPOOL_DISABLE_QUOTA_PROBE === '1' ? 0 : (config.quotaProbeSeconds || 0);
|
|
566
566
|
const prober = new Prober(accountManager, { intervalMs: probeSeconds * 1000 });
|
|
567
|
+
// Tell the AM the probe cadence so the TUI can flag a scoped/provider tag whose
|
|
568
|
+
// background probe has gone stale (> 2× interval since last success).
|
|
569
|
+
accountManager.quotaProbeIntervalMs = probeSeconds * 1000;
|
|
567
570
|
|
|
568
571
|
// Persist refreshed tokens back to config. Defense-in-depth: the updater reads
|
|
569
572
|
// the on-disk refresh token and SKIPS the rotation if a fresher writer already
|
package/src/oauth.js
CHANGED
|
@@ -245,6 +245,61 @@ export async function fetchUsage(accessToken) {
|
|
|
245
245
|
}
|
|
246
246
|
}
|
|
247
247
|
|
|
248
|
+
// z.ai quota monitor — a DIFFERENT host/path than the message upstream
|
|
249
|
+
// (account.upstream is the Anthropic-compat endpoint). Zero-spend read.
|
|
250
|
+
const ZAI_QUOTA_URL = 'https://api.z.ai/api/monitor/usage/quota/limit';
|
|
251
|
+
|
|
252
|
+
/** Classify one z.ai `limits[]` entry into a Ses (5h) or Wk (weekly) TOKEN window.
|
|
253
|
+
* Only `TOKENS_LIMIT` maps to the quota bars; `TIME_LIMIT` is a tool-call cap
|
|
254
|
+
* (web-search/reader counts) and is intentionally ignored. `unit` is z.ai's
|
|
255
|
+
* window enum (3 = 5-hour session, 6 = weekly); we fall back to reset-distance
|
|
256
|
+
* when the code is unfamiliar so a new plan tier still classifies sanely. */
|
|
257
|
+
export function classifyZaiLimit(l, now = Date.now()) {
|
|
258
|
+
if (!l || l.type !== 'TOKENS_LIMIT') return null;
|
|
259
|
+
const reset = Number(l.nextResetTime);
|
|
260
|
+
const resetAt = Number.isFinite(reset) && reset > 0 ? reset : null;
|
|
261
|
+
const pct = typeof l.percentage === 'number' ? l.percentage : parseFloat(l.percentage);
|
|
262
|
+
const utilization = Number.isFinite(pct) ? Math.max(0, Math.min(1, pct / 100)) : null;
|
|
263
|
+
let bucket;
|
|
264
|
+
if (l.unit === 3) bucket = 'ses';
|
|
265
|
+
else if (l.unit === 6) bucket = 'wk';
|
|
266
|
+
else if (resetAt) bucket = (resetAt - now) <= 12 * 60 * 60 * 1000 ? 'ses' : 'wk';
|
|
267
|
+
else bucket = 'ses';
|
|
268
|
+
return { bucket, utilization, resetAt };
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
/** Read a provider account's quota. z.ai has a pollable monitor endpoint mapped
|
|
272
|
+
* to Ses/Wk token windows; Kimi (Moonshot coding key) has NO pollable quota
|
|
273
|
+
* (web console only), so it returns a `console-only` marker instead of fake
|
|
274
|
+
* bars. Returns { ses, wk, level } | { error, status?, source? }. */
|
|
275
|
+
export async function fetchProviderUsage(account) {
|
|
276
|
+
const provider = account?.provider;
|
|
277
|
+
const token = account?.credential;
|
|
278
|
+
if (provider === 'kimi') return { error: 'unsupported', source: 'console-only' };
|
|
279
|
+
if (provider !== 'zai' || !token) return { error: 'unsupported', source: null };
|
|
280
|
+
try {
|
|
281
|
+
const res = await fetch(ZAI_QUOTA_URL, {
|
|
282
|
+
headers: { 'Authorization': `Bearer ${token}`, 'Accept': 'application/json' },
|
|
283
|
+
});
|
|
284
|
+
if (!res.ok) return { error: `HTTP ${res.status}`, status: res.status };
|
|
285
|
+
const data = await res.json();
|
|
286
|
+
if (data?.code !== 200 || !data?.data) {
|
|
287
|
+
return { error: `bad_envelope${data?.code != null ? ' code=' + data.code : ''}` };
|
|
288
|
+
}
|
|
289
|
+
const now = Date.now();
|
|
290
|
+
let ses = null, wk = null;
|
|
291
|
+
for (const l of (Array.isArray(data.data.limits) ? data.data.limits : [])) {
|
|
292
|
+
const c = classifyZaiLimit(l, now);
|
|
293
|
+
if (!c) continue;
|
|
294
|
+
if (c.bucket === 'ses') ses = { utilization: c.utilization, resetAt: c.resetAt };
|
|
295
|
+
else if (c.bucket === 'wk') wk = { utilization: c.utilization, resetAt: c.resetAt };
|
|
296
|
+
}
|
|
297
|
+
return { ses, wk, level: data.data.level || null, source: 'zai' };
|
|
298
|
+
} catch (err) {
|
|
299
|
+
return { error: err.message || String(err), status: null };
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
|
|
248
303
|
// OAuth config (extracted from Claude Code)
|
|
249
304
|
const OAUTH_CLIENT_ID = '9d1c250a-e61b-44d9-88ed-5944d1962f5e';
|
|
250
305
|
const OAUTH_AUTHORIZE = 'https://claude.ai/oauth/authorize';
|
package/src/prober.js
CHANGED
|
@@ -7,13 +7,14 @@
|
|
|
7
7
|
// is blind to an account it isn't actively routing to and will pile traffic onto it.
|
|
8
8
|
// This is the one sanctioned active-upstream feature; the proxy is otherwise passive.
|
|
9
9
|
|
|
10
|
-
import { fetchUsage } from './oauth.js';
|
|
10
|
+
import { fetchUsage, fetchProviderUsage } from './oauth.js';
|
|
11
11
|
|
|
12
12
|
export class Prober {
|
|
13
|
-
constructor(accountManager, { intervalMs = 0, probeFn = fetchUsage, timeoutMs = 10_000, log = console.log } = {}) {
|
|
13
|
+
constructor(accountManager, { intervalMs = 0, probeFn = fetchUsage, providerProbeFn = fetchProviderUsage, timeoutMs = 10_000, log = console.log } = {}) {
|
|
14
14
|
this.am = accountManager;
|
|
15
15
|
this.intervalMs = intervalMs;
|
|
16
16
|
this.probeFn = probeFn;
|
|
17
|
+
this.providerProbeFn = providerProbeFn;
|
|
17
18
|
this.timeoutMs = timeoutMs;
|
|
18
19
|
this.log = log;
|
|
19
20
|
this.timer = null;
|
|
@@ -49,15 +50,21 @@ export class Prober {
|
|
|
49
50
|
if (this._inflight) { try { await this._inflight; } catch { /* swallow */ } }
|
|
50
51
|
}
|
|
51
52
|
|
|
52
|
-
/** Probe every OAuth account
|
|
53
|
-
*
|
|
53
|
+
/** Probe every OAuth account (usage endpoint) and every provider account with a
|
|
54
|
+
* pollable/known quota source (z.ai monitor; Kimi → console-only marker) once.
|
|
55
|
+
* Overlapping cycles are skipped. The active cycle is tracked on `_inflight` so
|
|
56
|
+
* stop() can await it. */
|
|
54
57
|
probeAll() {
|
|
55
58
|
if (this._running) return this._inflight || Promise.resolve();
|
|
56
59
|
this._running = true;
|
|
57
60
|
this._inflight = (async () => {
|
|
58
61
|
try {
|
|
59
|
-
const
|
|
60
|
-
|
|
62
|
+
const oauth = this.am.accounts.filter(a => a.type === 'oauth' && a.credential);
|
|
63
|
+
const providers = this.am.accounts.filter(a => a.type === 'provider' && a.credential);
|
|
64
|
+
await Promise.all([
|
|
65
|
+
...oauth.map(a => this.probeOne(a)),
|
|
66
|
+
...providers.map(a => this.probeProvider(a)),
|
|
67
|
+
]);
|
|
61
68
|
} finally {
|
|
62
69
|
this._running = false;
|
|
63
70
|
this._inflight = null;
|
|
@@ -66,6 +73,16 @@ export class Prober {
|
|
|
66
73
|
return this._inflight;
|
|
67
74
|
}
|
|
68
75
|
|
|
76
|
+
/** Probe one PROVIDER account. Provider tokens are static API keys (no OAuth
|
|
77
|
+
* refresh). Best-effort; never throws. */
|
|
78
|
+
async probeProvider(account) {
|
|
79
|
+
try {
|
|
80
|
+
const usage = await this._withTimeout(this.providerProbeFn(account));
|
|
81
|
+
if (!usage) return; // timed out — try again next cycle
|
|
82
|
+
this.am.applyProviderUsage(account.index, usage);
|
|
83
|
+
} catch { /* best-effort; never let a probe throw */ }
|
|
84
|
+
}
|
|
85
|
+
|
|
69
86
|
async probeOne(account) {
|
|
70
87
|
try {
|
|
71
88
|
await this.am.ensureTokenFresh(account.index);
|
package/src/server.js
CHANGED
|
@@ -736,8 +736,14 @@ async function forwardRequest(
|
|
|
736
736
|
|
|
737
737
|
if (upstreamRes.status >= 400 && upstreamRes.status < 500) {
|
|
738
738
|
const errorBody = await readErrorBody(upstreamRes);
|
|
739
|
+
// A transcript a lenient provider (GLM/Kimi) produced can be rejected by
|
|
740
|
+
// Anthropic on replay (a non-srvtoolu_ server_tool_use id, or a thinking
|
|
741
|
+
// signature it can't validate). Detect it on an Anthropic account so we can
|
|
742
|
+
// self-heal onto a provider instead of surfacing the 400.
|
|
743
|
+
const anthropicIncompat = account.type !== 'provider' && isAnthropicIncompatBody(errorBody);
|
|
739
744
|
const errorType = errorBody.includes('Invalid `signature` in `thinking` block')
|
|
740
745
|
? 'invalid_thinking_signature'
|
|
746
|
+
: anthropicIncompat ? 'anthropic_incompatible_transcript'
|
|
741
747
|
: `HTTP ${upstreamRes.status}`;
|
|
742
748
|
accountManager.releaseAccount(lease, { status: upstreamRes.status, error: errorType });
|
|
743
749
|
|
|
@@ -768,6 +774,24 @@ async function forwardRequest(
|
|
|
768
774
|
}
|
|
769
775
|
}
|
|
770
776
|
|
|
777
|
+
// React-and-heal: this transcript can't run on Claude (foreign server_tool_use
|
|
778
|
+
// id / thinking Anthropic can't validate). The 400 is pre-stream and the body
|
|
779
|
+
// is buffered, so latch the session Anthropic-incompatible (sticky → once per
|
|
780
|
+
// session) and retry PROVIDER-only, rather than surfacing the 400. Only worth
|
|
781
|
+
// it when a provider can actually serve it (profile=all with a GLM/Kimi token).
|
|
782
|
+
const providerAvailable = accountManager.accounts?.some(a => a.type === 'provider' && a.enabled !== false);
|
|
783
|
+
if (anthropicIncompat && requestInfo.sessionKey && providerAvailable
|
|
784
|
+
&& canRetryBufferedBody && retryCount + 1 < maxAttempts && !res.headersSent) {
|
|
785
|
+
accountManager.markSessionIncompatible?.(requestInfo.sessionKey, requestInfo.homeProvider);
|
|
786
|
+
excludedIndexes.add(account.index);
|
|
787
|
+
console.log(`[Maxpool] Anthropic rejected this transcript (server_tool_use/thinking); pinning session to GLM/Kimi and retrying`);
|
|
788
|
+
return forwardRequest(
|
|
789
|
+
req, res, body, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
|
|
790
|
+
retryConfig, queueConfig, { ...requestInfo, anthropicIncompatible: true },
|
|
791
|
+
canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
|
|
792
|
+
);
|
|
793
|
+
}
|
|
794
|
+
|
|
771
795
|
ctx.status = upstreamRes.status;
|
|
772
796
|
sendErrorBody(res, requestInfo, upstreamRes.status, errorBody, upstreamRes.headers);
|
|
773
797
|
return;
|
|
@@ -987,15 +1011,15 @@ function computeQueueWindowMs({
|
|
|
987
1011
|
}
|
|
988
1012
|
|
|
989
1013
|
function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRecoverSoon = true) {
|
|
990
|
-
const
|
|
1014
|
+
const incompat = accountManager._effectiveIncompatible?.(requestInfo) || { incompatible: false, homeProvider: null };
|
|
991
1015
|
|
|
992
|
-
//
|
|
993
|
-
//
|
|
994
|
-
//
|
|
995
|
-
if (
|
|
996
|
-
const fam =
|
|
1016
|
+
// An Anthropic-incompatible session is pinned to provider accounts — Anthropic
|
|
1017
|
+
// 400s on its server_tool_use id / foreign thinking. "All Claude at limit" would
|
|
1018
|
+
// be the wrong story (Claude is irrelevant to it); say what actually blocks it.
|
|
1019
|
+
if (incompat.incompatible) {
|
|
1020
|
+
const fam = incompat.homeProvider === 'zai' ? 'GLM' : incompat.homeProvider === 'kimi' ? 'Kimi' : 'GLM/Kimi';
|
|
997
1021
|
const eta = Number.isFinite(retryAfter) && retryAfter > 0 ? ` Retry in ${retryAfter}s.` : '';
|
|
998
|
-
return `This session
|
|
1022
|
+
return `This session's transcript can only run on GLM/Kimi — Claude rejects its server-tool ids/thinking on replay. No GLM/Kimi provider is available right now.${eta} Check the x-maxpool-zai-token / x-maxpool-kimi-token headers, or resume with 'cc ${incompat.homeProvider === 'kimi' ? 'kimi' : 'glm'}'.`;
|
|
999
1023
|
}
|
|
1000
1024
|
|
|
1001
1025
|
const thinking = requestInfo.requiresAnthropicThinkingIntegrity
|
|
@@ -1021,7 +1045,7 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
|
|
|
1021
1045
|
return `All ${n} accounts exhausted. Retry in ${retryAfter}s.`;
|
|
1022
1046
|
}
|
|
1023
1047
|
|
|
1024
|
-
export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin };
|
|
1048
|
+
export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody };
|
|
1025
1049
|
|
|
1026
1050
|
async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
|
|
1027
1051
|
if (!upstreamRes.body) return '';
|
|
@@ -1489,15 +1513,13 @@ function describeRequest(req, body) {
|
|
|
1489
1513
|
// rebalancing); an unparsed body leaves this false → treated as NOT safe
|
|
1490
1514
|
// (fail-closed) so we never replay a signed thinking block to a new account.
|
|
1491
1515
|
info.bodyThinkingScanned = true;
|
|
1492
|
-
//
|
|
1493
|
-
//
|
|
1494
|
-
//
|
|
1495
|
-
//
|
|
1496
|
-
// an unparsed body; both fail OPEN in routing (never stranded), and stickiness
|
|
1497
|
-
// (once a foreign id is seen) keeps a latched session pinned regardless.
|
|
1516
|
+
// Whether the resumed transcript can replay to a Claude account. A session with
|
|
1517
|
+
// a foreign server_tool_use id CANNOT (Anthropic 400s) → routing pins it to
|
|
1518
|
+
// providers. Everything else stays Claude-eligible (Kimi/GLM without server-tools
|
|
1519
|
+
// replay fine); a rare rejected-thinking 400 self-heals via react-and-heal.
|
|
1498
1520
|
const origin = detectTranscriptOrigin(json);
|
|
1499
|
-
if (origin.
|
|
1500
|
-
if (origin.
|
|
1521
|
+
if (origin.anthropicIncompatible) info.anthropicIncompatible = true;
|
|
1522
|
+
if (origin.homeProvider) info.homeProvider = origin.homeProvider;
|
|
1501
1523
|
} catch {
|
|
1502
1524
|
// Non-JSON requests are rare; body size still gives a useful load signal.
|
|
1503
1525
|
}
|
|
@@ -1517,39 +1539,57 @@ function requiresAnthropicThinkingIntegrity(json) {
|
|
|
1517
1539
|
// anything NOT matching this is treated as foreign (fail-closed classification).
|
|
1518
1540
|
const ANTHROPIC_TOOL_ID = /^(toolu|srvtoolu)_/;
|
|
1519
1541
|
|
|
1520
|
-
//
|
|
1521
|
-
//
|
|
1522
|
-
//
|
|
1523
|
-
//
|
|
1524
|
-
//
|
|
1525
|
-
//
|
|
1526
|
-
//
|
|
1542
|
+
// Decide whether a resumed transcript can replay to an Anthropic (Claude) account,
|
|
1543
|
+
// from the tool-use id shapes in its `messages`. Returns { anthropicIncompatible,
|
|
1544
|
+
// homeProvider }.
|
|
1545
|
+
// anthropicIncompatible = the transcript has a `server_tool_use` id NOT matching
|
|
1546
|
+
// ^srvtoolu_ — the ONE DETERMINISTIC incompatibility (Anthropic 400s on replay,
|
|
1547
|
+
// the reported bug). Anthropic validates a client `tool_use.id` LOOSELY
|
|
1548
|
+
// (^[a-zA-Z0-9_-]+$), so GLM `call_…` / Kimi `tool_…` client ids PASS and a
|
|
1549
|
+
// Kimi/GLM session without server-tools is FINE on Claude — we do NOT predict
|
|
1550
|
+
// those (a rejected thinking signature, if it ever happens, self-heals via the
|
|
1551
|
+
// 4xx react-and-heal in forwardRequest). server_tool_use is rare (GLM ~15%,
|
|
1552
|
+
// Kimi 0%), so most sessions are compatible.
|
|
1553
|
+
// homeProvider = the first foreign tool-use id shape (call_→zai, tool_→kimi), a
|
|
1554
|
+
// SOFT hint for the 'never' same-family preference only; ambiguous/none → null.
|
|
1555
|
+
// The request MODEL is not used — Claude Code rewrites it to opus on resume.
|
|
1527
1556
|
function detectTranscriptOrigin(json) {
|
|
1528
|
-
if (!json || typeof json !== 'object') return {
|
|
1529
|
-
let
|
|
1530
|
-
let
|
|
1531
|
-
const
|
|
1532
|
-
|
|
1557
|
+
if (!json || typeof json !== 'object') return { anthropicIncompatible: false, homeProvider: null };
|
|
1558
|
+
let incompatible = false;
|
|
1559
|
+
let homeProvider = null;
|
|
1560
|
+
const fam = id => id.startsWith('call_') ? 'zai' : id.startsWith('tool_') ? 'kimi' : null;
|
|
1561
|
+
const noteForeign = id => { if (!homeProvider) homeProvider = fam(id); };
|
|
1533
1562
|
const visit = value => {
|
|
1534
|
-
if (
|
|
1535
|
-
if (Array.isArray(value)) { for (const v of value) { visit(v); if (
|
|
1563
|
+
if (incompatible) return; // server_tool_use is decisive — stop
|
|
1564
|
+
if (Array.isArray(value)) { for (const v of value) { visit(v); if (incompatible) return; } return; }
|
|
1536
1565
|
if (!value || typeof value !== 'object') return;
|
|
1537
1566
|
const t = value.type;
|
|
1538
|
-
if (
|
|
1539
|
-
|
|
1540
|
-
else { foreignProvider = classifyForeign(value.id); return; }
|
|
1541
|
-
}
|
|
1542
|
-
if (t === 'tool_result' && typeof value.tool_use_id === 'string') {
|
|
1543
|
-
if (ANTHROPIC_TOOL_ID.test(value.tool_use_id)) sawAnthropic = true;
|
|
1544
|
-
else { foreignProvider = classifyForeign(value.tool_use_id); return; }
|
|
1567
|
+
if (t === 'server_tool_use' && typeof value.id === 'string' && !ANTHROPIC_TOOL_ID.test(value.id)) {
|
|
1568
|
+
incompatible = true; noteForeign(value.id); return;
|
|
1545
1569
|
}
|
|
1570
|
+
if (t === 'tool_use' && typeof value.id === 'string' && !ANTHROPIC_TOOL_ID.test(value.id)) noteForeign(value.id);
|
|
1571
|
+
if (t === 'tool_result' && typeof value.tool_use_id === 'string' && !ANTHROPIC_TOOL_ID.test(value.tool_use_id)) noteForeign(value.tool_use_id);
|
|
1546
1572
|
if (value.content) visit(value.content);
|
|
1547
1573
|
if (value.messages) visit(value.messages);
|
|
1548
1574
|
};
|
|
1549
1575
|
visit(json.messages);
|
|
1550
|
-
|
|
1551
|
-
|
|
1552
|
-
|
|
1576
|
+
return { anthropicIncompatible: incompatible, homeProvider };
|
|
1577
|
+
}
|
|
1578
|
+
|
|
1579
|
+
// Does this upstream 4xx error body indicate the transcript is un-replayable to
|
|
1580
|
+
// Anthropic (a lenient GLM/Kimi provider produced content Anthropic rejects on
|
|
1581
|
+
// replay)? Matches the three known shapes: a non-srvtoolu_ server_tool_use id, or a
|
|
1582
|
+
// thinking-block signature Anthropic can't validate (invalid OR missing). Used to
|
|
1583
|
+
// self-heal onto a provider instead of surfacing the 400.
|
|
1584
|
+
function isAnthropicIncompatBody(body) {
|
|
1585
|
+
if (!body) return false;
|
|
1586
|
+
// The deterministic server-tool-id 400, the known invalid-signature error, or a
|
|
1587
|
+
// thinking-block signature VALIDATION error (missing/invalid) — the last gated on
|
|
1588
|
+
// a validation verb so a user message merely echoing "thinking"/"signature" can't
|
|
1589
|
+
// false-latch the session provider-pinned for life.
|
|
1590
|
+
return /srvtoolu_|server_tool_use/.test(body)
|
|
1591
|
+
|| /invalid `signature` in `thinking`/i.test(body)
|
|
1592
|
+
|| (/thinking/.test(body) && /signature/.test(body) && /(should match|required|invalid|expected|must )/i.test(body));
|
|
1553
1593
|
}
|
|
1554
1594
|
|
|
1555
1595
|
function containsThinkingBlock(value) {
|
package/src/tui.js
CHANGED
|
@@ -864,6 +864,11 @@ export class TUI {
|
|
|
864
864
|
lines.push(yellow(' No accounts configured. Press [a] to add one.'));
|
|
865
865
|
} else {
|
|
866
866
|
lines.push('');
|
|
867
|
+
// Column legend — the per-row numbers are otherwise cryptic. Ses/Wk are the
|
|
868
|
+
// two quota bars; Now is live concurrency; 15m/1h are recent throughput.
|
|
869
|
+
if (W >= 88) {
|
|
870
|
+
lines.push(' ' + dim('Ses/Wk = 5h/7d quota (used% · resets-in) · Now = in-flight (weight) · 15m/1h = requests served (avg latency · Nf=fails)'));
|
|
871
|
+
}
|
|
867
872
|
const showBoth = W >= 70;
|
|
868
873
|
const bw = showBoth
|
|
869
874
|
? Math.max(5, Math.min(20, Math.floor((W - 56) / 2)))
|
|
@@ -979,7 +984,7 @@ export class TUI {
|
|
|
979
984
|
status = rpad(status, 13);
|
|
980
985
|
|
|
981
986
|
if (a.type === 'provider') {
|
|
982
|
-
return this._renderProviderAcct(sel, cur, name, type, status, a);
|
|
987
|
+
return this._renderProviderAcct(sel, cur, name, type, status, a, bw, showBoth);
|
|
983
988
|
}
|
|
984
989
|
|
|
985
990
|
// Quota ratios — prefer unified (Claude Max), fall back to standard (API key)
|
|
@@ -1008,25 +1013,53 @@ export class TUI {
|
|
|
1008
1013
|
}
|
|
1009
1014
|
const weekly = weeklyPolicyText(this.am, a);
|
|
1010
1015
|
if (weekly) line += ` ${weekly}`;
|
|
1011
|
-
// Per-model weekly caps (e.g. Fable
|
|
1012
|
-
// headroom)
|
|
1013
|
-
//
|
|
1016
|
+
// Per-model weekly caps (e.g. Fable, while the unified weekly still has
|
|
1017
|
+
// headroom). Show the ACTUAL utilization — "Fable 90%" (yellow) while high but
|
|
1018
|
+
// still usable, "Fable maxed" (red) ONLY at genuine exhaustion. This is the
|
|
1019
|
+
// SAME predicate the router benches on (_scopedExhausted), so "maxed" renders
|
|
1020
|
+
// iff the model is actually benched — 90%/critical is no longer mislabelled.
|
|
1014
1021
|
const exhaustedFloor = this.am.scheduler?.weeklyExhaustedThreshold ?? 0.985;
|
|
1015
|
-
const
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1022
|
+
const reserveFloor = this.am.scheduler?.weeklyReserveThreshold ?? 0.85;
|
|
1023
|
+
const scopedTags = [];
|
|
1024
|
+
for (const [fam, e] of Object.entries(q.scopedWeekly || {})) {
|
|
1025
|
+
if (!e || e.isActive === false || e.utilization == null) continue;
|
|
1026
|
+
if (e.utilization < reserveFloor) continue;
|
|
1027
|
+
const Fam = fam.charAt(0).toUpperCase() + fam.slice(1);
|
|
1028
|
+
scopedTags.push(e.utilization >= exhaustedFloor
|
|
1029
|
+
? red(`${Fam} maxed`)
|
|
1030
|
+
: yellow(`${Fam} ${Math.round(e.utilization * 100)}%`));
|
|
1031
|
+
}
|
|
1032
|
+
if (scopedTags.length) line += ` ${scopedTags.join(' ')}`;
|
|
1033
|
+
// Freshness: scoped caps are refreshed ONLY by the background probe (response
|
|
1034
|
+
// headers don't carry them). If the probe has gone stale (> 2× interval since
|
|
1035
|
+
// last success), say so rather than imply the last-known value is current.
|
|
1036
|
+
if (this.am._quotaProbeStale?.(a)) line += ` ${dim('stale')}`;
|
|
1020
1037
|
line += ` ${dim(loadText(this._accountLoad(a)))}`;
|
|
1021
1038
|
return line;
|
|
1022
1039
|
}
|
|
1023
1040
|
|
|
1024
|
-
_renderProviderAcct(sel, cur, name, type, status, a) {
|
|
1041
|
+
_renderProviderAcct(sel, cur, name, type, status, a, bw = 11, showBoth = true) {
|
|
1025
1042
|
const completed = a.completedRequests || 0;
|
|
1026
1043
|
const failed = a.failedRequests || 0;
|
|
1027
1044
|
const active = a.inFlight || 0;
|
|
1028
1045
|
const last = a.lastStatus ? `${statusColor(a.lastStatus)} ${formatMs(a.lastResponseMs)}` : '-';
|
|
1029
1046
|
const q = a.quota || {};
|
|
1047
|
+
|
|
1048
|
+
// Quota segment. z.ai has a pollable monitor endpoint → real Ses/Wk token bars,
|
|
1049
|
+
// same rendering as OAuth accounts. Kimi (Moonshot coding key) has NO pollable
|
|
1050
|
+
// quota (web console only) → an honest label, never a fake bar. Fields are the
|
|
1051
|
+
// SEPARATE provider* set, so a provider reading never leaks into an OAuth bar.
|
|
1052
|
+
let quotaSeg = '';
|
|
1053
|
+
if (q.providerSes != null || q.providerWk != null) {
|
|
1054
|
+
quotaSeg = ` Ses ${bar(q.providerSes, bw, q.providerSesReset)}`;
|
|
1055
|
+
if (showBoth && q.providerWk != null) quotaSeg += ` Wk ${bar(q.providerWk, bw, q.providerWkReset)}`;
|
|
1056
|
+
if (this.am._quotaProbeStale?.(a)) quotaSeg += ` ${dim('stale')}`;
|
|
1057
|
+
} else if (q.providerQuotaSource === 'console-only') {
|
|
1058
|
+
quotaSeg = ` ${dim('Quota console-only')}`;
|
|
1059
|
+
} else if (q.providerQuotaSource === 'zai') {
|
|
1060
|
+
quotaSeg = ` ${dim('Ses/Wk probing')}`;
|
|
1061
|
+
}
|
|
1062
|
+
|
|
1030
1063
|
let limit = '';
|
|
1031
1064
|
if (q.genericLimit != null && q.genericRemaining != null) {
|
|
1032
1065
|
const used = q.genericLimit - q.genericRemaining;
|
|
@@ -1034,7 +1067,7 @@ export class TUI {
|
|
|
1034
1067
|
limit = ` Lim ${used}/${q.genericLimit}${reset ? ` ${reset}` : ''}`;
|
|
1035
1068
|
}
|
|
1036
1069
|
const err = a.lastError ? ` Err ${String(a.lastError).slice(0, 18)}` : '';
|
|
1037
|
-
return ` ${sel}${cur} ${name} ${type} ${status} Act ${String(active).padStart(2)} OK ${String(completed).padStart(3)} Fail ${String(failed).padStart(2)} Last ${last} ${dim(loadText(this._accountLoad(a)))}${limit}${err}`;
|
|
1070
|
+
return ` ${sel}${cur} ${name} ${type} ${status} Act ${String(active).padStart(2)} OK ${String(completed).padStart(3)} Fail ${String(failed).padStart(2)} Last ${last}${quotaSeg} ${dim(loadText(this._accountLoad(a)))}${limit}${err}`;
|
|
1038
1071
|
}
|
|
1039
1072
|
|
|
1040
1073
|
_accountLoad(account) {
|