maxpool 1.5.42 → 1.5.43
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +1 -1
- package/src/account-manager.js +6 -1
- package/src/config.js +3 -1
- package/src/server.js +9 -3
- package/src/tui.js +1 -1
package/README.md
CHANGED
|
@@ -270,7 +270,7 @@ TEAMCLAUDE_CONFIG=./my-config.json maxpool server
|
|
|
270
270
|
"weeklySoftThreshold": 0.65,
|
|
271
271
|
"weeklyReserveThreshold": 0.85,
|
|
272
272
|
"weeklyCriticalThreshold": 0.95,
|
|
273
|
-
"weeklyExhaustedThreshold": 0.
|
|
273
|
+
"weeklyExhaustedThreshold": 0.999,
|
|
274
274
|
"weeklyBurnDebtWeight": 0.6
|
|
275
275
|
},
|
|
276
276
|
"retry": {
|
package/package.json
CHANGED
package/src/account-manager.js
CHANGED
|
@@ -91,7 +91,12 @@ const DEFAULT_SCHEDULER = {
|
|
|
91
91
|
weeklySoftThreshold: 0.65,
|
|
92
92
|
weeklyReserveThreshold: 0.85,
|
|
93
93
|
weeklyCriticalThreshold: 0.95,
|
|
94
|
-
|
|
94
|
+
// Use-it-or-lose-it (2026-07-24): bench only at 99.9%, not 98.5% — the weekly quota
|
|
95
|
+
// resets, so reserving the top ~1.5% just wastes it. The thin 0.1% floor is the ONE
|
|
96
|
+
// remaining guard: don't fire a request at an account whose own header already says
|
|
97
|
+
// it's essentially full (a near-guaranteed-waste hard-429). A real 429 sets util≈1.0
|
|
98
|
+
// and still benches here. Critical (0.95-0.999) stays last-resort-only (pass 2).
|
|
99
|
+
weeklyExhaustedThreshold: 0.999,
|
|
95
100
|
weeklyBurnDebtWeight: 0.6,
|
|
96
101
|
// Routing-cost tuning (lower cost = preferred). The goal is to AVOID
|
|
97
102
|
// short-term (rate/concurrency) throttling by spreading load across healthy
|
package/src/config.js
CHANGED
|
@@ -90,7 +90,9 @@ export function createDefaultConfig() {
|
|
|
90
90
|
weeklySoftThreshold: 0.65,
|
|
91
91
|
weeklyReserveThreshold: 0.85,
|
|
92
92
|
weeklyCriticalThreshold: 0.95,
|
|
93
|
-
|
|
93
|
+
// Use-it-or-lose-it: bench only at 99.9% (the weekly quota resets, so reserving
|
|
94
|
+
// the top ~1.5% wastes it). A real 429 sets util≈1.0 and still benches.
|
|
95
|
+
weeklyExhaustedThreshold: 0.999,
|
|
94
96
|
// Cross-PROVIDER fallback policy for 'cc all' (profile=all): whether a session
|
|
95
97
|
// may be served by a provider family other than its home (Claude → GLM/Kimi).
|
|
96
98
|
// 'never' — (DEFAULT) a Claude Code session stays on Anthropic; no
|
package/src/server.js
CHANGED
|
@@ -1348,16 +1348,22 @@ function classifyRateLimit(account, headers, body, opts = {}) {
|
|
|
1348
1348
|
// genuine account cap with stripped headers must bench the whole account, not one
|
|
1349
1349
|
// model). Neither bucket may be at/above the exhaustion floor.
|
|
1350
1350
|
const haveUnifiedEvidence = Number.isFinite(weekly) || Number.isFinite(fiveHour);
|
|
1351
|
+
// Exhaustion floor for 429-SCOPE classification — kept in sync with the scheduler's
|
|
1352
|
+
// weeklyExhaustedThreshold (0.999, use-it-or-lose-it). A 429 whose unified buckets are
|
|
1353
|
+
// still below the floor isn't weekly-exhaustion, so a model-family + long-retry-after
|
|
1354
|
+
// 429 benches only that model, not the whole (still-usable) account — matching routing,
|
|
1355
|
+
// which now treats 0.95-0.999 as usable (critical), not benched.
|
|
1356
|
+
const EXHAUSTION_FLOOR = 0.999;
|
|
1351
1357
|
const unifiedNotExhausted =
|
|
1352
|
-
(!Number.isFinite(weekly) || weekly <
|
|
1358
|
+
(!Number.isFinite(weekly) || weekly < EXHAUSTION_FLOOR) && (!Number.isFinite(fiveHour) || fiveHour < EXHAUSTION_FLOOR);
|
|
1353
1359
|
const modelScope =
|
|
1354
1360
|
(fam && haveUnifiedEvidence && unifiedNotExhausted && Number.isFinite(retryAfter) && retryAfter >= 30 * 60)
|
|
1355
1361
|
? fam : null;
|
|
1356
1362
|
|
|
1357
1363
|
const quotaHeaderExhaustion =
|
|
1358
1364
|
unifiedStatus === 'rejected'
|
|
1359
|
-
|| (Number.isFinite(fiveHour) && fiveHour >=
|
|
1360
|
-
|| (Number.isFinite(weekly) && weekly >=
|
|
1365
|
+
|| (Number.isFinite(fiveHour) && fiveHour >= EXHAUSTION_FLOOR)
|
|
1366
|
+
|| (Number.isFinite(weekly) && weekly >= EXHAUSTION_FLOOR)
|
|
1361
1367
|
|| (headers['anthropic-ratelimit-tokens-remaining'] != null && tokensRemaining <= 0)
|
|
1362
1368
|
|| (headers['anthropic-ratelimit-requests-remaining'] != null && requestsRemaining <= 0);
|
|
1363
1369
|
if (quotaHeaderExhaustion) return { scope: 'account', fingerprint: null, modelScope };
|
package/src/tui.js
CHANGED
|
@@ -1294,7 +1294,7 @@ export class TUI {
|
|
|
1294
1294
|
// still usable, "Fable maxed" (red) ONLY at genuine exhaustion. This is the
|
|
1295
1295
|
// SAME predicate the router benches on (_scopedExhausted), so "maxed" renders
|
|
1296
1296
|
// iff the model is actually benched — 90%/critical is no longer mislabelled.
|
|
1297
|
-
const exhaustedFloor = this.am.scheduler?.weeklyExhaustedThreshold ?? 0.
|
|
1297
|
+
const exhaustedFloor = this.am.scheduler?.weeklyExhaustedThreshold ?? 0.999;
|
|
1298
1298
|
const reserveFloor = this.am.scheduler?.weeklyReserveThreshold ?? 0.85;
|
|
1299
1299
|
const scopedTags = [];
|
|
1300
1300
|
for (const [fam, e] of Object.entries(q.scopedWeekly || {})) {
|