maxpool 1.5.54 → 1.5.56
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/server.js +54 -20
package/package.json
CHANGED
package/src/server.js
CHANGED
|
@@ -91,10 +91,14 @@ const DEFAULT_QUEUE = {
|
|
|
91
91
|
// exports CLAUDE_STREAM_IDLE_TIMEOUT_MS=3h, but a session started any other way keeps the
|
|
92
92
|
// 300s floor — holding its request for hours just parks a caller that left at 5 minutes.
|
|
93
93
|
// Derive from the env we can observe; otherwise stay under the real floor.
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
94
|
+
// How long the CLIENT will tolerate a held stream. This is a fact about the PEER, so it
|
|
95
|
+
// is read per-request from `x-maxpool-client-stream-idle-ms` (the cc alias forwards its
|
|
96
|
+
// own CLAUDE_STREAM_IDLE_TIMEOUT_MS). Reading maxpool's OWN env was a category error: the
|
|
97
|
+
// alias exports that variable to the Claude Code process, never to this one, so it always
|
|
98
|
+
// fell to 240s and clamped every hold to 4 minutes despite a configured 24h.
|
|
99
|
+
// The 240s default is CORRECT for a bare `claude` — without the alias the client dies at
|
|
100
|
+
// a hard 300s and no keepalive can extend it — so it stays as the conservative floor.
|
|
101
|
+
streamClientToleranceMs: Math.max(60_000, Number(process.env.MAXPOOL_STREAM_CLIENT_TOLERANCE_MS) || 240_000),
|
|
98
102
|
// Non-streaming requests have no SSE heartbeat to keep them alive, so a long
|
|
99
103
|
// hold would die on the client timeout anyway. Cap their wait conservatively.
|
|
100
104
|
nonStreamMaxWaitMs: 5 * 60 * 1000,
|
|
@@ -277,6 +281,16 @@ export function createProxyServer(accountManager, config, hooks = {}) {
|
|
|
277
281
|
}
|
|
278
282
|
requestInfo.profile = getMaxpoolProfile(req.headers);
|
|
279
283
|
requestInfo.sessionKey = headerValue(req.headers, 'x-maxpool-session');
|
|
284
|
+
// The CLIENT tells us how long it will wait — the only source that is actually
|
|
285
|
+
// true. A `cc` session exports CLAUDE_STREAM_IDLE_TIMEOUT_MS=3h and forwards it
|
|
286
|
+
// here; a bare `claude` sends nothing and keeps the conservative 240s default,
|
|
287
|
+
// which is correct for it (its watchdog dies at a hard 300s regardless).
|
|
288
|
+
// Held at 80% so maxpool always gives up fractionally BEFORE the client does,
|
|
289
|
+
// turning a silent client-side death into an honest retryable 429.
|
|
290
|
+
const clientIdleMs = Number(headerValue(req.headers, 'x-maxpool-client-stream-idle-ms'));
|
|
291
|
+
if (Number.isFinite(clientIdleMs) && clientIdleMs > 300_000) {
|
|
292
|
+
requestInfo.clientToleranceMs = Math.floor(clientIdleMs * 0.8);
|
|
293
|
+
}
|
|
280
294
|
// (Removed a FALSE "provider fallback disabled for signed thinking" log here: it
|
|
281
295
|
// fired on every thinking `all` request but was untrue under the default
|
|
282
296
|
// when-exhausted/always policies — providers DO serve thinking requests — and it
|
|
@@ -1419,7 +1433,13 @@ function computeQueueWindowMs({
|
|
|
1419
1433
|
// hard, independent of how patient the client is — a raised CLAUDE_STREAM_IDLE_TIMEOUT_MS
|
|
1420
1434
|
// (the `cc` alias sets 3h) otherwise licenses a multi-hour hold on a connection that is
|
|
1421
1435
|
// simply gone. Error-fast + client reconnect beats an unattended hold.
|
|
1422
|
-
|
|
1436
|
+
// Network holds are NOT special-cased short any more. maxpool already re-polls every ~1s
|
|
1437
|
+
// and each retry issues a FRESH fetch, so a hold IS "keep probing, resume the moment any
|
|
1438
|
+
// route returns" — exactly what an unattended agent needs to survive a connectivity blip.
|
|
1439
|
+
// Failing fast at 2 minutes handed the turn to Claude Code's retry loop, which is the
|
|
1440
|
+
// thing that loses accumulated work. Visibility is paid for by logging/TUI, not by
|
|
1441
|
+
// truncating the wait.
|
|
1442
|
+
if (cause === 'network' && networkMaxWaitMs != null) windowMs = Math.min(windowMs, Math.max(networkMaxWaitMs, streamClientToleranceMs || 0));
|
|
1423
1443
|
return windowMs;
|
|
1424
1444
|
}
|
|
1425
1445
|
|
|
@@ -1446,11 +1466,18 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
|
|
|
1446
1466
|
}
|
|
1447
1467
|
|
|
1448
1468
|
const claudeCount = accountManager.accounts.filter(a => a.type !== 'provider').length;
|
|
1449
|
-
// Only name the providers when this pool
|
|
1450
|
-
//
|
|
1451
|
-
//
|
|
1452
|
-
|
|
1469
|
+
// Only name the providers when this pool HAS them AND they are actually allowed to
|
|
1470
|
+
// serve this request. With crossProviderFallbackPolicy 'never' (the default) they are
|
|
1471
|
+
// barred by POLICY, not saturated — saying they are "at their limit" is a lie that
|
|
1472
|
+
// hides the real, one-keypress fix. Name the switch instead.
|
|
1473
|
+
const providerAccounts = accountManager.accounts.filter(a => a.type === 'provider');
|
|
1474
|
+
const providersUsable = providerAccounts.some(a => a.enabled !== false
|
|
1475
|
+
&& accountManager._claudeFallbackFor?.(a.provider) !== 'never');
|
|
1476
|
+
const providersClause = providerAccounts.length && providersUsable
|
|
1453
1477
|
? ' and the GLM/Kimi providers' : '';
|
|
1478
|
+
const providersBarredHint = providerAccounts.length && !providersUsable
|
|
1479
|
+
? ' GLM and Kimi are switched off for Claude sessions — turn one on with m then g if you want them to cover this.'
|
|
1480
|
+
: '';
|
|
1454
1481
|
|
|
1455
1482
|
// No route is expected to recover within the queue window — i.e. every Claude
|
|
1456
1483
|
// account is at its own 5h/weekly limit. A short "retry in Ns" would be a lie;
|
|
@@ -1459,10 +1486,14 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
|
|
|
1459
1486
|
const eta = Number.isFinite(retryAfter) && retryAfter > 0
|
|
1460
1487
|
? ` Soonest reset in ~${formatRetryDuration(retryAfter)}, beyond the hold window.`
|
|
1461
1488
|
: '';
|
|
1462
|
-
return `No account can take this request — all ${claudeCount} Claude accounts${providersClause} are at their limit.${eta} Add another Claude account or wait for a quota reset
|
|
1489
|
+
return `No account can take this request — all ${claudeCount} Claude accounts${providersClause} are at their limit.${eta} Add another Claude account or wait for a quota reset.${providersBarredHint}`;
|
|
1463
1490
|
}
|
|
1464
1491
|
|
|
1465
|
-
|
|
1492
|
+
// "momentarily ... Retry in 2282s" was two bugs in one breath: 2282 seconds is 38
|
|
1493
|
+
// MINUTES (not momentary), and raw seconds are unreadable. Scale the wording to the
|
|
1494
|
+
// actual wait and render it in human units, the way every other branch already does.
|
|
1495
|
+
const waitLong = Number.isFinite(retryAfter) && retryAfter >= 120;
|
|
1496
|
+
return `No account can take this request right now — all ${claudeCount} Claude accounts${providersClause} are ${waitLong ? 'at their limit' : 'momentarily at their limit'}. Retry in ~${formatRetryDuration(retryAfter)}.${providersBarredHint}`;
|
|
1466
1497
|
}
|
|
1467
1498
|
|
|
1468
1499
|
// A provider (GLM/Kimi) rejecting a request whose token count exceeds its context
|
|
@@ -1959,8 +1990,9 @@ async function queueAndRetry(
|
|
|
1959
1990
|
const streamHoldMaxMs = queueConfig.streamHoldMaxMs == null
|
|
1960
1991
|
? 7 * 24 * 60 * 60 * 1000
|
|
1961
1992
|
: Math.max(0, Number(queueConfig.streamHoldMaxMs) || 0);
|
|
1962
|
-
|
|
1963
|
-
|
|
1993
|
+
// Per-request (from the client's own header) wins over the conservative default.
|
|
1994
|
+
const streamClientToleranceMs = Number.isFinite(requestInfo.clientToleranceMs)
|
|
1995
|
+
? requestInfo.clientToleranceMs
|
|
1964
1996
|
: Math.max(0, Number(queueConfig.streamClientToleranceMs) || 0);
|
|
1965
1997
|
const networkMaxWaitMs = queueConfig.networkMaxWaitMs == null
|
|
1966
1998
|
? 2 * 60 * 1000
|
|
@@ -2417,14 +2449,16 @@ function startIdleRequestReaper(res, reqId, idleMs, { now = Date.now, setInterva
|
|
|
2417
2449
|
// "in-flight" on one account for up to 6.7h, serving zero, which distorted the load
|
|
2418
2450
|
// balancer into avoiding a healthy account. A held request is progressing only if the
|
|
2419
2451
|
// UPSTREAM produced something; heartbeat bytes prove nothing.
|
|
2420
|
-
|
|
2452
|
+
// A QUEUE-HELD request is exempt. It has ALREADY released its account lease before
|
|
2453
|
+
// queueing, so reaping it frees no capacity — the thing this reaper exists to protect.
|
|
2454
|
+
// Its wait is bounded by its own queue ticket deadline instead. Reaping it here was
|
|
2455
|
+
// the blocker that made a longer hold window inert: the window can be hours, but the
|
|
2456
|
+
// socket was destroyed at 20 minutes with no error frame, just a reset.
|
|
2457
|
+
// (The 2026-07-29 case this reaper caught — 50 requests pinned on one account for 6.7h
|
|
2458
|
+
// — were IN-FLIGHT holding leases, not queue-held, so they are still reaped below.)
|
|
2459
|
+
if (getRequestInfo?.()?.queueHeartbeatActive) { lastProgressAt = now(); return; }
|
|
2421
2460
|
const bytes = res.socket?.bytesWritten ?? lastBytes;
|
|
2422
|
-
if (bytes !== lastBytes) {
|
|
2423
|
-
lastBytes = bytes;
|
|
2424
|
-
if (!heldOnHeartbeat) { lastProgressAt = now(); return; }
|
|
2425
|
-
// Held: those bytes were OUR keepalive, so they are not progress — deliberately
|
|
2426
|
-
// fall through to the staleness check rather than returning.
|
|
2427
|
-
}
|
|
2461
|
+
if (bytes !== lastBytes) { lastBytes = bytes; lastProgressAt = now(); return; }
|
|
2428
2462
|
if (now() - lastProgressAt >= idleMs && !res.writableEnded && !res.destroyed) {
|
|
2429
2463
|
console.error(`[Maxpool] Request ${reqId} — no write progress for ${Math.round(idleMs / 1000)}s (backstop reaper); force-aborting a stuck request to free its account slot`);
|
|
2430
2464
|
res.destroy();
|