maxpool 1.5.50 → 1.5.52
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/event-log.js +12 -2
- package/src/server.js +75 -12
- package/src/sleep-guard.js +6 -1
package/package.json
CHANGED
package/src/event-log.js
CHANGED
|
@@ -7,7 +7,11 @@
|
|
|
7
7
|
|
|
8
8
|
import { appendFile, stat, rename, chmod } from 'node:fs/promises';
|
|
9
9
|
|
|
10
|
-
const MAX_BYTES = 5 * 1024 * 1024; // rotate at ~5 MB
|
|
10
|
+
const MAX_BYTES = 5 * 1024 * 1024; // rotate at ~5 MB
|
|
11
|
+
// Keep 5 generations (~30 MB total), not 1. With a single backup the log rotated away the
|
|
12
|
+
// evidence MID-INVESTIGATION on 2026-07-28 — the reported error's own line was already gone
|
|
13
|
+
// before it could be read. Diagnosing a stall needs hours of history, not minutes.
|
|
14
|
+
const MAX_GENERATIONS = Math.max(1, Number(process.env.MAXPOOL_LOG_GENERATIONS) || 5);
|
|
11
15
|
const ROTATE_CHECK_MS = 2000; // rotation-owner size-check cadence
|
|
12
16
|
// Cap each line below the platform PIPE_BUF (512 B on macOS) so concurrent
|
|
13
17
|
// O_APPEND writes from coexisting processes during a reload stay atomic (never
|
|
@@ -91,7 +95,13 @@ export async function rotateIfNeeded() {
|
|
|
91
95
|
const path = logPath;
|
|
92
96
|
if (!path) return;
|
|
93
97
|
const st = await stat(path).catch(() => null);
|
|
94
|
-
if (st && st.size > MAX_BYTES)
|
|
98
|
+
if (st && st.size > MAX_BYTES) {
|
|
99
|
+
// Cascade .N-1 -> .N (oldest first) so N generations survive instead of one.
|
|
100
|
+
for (let i = MAX_GENERATIONS - 1; i >= 1; i--) {
|
|
101
|
+
await rename(`${path}.${i}`, `${path}.${i + 1}`).catch(() => {});
|
|
102
|
+
}
|
|
103
|
+
await rename(path, `${path}.1`).catch(() => {});
|
|
104
|
+
}
|
|
95
105
|
}
|
|
96
106
|
|
|
97
107
|
/**
|
package/src/server.js
CHANGED
|
@@ -22,7 +22,25 @@ const DEFAULT_RETRY = {
|
|
|
22
22
|
// serves EVERY provider (Anthropic/GLM/Kimi), so the idle gap is generous enough
|
|
23
23
|
// that a legitimately-slow-but-alive stream is never cut (each chunk resets it).
|
|
24
24
|
const UPSTREAM_TTFB_MS = Math.max(5_000, Number(process.env.MAXPOOL_TTFB_MS) || 120_000); // headers must arrive within this
|
|
25
|
-
|
|
25
|
+
// A real SSE EVENT, not a `:` comment. Claude Code's stall watchdog is reset only when its
|
|
26
|
+
// SSE iterator YIELDS an event; per the spec a comment line is discarded by the parser and
|
|
27
|
+
// never yields, so a comment keepalive resets nothing. That is why held requests died at
|
|
28
|
+
// EXACTLY 300.0s — the client's floor — despite a 10s heartbeat (60 such deaths in 2.4
|
|
29
|
+
// days). `ping` is an unknown event type the client ignores semantically while still
|
|
30
|
+
// counting as traffic.
|
|
31
|
+
const NETWORK_ERROR_CODES = new Set([
|
|
32
|
+
'ECONNRESET', 'ECONNREFUSED', 'ETIMEDOUT', 'UND_ERR_CONNECT_TIMEOUT',
|
|
33
|
+
'ENOTFOUND', 'EAI_AGAIN', 'EPIPE', 'ECONNABORTED', 'UND_ERR_SOCKET', 'UND_ERR_HEADERS_TIMEOUT',
|
|
34
|
+
]);
|
|
35
|
+
const isNetworkCode = c => Boolean(c) && NETWORK_ERROR_CODES.has(c);
|
|
36
|
+
const QUEUE_KEEPALIVE = 'event: ping\ndata: {}\n\n';
|
|
37
|
+
// 240s, strictly BELOW Claude Code's hard 300s stall floor. At the old 300_000 the two
|
|
38
|
+
// timers were a dead heat and the client always won — maxpool's clock starts when a chunk
|
|
39
|
+
// is READ from upstream, strictly before the client parses it. Result: `upstream idle
|
|
40
|
+
// timeout` fired 0 times in 2.4 days while 60 requests died silently client-side. Below the
|
|
41
|
+
// floor maxpool wins and turns a silent stall into a labelled error. Anthropic pings during
|
|
42
|
+
// extended thinking, so real gaps never approach 4 minutes.
|
|
43
|
+
const STREAM_IDLE_MS = Math.max(30_000, Number(process.env.MAXPOOL_STREAM_IDLE_MS) || 240_000); // max gap BETWEEN streamed chunks (reset per chunk)
|
|
26
44
|
const UPSTREAM_BODY_MS = Math.max(30_000, Number(process.env.MAXPOOL_BODY_MS) || 300_000); // non-streaming body read
|
|
27
45
|
const CLIENT_DRAIN_MS = Math.max(5_000, Number(process.env.MAXPOOL_DRAIN_MS) || 60_000); // max wait for a backpressured client to drain (half-open client → free the lease)
|
|
28
46
|
// A provider 403 is (unlike a 401) almost always transient QUOTA/PLAN exhaustion — cool
|
|
@@ -42,6 +60,13 @@ const DEFAULT_QUEUE = {
|
|
|
42
60
|
maxWaitMs: 24 * 60 * 60 * 1000,
|
|
43
61
|
autoMaxWaitMs: null,
|
|
44
62
|
capacityMaxWaitMs: 15 * 60 * 1000,
|
|
63
|
+
// NETWORK-cause hold ceiling. A quota/capacity hold is worth waiting out — a reset is
|
|
64
|
+
// genuinely coming. A NETWORK hold is not: the connection is dead, and holding it just
|
|
65
|
+
// parks the caller on a socket nothing is watching. On 2026-07-28 a request was held
|
|
66
|
+
// 10,445s (2h54m) having produced ZERO bytes; nothing aborted it until the user touched
|
|
67
|
+
// the keyboard. Give up quickly instead and return a retryable 429, so the client
|
|
68
|
+
// reconnects on a FRESH connection — which is what actually self-heals a dead route.
|
|
69
|
+
networkMaxWaitMs: 2 * 60 * 1000,
|
|
45
70
|
weeklyMaxWaitMs: 24 * 60 * 60 * 1000, // legacy bound; streaming holds use streamHoldMaxMs
|
|
46
71
|
// STREAMING hold ceiling: how long a streaming request may be HELD ALIVE on
|
|
47
72
|
// the SSE heartbeat waiting for any account to free up. Defaults to 7d (the
|
|
@@ -62,7 +87,14 @@ const DEFAULT_QUEUE = {
|
|
|
62
87
|
// (front-loaded work waiting for a free account ≈ a few hours) so beyond that the
|
|
63
88
|
// request error-fasts with an honest retryable 429 instead of a silent multi-day park.
|
|
64
89
|
// Pair with a raised client watchdog (the cc launch sets CLAUDE_STREAM_IDLE_TIMEOUT_MS).
|
|
65
|
-
|
|
90
|
+
// Only trust a long hold when the client's watchdog was ACTUALLY raised. The `cc` alias
|
|
91
|
+
// exports CLAUDE_STREAM_IDLE_TIMEOUT_MS=3h, but a session started any other way keeps the
|
|
92
|
+
// 300s floor — holding its request for hours just parks a caller that left at 5 minutes.
|
|
93
|
+
// Derive from the env we can observe; otherwise stay under the real floor.
|
|
94
|
+
streamClientToleranceMs: Math.max(60_000, Number(process.env.MAXPOOL_STREAM_CLIENT_TOLERANCE_MS)
|
|
95
|
+
|| (Number(process.env.CLAUDE_STREAM_IDLE_TIMEOUT_MS) > 300_000
|
|
96
|
+
? Math.floor(Number(process.env.CLAUDE_STREAM_IDLE_TIMEOUT_MS) * 0.8)
|
|
97
|
+
: 240_000)),
|
|
66
98
|
// Non-streaming requests have no SSE heartbeat to keep them alive, so a long
|
|
67
99
|
// hold would die on the client timeout anyway. Cap their wait conservatively.
|
|
68
100
|
nonStreamMaxWaitMs: 5 * 60 * 1000,
|
|
@@ -1226,10 +1258,24 @@ async function forwardRequest(
|
|
|
1226
1258
|
// lease (free the scarce account) and STOP: no retry (the client is gone), no
|
|
1227
1259
|
// write (the socket is dead).
|
|
1228
1260
|
if (clientGone.signal.aborted) {
|
|
1261
|
+
// LOG IT. This silent return hid every client-side give-up: 60 requests in 2.4 days
|
|
1262
|
+
// died here leaving only an indistinguishable `(null, 300.0s)` line. `committed`
|
|
1263
|
+
// separates "the user saw a partial answer" (real harm, unretryable) from "the user
|
|
1264
|
+
// was still waiting" — and the elapsed time is what exposes a client watchdog firing
|
|
1265
|
+
// at its floor while maxpool was still happily holding the request.
|
|
1266
|
+
const heldMs = Date.now() - (requestInfo.startedAt || Date.now());
|
|
1267
|
+
console.log(`[Maxpool] Client left after ${(heldMs / 1000).toFixed(1)}s on "${account.name}" `
|
|
1268
|
+
+ `(${res.headersSent ? 'mid-response — output already sent' : 'still waiting, nothing sent'})`);
|
|
1229
1269
|
releaseOnClientGone();
|
|
1230
1270
|
return;
|
|
1231
1271
|
}
|
|
1232
|
-
|
|
1272
|
+
// undici reports every socket/DNS/TLS failure as the bare string "fetch failed" and
|
|
1273
|
+
// hangs the REAL reason off err.cause. Logging err.message alone threw that away: 588
|
|
1274
|
+
// "fetch failed" lines and ZERO ECONNRESET/ENOTFOUND/UND_ERR in the whole log, leaving
|
|
1275
|
+
// every incident unattributable (DNS? TLS? socket exhaustion?).
|
|
1276
|
+
const rootCause = err.cause?.code || err.cause?.message || err.code || '';
|
|
1277
|
+
console.error(`[Maxpool] Upstream error (account "${account.name}"):`, err.message
|
|
1278
|
+
+ (rootCause && !String(err.message).includes(rootCause) ? ` (cause: ${rootCause})` : ''));
|
|
1233
1279
|
|
|
1234
1280
|
if (logDir) {
|
|
1235
1281
|
logSections.push(`=== ERROR ===\n${err.stack || err.message}`);
|
|
@@ -1238,8 +1284,10 @@ async function forwardRequest(
|
|
|
1238
1284
|
|
|
1239
1285
|
const isTransient = err instanceof Error &&
|
|
1240
1286
|
(err.message.includes('fetch failed') ||
|
|
1241
|
-
|
|
1242
|
-
err.
|
|
1287
|
+
// Read BOTH: on an undici fetch rejection err.code is undefined and the code lives
|
|
1288
|
+
// on err.cause, so the bare err.code branches were dead — only the 'fetch failed'
|
|
1289
|
+
// string match kept this classification alive.
|
|
1290
|
+
isNetworkCode(err.code) || isNetworkCode(err.cause?.code) ||
|
|
1243
1291
|
// Upstream-hang guards: a stalled connection is network-class, not the
|
|
1244
1292
|
// account's fault → 5s cooldown + release + (pre-headers) retry elsewhere;
|
|
1245
1293
|
// once committed, isTransient falls through to sendErrorResponse which ends
|
|
@@ -1346,7 +1394,7 @@ function formatRetryDuration(seconds) {
|
|
|
1346
1394
|
function computeQueueWindowMs({
|
|
1347
1395
|
cause, stream, retryPlanCause,
|
|
1348
1396
|
maxWaitMs, capacityMaxWaitMs, nonStreamMaxWaitMs, streamHoldMaxMs, streamClientToleranceMs,
|
|
1349
|
-
isCountTokens, countTokensMaxWaitMs,
|
|
1397
|
+
isCountTokens, countTokensMaxWaitMs, networkMaxWaitMs,
|
|
1350
1398
|
}) {
|
|
1351
1399
|
let windowMs;
|
|
1352
1400
|
if (!stream) {
|
|
@@ -1367,6 +1415,11 @@ function computeQueueWindowMs({
|
|
|
1367
1415
|
// the upstream processing once acquired) so a non-heartbeated metadata call fast-
|
|
1368
1416
|
// fails with a retryable 429 instead of hanging past the client's idle window.
|
|
1369
1417
|
if (isCountTokens && countTokensMaxWaitMs != null) windowMs = Math.min(windowMs, countTokensMaxWaitMs);
|
|
1418
|
+
// A dead ROUTE is not a scheduled reset: holding it out is waiting for nothing. Cap it
|
|
1419
|
+
// hard, independent of how patient the client is — a raised CLAUDE_STREAM_IDLE_TIMEOUT_MS
|
|
1420
|
+
// (the `cc` alias sets 3h) otherwise licenses a multi-hour hold on a connection that is
|
|
1421
|
+
// simply gone. Error-fast + client reconnect beats an unattended hold.
|
|
1422
|
+
if (cause === 'network' && networkMaxWaitMs != null) windowMs = Math.min(windowMs, networkMaxWaitMs);
|
|
1370
1423
|
return windowMs;
|
|
1371
1424
|
}
|
|
1372
1425
|
|
|
@@ -1803,8 +1856,14 @@ function sendErrorResponse(res, requestInfo, status, payload, headers = {}) {
|
|
|
1803
1856
|
if (requestInfo.queueHeartbeatActive || res.headersSent) {
|
|
1804
1857
|
clearQueueHeartbeat(requestInfo);
|
|
1805
1858
|
if (!res.destroyed && !res.writableEnded) {
|
|
1806
|
-
|
|
1807
|
-
|
|
1859
|
+
// Guarded: a write onto a half-dead socket throws, and there is no res.on('error')
|
|
1860
|
+
// anywhere here — an uncaught one reaches the worker's uncaughtException handler,
|
|
1861
|
+
// which process.exit()s and bounces EVERY other in-flight stream. ensureQueueHeartbeat
|
|
1862
|
+
// already guards its identical write; this one did not.
|
|
1863
|
+
try {
|
|
1864
|
+
res.write(`event: error\ndata: ${JSON.stringify(payload)}\n\n`);
|
|
1865
|
+
} catch { /* peer vanished mid-write — nothing to deliver, fall through to end() */ }
|
|
1866
|
+
try { res.end(); } catch { /* already torn down */ }
|
|
1808
1867
|
}
|
|
1809
1868
|
return;
|
|
1810
1869
|
}
|
|
@@ -1903,6 +1962,9 @@ async function queueAndRetry(
|
|
|
1903
1962
|
const streamClientToleranceMs = queueConfig.streamClientToleranceMs == null
|
|
1904
1963
|
? 3 * 60 * 60 * 1000
|
|
1905
1964
|
: Math.max(0, Number(queueConfig.streamClientToleranceMs) || 0);
|
|
1965
|
+
const networkMaxWaitMs = queueConfig.networkMaxWaitMs == null
|
|
1966
|
+
? 2 * 60 * 1000
|
|
1967
|
+
: Math.max(0, Number(queueConfig.networkMaxWaitMs) || 0);
|
|
1906
1968
|
const queueWindowMs = computeQueueWindowMs({
|
|
1907
1969
|
cause,
|
|
1908
1970
|
stream: Boolean(requestInfo.stream),
|
|
@@ -1910,6 +1972,7 @@ async function queueAndRetry(
|
|
|
1910
1972
|
maxWaitMs,
|
|
1911
1973
|
capacityMaxWaitMs,
|
|
1912
1974
|
nonStreamMaxWaitMs,
|
|
1975
|
+
networkMaxWaitMs,
|
|
1913
1976
|
streamHoldMaxMs,
|
|
1914
1977
|
streamClientToleranceMs,
|
|
1915
1978
|
isCountTokens: Boolean(requestInfo.isCountTokens),
|
|
@@ -1972,7 +2035,7 @@ async function queueAndRetry(
|
|
|
1972
2035
|
// committed stream and DROP the held session. Keeping the heartbeat active lets
|
|
1973
2036
|
// it re-hold. The heartbeat is instead stopped the instant real upstream bytes
|
|
1974
2037
|
// start flowing, inside streamResponse — that prevents the Bug A interleave
|
|
1975
|
-
// (
|
|
2038
|
+
// (queue keepalive pings injected between real SSE events) without losing
|
|
1976
2039
|
// re-holdability on a post-resume failover.
|
|
1977
2040
|
return forwardRequest(
|
|
1978
2041
|
req, res, body, accountManager, upstream, 0, hooks, reqId, ctx, logDir,
|
|
@@ -2024,7 +2087,7 @@ function ensureQueueHeartbeat(res, requestInfo, queueConfig, accountManager) {
|
|
|
2024
2087
|
'X-Accel-Buffering': 'no',
|
|
2025
2088
|
});
|
|
2026
2089
|
res.flushHeaders?.();
|
|
2027
|
-
res.write(
|
|
2090
|
+
res.write(QUEUE_KEEPALIVE);
|
|
2028
2091
|
} catch {
|
|
2029
2092
|
reapDead();
|
|
2030
2093
|
return;
|
|
@@ -2033,7 +2096,7 @@ function ensureQueueHeartbeat(res, requestInfo, queueConfig, accountManager) {
|
|
|
2033
2096
|
requestInfo.queueHeartbeatTimer = setInterval(() => {
|
|
2034
2097
|
if (res.destroyed || res.writableEnded) { reapDead(); return; }
|
|
2035
2098
|
try {
|
|
2036
|
-
res.write(
|
|
2099
|
+
res.write(QUEUE_KEEPALIVE);
|
|
2037
2100
|
} catch {
|
|
2038
2101
|
reapDead();
|
|
2039
2102
|
}
|
|
@@ -2369,7 +2432,7 @@ async function streamResponse(webStream, res, status, responseHeaders, accountIn
|
|
|
2369
2432
|
// We're now committed to streaming a real upstream response body onto this
|
|
2370
2433
|
// response — there is no more failover for this forward. Stop the queue
|
|
2371
2434
|
// heartbeat (if this was a resumed held stream) BEFORE the first real byte, so
|
|
2372
|
-
// the setInterval can't inject
|
|
2435
|
+
// the setInterval can't inject queue keepalive pings between live SSE
|
|
2373
2436
|
// events (Bug A). It is deliberately NOT cleared earlier (on resume), so a
|
|
2374
2437
|
// pre-byte failover can still re-hold the session via queueAndRetry.
|
|
2375
2438
|
clearQueueHeartbeat(requestInfo);
|
package/src/sleep-guard.js
CHANGED
|
@@ -11,7 +11,12 @@
|
|
|
11
11
|
import { spawn as realSpawn } from 'node:child_process';
|
|
12
12
|
|
|
13
13
|
const POLL_MS = 5_000; // re-check the in-flight signal (idle-sleep timers are minutes)
|
|
14
|
-
|
|
14
|
+
// Stay awake this long after the last request. 60s lost a measured race on 2026-07-28: a
|
|
15
|
+
// network outage drained all in-flight work, the guard released exactly 74s later, and
|
|
16
|
+
// macOS (idle-sleep timer) slept in the SAME SECOND — so the machine went down while the
|
|
17
|
+
// outage was still in progress and recovery had to wait for a keystroke. 5 minutes bridges
|
|
18
|
+
// an outage-shaped gap without meaningfully delaying a genuinely idle sleep.
|
|
19
|
+
const GRACE_MS = 300_000;
|
|
15
20
|
|
|
16
21
|
export class SleepGuard {
|
|
17
22
|
constructor({ getWorkPending, log = () => {}, spawn = realSpawn, enabled, pollMs = POLL_MS, graceMs = GRACE_MS } = {}) {
|