maxpool 1.5.55 → 1.5.57

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.55",
3
+ "version": "1.5.57",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -1785,13 +1785,14 @@ export class AccountManager {
1785
1785
  // are all unavailable the request HOLDS/queues (recoverable) rather than 400ing.
1786
1786
  if (requestInfo.hasImage && account.provider === 'kimi') return false;
1787
1787
 
1788
- // A large-context session: a provider already rejected this request with a
1789
- // context-length 400. The GLM/Kimi *coding* endpoints serve a fixed model capped
1790
- // at ~256K and IGNORE the model id (so a K3/GLM 1M PLAN doesn't lift the coding
1791
- // leg's ceiling). Only a 1M-context Claude account can hold it — bench the
1792
- // providers for this session so it routes to Claude, or HOLDS for one, instead of
1793
- // re-404ing on a too-small leg. Sticky per session (context only grows turn over
1794
- // turn), so no follow-up turn re-pays the wasted attempt.
1788
+ // A large-context session: a provider already rejected THIS request with a
1789
+ // context-length 400. Deliberately REACTIVE — it never assumes a ceiling, it learns
1790
+ // one from an actual rejection, so it self-corrects as providers grow. That matters:
1791
+ // the old ~256K coding-endpoint cap is gone (verified 2026-08-02 — GLM 5.2 and Kimi
1792
+ // K3 both accepted a ~400K-token payload and both honoured the requested model id),
1793
+ // so this branch simply stops firing rather than needing a new constant.
1794
+ // Sticky per session (context only grows turn over turn) so no follow-up turn re-pays
1795
+ // the wasted attempt.
1795
1796
  if (account.type === 'provider' && this._isSessionLargeContext(requestInfo)) return false;
1796
1797
 
1797
1798
  const { incompatible, homeProvider } = this._effectiveIncompatible(requestInfo);
package/src/server.js CHANGED
@@ -91,10 +91,14 @@ const DEFAULT_QUEUE = {
91
91
  // exports CLAUDE_STREAM_IDLE_TIMEOUT_MS=3h, but a session started any other way keeps the
92
92
  // 300s floor — holding its request for hours just parks a caller that left at 5 minutes.
93
93
  // Derive from the env we can observe; otherwise stay under the real floor.
94
- streamClientToleranceMs: Math.max(60_000, Number(process.env.MAXPOOL_STREAM_CLIENT_TOLERANCE_MS)
95
- || (Number(process.env.CLAUDE_STREAM_IDLE_TIMEOUT_MS) > 300_000
96
- ? Math.floor(Number(process.env.CLAUDE_STREAM_IDLE_TIMEOUT_MS) * 0.8)
97
- : 240_000)),
94
+ // How long the CLIENT will tolerate a held stream. This is a fact about the PEER, so it
95
+ // is read per-request from `x-maxpool-client-stream-idle-ms` (the cc alias forwards its
96
+ // own CLAUDE_STREAM_IDLE_TIMEOUT_MS). Reading maxpool's OWN env was a category error: the
97
+ // alias exports that variable to the Claude Code process, never to this one, so it always
98
+ // fell to 240s and clamped every hold to 4 minutes despite a configured 24h.
99
+ // The 240s default is CORRECT for a bare `claude` — without the alias the client dies at
100
+ // a hard 300s and no keepalive can extend it — so it stays as the conservative floor.
101
+ streamClientToleranceMs: Math.max(60_000, Number(process.env.MAXPOOL_STREAM_CLIENT_TOLERANCE_MS) || 240_000),
98
102
  // Non-streaming requests have no SSE heartbeat to keep them alive, so a long
99
103
  // hold would die on the client timeout anyway. Cap their wait conservatively.
100
104
  nonStreamMaxWaitMs: 5 * 60 * 1000,
@@ -277,6 +281,16 @@ export function createProxyServer(accountManager, config, hooks = {}) {
277
281
  }
278
282
  requestInfo.profile = getMaxpoolProfile(req.headers);
279
283
  requestInfo.sessionKey = headerValue(req.headers, 'x-maxpool-session');
284
+ // The CLIENT tells us how long it will wait — the only source that is actually
285
+ // true. A `cc` session exports CLAUDE_STREAM_IDLE_TIMEOUT_MS=3h and forwards it
286
+ // here; a bare `claude` sends nothing and keeps the conservative 240s default,
287
+ // which is correct for it (its watchdog dies at a hard 300s regardless).
288
+ // Held at 80% so maxpool always gives up fractionally BEFORE the client does,
289
+ // turning a silent client-side death into an honest retryable 429.
290
+ const clientIdleMs = Number(headerValue(req.headers, 'x-maxpool-client-stream-idle-ms'));
291
+ if (Number.isFinite(clientIdleMs) && clientIdleMs > 300_000) {
292
+ requestInfo.clientToleranceMs = Math.floor(clientIdleMs * 0.8);
293
+ }
280
294
  // (Removed a FALSE "provider fallback disabled for signed thinking" log here: it
281
295
  // fired on every thinking `all` request but was untrue under the default
282
296
  // when-exhausted/always policies — providers DO serve thinking requests — and it
@@ -1419,7 +1433,13 @@ function computeQueueWindowMs({
1419
1433
  // hard, independent of how patient the client is — a raised CLAUDE_STREAM_IDLE_TIMEOUT_MS
1420
1434
  // (the `cc` alias sets 3h) otherwise licenses a multi-hour hold on a connection that is
1421
1435
  // simply gone. Error-fast + client reconnect beats an unattended hold.
1422
- if (cause === 'network' && networkMaxWaitMs != null) windowMs = Math.min(windowMs, networkMaxWaitMs);
1436
+ // Network holds are NOT special-cased short any more. maxpool already re-polls every ~1s
1437
+ // and each retry issues a FRESH fetch, so a hold IS "keep probing, resume the moment any
1438
+ // route returns" — exactly what an unattended agent needs to survive a connectivity blip.
1439
+ // Failing fast at 2 minutes handed the turn to Claude Code's retry loop, which is the
1440
+ // thing that loses accumulated work. Visibility is paid for by logging/TUI, not by
1441
+ // truncating the wait.
1442
+ if (cause === 'network' && networkMaxWaitMs != null) windowMs = Math.min(windowMs, Math.max(networkMaxWaitMs, streamClientToleranceMs || 0));
1423
1443
  return windowMs;
1424
1444
  }
1425
1445
 
@@ -1970,8 +1990,9 @@ async function queueAndRetry(
1970
1990
  const streamHoldMaxMs = queueConfig.streamHoldMaxMs == null
1971
1991
  ? 7 * 24 * 60 * 60 * 1000
1972
1992
  : Math.max(0, Number(queueConfig.streamHoldMaxMs) || 0);
1973
- const streamClientToleranceMs = queueConfig.streamClientToleranceMs == null
1974
- ? 3 * 60 * 60 * 1000
1993
+ // Per-request (from the client's own header) wins over the conservative default.
1994
+ const streamClientToleranceMs = Number.isFinite(requestInfo.clientToleranceMs)
1995
+ ? requestInfo.clientToleranceMs
1975
1996
  : Math.max(0, Number(queueConfig.streamClientToleranceMs) || 0);
1976
1997
  const networkMaxWaitMs = queueConfig.networkMaxWaitMs == null
1977
1998
  ? 2 * 60 * 1000
@@ -2349,7 +2370,10 @@ function prepareRuntimeProviders(accountManager, headers) {
2349
2370
 
2350
2371
  const kimiToken = headerValue(headers, 'x-maxpool-kimi-token');
2351
2372
  if (kimiToken) {
2352
- const model = headerValue(headers, 'x-maxpool-kimi-model') || 'kimi-k2.7';
2373
+ // Fallback only — `cc all` always sends x-maxpool-kimi-model from the llm_config SSOT,
2374
+ // so this is what a bare/older client gets. Kept current deliberately: it read
2375
+ // 'kimi-k2.7' while the fleet had moved to k3.
2376
+ const model = headerValue(headers, 'x-maxpool-kimi-model') || 'kimi-k3';
2353
2377
  accountManager.upsertRuntimeAccount({
2354
2378
  name: 'kimi-fallback',
2355
2379
  type: 'provider',
@@ -2428,14 +2452,16 @@ function startIdleRequestReaper(res, reqId, idleMs, { now = Date.now, setInterva
2428
2452
  // "in-flight" on one account for up to 6.7h, serving zero, which distorted the load
2429
2453
  // balancer into avoiding a healthy account. A held request is progressing only if the
2430
2454
  // UPSTREAM produced something; heartbeat bytes prove nothing.
2431
- const heldOnHeartbeat = Boolean(getRequestInfo?.()?.queueHeartbeatActive);
2455
+ // A QUEUE-HELD request is exempt. It has ALREADY released its account lease before
2456
+ // queueing, so reaping it frees no capacity — the thing this reaper exists to protect.
2457
+ // Its wait is bounded by its own queue ticket deadline instead. Reaping it here was
2458
+ // the blocker that made a longer hold window inert: the window can be hours, but the
2459
+ // socket was destroyed at 20 minutes with no error frame, just a reset.
2460
+ // (The 2026-07-29 case this reaper caught — 50 requests pinned on one account for 6.7h
2461
+ // — were IN-FLIGHT holding leases, not queue-held, so they are still reaped below.)
2462
+ if (getRequestInfo?.()?.queueHeartbeatActive) { lastProgressAt = now(); return; }
2432
2463
  const bytes = res.socket?.bytesWritten ?? lastBytes;
2433
- if (bytes !== lastBytes) {
2434
- lastBytes = bytes;
2435
- if (!heldOnHeartbeat) { lastProgressAt = now(); return; }
2436
- // Held: those bytes were OUR keepalive, so they are not progress — deliberately
2437
- // fall through to the staleness check rather than returning.
2438
- }
2464
+ if (bytes !== lastBytes) { lastBytes = bytes; lastProgressAt = now(); return; }
2439
2465
  if (now() - lastProgressAt >= idleMs && !res.writableEnded && !res.destroyed) {
2440
2466
  console.error(`[Maxpool] Request ${reqId} — no write progress for ${Math.round(idleMs / 1000)}s (backstop reaper); force-aborting a stuck request to free its account slot`);
2441
2467
  res.destroy();
package/src/tui.js CHANGED
@@ -1152,16 +1152,31 @@ export class TUI {
1152
1152
  }
1153
1153
  lines.push(` Routing ${cyan(routing)}${xpText}`);
1154
1154
  const queuedCount = this.am.queueState?.waiting?.length || 0;
1155
- if (this.am._isUpstreamThrottleBlocking?.() || queuedCount) {
1155
+ // Throttle and WAITING are different things and used to share one line labelled
1156
+ // "Anthropic upstream throttled" — so a request parked purely because every account
1157
+ // is at its quota was either invisible (no throttle) or described as a throttle it
1158
+ // wasn't. Waiting is the headline state of this proxy; it gets its own line, always
1159
+ // shown whenever anything is parked, naming what it is waiting FOR.
1160
+ if (this.am._isUpstreamThrottleBlocking?.()) {
1156
1161
  const throttle = this.am.upstreamThrottle;
1157
1162
  const remaining = throttle.until ? Math.max(0, Math.ceil((throttle.until - Date.now()) / 1000)) : 0;
1158
- const state = this.am._isUpstreamThrottleBlocking?.()
1159
- ? throttle.probeInFlight ? 'probing recovery' : `retry in ${remaining}s`
1160
- : 'recovering';
1161
- const queued = queuedCount;
1162
- const oldest = queued ? Math.max(0, Date.now() - this.am.queueState.waiting[0].queuedAt) : 0;
1163
- const queueText = queued ? ` queued ${queued} oldest ${formatMs(oldest)}` : '';
1164
- lines.push(` ${yellow(' Anthropic upstream throttled')} ${dim(state + queueText)}`);
1163
+ const state = throttle.probeInFlight ? 'probing recovery' : `retry in ${remaining}s`;
1164
+ lines.push(` ${yellow(' Anthropic upstream throttled')} ${dim(state)}`);
1165
+ }
1166
+ if (queuedCount) {
1167
+ const oldest = Math.max(0, Date.now() - this.am.queueState.waiting[0].queuedAt);
1168
+ // Name the soonest thing that would release them, so a long wait reads as
1169
+ // "waiting for a known reset" rather than "hung".
1170
+ let why = 'waiting for capacity';
1171
+ try {
1172
+ const plan = this.am.nextRetryForRequest?.({}, new Set()) || {};
1173
+ if (Number.isFinite(plan.retryAfterMs) && plan.retryAfterMs > 0) {
1174
+ why = `next account frees in ~${formatMs(plan.retryAfterMs)}`;
1175
+ } else if (plan.cause) {
1176
+ why = `waiting (${plan.cause.replace(/_/g, ' ')})`;
1177
+ }
1178
+ } catch { /* display-only; never let the oracle break the render */ }
1179
+ lines.push(` ${cyan(' Parked')} ${dim(`${queuedCount} request${queuedCount === 1 ? '' : 's'} held oldest ${formatMs(oldest)} · ${why}`)}`);
1165
1180
  }
1166
1181
 
1167
1182
  // ── Accounts