maxpool 1.5.35 → 1.5.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/package.json +1 -1
  2. package/src/server.js +46 -5
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.35",
3
+ "version": "1.5.37",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
package/src/server.js CHANGED
@@ -25,6 +25,11 @@ const UPSTREAM_TTFB_MS = Math.max(5_000, Number(process.env.MAXPOOL_TTFB_MS) ||
25
25
  const STREAM_IDLE_MS = Math.max(30_000, Number(process.env.MAXPOOL_STREAM_IDLE_MS) || 300_000); // max gap BETWEEN streamed chunks (reset per chunk)
26
26
  const UPSTREAM_BODY_MS = Math.max(30_000, Number(process.env.MAXPOOL_BODY_MS) || 300_000); // non-streaming body read
27
27
  const CLIENT_DRAIN_MS = Math.max(5_000, Number(process.env.MAXPOOL_DRAIN_MS) || 60_000); // max wait for a backpressured client to drain (half-open client → free the lease)
28
+ // A provider 403 is (unlike a 401) almost always transient QUOTA/PLAN exhaustion — cool
29
+ // the provider down RECOVERABLY for this window, then re-probe, instead of permanently
30
+ // disabling it. Short (not reset-length) so a provider-pinned session's hold window
31
+ // isn't blown, and a still-exhausted provider just re-benches on the next single retry.
32
+ const PROVIDER_FORBIDDEN_COOLDOWN_SEC = Math.max(60, Number(process.env.MAXPOOL_PROVIDER_403_COOLDOWN_SEC) || 1800);
28
33
  // Backstop reaper idle ceiling. Floored WELL above the longest a legit NON-streaming
29
34
  // request can go silent (no heartbeat there): TTFB 120s + nonStreamMaxWaitMs 300s +
30
35
  // UPSTREAM_BODY_MS 300s ≈ 720s — so a slow-but-alive non-streaming request is never
@@ -51,6 +56,11 @@ const DEFAULT_QUEUE = {
51
56
  // Non-streaming requests have no SSE heartbeat to keep them alive, so a long
52
57
  // hold would die on the client timeout anyway. Cap their wait conservatively.
53
58
  nonStreamMaxWaitMs: 5 * 60 * 1000,
59
+ // count_tokens is cheap non-streaming metadata — cap its queue wait VERY low so it
60
+ // fast-fails with a retryable 429 instead of hanging silently past the client's idle
61
+ // window. Kept well under any plausible client idle timeout (observed errors as low
62
+ // as ~23s). Env-overridable for tuning as the "Stream idle timeout" reports resolve.
63
+ countTokensMaxWaitMs: Math.max(1000, Number(process.env.MAXPOOL_COUNT_TOKENS_MAX_WAIT_MS) || 8000),
54
64
  // Backpressure: holds used to be 0ms, now they can be hours. Bound the queue
55
65
  // so 22 retrying agents can't grow the heap without limit.
56
66
  maxConcurrentQueued: 64,
@@ -84,6 +94,11 @@ export function createProxyServer(accountManager, config, hooks = {}) {
84
94
 
85
95
  const server = http.createServer(async (req, res) => {
86
96
  try {
97
+ // Disable Nagle on the client socket: flush each SSE event immediately instead
98
+ // of coalescing small writes into bursts. Smooths streaming cadence toward the
99
+ // HTTP/2-direct shape, reducing how often a client terminal's re-render churns
100
+ // on long streams. Same underlying socket as res, so this covers the write path.
101
+ try { req.socket?.setNoDelay(true); } catch { /* socket may already be gone */ }
87
102
  try { res.setHeader('x-maxpool-worker', workerStamp); } catch { /* headers sent */ }
88
103
  if (draining) {
89
104
  try { res.setHeader('Connection', 'close'); } catch { /* headers may be sent */ }
@@ -654,16 +669,27 @@ async function forwardRequest(
654
669
 
655
670
  if (account.type === 'provider' && isProviderAuthStatus(upstreamRes.status)) {
656
671
  const errorBody = await readErrorBody(upstreamRes);
657
- const reason = upstreamRes.status === 401 ? 'auth_failed' : 'forbidden';
658
- accountManager.markAuthFailed(account.index, upstreamRes.status, reason);
659
- accountManager.releaseAccount(lease, { status: upstreamRes.status, error: reason });
672
+ const forbidden = upstreamRes.status === 403;
673
+ if (forbidden) {
674
+ // 403 = almost always QUOTA/PLAN exhaustion (e.g. Kimi Coding-Plan weekly maxed),
675
+ // NOT bad credentials — the usage probe still succeeds with the same key. Cool it
676
+ // down RECOVERABLY (rides the rate-limit recovery in _isAvailable → auto-un-benches
677
+ // at expiry) instead of the permanent auth-disable that stranded it until restart.
678
+ // `neutral` release so a transient quota-403 doesn't bump failure/scoring counters.
679
+ accountManager.markRateLimited(account.index, PROVIDER_FORBIDDEN_COOLDOWN_SEC, { status: 403, recordFailure: false });
680
+ accountManager.releaseAccount(lease, { neutral: true });
681
+ } else {
682
+ // 401 = genuinely bad/missing credentials → permanent disable (needs re-auth).
683
+ accountManager.markAuthFailed(account.index, upstreamRes.status, 'auth_failed');
684
+ accountManager.releaseAccount(lease, { status: upstreamRes.status, error: 'auth_failed' });
685
+ }
660
686
  excludedIndexes.add(account.index);
661
687
 
662
688
  if (logDir) {
663
- logSections.push(`=== RESPONSE ${upstreamRes.status} — "${account.name}" disabled (${reason}), failing over ===\n${formatHeaders(upstreamRes.headers)}`);
689
+ logSections.push(`=== RESPONSE ${upstreamRes.status} — "${account.name}" ${forbidden ? `cooled down ${PROVIDER_FORBIDDEN_COOLDOWN_SEC}s (quota/forbidden)` : 'disabled (auth)'}, failing over ===\n${formatHeaders(upstreamRes.headers)}`);
664
690
  if (errorBody) logSections.push(`=== ERROR BODY ===\n${errorBody}`);
665
691
  }
666
- console.log(`[Maxpool] ${upstreamRes.status} on provider "${account.name}" — disabled and failing over before first byte`);
692
+ console.log(`[Maxpool] ${upstreamRes.status} on provider "${account.name}" — ${forbidden ? `cooled down ${PROVIDER_FORBIDDEN_COOLDOWN_SEC}s (recoverable)` : 'disabled'} and failing over before first byte`);
667
693
 
668
694
  if (
669
695
  canRetryBufferedBody &&
@@ -1074,6 +1100,7 @@ function formatRetryDuration(seconds) {
1074
1100
  function computeQueueWindowMs({
1075
1101
  cause, stream, retryPlanCause,
1076
1102
  maxWaitMs, capacityMaxWaitMs, nonStreamMaxWaitMs, streamHoldMaxMs,
1103
+ isCountTokens, countTokensMaxWaitMs,
1077
1104
  }) {
1078
1105
  let windowMs;
1079
1106
  if (!stream) {
@@ -1084,6 +1111,10 @@ function computeQueueWindowMs({
1084
1111
  windowMs = streamHoldMaxMs;
1085
1112
  }
1086
1113
  if (retryPlanCause === 'concurrency_cap') windowMs = Math.min(windowMs, capacityMaxWaitMs);
1114
+ // count_tokens: cap the QUEUE wait low (bounds only the wait-for-an-account, never
1115
+ // the upstream processing once acquired) so a non-heartbeated metadata call fast-
1116
+ // fails with a retryable 429 instead of hanging past the client's idle window.
1117
+ if (isCountTokens && countTokensMaxWaitMs != null) windowMs = Math.min(windowMs, countTokensMaxWaitMs);
1087
1118
  return windowMs;
1088
1119
  }
1089
1120
 
@@ -1371,6 +1402,9 @@ async function queueAndRetry(
1371
1402
  const nonStreamMaxWaitMs = queueConfig.nonStreamMaxWaitMs == null
1372
1403
  ? 5 * 60_000
1373
1404
  : Math.max(0, Number(queueConfig.nonStreamMaxWaitMs) || 0);
1405
+ const countTokensMaxWaitMs = queueConfig.countTokensMaxWaitMs == null
1406
+ ? 8000
1407
+ : Math.max(0, Number(queueConfig.countTokensMaxWaitMs) || 0);
1374
1408
  const retryPlan = accountManager.nextRetryForRequest?.(requestInfo, new Set()) || {
1375
1409
  retryAfterMs: Infinity,
1376
1410
  cause: 'unavailable',
@@ -1405,6 +1439,8 @@ async function queueAndRetry(
1405
1439
  capacityMaxWaitMs,
1406
1440
  nonStreamMaxWaitMs,
1407
1441
  streamHoldMaxMs,
1442
+ isCountTokens: Boolean(requestInfo.isCountTokens),
1443
+ countTokensMaxWaitMs,
1408
1444
  });
1409
1445
 
1410
1446
  if (queueWindowMs <= 0) return finishQueuedStreamIfNeeded(res, requestInfo, honestMessage);
@@ -1575,6 +1611,11 @@ function describeRequest(req, body) {
1575
1611
  path: req.url,
1576
1612
  bodyBytes: body.length,
1577
1613
  weight,
1614
+ // count_tokens is cheap non-streaming metadata with NO SSE heartbeat, so a long
1615
+ // queue-hold just dies on the client's idle window ("Stream idle timeout - no
1616
+ // chunks received"). Flag it (URL-based, so it's set regardless of body parse) to
1617
+ // cap its queue wait short and fast-fail with a retryable 429 instead of hanging.
1618
+ isCountTokens: /\/v1\/messages\/count_tokens\b/.test(req.url),
1578
1619
  };
1579
1620
  try {
1580
1621
  const json = JSON.parse(body.toString());