maxpool 1.5.40 → 1.5.41

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.40",
3
+ "version": "1.5.41",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -1699,6 +1699,15 @@ export class AccountManager {
1699
1699
  // are all unavailable the request HOLDS/queues (recoverable) rather than 400ing.
1700
1700
  if (requestInfo.hasImage && account.provider === 'kimi') return false;
1701
1701
 
1702
+ // A large-context session: a provider already rejected this request with a
1703
+ // context-length 400. The GLM/Kimi *coding* endpoints serve a fixed model capped
1704
+ // at ~256K and IGNORE the model id (so a K3/GLM 1M PLAN doesn't lift the coding
1705
+ // leg's ceiling). Only a 1M-context Claude account can hold it — bench the
1706
+ // providers for this session so it routes to Claude, or HOLDS for one, instead of
1707
+ // re-404ing on a too-small leg. Sticky per session (context only grows turn over
1708
+ // turn), so no follow-up turn re-pays the wasted attempt.
1709
+ if (account.type === 'provider' && this._isSessionLargeContext(requestInfo)) return false;
1710
+
1702
1711
  const { incompatible, homeProvider } = this._effectiveIncompatible(requestInfo);
1703
1712
  const policy = this._crossProviderFallbackPolicy();
1704
1713
 
@@ -1732,6 +1741,9 @@ export class AccountManager {
1732
1741
  if (requestInfo.anthropicIncompatible) {
1733
1742
  this.markSessionIncompatible(requestInfo.sessionKey, requestInfo.homeProvider);
1734
1743
  }
1744
+ if (requestInfo.largeContext) {
1745
+ this.markSessionLargeContext(requestInfo.sessionKey);
1746
+ }
1735
1747
  }
1736
1748
 
1737
1749
  // Latch a session as Anthropic-incompatible (a foreign server_tool_use id, or a
@@ -1750,6 +1762,25 @@ export class AccountManager {
1750
1762
  });
1751
1763
  }
1752
1764
 
1765
+ // Latch a session as large-context: a provider (the GLM/Kimi coding endpoint, fixed
1766
+ // ~256K) rejected a request with a context-length 400 that only a 1M Claude can hold.
1767
+ // Sticky + never-downgrades so every follow-up turn (context only grows) skips the
1768
+ // too-small providers instead of re-paying a wasted 400. Cleared only by a new session.
1769
+ markSessionLargeContext(sessionKey) {
1770
+ if (!sessionKey) return;
1771
+ const existing = this.sessionPolicies.get(sessionKey) || {};
1772
+ if (!existing.largeContext) {
1773
+ console.log(`[Maxpool] Session "${sessionKey}" exceeds provider context limits — pinned to Claude (GLM/Kimi benched for this session)`);
1774
+ }
1775
+ this.sessionPolicies.set(sessionKey, { ...existing, largeContext: true });
1776
+ }
1777
+
1778
+ _isSessionLargeContext(requestInfo = {}) {
1779
+ if (requestInfo.largeContext) return true;
1780
+ if (!requestInfo.sessionKey) return false;
1781
+ return Boolean(this.sessionPolicies.get(requestInfo.sessionKey)?.largeContext);
1782
+ }
1783
+
1753
1784
  // Marks a session as containing Anthropic signed thinking. This no longer bars
1754
1785
  // provider fallback (a lenient provider accepts an Anthropic signature) — it only
1755
1786
  // keeps the session's live cross-account MIGRATION on Claude (the rebalance guard),
@@ -2587,6 +2618,10 @@ export class AccountManager {
2587
2618
  account.modelMap = acctData.modelMap || account.modelMap;
2588
2619
  account.stripBetaHeaders = Boolean(acctData.stripBetaHeaders);
2589
2620
  account.runtime = true;
2621
+ // Restore path carries an explicit enabled (persisted disable); honor it. The `cc
2622
+ // all` header path (prepareRuntimeProviders) omits enabled, so a re-sent token
2623
+ // NEVER silently re-enables a provider the user benched in the TUI.
2624
+ if (acctData.enabled !== undefined) account.enabled = acctData.enabled !== false;
2590
2625
  if (account.status === 'error' && changed) {
2591
2626
  account.status = 'active';
2592
2627
  account.lastError = null;
@@ -2620,6 +2655,10 @@ export class AccountManager {
2620
2655
  model: a.model,
2621
2656
  modelMap: a.modelMap,
2622
2657
  stripBetaHeaders: a.stripBetaHeaders,
2658
+ // Persist the user's enable/disable so a provider they benched in the TUI stays
2659
+ // benched across a restart — without this an intentionally-disabled GLM/Kimi
2660
+ // silently comes back enabled on the next boot (restore defaults enabled:true).
2661
+ enabled: a.enabled,
2623
2662
  }));
2624
2663
  }
2625
2664
 
@@ -2797,6 +2836,7 @@ export class AccountManager {
2797
2836
  stickyBindings: this.sessionBindings.size,
2798
2837
  thinkingProtected: [...this.sessionPolicies.values()].filter(p => p.requiresAnthropicThinkingIntegrity).length,
2799
2838
  providerPinned: [...this.sessionPolicies.values()].filter(p => p.anthropicIncompatible).length,
2839
+ largeContextPinned: [...this.sessionPolicies.values()].filter(p => p.largeContext).length,
2800
2840
  },
2801
2841
  };
2802
2842
  }
package/src/prober.js CHANGED
@@ -70,7 +70,11 @@ export class Prober {
70
70
  this._stopping = false;
71
71
  this._inflight = (async () => {
72
72
  try {
73
- // Skip auth-dead accounts (dead refresh token): probing them just re-POSTs
73
+ // DISABLED accounts are INTENTIONALLY still probed (no `a.enabled` filter):
74
+ // a user often disables an account precisely BECAUSE it's exhausted, and still
75
+ // wants to see its quota recover — so keep refreshing its usage for visibility
76
+ // even though routing skips it. Do NOT add an enabled gate here.
77
+ // Skip only auth-dead accounts (dead refresh token): probing them just re-POSTs
74
78
  // the rejected token every cycle — a 400 storm. They recover only on re-auth.
75
79
  const oauth = this.am.accounts.filter(a => a.type === 'oauth' && a.credential && !a.refreshDead);
76
80
  const providers = this.am.accounts.filter(a => a.type === 'provider' && a.credential);
package/src/server.js CHANGED
@@ -53,6 +53,16 @@ const DEFAULT_QUEUE = {
53
53
  // resets idle-gap client timeouts; if a client uses a wall-clock total-request
54
54
  // deadline, lower this to just under it.
55
55
  streamHoldMaxMs: 7 * 24 * 60 * 60 * 1000,
56
+ // HARD ceiling on EVERY streaming hold (capacity/quota/throttle/concurrency alike).
57
+ // maxpool CANNOT keep a client alive past its own stream watchdog — Claude Code aborts
58
+ // "Stream idle timeout - no chunks received" after CLAUDE_STREAM_IDLE_TIMEOUT_MS of no
59
+ // real content EVENTS, and it drops SSE ping/comment keep-alives before the watchdog
60
+ // sees them (anthropic-sdk-typescript#998). So a 7-day/24h server-side hold only parks
61
+ // a request the client already abandoned. Bound it to the wait the user actually wants
62
+ // (front-loaded work waiting for a free account ≈ a few hours) so beyond that the
63
+ // request error-fasts with an honest retryable 429 instead of a silent multi-day park.
64
+ // Pair with a raised client watchdog (the cc launch sets CLAUDE_STREAM_IDLE_TIMEOUT_MS).
65
+ streamClientToleranceMs: Math.max(60_000, Number(process.env.MAXPOOL_STREAM_CLIENT_TOLERANCE_MS) || 3 * 60 * 60 * 1000),
56
66
  // Non-streaming requests have no SSE heartbeat to keep them alive, so a long
57
67
  // hold would die on the client timeout anyway. Cap their wait conservatively.
58
68
  nonStreamMaxWaitMs: 5 * 60 * 1000,
@@ -235,9 +245,10 @@ export function createProxyServer(accountManager, config, hooks = {}) {
235
245
  }
236
246
  requestInfo.profile = getMaxpoolProfile(req.headers);
237
247
  requestInfo.sessionKey = headerValue(req.headers, 'x-maxpool-session');
238
- if (requestInfo.requiresAnthropicThinkingIntegrity && requestInfo.profile === 'all') {
239
- console.log('[Maxpool] Anthropic thinking detected; provider fallback disabled for this session/request');
240
- }
248
+ // (Removed a FALSE "provider fallback disabled for signed thinking" log here: it
249
+ // fired on every thinking `all` request but was untrue under the default
250
+ // when-exhausted/always policies — providers DO serve thinking requests — and it
251
+ // repeatedly misdirected diagnosis of the "Stream idle timeout" reports.)
241
252
  prepareRuntimeProviders(accountManager, req.headers);
242
253
 
243
254
  await forwardRequest(
@@ -856,9 +867,15 @@ async function forwardRequest(
856
867
  // signature it can't validate). Detect it on an Anthropic account so we can
857
868
  // self-heal onto a provider instead of surfacing the 400.
858
869
  const anthropicIncompat = account.type !== 'provider' && isAnthropicIncompatBody(errorBody);
870
+ // A provider (GLM ~200K / Kimi 256K context) rejecting an oversized request that
871
+ // only a large-context Claude (1M) can hold — e.g. "exceeded model token limit:
872
+ // 262144". Detect it ONLY on a provider (a Claude account's context-length 400 is
873
+ // terminal — nothing bigger to fall to) so we can pin the session to Claude.
874
+ const providerTooSmall = account.type === 'provider' && isContextLengthError(errorBody);
859
875
  const errorType = errorBody.includes('Invalid `signature` in `thinking` block')
860
876
  ? 'invalid_thinking_signature'
861
877
  : anthropicIncompat ? 'anthropic_incompatible_transcript'
878
+ : providerTooSmall ? 'provider_context_too_small'
862
879
  : `HTTP ${upstreamRes.status}`;
863
880
  accountManager.releaseAccount(lease, { status: upstreamRes.status, error: errorType });
864
881
 
@@ -907,6 +924,33 @@ async function forwardRequest(
907
924
  );
908
925
  }
909
926
 
927
+ // React-and-heal: a PROVIDER (GLM/Kimi coding endpoint, fixed ~256K context)
928
+ // rejected an oversized request only a 1M-context Claude can hold — e.g. Kimi's
929
+ // "exceeded model token limit: 262144 (requested: 643557)". The coding leg IGNORES
930
+ // the model id, so a K3/GLM-1M plan doesn't lift its ceiling. Latch the session
931
+ // large-context (so its follow-up turns skip the too-small providers) and retry
932
+ // EXCLUDING every provider → routes to Claude, or HOLDS for a Claude account,
933
+ // instead of surfacing a 400 the client just retry-loops on. Only worth it when a
934
+ // Claude account exists to serve it; with none, the 400 surfaces (nothing bigger).
935
+ const claudeAvailable = accountManager.accounts?.some(a => a.type !== 'provider' && a.enabled !== false);
936
+ if (providerTooSmall && requestInfo.sessionKey && claudeAvailable
937
+ && canRetryBufferedBody && retryCount + 1 < maxAttempts && !res.headersSent) {
938
+ accountManager.markSessionLargeContext?.(requestInfo.sessionKey);
939
+ // Bench EVERY provider: today's GLM + Kimi coding legs both cap at ~256K, so once
940
+ // one 400s on size the others can't hold it either. If a genuine 1M-context
941
+ // provider is ever added, make this exclusion context-limit-aware instead of
942
+ // type-wide (the _isRequestCompatible gate would need the same treatment).
943
+ for (const a of (accountManager.accounts || [])) {
944
+ if (a.type === 'provider') excludedIndexes.add(a.index);
945
+ }
946
+ console.log(`[Maxpool] Provider "${account.name}" context too small for this request; pinning session to Claude and retrying`);
947
+ return forwardRequest(
948
+ req, res, body, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
949
+ retryConfig, queueConfig, { ...requestInfo, largeContext: true },
950
+ canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
951
+ );
952
+ }
953
+
910
954
  ctx.status = upstreamRes.status;
911
955
  sendErrorBody(res, requestInfo, upstreamRes.status, errorBody, upstreamRes.headers);
912
956
  return;
@@ -1135,7 +1179,7 @@ function formatRetryDuration(seconds) {
1135
1179
  */
1136
1180
  function computeQueueWindowMs({
1137
1181
  cause, stream, retryPlanCause,
1138
- maxWaitMs, capacityMaxWaitMs, nonStreamMaxWaitMs, streamHoldMaxMs,
1182
+ maxWaitMs, capacityMaxWaitMs, nonStreamMaxWaitMs, streamHoldMaxMs, streamClientToleranceMs,
1139
1183
  isCountTokens, countTokensMaxWaitMs,
1140
1184
  }) {
1141
1185
  let windowMs;
@@ -1147,6 +1191,12 @@ function computeQueueWindowMs({
1147
1191
  windowMs = streamHoldMaxMs;
1148
1192
  }
1149
1193
  if (retryPlanCause === 'concurrency_cap') windowMs = Math.min(windowMs, capacityMaxWaitMs);
1194
+ // Bound EVERY streaming cause to the client-tolerance ceiling — the client's own
1195
+ // watchdog kills the stream well before a 24h/7d server hold, so anything past this
1196
+ // just parks an abandoned request. A finite reset WITHIN the ceiling still holds +
1197
+ // resumes (via the nextRetryForRequest oracle); a reset beyond it error-fasts (a real
1198
+ // retryable 429 at the pre-heartbeat gate) instead of hanging.
1199
+ if (stream && streamClientToleranceMs != null) windowMs = Math.min(windowMs, streamClientToleranceMs);
1150
1200
  // count_tokens: cap the QUEUE wait low (bounds only the wait-for-an-account, never
1151
1201
  // the upstream processing once acquired) so a non-heartbeated metadata call fast-
1152
1202
  // fails with a retryable 429 instead of hanging past the client's idle window.
@@ -1166,9 +1216,22 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
1166
1216
  return `This session's transcript can only run on ${fam} — Claude rejects its server-tool ids/thinking on replay. No ${fam} provider is available right now.${eta} Check the x-maxpool-zai-token / x-maxpool-kimi-token headers, or resume with 'cc ${incompat.homeProvider === 'kimi' ? 'kimi' : 'glm'}'.`;
1167
1217
  }
1168
1218
 
1169
- const thinking = requestInfo.requiresAnthropicThinkingIntegrity
1170
- || accountManager._requiresAnthropicThinkingIntegrity?.(requestInfo);
1171
- const n = accountManager.accounts.length;
1219
+ // A large-context session (a provider already 400'd it as too big for its ~256K leg):
1220
+ // only a 1M-context Claude can hold it — the providers are structurally BARRED, not
1221
+ // merely "at their limit". Say the oversized truth + the two real ways out (wait for a
1222
+ // Claude account, or /compact), instead of the misleading "providers at limit" line.
1223
+ if (accountManager._isSessionLargeContext?.(requestInfo)) {
1224
+ const eta = Number.isFinite(retryAfter) && retryAfter > 0
1225
+ ? ` A Claude account should free in ~${formatRetryDuration(retryAfter)}.` : '';
1226
+ return `This session is too large for the GLM/Kimi fallbacks (their ~256K limit) — it needs a 1M-context Claude account, and they're all busy right now.${eta} It sends as soon as one frees; /compact shortens the session if you'd rather not wait.`;
1227
+ }
1228
+
1229
+ const claudeCount = accountManager.accounts.filter(a => a.type !== 'provider').length;
1230
+ // Only name the providers when this pool actually HAS them (`cc all`). On `cc ma`
1231
+ // (Claude-only) there are none, so the old hardcoded "and the GLM/Kimi providers"
1232
+ // was a lie. When present they DO serve (not barred), so they're saturated too.
1233
+ const providersClause = accountManager.accounts.some(a => a.type === 'provider')
1234
+ ? ' and the GLM/Kimi providers' : '';
1172
1235
 
1173
1236
  // No route is expected to recover within the queue window — i.e. every Claude
1174
1237
  // account is at its own 5h/weekly limit. A short "retry in Ns" would be a lie;
@@ -1177,19 +1240,24 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
1177
1240
  const eta = Number.isFinite(retryAfter) && retryAfter > 0
1178
1241
  ? ` Soonest reset in ~${formatRetryDuration(retryAfter)}, beyond the hold window.`
1179
1242
  : '';
1180
- const base = `No Claude account can take this request — all ${n} are at their 5h or weekly limit.${eta} Add another Claude account or wait for a quota reset.`;
1181
- return thinking
1182
- ? `${base} GLM/Kimi fallback is unavailable because this session contains Anthropic signed thinking blocks; start a fresh non-thinking session to use them.`
1183
- : base;
1243
+ return `No account can take this request — all ${claudeCount} Claude accounts${providersClause} are at their limit.${eta} Add another Claude account or wait for a quota reset.`;
1184
1244
  }
1185
1245
 
1186
- if (thinking) {
1187
- return `No Claude account could accept this request. Non-Claude fallback is disabled because this session contains Anthropic signed thinking blocks. Retry in ${retryAfter}s, wait for Claude capacity, or start a fresh non-thinking session to use GLM/Kimi.`;
1188
- }
1189
- return `All ${n} accounts exhausted. Retry in ${retryAfter}s.`;
1246
+ return `No account can take this request right now — all ${claudeCount} Claude accounts${providersClause} are momentarily at their limit. Retry in ${retryAfter}s.`;
1247
+ }
1248
+
1249
+ // A provider (GLM/Kimi) rejecting a request whose token count exceeds its context
1250
+ // window — Kimi's coding leg ("exceeded model token limit: 262144"), GLM's, or an
1251
+ // OpenAI-style "maximum context length". Kept narrow (specific context-overflow
1252
+ // phrasings, NOT a bare "token limit" which a rate-limit body also carries) so the
1253
+ // pin-to-Claude heal only fires on a genuine size overflow, not any 400. Rate-limit
1254
+ // 429s are intercepted earlier (classifyRateLimit) and never reach this check.
1255
+ function isContextLengthError(errorBody) {
1256
+ if (!errorBody) return false;
1257
+ return /exceeded model token limit|maximum context length|context length exceeded|context window (?:size )?(?:exceeded|too)|prompt is too long|input is too long|reduce the length of|too many (?:input )?tokens|request too large/i.test(errorBody);
1190
1258
  }
1191
1259
 
1192
- export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, streamResponse, startIdleRequestReaper };
1260
+ export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
1193
1261
 
1194
1262
  async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
1195
1263
  if (!upstreamRes.body) return '';
@@ -1467,6 +1535,9 @@ async function queueAndRetry(
1467
1535
  const streamHoldMaxMs = queueConfig.streamHoldMaxMs == null
1468
1536
  ? 7 * 24 * 60 * 60 * 1000
1469
1537
  : Math.max(0, Number(queueConfig.streamHoldMaxMs) || 0);
1538
+ const streamClientToleranceMs = queueConfig.streamClientToleranceMs == null
1539
+ ? 3 * 60 * 60 * 1000
1540
+ : Math.max(0, Number(queueConfig.streamClientToleranceMs) || 0);
1470
1541
  const queueWindowMs = computeQueueWindowMs({
1471
1542
  cause,
1472
1543
  stream: Boolean(requestInfo.stream),
@@ -1475,6 +1546,7 @@ async function queueAndRetry(
1475
1546
  capacityMaxWaitMs,
1476
1547
  nonStreamMaxWaitMs,
1477
1548
  streamHoldMaxMs,
1549
+ streamClientToleranceMs,
1478
1550
  isCountTokens: Boolean(requestInfo.isCountTokens),
1479
1551
  countTokensMaxWaitMs,
1480
1552
  });
package/src/tui.js CHANGED
@@ -586,6 +586,9 @@ export class TUI {
586
586
  .map(index => ({ account: this.am.accounts[index], index }))
587
587
  .filter(({ account }) => {
588
588
  if (action === 'prefer') return account.type !== 'provider' && account.enabled;
589
+ // Enable/disable also works on runtime providers (GLM/Kimi) — a session-only
590
+ // toggle, since they're not in config. Rename/delete stay config-account-only.
591
+ if (action === 'toggle') return this._configAccountIndex(account) >= 0 || account.type === 'provider';
589
592
  return this._configAccountIndex(account) >= 0;
590
593
  })
591
594
  .map(({ index }) => index);
@@ -872,7 +875,18 @@ export class TUI {
872
875
  if (!account) return;
873
876
  const configIndex = this._configAccountIndex(account);
874
877
  if (configIndex < 0) {
875
- this._addLog(`Cannot ${enabled ? 'enable' : 'disable'} runtime provider "${account.name}" here`);
878
+ // A runtime provider (GLM/Kimi) isn't in config — it's re-created from the `cc all`
879
+ // request headers — but its enable/disable IS durable: the flag persists in memory
880
+ // across `cc all` requests (the header upsert never re-enables it) and to state.json
881
+ // on the next save, so it stays benched across a restart too. Re-enable it here the
882
+ // same way whenever the user wants it back — there's no "removed forever" state.
883
+ if (account.type === 'provider') {
884
+ this.am.setAccountEnabled(idx, enabled);
885
+ if (!enabled && this.am.preferredAccountName === account.name) this.am.setRoutingMode?.('automatic');
886
+ this._addLog(`${enabled ? 'Enabled' : 'Disabled'} provider "${account.name}" — ${enabled ? 'routing resumed' : 'benched (stays off across cc all + restart; re-enable here anytime)'}`);
887
+ return;
888
+ }
889
+ this._addLog(`Cannot ${enabled ? 'enable' : 'disable'} "${account.name}" here (not in config)`);
876
890
  return;
877
891
  }
878
892
  const previous = this.config.accounts[configIndex].enabled;