maxpool 1.5.56 → 1.5.58

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.56",
3
+ "version": "1.5.58",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -1785,13 +1785,14 @@ export class AccountManager {
1785
1785
  // are all unavailable the request HOLDS/queues (recoverable) rather than 400ing.
1786
1786
  if (requestInfo.hasImage && account.provider === 'kimi') return false;
1787
1787
 
1788
- // A large-context session: a provider already rejected this request with a
1789
- // context-length 400. The GLM/Kimi *coding* endpoints serve a fixed model capped
1790
- // at ~256K and IGNORE the model id (so a K3/GLM 1M PLAN doesn't lift the coding
1791
- // leg's ceiling). Only a 1M-context Claude account can hold it — bench the
1792
- // providers for this session so it routes to Claude, or HOLDS for one, instead of
1793
- // re-404ing on a too-small leg. Sticky per session (context only grows turn over
1794
- // turn), so no follow-up turn re-pays the wasted attempt.
1788
+ // A large-context session: a provider already rejected THIS request with a
1789
+ // context-length 400. Deliberately REACTIVE — it never assumes a ceiling, it learns
1790
+ // one from an actual rejection, so it self-corrects as providers grow. That matters:
1791
+ // the old ~256K coding-endpoint cap is gone (verified 2026-08-02 — GLM 5.2 and Kimi
1792
+ // K3 both accepted a ~400K-token payload and both honoured the requested model id),
1793
+ // so this branch simply stops firing rather than needing a new constant.
1794
+ // Sticky per session (context only grows turn over turn) so no follow-up turn re-pays
1795
+ // the wasted attempt.
1795
1796
  if (account.type === 'provider' && this._isSessionLargeContext(requestInfo)) return false;
1796
1797
 
1797
1798
  const { incompatible, homeProvider } = this._effectiveIncompatible(requestInfo);
package/src/oauth.js CHANGED
@@ -27,14 +27,17 @@ export function tokenFingerprint(token) {
27
27
  * Refresh an expired OAuth access token using the refresh token.
28
28
  * Retries on 5xx and network errors with exponential backoff.
29
29
  */
30
+ // How long to wait for the user to finish the browser login. Env-overridable.
31
+ const LOGIN_TIMEOUT_MS = Math.max(60_000, Number(process.env.MAXPOOL_LOGIN_TIMEOUT_MS) || 300_000);
32
+
30
33
  export async function refreshAccessToken(refreshToken, endpoint = DEFAULT_TOKEN_ENDPOINT) {
31
34
  const maxRetries = 2;
32
35
  const baseDelayMs = 500;
33
- // Bound each refresh POST so a hung connect during an outage can't pin the
34
- // single-flight _refreshPromise indefinitely (which would stall that account's
35
- // recovery). Comfortably above a normal refresh latency; a timeout is classified
36
- // as a network error below and retried / short-cooled.
37
- const perAttemptTimeoutMs = 10_000;
36
+ // 30s, not 10s. A refresh aborted by OUR timeout is the dangerous case (see below), so
37
+ // the timeout must be generous enough that a merely slow network never triggers it.
38
+ // Measured 2026-08-03: refreshes aborted at 10s during a degraded-network window, and
39
+ // the retries that followed destroyed 4 accounts.
40
+ const perAttemptTimeoutMs = Math.max(10_000, Number(process.env.MAXPOOL_TOKEN_REFRESH_TIMEOUT_MS) || 30_000);
38
41
 
39
42
  for (let attempt = 0; attempt <= maxRetries; attempt++) {
40
43
  try {
@@ -83,10 +86,26 @@ export async function refreshAccessToken(refreshToken, endpoint = DEFAULT_TOKEN_
83
86
  (err.code === 'ECONNRESET' || err.code === 'ECONNREFUSED' ||
84
87
  err.code === 'ETIMEDOUT' || err.code === 'UND_ERR_CONNECT_TIMEOUT'));
85
88
 
86
- if (attempt < maxRetries && isNetworkError) {
89
+ // NEVER re-send a single-use refresh token after an AMBIGUOUS failure. A timeout,
90
+ // an abort, or a reset AFTER the request left us means the server may have processed
91
+ // it and rotated the token — retrying then spends a token that is already dead and
92
+ // the account is destroyed with invalid_grant. Only a failure that provably happened
93
+ // BEFORE the server saw the request (connection refused, DNS) is safe to retry.
94
+ //
95
+ // Measured 2026-08-03: a degraded network aborted refreshes at 10s; the retry loop
96
+ // re-sent each token up to 3 times within seconds, and 4 accounts died with
97
+ // "Refresh token not found or invalid" — requiring a manual re-login each.
98
+ const sentToServer = !(err.code === 'ECONNREFUSED' || err.code === 'ENOTFOUND'
99
+ || err.code === 'EAI_AGAIN' || err.code === 'UND_ERR_CONNECT_TIMEOUT');
100
+ if (attempt < maxRetries && isNetworkError && !sentToServer) {
87
101
  continue;
88
102
  }
89
- if (isNetworkError) err.retryable = true;
103
+ if (isNetworkError) {
104
+ err.retryable = true;
105
+ // Tell the caller the token's fate is UNKNOWN, so it cools the account and tries
106
+ // once later instead of treating this as a clean failure.
107
+ err.ambiguousRefresh = sentToServer;
108
+ }
90
109
  throw err;
91
110
  }
92
111
  }
@@ -555,11 +574,13 @@ function startCallbackServer(expectedState) {
555
574
  });
556
575
  server.on('error', reject);
557
576
 
558
- // Timeout after 2 minutes (unref so it doesn't keep the process alive)
577
+ // Bounded so a forgotten browser tab can't pin the process. 5 minutes, not 2:
578
+ // reported 2026-08-03 that a real login (password + 2FA, several accounts in a row)
579
+ // repeatedly overran 2 minutes and had to be restarted from scratch.
559
580
  const timer = setTimeout(() => {
560
- rejectCode(new Error('Login timed out after 2 minutes'));
581
+ rejectCode(new Error(`Login timed out after ${Math.round(LOGIN_TIMEOUT_MS / 60_000)} minutes`));
561
582
  server.close();
562
- }, 120_000);
583
+ }, LOGIN_TIMEOUT_MS);
563
584
  timer.unref();
564
585
  });
565
586
  }
package/src/server.js CHANGED
@@ -2370,7 +2370,10 @@ function prepareRuntimeProviders(accountManager, headers) {
2370
2370
 
2371
2371
  const kimiToken = headerValue(headers, 'x-maxpool-kimi-token');
2372
2372
  if (kimiToken) {
2373
- const model = headerValue(headers, 'x-maxpool-kimi-model') || 'kimi-k2.7';
2373
+ // Fallback only — `cc all` always sends x-maxpool-kimi-model from the llm_config SSOT,
2374
+ // so this is what a bare/older client gets. Kept current deliberately: it read
2375
+ // 'kimi-k2.7' while the fleet had moved to k3.
2376
+ const model = headerValue(headers, 'x-maxpool-kimi-model') || 'kimi-k3';
2374
2377
  accountManager.upsertRuntimeAccount({
2375
2378
  name: 'kimi-fallback',
2376
2379
  type: 'provider',
package/src/tui.js CHANGED
@@ -1152,16 +1152,31 @@ export class TUI {
1152
1152
  }
1153
1153
  lines.push(` Routing ${cyan(routing)}${xpText}`);
1154
1154
  const queuedCount = this.am.queueState?.waiting?.length || 0;
1155
- if (this.am._isUpstreamThrottleBlocking?.() || queuedCount) {
1155
+ // Throttle and WAITING are different things and used to share one line labelled
1156
+ // "Anthropic upstream throttled" — so a request parked purely because every account
1157
+ // is at its quota was either invisible (no throttle) or described as a throttle it
1158
+ // wasn't. Waiting is the headline state of this proxy; it gets its own line, always
1159
+ // shown whenever anything is parked, naming what it is waiting FOR.
1160
+ if (this.am._isUpstreamThrottleBlocking?.()) {
1156
1161
  const throttle = this.am.upstreamThrottle;
1157
1162
  const remaining = throttle.until ? Math.max(0, Math.ceil((throttle.until - Date.now()) / 1000)) : 0;
1158
- const state = this.am._isUpstreamThrottleBlocking?.()
1159
- ? throttle.probeInFlight ? 'probing recovery' : `retry in ${remaining}s`
1160
- : 'recovering';
1161
- const queued = queuedCount;
1162
- const oldest = queued ? Math.max(0, Date.now() - this.am.queueState.waiting[0].queuedAt) : 0;
1163
- const queueText = queued ? ` queued ${queued} oldest ${formatMs(oldest)}` : '';
1164
- lines.push(` ${yellow(' Anthropic upstream throttled')} ${dim(state + queueText)}`);
1163
+ const state = throttle.probeInFlight ? 'probing recovery' : `retry in ${remaining}s`;
1164
+ lines.push(` ${yellow(' Anthropic upstream throttled')} ${dim(state)}`);
1165
+ }
1166
+ if (queuedCount) {
1167
+ const oldest = Math.max(0, Date.now() - this.am.queueState.waiting[0].queuedAt);
1168
+ // Name the soonest thing that would release them, so a long wait reads as
1169
+ // "waiting for a known reset" rather than "hung".
1170
+ let why = 'waiting for capacity';
1171
+ try {
1172
+ const plan = this.am.nextRetryForRequest?.({}, new Set()) || {};
1173
+ if (Number.isFinite(plan.retryAfterMs) && plan.retryAfterMs > 0) {
1174
+ why = `next account frees in ~${formatMs(plan.retryAfterMs)}`;
1175
+ } else if (plan.cause) {
1176
+ why = `waiting (${plan.cause.replace(/_/g, ' ')})`;
1177
+ }
1178
+ } catch { /* display-only; never let the oracle break the render */ }
1179
+ lines.push(` ${cyan(' Parked')} ${dim(`${queuedCount} request${queuedCount === 1 ? '' : 's'} held oldest ${formatMs(oldest)} · ${why}`)}`);
1165
1180
  }
1166
1181
 
1167
1182
  // ── Accounts