maxpool 1.5.39 → 1.5.41

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.39",
3
+ "version": "1.5.41",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -1699,6 +1699,15 @@ export class AccountManager {
1699
1699
  // are all unavailable the request HOLDS/queues (recoverable) rather than 400ing.
1700
1700
  if (requestInfo.hasImage && account.provider === 'kimi') return false;
1701
1701
 
1702
+ // A large-context session: a provider already rejected this request with a
1703
+ // context-length 400. The GLM/Kimi *coding* endpoints serve a fixed model capped
1704
+ // at ~256K and IGNORE the model id (so a K3/GLM 1M PLAN doesn't lift the coding
1705
+ // leg's ceiling). Only a 1M-context Claude account can hold it — bench the
1706
+ // providers for this session so it routes to Claude, or HOLDS for one, instead of
1707
+ // re-404ing on a too-small leg. Sticky per session (context only grows turn over
1708
+ // turn), so no follow-up turn re-pays the wasted attempt.
1709
+ if (account.type === 'provider' && this._isSessionLargeContext(requestInfo)) return false;
1710
+
1702
1711
  const { incompatible, homeProvider } = this._effectiveIncompatible(requestInfo);
1703
1712
  const policy = this._crossProviderFallbackPolicy();
1704
1713
 
@@ -1732,6 +1741,9 @@ export class AccountManager {
1732
1741
  if (requestInfo.anthropicIncompatible) {
1733
1742
  this.markSessionIncompatible(requestInfo.sessionKey, requestInfo.homeProvider);
1734
1743
  }
1744
+ if (requestInfo.largeContext) {
1745
+ this.markSessionLargeContext(requestInfo.sessionKey);
1746
+ }
1735
1747
  }
1736
1748
 
1737
1749
  // Latch a session as Anthropic-incompatible (a foreign server_tool_use id, or a
@@ -1750,6 +1762,25 @@ export class AccountManager {
1750
1762
  });
1751
1763
  }
1752
1764
 
1765
+ // Latch a session as large-context: a provider (the GLM/Kimi coding endpoint, fixed
1766
+ // ~256K) rejected a request with a context-length 400 that only a 1M Claude can hold.
1767
+ // Sticky + never-downgrades so every follow-up turn (context only grows) skips the
1768
+ // too-small providers instead of re-paying a wasted 400. Cleared only by a new session.
1769
+ markSessionLargeContext(sessionKey) {
1770
+ if (!sessionKey) return;
1771
+ const existing = this.sessionPolicies.get(sessionKey) || {};
1772
+ if (!existing.largeContext) {
1773
+ console.log(`[Maxpool] Session "${sessionKey}" exceeds provider context limits — pinned to Claude (GLM/Kimi benched for this session)`);
1774
+ }
1775
+ this.sessionPolicies.set(sessionKey, { ...existing, largeContext: true });
1776
+ }
1777
+
1778
+ _isSessionLargeContext(requestInfo = {}) {
1779
+ if (requestInfo.largeContext) return true;
1780
+ if (!requestInfo.sessionKey) return false;
1781
+ return Boolean(this.sessionPolicies.get(requestInfo.sessionKey)?.largeContext);
1782
+ }
1783
+
1753
1784
  // Marks a session as containing Anthropic signed thinking. This no longer bars
1754
1785
  // provider fallback (a lenient provider accepts an Anthropic signature) — it only
1755
1786
  // keeps the session's live cross-account MIGRATION on Claude (the rebalance guard),
@@ -2587,6 +2618,10 @@ export class AccountManager {
2587
2618
  account.modelMap = acctData.modelMap || account.modelMap;
2588
2619
  account.stripBetaHeaders = Boolean(acctData.stripBetaHeaders);
2589
2620
  account.runtime = true;
2621
+ // Restore path carries an explicit enabled (persisted disable); honor it. The `cc
2622
+ // all` header path (prepareRuntimeProviders) omits enabled, so a re-sent token
2623
+ // NEVER silently re-enables a provider the user benched in the TUI.
2624
+ if (acctData.enabled !== undefined) account.enabled = acctData.enabled !== false;
2590
2625
  if (account.status === 'error' && changed) {
2591
2626
  account.status = 'active';
2592
2627
  account.lastError = null;
@@ -2620,6 +2655,10 @@ export class AccountManager {
2620
2655
  model: a.model,
2621
2656
  modelMap: a.modelMap,
2622
2657
  stripBetaHeaders: a.stripBetaHeaders,
2658
+ // Persist the user's enable/disable so a provider they benched in the TUI stays
2659
+ // benched across a restart — without this an intentionally-disabled GLM/Kimi
2660
+ // silently comes back enabled on the next boot (restore defaults enabled:true).
2661
+ enabled: a.enabled,
2623
2662
  }));
2624
2663
  }
2625
2664
 
@@ -2797,6 +2836,7 @@ export class AccountManager {
2797
2836
  stickyBindings: this.sessionBindings.size,
2798
2837
  thinkingProtected: [...this.sessionPolicies.values()].filter(p => p.requiresAnthropicThinkingIntegrity).length,
2799
2838
  providerPinned: [...this.sessionPolicies.values()].filter(p => p.anthropicIncompatible).length,
2839
+ largeContextPinned: [...this.sessionPolicies.values()].filter(p => p.largeContext).length,
2800
2840
  },
2801
2841
  };
2802
2842
  }
package/src/index.js CHANGED
@@ -12,7 +12,7 @@ import { loginOAuth, fetchProfile, refreshAccessToken, isTokenExpiringSoon, toke
12
12
  import { TUI } from './tui.js';
13
13
  import { RestartController } from './restart-controller.js';
14
14
  import { resolveAccounts } from './account-config.js';
15
- import { maybeCheckForUpdate, getCurrentVersion } from './updater.js';
15
+ import { maybeCheckForUpdate, getCurrentVersion, markApplied } from './updater.js';
16
16
  import {
17
17
  runReloadBaton,
18
18
  RELOAD_SWAPPED, RELOAD_ROLLED_BACK,
@@ -1025,11 +1025,31 @@ async function serverWorkerCommand() {
1025
1025
  // would permanently disable all future update detection. It only refreshes
1026
1026
  // versionInfo (the persistent TUI banner is the reminder, so no repeated log spam);
1027
1027
  // autoUpdate progress lines still surface. unref so it never blocks a clean exit.
1028
+ const notifyUpdate = msg => (tui?._addLog ? tui._addLog(msg) : console.log(`[Maxpool] ${msg}`));
1029
+ // Fully-automatic apply (opt-in autoApply): mark the version attempted, THEN seamlessly
1030
+ // reload into the freshly-installed code (sessions survive). Marking here — at the
1031
+ // moment we act — is what makes the quarantine truthful: a boot-broken release is
1032
+ // attempted once, then blocked until an even newer version appears (no reload loop, no
1033
+ // stranded download). Shared by the startup one-shot AND the periodic timer.
1034
+ const applyUpdateIfReady = r => {
1035
+ if (r?.applicable && config?.autoApply && restartController) {
1036
+ markApplied(r.installedVersion);
1037
+ notifyUpdate('Applying update — seamless reload…');
1038
+ restartController.requestRestart();
1039
+ }
1040
+ };
1028
1041
  const updateIntervalMs = Math.max(60_000, Number(process.env.MAXPOOL_UPDATE_CHECK_INTERVAL_MS) || 6 * 60 * 60 * 1000);
1029
1042
  updateTimer = setInterval(() => {
1030
- if (config?.updateCheck === false) return;
1031
- const autoNotify = msg => { if (config?.autoUpdate) (tui?._addLog ? tui._addLog(msg) : console.log(`[Maxpool] ${msg}`)); };
1032
- maybeCheckForUpdate(config, autoNotify, info => { accountManager.versionInfo = info; }).catch(() => {});
1043
+ // ONLY the lease-holding primary self-installs — the timer is unconditional across
1044
+ // EVERY worker (incl. a headless reload worker), so gate on hasLease here to avoid
1045
+ // two `npm i -g` racing during a reload overlap (global-package corruption). A
1046
+ // headless reload worker never holds the lease, so it never installs/auto-applies.
1047
+ if (config?.updateCheck === false || !hasLease) return;
1048
+ // announce:false — the persistent TUI banner is the passive reminder; only real
1049
+ // actions (installing / applying) log here, so a pending update never churns the log.
1050
+ maybeCheckForUpdate(config, notifyUpdate, info => { accountManager.versionInfo = info; }, { announce: false })
1051
+ .then(applyUpdateIfReady)
1052
+ .catch(() => {});
1033
1053
  }, updateIntervalMs);
1034
1054
  updateTimer.unref();
1035
1055
 
@@ -1079,8 +1099,11 @@ async function serverWorkerCommand() {
1079
1099
  // A reload-spawned/takeover worker must NEVER re-probe (1x not 2x traffic).
1080
1100
  if (!viaTakeover && !isReloadWorker) {
1081
1101
  if (process.env.MAXPOOL_TEST_LOG_UPDATE_CHECK === '1') console.log('[Maxpool] UPDATE_CHECK_FIRED');
1082
- const notify = msg => (tui?._addLog ? tui._addLog(msg) : console.log(`[Maxpool] ${msg}`));
1083
- maybeCheckForUpdate(config, notify, info => { accountManager.versionInfo = info; }).catch(() => {});
1102
+ // Cold-start-behind: download AND (autoApply) self-apply via the SAME helper as the
1103
+ // periodic path — so the version is applied, not marked-attempted-then-stranded.
1104
+ maybeCheckForUpdate(config, notifyUpdate, info => { accountManager.versionInfo = info; })
1105
+ .then(applyUpdateIfReady)
1106
+ .catch(() => {});
1084
1107
  } else if (process.env.MAXPOOL_TEST_LOG_UPDATE_CHECK === '1') {
1085
1108
  console.log('[Maxpool] UPDATE_CHECK_SKIPPED (reload)');
1086
1109
  }
package/src/prober.js CHANGED
@@ -70,7 +70,11 @@ export class Prober {
70
70
  this._stopping = false;
71
71
  this._inflight = (async () => {
72
72
  try {
73
- // Skip auth-dead accounts (dead refresh token): probing them just re-POSTs
73
+ // DISABLED accounts are INTENTIONALLY still probed (no `a.enabled` filter):
74
+ // a user often disables an account precisely BECAUSE it's exhausted, and still
75
+ // wants to see its quota recover — so keep refreshing its usage for visibility
76
+ // even though routing skips it. Do NOT add an enabled gate here.
77
+ // Skip only auth-dead accounts (dead refresh token): probing them just re-POSTs
74
78
  // the rejected token every cycle — a 400 storm. They recover only on re-auth.
75
79
  const oauth = this.am.accounts.filter(a => a.type === 'oauth' && a.credential && !a.refreshDead);
76
80
  const providers = this.am.accounts.filter(a => a.type === 'provider' && a.credential);
@@ -67,7 +67,10 @@ export class RestartController {
67
67
  }
68
68
 
69
69
  requestRestart() {
70
- if (this.restarting) return;
70
+ // Idempotent against a restart ALREADY in progress — restarting OR pending (drain).
71
+ // Without the `pending` guard a second call (e.g. auto-apply firing while a manual
72
+ // restart is draining) would orphan the first _drainTimer and re-log.
73
+ if (this.restarting || this.pending) return;
71
74
  this.pauseAdmission();
72
75
  if (this.upstreamRequests.size === 0) {
73
76
  this._restart();
package/src/server.js CHANGED
@@ -53,6 +53,16 @@ const DEFAULT_QUEUE = {
53
53
  // resets idle-gap client timeouts; if a client uses a wall-clock total-request
54
54
  // deadline, lower this to just under it.
55
55
  streamHoldMaxMs: 7 * 24 * 60 * 60 * 1000,
56
+ // HARD ceiling on EVERY streaming hold (capacity/quota/throttle/concurrency alike).
57
+ // maxpool CANNOT keep a client alive past its own stream watchdog — Claude Code aborts
58
+ // "Stream idle timeout - no chunks received" after CLAUDE_STREAM_IDLE_TIMEOUT_MS of no
59
+ // real content EVENTS, and it drops SSE ping/comment keep-alives before the watchdog
60
+ // sees them (anthropic-sdk-typescript#998). So a 7-day/24h server-side hold only parks
61
+ // a request the client already abandoned. Bound it to the wait the user actually wants
62
+ // (front-loaded work waiting for a free account ≈ a few hours) so beyond that the
63
+ // request error-fasts with an honest retryable 429 instead of a silent multi-day park.
64
+ // Pair with a raised client watchdog (the cc launch sets CLAUDE_STREAM_IDLE_TIMEOUT_MS).
65
+ streamClientToleranceMs: Math.max(60_000, Number(process.env.MAXPOOL_STREAM_CLIENT_TOLERANCE_MS) || 3 * 60 * 60 * 1000),
56
66
  // Non-streaming requests have no SSE heartbeat to keep them alive, so a long
57
67
  // hold would die on the client timeout anyway. Cap their wait conservatively.
58
68
  nonStreamMaxWaitMs: 5 * 60 * 1000,
@@ -235,9 +245,10 @@ export function createProxyServer(accountManager, config, hooks = {}) {
235
245
  }
236
246
  requestInfo.profile = getMaxpoolProfile(req.headers);
237
247
  requestInfo.sessionKey = headerValue(req.headers, 'x-maxpool-session');
238
- if (requestInfo.requiresAnthropicThinkingIntegrity && requestInfo.profile === 'all') {
239
- console.log('[Maxpool] Anthropic thinking detected; provider fallback disabled for this session/request');
240
- }
248
+ // (Removed a FALSE "provider fallback disabled for signed thinking" log here: it
249
+ // fired on every thinking `all` request but was untrue under the default
250
+ // when-exhausted/always policies — providers DO serve thinking requests — and it
251
+ // repeatedly misdirected diagnosis of the "Stream idle timeout" reports.)
241
252
  prepareRuntimeProviders(accountManager, req.headers);
242
253
 
243
254
  await forwardRequest(
@@ -856,9 +867,15 @@ async function forwardRequest(
856
867
  // signature it can't validate). Detect it on an Anthropic account so we can
857
868
  // self-heal onto a provider instead of surfacing the 400.
858
869
  const anthropicIncompat = account.type !== 'provider' && isAnthropicIncompatBody(errorBody);
870
+ // A provider (GLM ~200K / Kimi 256K context) rejecting an oversized request that
871
+ // only a large-context Claude (1M) can hold — e.g. "exceeded model token limit:
872
+ // 262144". Detect it ONLY on a provider (a Claude account's context-length 400 is
873
+ // terminal — nothing bigger to fall to) so we can pin the session to Claude.
874
+ const providerTooSmall = account.type === 'provider' && isContextLengthError(errorBody);
859
875
  const errorType = errorBody.includes('Invalid `signature` in `thinking` block')
860
876
  ? 'invalid_thinking_signature'
861
877
  : anthropicIncompat ? 'anthropic_incompatible_transcript'
878
+ : providerTooSmall ? 'provider_context_too_small'
862
879
  : `HTTP ${upstreamRes.status}`;
863
880
  accountManager.releaseAccount(lease, { status: upstreamRes.status, error: errorType });
864
881
 
@@ -907,6 +924,33 @@ async function forwardRequest(
907
924
  );
908
925
  }
909
926
 
927
+ // React-and-heal: a PROVIDER (GLM/Kimi coding endpoint, fixed ~256K context)
928
+ // rejected an oversized request only a 1M-context Claude can hold — e.g. Kimi's
929
+ // "exceeded model token limit: 262144 (requested: 643557)". The coding leg IGNORES
930
+ // the model id, so a K3/GLM-1M plan doesn't lift its ceiling. Latch the session
931
+ // large-context (so its follow-up turns skip the too-small providers) and retry
932
+ // EXCLUDING every provider → routes to Claude, or HOLDS for a Claude account,
933
+ // instead of surfacing a 400 the client just retry-loops on. Only worth it when a
934
+ // Claude account exists to serve it; with none, the 400 surfaces (nothing bigger).
935
+ const claudeAvailable = accountManager.accounts?.some(a => a.type !== 'provider' && a.enabled !== false);
936
+ if (providerTooSmall && requestInfo.sessionKey && claudeAvailable
937
+ && canRetryBufferedBody && retryCount + 1 < maxAttempts && !res.headersSent) {
938
+ accountManager.markSessionLargeContext?.(requestInfo.sessionKey);
939
+ // Bench EVERY provider: today's GLM + Kimi coding legs both cap at ~256K, so once
940
+ // one 400s on size the others can't hold it either. If a genuine 1M-context
941
+ // provider is ever added, make this exclusion context-limit-aware instead of
942
+ // type-wide (the _isRequestCompatible gate would need the same treatment).
943
+ for (const a of (accountManager.accounts || [])) {
944
+ if (a.type === 'provider') excludedIndexes.add(a.index);
945
+ }
946
+ console.log(`[Maxpool] Provider "${account.name}" context too small for this request; pinning session to Claude and retrying`);
947
+ return forwardRequest(
948
+ req, res, body, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
949
+ retryConfig, queueConfig, { ...requestInfo, largeContext: true },
950
+ canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
951
+ );
952
+ }
953
+
910
954
  ctx.status = upstreamRes.status;
911
955
  sendErrorBody(res, requestInfo, upstreamRes.status, errorBody, upstreamRes.headers);
912
956
  return;
@@ -1135,7 +1179,7 @@ function formatRetryDuration(seconds) {
1135
1179
  */
1136
1180
  function computeQueueWindowMs({
1137
1181
  cause, stream, retryPlanCause,
1138
- maxWaitMs, capacityMaxWaitMs, nonStreamMaxWaitMs, streamHoldMaxMs,
1182
+ maxWaitMs, capacityMaxWaitMs, nonStreamMaxWaitMs, streamHoldMaxMs, streamClientToleranceMs,
1139
1183
  isCountTokens, countTokensMaxWaitMs,
1140
1184
  }) {
1141
1185
  let windowMs;
@@ -1147,6 +1191,12 @@ function computeQueueWindowMs({
1147
1191
  windowMs = streamHoldMaxMs;
1148
1192
  }
1149
1193
  if (retryPlanCause === 'concurrency_cap') windowMs = Math.min(windowMs, capacityMaxWaitMs);
1194
+ // Bound EVERY streaming cause to the client-tolerance ceiling — the client's own
1195
+ // watchdog kills the stream well before a 24h/7d server hold, so anything past this
1196
+ // just parks an abandoned request. A finite reset WITHIN the ceiling still holds +
1197
+ // resumes (via the nextRetryForRequest oracle); a reset beyond it error-fasts (a real
1198
+ // retryable 429 at the pre-heartbeat gate) instead of hanging.
1199
+ if (stream && streamClientToleranceMs != null) windowMs = Math.min(windowMs, streamClientToleranceMs);
1150
1200
  // count_tokens: cap the QUEUE wait low (bounds only the wait-for-an-account, never
1151
1201
  // the upstream processing once acquired) so a non-heartbeated metadata call fast-
1152
1202
  // fails with a retryable 429 instead of hanging past the client's idle window.
@@ -1166,9 +1216,22 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
1166
1216
  return `This session's transcript can only run on ${fam} — Claude rejects its server-tool ids/thinking on replay. No ${fam} provider is available right now.${eta} Check the x-maxpool-zai-token / x-maxpool-kimi-token headers, or resume with 'cc ${incompat.homeProvider === 'kimi' ? 'kimi' : 'glm'}'.`;
1167
1217
  }
1168
1218
 
1169
- const thinking = requestInfo.requiresAnthropicThinkingIntegrity
1170
- || accountManager._requiresAnthropicThinkingIntegrity?.(requestInfo);
1171
- const n = accountManager.accounts.length;
1219
+ // A large-context session (a provider already 400'd it as too big for its ~256K leg):
1220
+ // only a 1M-context Claude can hold it — the providers are structurally BARRED, not
1221
+ // merely "at their limit". Say the oversized truth + the two real ways out (wait for a
1222
+ // Claude account, or /compact), instead of the misleading "providers at limit" line.
1223
+ if (accountManager._isSessionLargeContext?.(requestInfo)) {
1224
+ const eta = Number.isFinite(retryAfter) && retryAfter > 0
1225
+ ? ` A Claude account should free in ~${formatRetryDuration(retryAfter)}.` : '';
1226
+ return `This session is too large for the GLM/Kimi fallbacks (their ~256K limit) — it needs a 1M-context Claude account, and they're all busy right now.${eta} It sends as soon as one frees; /compact shortens the session if you'd rather not wait.`;
1227
+ }
1228
+
1229
+ const claudeCount = accountManager.accounts.filter(a => a.type !== 'provider').length;
1230
+ // Only name the providers when this pool actually HAS them (`cc all`). On `cc ma`
1231
+ // (Claude-only) there are none, so the old hardcoded "and the GLM/Kimi providers"
1232
+ // was a lie. When present they DO serve (not barred), so they're saturated too.
1233
+ const providersClause = accountManager.accounts.some(a => a.type === 'provider')
1234
+ ? ' and the GLM/Kimi providers' : '';
1172
1235
 
1173
1236
  // No route is expected to recover within the queue window — i.e. every Claude
1174
1237
  // account is at its own 5h/weekly limit. A short "retry in Ns" would be a lie;
@@ -1177,19 +1240,24 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
1177
1240
  const eta = Number.isFinite(retryAfter) && retryAfter > 0
1178
1241
  ? ` Soonest reset in ~${formatRetryDuration(retryAfter)}, beyond the hold window.`
1179
1242
  : '';
1180
- const base = `No Claude account can take this request — all ${n} are at their 5h or weekly limit.${eta} Add another Claude account or wait for a quota reset.`;
1181
- return thinking
1182
- ? `${base} GLM/Kimi fallback is unavailable because this session contains Anthropic signed thinking blocks; start a fresh non-thinking session to use them.`
1183
- : base;
1243
+ return `No account can take this request — all ${claudeCount} Claude accounts${providersClause} are at their limit.${eta} Add another Claude account or wait for a quota reset.`;
1184
1244
  }
1185
1245
 
1186
- if (thinking) {
1187
- return `No Claude account could accept this request. Non-Claude fallback is disabled because this session contains Anthropic signed thinking blocks. Retry in ${retryAfter}s, wait for Claude capacity, or start a fresh non-thinking session to use GLM/Kimi.`;
1188
- }
1189
- return `All ${n} accounts exhausted. Retry in ${retryAfter}s.`;
1246
+ return `No account can take this request right now — all ${claudeCount} Claude accounts${providersClause} are momentarily at their limit. Retry in ${retryAfter}s.`;
1247
+ }
1248
+
1249
+ // A provider (GLM/Kimi) rejecting a request whose token count exceeds its context
1250
+ // window — Kimi's coding leg ("exceeded model token limit: 262144"), GLM's, or an
1251
+ // OpenAI-style "maximum context length". Kept narrow (specific context-overflow
1252
+ // phrasings, NOT a bare "token limit" which a rate-limit body also carries) so the
1253
+ // pin-to-Claude heal only fires on a genuine size overflow, not any 400. Rate-limit
1254
+ // 429s are intercepted earlier (classifyRateLimit) and never reach this check.
1255
+ function isContextLengthError(errorBody) {
1256
+ if (!errorBody) return false;
1257
+ return /exceeded model token limit|maximum context length|context length exceeded|context window (?:size )?(?:exceeded|too)|prompt is too long|input is too long|reduce the length of|too many (?:input )?tokens|request too large/i.test(errorBody);
1190
1258
  }
1191
1259
 
1192
- export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, streamResponse, startIdleRequestReaper };
1260
+ export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
1193
1261
 
1194
1262
  async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
1195
1263
  if (!upstreamRes.body) return '';
@@ -1467,6 +1535,9 @@ async function queueAndRetry(
1467
1535
  const streamHoldMaxMs = queueConfig.streamHoldMaxMs == null
1468
1536
  ? 7 * 24 * 60 * 60 * 1000
1469
1537
  : Math.max(0, Number(queueConfig.streamHoldMaxMs) || 0);
1538
+ const streamClientToleranceMs = queueConfig.streamClientToleranceMs == null
1539
+ ? 3 * 60 * 60 * 1000
1540
+ : Math.max(0, Number(queueConfig.streamClientToleranceMs) || 0);
1470
1541
  const queueWindowMs = computeQueueWindowMs({
1471
1542
  cause,
1472
1543
  stream: Boolean(requestInfo.stream),
@@ -1475,6 +1546,7 @@ async function queueAndRetry(
1475
1546
  capacityMaxWaitMs,
1476
1547
  nonStreamMaxWaitMs,
1477
1548
  streamHoldMaxMs,
1549
+ streamClientToleranceMs,
1478
1550
  isCountTokens: Boolean(requestInfo.isCountTokens),
1479
1551
  countTokensMaxWaitMs,
1480
1552
  });
package/src/tui.js CHANGED
@@ -586,6 +586,9 @@ export class TUI {
586
586
  .map(index => ({ account: this.am.accounts[index], index }))
587
587
  .filter(({ account }) => {
588
588
  if (action === 'prefer') return account.type !== 'provider' && account.enabled;
589
+ // Enable/disable also works on runtime providers (GLM/Kimi) — a session-only
590
+ // toggle, since they're not in config. Rename/delete stay config-account-only.
591
+ if (action === 'toggle') return this._configAccountIndex(account) >= 0 || account.type === 'provider';
589
592
  return this._configAccountIndex(account) >= 0;
590
593
  })
591
594
  .map(({ index }) => index);
@@ -872,7 +875,18 @@ export class TUI {
872
875
  if (!account) return;
873
876
  const configIndex = this._configAccountIndex(account);
874
877
  if (configIndex < 0) {
875
- this._addLog(`Cannot ${enabled ? 'enable' : 'disable'} runtime provider "${account.name}" here`);
878
+ // A runtime provider (GLM/Kimi) isn't in config — it's re-created from the `cc all`
879
+ // request headers — but its enable/disable IS durable: the flag persists in memory
880
+ // across `cc all` requests (the header upsert never re-enables it) and to state.json
881
+ // on the next save, so it stays benched across a restart too. Re-enable it here the
882
+ // same way whenever the user wants it back — there's no "removed forever" state.
883
+ if (account.type === 'provider') {
884
+ this.am.setAccountEnabled(idx, enabled);
885
+ if (!enabled && this.am.preferredAccountName === account.name) this.am.setRoutingMode?.('automatic');
886
+ this._addLog(`${enabled ? 'Enabled' : 'Disabled'} provider "${account.name}" — ${enabled ? 'routing resumed' : 'benched (stays off across cc all + restart; re-enable here anytime)'}`);
887
+ return;
888
+ }
889
+ this._addLog(`Cannot ${enabled ? 'enable' : 'disable'} "${account.name}" here (not in config)`);
876
890
  return;
877
891
  }
878
892
  const previous = this.config.accounts[configIndex].enabled;
package/src/updater.js CHANGED
@@ -68,43 +68,99 @@ export async function selfUpdate({ timeoutMs = 120_000 } = {}) {
68
68
  }
69
69
  }
70
70
 
71
+ // The version of the CODE currently EXECUTING, captured ONCE before any self-install.
72
+ // getCurrentVersion() reads package.json FROM DISK, which `npm i -g` rewrites — so
73
+ // after a background self-install the disk version != the running version. Every
74
+ // "am I behind?" and loop-guard decision keys on THIS fixed value, not the disk read.
75
+ let _bootVersion;
76
+ // The newest version this process has already ATTEMPTED to auto-apply. The loop guard:
77
+ // a version that installs fine but fails to BOOT rolls back to the old worker, which is
78
+ // still running _bootVersion with its timer armed — without this it would re-detect the
79
+ // on-disk version every check and re-apply forever (a boot-broken release → infinite
80
+ // reload loop, ~30s of refused admission each cycle). We apply only versions strictly
81
+ // newer than max(_bootVersion, _lastAttemptedTarget), so a rolled-back target is
82
+ // quarantined until an even newer release appears.
83
+ let _lastAttemptedTarget = null;
84
+
85
+ /** Test-only: reset the module's version-tracking state between cases. */
86
+ export function __resetUpdaterState() { _bootVersion = undefined; _lastAttemptedTarget = null; }
87
+
88
+ /** Mark a version as ATTEMPTED-to-apply. The caller calls this at the moment it triggers
89
+ * the reload — BEFORE the reload — so a rolled-back target is quarantined (advance-only).
90
+ * Kept as the caller's action (not a side effect of maybeCheckForUpdate) so a caller that
91
+ * ignores `applicable` can never strand a version or poison the quarantine floor. */
92
+ export function markApplied(version) {
93
+ if (version && (!_lastAttemptedTarget || compareVersions(version, _lastAttemptedTarget) > 0)) {
94
+ _lastAttemptedTarget = version;
95
+ }
96
+ }
97
+
71
98
  /**
72
- * Startup hook: check for an update and either notify (default) or self-install
73
- * (config.autoUpdate). Never auto-restarts a running proxy — the new version
74
- * applies on the next restart, so in-flight sessions are never interrupted.
75
- * Fire-and-forget; all failures are swallowed.
99
+ * Check for an update; with `config.autoUpdate` also self-install; with
100
+ * `config.autoApply` SIGNAL that the caller should seamlessly reload to APPLY it
101
+ * (returns `applicable:true` — the CALLER then markApplied()+reloads). Failures swallowed.
102
+ * Deps are injectable for tests.
103
+ *
104
+ * `onVersionInfo` is ALWAYS invoked with { current, latest, hasUpdate, checkedAt } —
105
+ * current + hasUpdate keyed on the RUNNING version (so the banner stays accurate after a
106
+ * background download, until the reload). `deps.announce===false` suppresses the passive
107
+ * "Update available" line (the periodic path — the persistent banner already shows it).
76
108
  *
77
- * `onVersionInfo` (optional) is ALWAYS invoked with { current, latest, hasUpdate,
78
- * checkedAt } — even when up-to-date, offline, or updateCheck is off — so the TUI
79
- * header / status can show the running version + whether an update is available.
109
+ * Returns { hasUpdate, applicable, installedVersion? }.
80
110
  */
81
- export async function maybeCheckForUpdate(config, notify, onVersionInfo) {
82
- const current = await getCurrentVersion();
83
- // Skip the npm round-trip when the user disabled update checks, but still report
84
- // the running version so the indicator can show it.
85
- const result = config?.updateCheck === false ? null : await checkForUpdate(current);
111
+ export async function maybeCheckForUpdate(config, notify, onVersionInfo, deps = {}) {
112
+ const _get = deps.getCurrentVersion || getCurrentVersion;
113
+ const _check = deps.checkForUpdate || checkForUpdate;
114
+ const _self = deps.selfUpdate || selfUpdate;
115
+ const announce = deps.announce !== false;
116
+
117
+ if (_bootVersion === undefined) _bootVersion = await _get();
118
+ const current = _bootVersion;
119
+ const result = config?.updateCheck === false ? null : await _check(current);
120
+ const hasUpdate = Boolean(result?.hasUpdate);
86
121
 
87
122
  if (onVersionInfo) {
88
123
  try {
89
- onVersionInfo({
90
- current,
91
- latest: result?.latest ?? null,
92
- hasUpdate: Boolean(result?.hasUpdate),
93
- checkedAt: Date.now(),
94
- });
95
- } catch { /* the indicator is best-effort; never break startup */ }
124
+ onVersionInfo({ current, latest: result?.latest ?? null, hasUpdate, checkedAt: Date.now() });
125
+ } catch { /* indicator is best-effort; never break startup */ }
96
126
  }
97
127
 
98
- if (!result || !result.hasUpdate) return;
128
+ if (!hasUpdate) return { hasUpdate: false, applicable: false };
129
+ if (announce) notify(`Update available: ${result.current} → ${result.latest}`);
99
130
 
100
- notify(`Update available: ${result.current} → ${result.latest}`);
101
- if (config?.autoUpdate) {
131
+ if (!config?.autoUpdate) {
132
+ if (announce) notify(`Run 'npm i -g ${PACKAGE}' to update, or set "autoUpdate": true in your config.`);
133
+ return { hasUpdate: true, applicable: false };
134
+ }
135
+
136
+ // Skip the reinstall if the latest is ALREADY on disk (a prior check downloaded it) —
137
+ // otherwise every periodic check re-runs `npm i -g` + re-logs until a manual restart.
138
+ const onDisk = await _get();
139
+ if (!onDisk || compareVersions(onDisk, result.latest) < 0) {
102
140
  notify(`Auto-updating to ${result.latest}…`);
103
- const r = await selfUpdate();
104
- notify(r.ok
105
- ? `Updated to ${result.latest}. Restart maxpool to apply (running sessions are not interrupted).`
106
- : `Auto-update failed: ${r.error}. Run: npm i -g ${PACKAGE}`);
107
- } else {
108
- notify(`Run 'npm i -g ${PACKAGE}' to update, or set "autoUpdate": true in your config.`);
141
+ const r = await _self();
142
+ if (!r.ok) {
143
+ notify(`Auto-update failed: ${r.error}. Run: npm i -g ${PACKAGE}`);
144
+ return { hasUpdate: true, applicable: false };
145
+ }
146
+ }
147
+
148
+ const installed = await _get();
149
+ // Loop/quarantine guard — a version is "applicable" only if it is strictly newer than
150
+ // the newest we've already booted-as OR already ATTEMPTED to apply (markApplied). This
151
+ // function has NO side effect on that floor: the CALLER marks it when it triggers the
152
+ // reload, so a caller that ignores `applicable` can't strand a version or poison state.
153
+ const floor = (_lastAttemptedTarget && compareVersions(_lastAttemptedTarget, current) > 0)
154
+ ? _lastAttemptedTarget : current;
155
+ const advanced = Boolean(installed) && compareVersions(installed, current) > 0;
156
+ const applicable = advanced && compareVersions(installed, floor) > 0;
157
+
158
+ // Passive notices only. The "Applying now…" line + the reload are the caller's (it
159
+ // logs them exactly when it acts), so nothing here can claim an apply that didn't happen.
160
+ if (advanced && !applicable) {
161
+ if (announce) notify(`Update ${installed} already attempted — staying on ${current}; will retry only a newer release.`);
162
+ } else if (advanced && !config?.autoApply) {
163
+ if (announce) notify(`Updated to ${installed}. Restart maxpool to apply (sessions are not interrupted).`);
109
164
  }
165
+ return { hasUpdate: true, installedVersion: installed, applicable };
110
166
  }