maxpool 1.5.56 → 1.5.57

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.56",
3
+ "version": "1.5.57",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -1785,13 +1785,14 @@ export class AccountManager {
1785
1785
  // are all unavailable the request HOLDS/queues (recoverable) rather than 400ing.
1786
1786
  if (requestInfo.hasImage && account.provider === 'kimi') return false;
1787
1787
 
1788
- // A large-context session: a provider already rejected this request with a
1789
- // context-length 400. The GLM/Kimi *coding* endpoints serve a fixed model capped
1790
- // at ~256K and IGNORE the model id (so a K3/GLM 1M PLAN doesn't lift the coding
1791
- // leg's ceiling). Only a 1M-context Claude account can hold it — bench the
1792
- // providers for this session so it routes to Claude, or HOLDS for one, instead of
1793
- // re-404ing on a too-small leg. Sticky per session (context only grows turn over
1794
- // turn), so no follow-up turn re-pays the wasted attempt.
1788
+ // A large-context session: a provider already rejected THIS request with a
1789
+ // context-length 400. Deliberately REACTIVE — it never assumes a ceiling, it learns
1790
+ // one from an actual rejection, so it self-corrects as providers grow. That matters:
1791
+ // the old ~256K coding-endpoint cap is gone (verified 2026-08-02 — GLM 5.2 and Kimi
1792
+ // K3 both accepted a ~400K-token payload and both honoured the requested model id),
1793
+ // so this branch simply stops firing rather than needing a new constant.
1794
+ // Sticky per session (context only grows turn over turn) so no follow-up turn re-pays
1795
+ // the wasted attempt.
1795
1796
  if (account.type === 'provider' && this._isSessionLargeContext(requestInfo)) return false;
1796
1797
 
1797
1798
  const { incompatible, homeProvider } = this._effectiveIncompatible(requestInfo);
package/src/server.js CHANGED
@@ -2370,7 +2370,10 @@ function prepareRuntimeProviders(accountManager, headers) {
2370
2370
 
2371
2371
  const kimiToken = headerValue(headers, 'x-maxpool-kimi-token');
2372
2372
  if (kimiToken) {
2373
- const model = headerValue(headers, 'x-maxpool-kimi-model') || 'kimi-k2.7';
2373
+ // Fallback only — `cc all` always sends x-maxpool-kimi-model from the llm_config SSOT,
2374
+ // so this is what a bare/older client gets. Kept current deliberately: it read
2375
+ // 'kimi-k2.7' while the fleet had moved to k3.
2376
+ const model = headerValue(headers, 'x-maxpool-kimi-model') || 'kimi-k3';
2374
2377
  accountManager.upsertRuntimeAccount({
2375
2378
  name: 'kimi-fallback',
2376
2379
  type: 'provider',
package/src/tui.js CHANGED
@@ -1152,16 +1152,31 @@ export class TUI {
1152
1152
  }
1153
1153
  lines.push(` Routing ${cyan(routing)}${xpText}`);
1154
1154
  const queuedCount = this.am.queueState?.waiting?.length || 0;
1155
- if (this.am._isUpstreamThrottleBlocking?.() || queuedCount) {
1155
+ // Throttle and WAITING are different things and used to share one line labelled
1156
+ // "Anthropic upstream throttled" — so a request parked purely because every account
1157
+ // is at its quota was either invisible (no throttle) or described as a throttle it
1158
+ // wasn't. Waiting is the headline state of this proxy; it gets its own line, always
1159
+ // shown whenever anything is parked, naming what it is waiting FOR.
1160
+ if (this.am._isUpstreamThrottleBlocking?.()) {
1156
1161
  const throttle = this.am.upstreamThrottle;
1157
1162
  const remaining = throttle.until ? Math.max(0, Math.ceil((throttle.until - Date.now()) / 1000)) : 0;
1158
- const state = this.am._isUpstreamThrottleBlocking?.()
1159
- ? throttle.probeInFlight ? 'probing recovery' : `retry in ${remaining}s`
1160
- : 'recovering';
1161
- const queued = queuedCount;
1162
- const oldest = queued ? Math.max(0, Date.now() - this.am.queueState.waiting[0].queuedAt) : 0;
1163
- const queueText = queued ? ` queued ${queued} oldest ${formatMs(oldest)}` : '';
1164
- lines.push(` ${yellow(' Anthropic upstream throttled')} ${dim(state + queueText)}`);
1163
+ const state = throttle.probeInFlight ? 'probing recovery' : `retry in ${remaining}s`;
1164
+ lines.push(` ${yellow(' Anthropic upstream throttled')} ${dim(state)}`);
1165
+ }
1166
+ if (queuedCount) {
1167
+ const oldest = Math.max(0, Date.now() - this.am.queueState.waiting[0].queuedAt);
1168
+ // Name the soonest thing that would release them, so a long wait reads as
1169
+ // "waiting for a known reset" rather than "hung".
1170
+ let why = 'waiting for capacity';
1171
+ try {
1172
+ const plan = this.am.nextRetryForRequest?.({}, new Set()) || {};
1173
+ if (Number.isFinite(plan.retryAfterMs) && plan.retryAfterMs > 0) {
1174
+ why = `next account frees in ~${formatMs(plan.retryAfterMs)}`;
1175
+ } else if (plan.cause) {
1176
+ why = `waiting (${plan.cause.replace(/_/g, ' ')})`;
1177
+ }
1178
+ } catch { /* display-only; never let the oracle break the render */ }
1179
+ lines.push(` ${cyan(' Parked')} ${dim(`${queuedCount} request${queuedCount === 1 ? '' : 's'} held oldest ${formatMs(oldest)} · ${why}`)}`);
1165
1180
  }
1166
1181
 
1167
1182
  // ── Accounts