maxpool 1.5.22 → 1.5.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.22",
3
+ "version": "1.5.23",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -37,6 +37,6 @@
37
37
  "url": "https://github.com/2solarmax/maxpool/issues"
38
38
  },
39
39
  "engines": {
40
- "node": ">=18.0.0"
40
+ "node": ">=20.3.0"
41
41
  }
42
42
  }
package/src/oauth.js CHANGED
@@ -268,14 +268,92 @@ export function classifyZaiLimit(l, now = Date.now()) {
268
268
  return { bucket, utilization, resetAt };
269
269
  }
270
270
 
271
- /** Read a provider account's quota. z.ai has a pollable monitor endpoint mapped
272
- * to Ses/Wk token windows; Kimi (Moonshot coding key) has NO pollable quota
273
- * (web console only), so it returns a `console-only` marker instead of fake
274
- * bars. Returns { ses, wk, level } | { error, status?, source? }. */
271
+ // Kimi Coding Plan usage — the coding key (sk-kimi-…) reads its own quota at
272
+ // <base>/v1/usages (the community `kimi-code-usage` tool's endpoint). NOTE the
273
+ // User-Agent: the endpoint is served by the Kimi Code platform, not the Open
274
+ // Platform (whose /users/me/balance 401s a coding key). Best-effort, non-load-
275
+ // bearing UA — the failure path degrades to last-known + the staleness marker.
276
+ const KIMI_USAGE_UA = 'KimiCLI/1.6';
277
+
278
+ /** ms of one Kimi limit `window`, normalized from its timeUnit so the SHORTEST
279
+ * window (the 5h session) is picked regardless of unit. null when unknown. */
280
+ export function kimiWindowMs(window) {
281
+ const dur = Number(window?.duration);
282
+ if (!Number.isFinite(dur) || dur <= 0) return null;
283
+ const unit = String(window?.timeUnit || window?.time_unit || '').toUpperCase();
284
+ const mult = unit.includes('SECOND') ? 1e3
285
+ : unit.includes('MINUTE') ? 60e3
286
+ : unit.includes('HOUR') ? 3600e3
287
+ : unit.includes('DAY') ? 86400e3
288
+ : unit.includes('WEEK') ? 7 * 86400e3
289
+ : unit.includes('MONTH') ? 30 * 86400e3
290
+ : null;
291
+ return mult == null ? null : dur * mult;
292
+ }
293
+
294
+ /** Normalize a Kimi usage row {limit,used,remaining,resetTime} (values are STRINGS)
295
+ * to { utilization, resetAt }. CRITICAL: utilization is `null` — never a computed
296
+ * NaN — when the inputs are absent or limit<=0, because clamp01(NaN)=0 downstream
297
+ * would render a blind account as "0% used = fully available" and route real
298
+ * traffic onto quota we can't see. */
299
+ export function parseKimiRow(d) {
300
+ if (!d || typeof d !== 'object') return null;
301
+ const limit = Number(d.limit ?? d.limit_amount);
302
+ let used = d.used != null ? Number(d.used ?? d.used_amount) : null;
303
+ const remaining = d.remaining != null ? Number(d.remaining) : null;
304
+ if (used == null && remaining != null && Number.isFinite(limit)) used = limit - remaining;
305
+ const utilization = (Number.isFinite(limit) && limit > 0 && used != null && Number.isFinite(used))
306
+ ? Math.max(0, Math.min(1, used / limit))
307
+ : null;
308
+ const resetRaw = d.resetTime || d.reset_at || d.reset_time;
309
+ const parsed = resetRaw ? Date.parse(resetRaw) : NaN;
310
+ return { utilization, resetAt: Number.isFinite(parsed) ? parsed : null };
311
+ }
312
+
313
+ /** Read the Kimi Coding Plan quota → { ses, wk, source:'kimi' } | { error, status? }.
314
+ * Weekly = the top-level `usage` rolling window; Ses (5h) = the shortest `limits[]`
315
+ * window. A transient/partial response returns {error} or null buckets so
316
+ * applyProviderUsage keeps the last-known values (never a phantom 0%). */
317
+ export async function fetchKimiUsage(account) {
318
+ const token = account?.credential;
319
+ if (!token) return { error: 'unsupported', source: null };
320
+ const base = String(account.upstream || 'https://api.kimi.com/coding').replace(/\/+$/, '');
321
+ const headers = { 'Authorization': `Bearer ${token}`, 'User-Agent': KIMI_USAGE_UA, 'Accept': 'application/json' };
322
+ try {
323
+ let res = await fetch(`${base}/v1/usages`, { headers, signal: AbortSignal.timeout(10_000) });
324
+ if (res.status === 404) res = await fetch(`${base}/v1/usage`, { headers, signal: AbortSignal.timeout(10_000) });
325
+ if (!res.ok) return { error: `HTTP ${res.status}`, status: res.status };
326
+ const data = await res.json();
327
+ // Weekly = top-level `usage` (ALWAYS present in a valid coding-plan response).
328
+ // If it's missing, the response is partial/malformed → return an error so
329
+ // applyProviderUsage keeps the last-known weekly instead of the falsy-wk branch
330
+ // BLANKING it (unlike z.ai, where a missing weekly genuinely means "no weekly").
331
+ const wkRow = parseKimiRow(data?.usage);
332
+ if (!wkRow || wkRow.utilization == null) return { error: 'partial_response', source: 'kimi' };
333
+ const wk = wkRow;
334
+ // Ses = the shortest limits[] window (the 5h). Fall back to first if durations unknown.
335
+ let ses = null, sesMs = Infinity;
336
+ for (const l of (Array.isArray(data?.limits) ? data.limits : [])) {
337
+ const detail = l && typeof l.detail === 'object' ? l.detail : l;
338
+ const row = parseKimiRow(detail);
339
+ if (!row || row.utilization == null) continue;
340
+ const ms = kimiWindowMs(l?.window);
341
+ if (ms != null && ms < sesMs) { sesMs = ms; ses = row; }
342
+ else if (ses == null && ms == null) ses = row;
343
+ }
344
+ return { ses, wk, source: 'kimi', level: data?.user?.membership?.level || null };
345
+ } catch (err) {
346
+ return { error: err.message || String(err), status: null };
347
+ }
348
+ }
349
+
350
+ /** Read a provider account's quota. z.ai and Kimi both have a pollable coding-plan
351
+ * usage endpoint mapped to Ses/Wk windows. Returns { ses, wk, source } |
352
+ * { error, status?, source? }. */
275
353
  export async function fetchProviderUsage(account) {
276
354
  const provider = account?.provider;
277
355
  const token = account?.credential;
278
- if (provider === 'kimi') return { error: 'unsupported', source: 'console-only' };
356
+ if (provider === 'kimi') return fetchKimiUsage(account);
279
357
  if (provider !== 'zai' || !token) return { error: 'unsupported', source: null };
280
358
  try {
281
359
  const res = await fetch(ZAI_QUOTA_URL, {
package/src/server.js CHANGED
@@ -15,6 +15,16 @@ const DEFAULT_RETRY = {
15
15
  maxRetryBufferBytes: 10 * 1024 * 1024,
16
16
  };
17
17
 
18
+ // Upstream-hang guards. A half-open upstream (headers then silence, never closes)
19
+ // would block the request forever and LEAK its in-flight lease — the account climbs
20
+ // to safetyMaxActivePerAccount and is falsely "full", breaking routing (the "106
21
+ // phantom active" incident). Env-overridable with conservative defaults: this path
22
+ // serves EVERY provider (Anthropic/GLM/Kimi), so the idle gap is generous enough
23
+ // that a legitimately-slow-but-alive stream is never cut (each chunk resets it).
24
+ const UPSTREAM_TTFB_MS = Math.max(5_000, Number(process.env.MAXPOOL_TTFB_MS) || 120_000); // headers must arrive within this
25
+ const STREAM_IDLE_MS = Math.max(30_000, Number(process.env.MAXPOOL_STREAM_IDLE_MS) || 300_000); // max gap BETWEEN streamed chunks (reset per chunk)
26
+ const UPSTREAM_BODY_MS = Math.max(30_000, Number(process.env.MAXPOOL_BODY_MS) || 300_000); // non-streaming body read
27
+
18
28
  const DEFAULT_QUEUE = {
19
29
  enabled: true,
20
30
  maxWaitMs: 24 * 60 * 60 * 1000,
@@ -429,14 +439,31 @@ async function forwardRequest(
429
439
  }
430
440
  }
431
441
 
442
+ // TTFB guard: an upstream that accepts the connection but never sends response
443
+ // headers would hang `await fetch` forever (signal only covers client-disconnect)
444
+ // and leak the lease. Abort via a DEDICATED controller (so the caller can tell a
445
+ // TTFB timeout apart from a client-gone abort) and CLEAR it the moment headers
446
+ // arrive — the streaming body is then governed by the per-chunk idle guard, never
447
+ // this fixed clock, so a long healthy stream is never cut.
448
+ const ttfbController = new AbortController();
449
+ const ttfbTimer = setTimeout(
450
+ () => ttfbController.abort(Object.assign(new Error('upstream TTFB timeout'), { code: 'UPSTREAM_TTFB' })),
451
+ UPSTREAM_TTFB_MS,
452
+ );
453
+ ttfbTimer.unref?.();
432
454
  try {
433
- const upstreamRes = await fetch(upstreamUrl, {
434
- method,
435
- headers,
436
- body: ['GET', 'HEAD'].includes(method) ? undefined : upstreamBody,
437
- redirect: 'manual',
438
- signal: clientGone.signal,
439
- });
455
+ let upstreamRes;
456
+ try {
457
+ upstreamRes = await fetch(upstreamUrl, {
458
+ method,
459
+ headers,
460
+ body: ['GET', 'HEAD'].includes(method) ? undefined : upstreamBody,
461
+ redirect: 'manual',
462
+ signal: AbortSignal.any([clientGone.signal, ttfbController.signal]),
463
+ });
464
+ } finally {
465
+ clearTimeout(ttfbTimer);
466
+ }
440
467
  // Response arrived — the pre-response leak window is over. Stop guarding for
441
468
  // client-disconnect via abort (streamResponse handles mid-stream disconnects).
442
469
  res.off('close', onClientClose);
@@ -840,7 +867,26 @@ async function forwardRequest(
840
867
  writeRequestLog(logDir, reqId, logSections);
841
868
  }
842
869
  } else {
843
- const buf = Buffer.from(await upstreamRes.arrayBuffer());
870
+ // Bound the non-streaming body read so a mid-body upstream stall can't hang
871
+ // the request forever and leak the lease (same class as the streaming idle
872
+ // guard). On timeout → UPSTREAM_BODY → the caller frees the lease.
873
+ const bodyP = upstreamRes.arrayBuffer();
874
+ bodyP.catch(() => {});
875
+ let bodyTimer;
876
+ const bodyTimeout = new Promise((_, reject) => {
877
+ bodyTimer = setTimeout(
878
+ () => reject(Object.assign(new Error('upstream body timeout'), { code: 'UPSTREAM_BODY' })),
879
+ UPSTREAM_BODY_MS,
880
+ );
881
+ bodyTimer.unref?.();
882
+ });
883
+ let arr;
884
+ try {
885
+ arr = await Promise.race([bodyP, bodyTimeout]);
886
+ } finally {
887
+ clearTimeout(bodyTimer);
888
+ }
889
+ const buf = Buffer.from(arr);
844
890
  extractUsageFromBody(buf, account.index, accountManager);
845
891
  markThinkingFromResponse(buf, accountManager, requestInfo);
846
892
  accountManager.releaseAccount(lease, { success: upstreamRes.status < 500, status: upstreamRes.status });
@@ -896,6 +942,11 @@ async function forwardRequest(
896
942
  (err.message.includes('fetch failed') ||
897
943
  err.code === 'ECONNRESET' || err.code === 'ECONNREFUSED' ||
898
944
  err.code === 'ETIMEDOUT' || err.code === 'UND_ERR_CONNECT_TIMEOUT' ||
945
+ // Upstream-hang guards: a stalled connection is network-class, not the
946
+ // account's fault → 5s cooldown + release + (pre-headers) retry elsewhere;
947
+ // once committed, isTransient falls through to sendErrorResponse which ends
948
+ // the client's SSE cleanly (frees the lease AND unhangs the client).
949
+ err.code === 'UPSTREAM_TTFB' || err.code === 'UPSTREAM_IDLE' || err.code === 'UPSTREAM_BODY' ||
899
950
  err.message.includes('terminated'));
900
951
 
901
952
  if (isTransient) {
@@ -1045,7 +1096,7 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
1045
1096
  return `All ${n} accounts exhausted. Retry in ${retryAfter}s.`;
1046
1097
  }
1047
1098
 
1048
- export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody };
1099
+ export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, streamResponse };
1049
1100
 
1050
1101
  async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
1051
1102
  if (!upstreamRes.body) return '';
@@ -1696,7 +1747,7 @@ function mappedModel(originalModel, account) {
1696
1747
  /**
1697
1748
  * Stream an SSE response to the client, parsing usage data along the way.
1698
1749
  */
1699
- async function streamResponse(webStream, res, status, responseHeaders, accountIndex, accountManager, streamLog, requestInfo = {}) {
1750
+ async function streamResponse(webStream, res, status, responseHeaders, accountIndex, accountManager, streamLog, requestInfo = {}, idleMs = STREAM_IDLE_MS) {
1700
1751
  const reader = webStream.getReader();
1701
1752
  const decoder = new TextDecoder();
1702
1753
  let sseBuffer = '';
@@ -1711,9 +1762,39 @@ async function streamResponse(webStream, res, status, responseHeaders, accountIn
1711
1762
  // pre-byte failover can still re-hold the session via queueAndRetry.
1712
1763
  clearQueueHeartbeat(requestInfo);
1713
1764
 
1765
+ // Client left mid-stream: the caller drops its disconnect-abort once headers
1766
+ // arrive, and the loop below only checks res.destroyed AFTER a read resolves — so
1767
+ // a BLOCKED read() would never notice. Cancel the reader on 'close' to unblock it
1768
+ // (resolves the pending read as done) → the loop breaks → the lease frees instead
1769
+ // of leaking until the (maybe never) upstream close.
1770
+ const onClose = () => { reader.cancel().catch(() => {}); };
1771
+ res.once('close', onClose);
1772
+
1714
1773
  try {
1715
1774
  while (true) {
1716
- const { done, value } = await reader.read();
1775
+ // Idle guard: a half-open upstream (headers, then silence, never closes) would
1776
+ // block reader.read() forever → the whole request handler hangs → neither the
1777
+ // lease nor onRequestEnd is released (the "106 phantom active" leak). Bound each
1778
+ // read by a per-chunk idle timeout, RESET every chunk so a healthy slow stream
1779
+ // is never cut; on a real stall, UPSTREAM_IDLE → the caller frees the lease and
1780
+ // ends the client SSE cleanly (isTransient → sendErrorResponse).
1781
+ const readP = reader.read();
1782
+ readP.catch(() => {}); // race-loser must not surface as an unhandled rejection
1783
+ let idleTimer;
1784
+ const idle = new Promise((_, reject) => {
1785
+ idleTimer = setTimeout(
1786
+ () => reject(Object.assign(new Error('upstream idle timeout'), { code: 'UPSTREAM_IDLE' })),
1787
+ idleMs,
1788
+ );
1789
+ idleTimer.unref?.();
1790
+ });
1791
+ let result;
1792
+ try {
1793
+ result = await Promise.race([readP, idle]);
1794
+ } finally {
1795
+ clearTimeout(idleTimer); // single live timer at a time — no per-chunk timer accumulation
1796
+ }
1797
+ const { done, value } = result;
1717
1798
  if (done) break;
1718
1799
 
1719
1800
  // Client disconnected — stop reading from upstream
@@ -1760,6 +1841,7 @@ async function streamResponse(webStream, res, status, responseHeaders, accountIn
1760
1841
  readFailed = true;
1761
1842
  throw err;
1762
1843
  } finally {
1844
+ res.off('close', onClose);
1763
1845
  // Cancel upstream reader to stop consuming data nobody needs
1764
1846
  reader.cancel().catch(() => {});
1765
1847
  if (!readFailed) {