maxpool 1.3.1 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/oauth.js CHANGED
@@ -4,7 +4,10 @@ import { createInterface } from 'node:readline';
4
4
  import http from 'node:http';
5
5
 
6
6
  const PROFILE_URL = 'https://api.anthropic.com/api/oauth/profile';
7
- const DEFAULT_TOKEN_ENDPOINT = 'https://platform.claude.com/v1/oauth/token';
7
+ // Token endpoint is overridable via env so a stub OAuth server can be pointed at
8
+ // for integration tests (e.g. the single-use-rotating-token reload torture test).
9
+ const DEFAULT_TOKEN_ENDPOINT = process.env.MAXPOOL_OAUTH_TOKEN_ENDPOINT
10
+ || 'https://platform.claude.com/v1/oauth/token';
8
11
  const DEFAULT_CLIENT_ID = '9d1c250a-e61b-44d9-88ed-5944d1962f5e';
9
12
 
10
13
  /**
package/src/prober.js CHANGED
@@ -40,20 +40,29 @@ export class Prober {
40
40
  }
41
41
  }
42
42
 
43
- stop() {
43
+ /** Stop scheduling AND await any probe cycle already in flight, so a caller
44
+ * (the baton release) can be sure no probe-driven token rotation is pending
45
+ * before it hands the writer lease to another worker. */
46
+ async stop() {
44
47
  if (this.timer) { clearInterval(this.timer); this.timer = null; }
48
+ if (this._inflight) { try { await this._inflight; } catch { /* swallow */ } }
45
49
  }
46
50
 
47
- /** Probe every OAuth account once. Overlapping cycles are skipped. */
48
- async probeAll() {
49
- if (this._running) return;
51
+ /** Probe every OAuth account once. Overlapping cycles are skipped. The active
52
+ * cycle is tracked on `_inflight` so stop() can await it. */
53
+ probeAll() {
54
+ if (this._running) return this._inflight || Promise.resolve();
50
55
  this._running = true;
51
- try {
52
- const accounts = this.am.accounts.filter(a => a.type === 'oauth' && a.credential);
53
- await Promise.all(accounts.map(a => this.probeOne(a)));
54
- } finally {
55
- this._running = false;
56
- }
56
+ this._inflight = (async () => {
57
+ try {
58
+ const accounts = this.am.accounts.filter(a => a.type === 'oauth' && a.credential);
59
+ await Promise.all(accounts.map(a => this.probeOne(a)));
60
+ } finally {
61
+ this._running = false;
62
+ this._inflight = null;
63
+ }
64
+ })();
65
+ return this._inflight;
57
66
  }
58
67
 
59
68
  async probeOne(account) {
@@ -0,0 +1,121 @@
1
+ // IPC message vocabulary + the supervisor-side baton orchestration for the
2
+ // near-zero-downtime reload (issue #6). Pure logic, no process side-effects, so
3
+ // the baton sequence is unit-testable against fake workers.
4
+ //
5
+ // Roles:
6
+ // Supervisor — owns the listening socket for life, never closes it, spawns
7
+ // workers and hands the socket HANDLE to exactly one acceptor at a time.
8
+ // Worker — accepts on the shared handle; exactly one worker holds the
9
+ // WRITER LEASE (refresh/probe/persist) at any instant.
10
+ //
11
+ // The baton (single-writer guarantee): old worker RELEASES (stops accepting +
12
+ // stops writing, flushes once) BEFORE the new worker ACQUIRES. Reads overlap;
13
+ // writes never do.
14
+
15
+ // Supervisor → worker
16
+ export const MSG_LISTEN = 'listen'; // (with handle) accept on this socket + take TUI/lease
17
+ export const MSG_RELEASE = 'release'; // stop accepting, stop writing, flush, give up TUI
18
+ export const MSG_TAKEOVER = 'takeover'; // (with handle) start accepting + acquire writer lease + TUI
19
+ export const MSG_PROBE_READY = 'probe-ready'; // ask a headless worker to confirm it booted OK
20
+
21
+ // Worker → supervisor
22
+ export const MSG_RELOAD_REQUEST = 'reload-request'; // primary worker asks for a seamless reload
23
+ export const MSG_READY = 'ready'; // headless worker booted new code successfully
24
+ export const MSG_FAILED = 'failed'; // worker failed to boot/bind/restore
25
+ export const MSG_RELEASED = 'released'; // old worker stopped accepting + writing, flushed
26
+ export const MSG_PRIMARY = 'primary'; // new worker is now sole acceptor + lease holder
27
+ export const MSG_QUOTA_STATE = 'quota-state'; // old worker hands its in-memory quota to supervisor
28
+
29
+ /**
30
+ * Outcome codes for a reload attempt, surfaced to the supervisor so it can log
31
+ * and decide fallback. SWAPPED = fully cut over; ROLLED_BACK = old worker stays
32
+ * primary (no disruption); FALLBACK = abrupt exit-75 restart path taken.
33
+ */
34
+ export const RELOAD_SWAPPED = 'swapped';
35
+ export const RELOAD_ROLLED_BACK = 'rolled-back';
36
+ export const RELOAD_FALLBACK = 'fallback';
37
+
38
+ /**
39
+ * Drive the baton handshake over two worker "channels" (anything with
40
+ * .send(msg[,handle]) and an async waitFor(type, timeoutMs) that resolves with
41
+ * the message payload or rejects on timeout/exit). Returns one of the RELOAD_*
42
+ * outcomes. The caller (supervisor) owns spawning, killing, reaping and the
43
+ * actual fallback restart; this function only sequences the protocol and tells
44
+ * the caller what happened.
45
+ *
46
+ * Sequence (matches design step 2a–2e):
47
+ * a. new worker already spawned headless (no lease); we ask it to confirm READY
48
+ * b. READY fails → ROLLED_BACK: caller kills new, old stays primary
49
+ * c. old RELEASE → waits RELEASED (old stopped accepting + writing, flushed)
50
+ * d. new TAKEOVER → waits PRIMARY (new is sole acceptor + lease holder)
51
+ * e. caller drains+reaps old
52
+ *
53
+ * Any throw → FALLBACK (caller does the tested abrupt restart). All-or-nothing:
54
+ * we never leave both accepting or both writing.
55
+ */
56
+ export async function runReloadBaton({
57
+ oldWorker,
58
+ newWorker,
59
+ handle,
60
+ prepareHandle = async h => h,
61
+ readyTimeoutMs = 10_000,
62
+ releaseTimeoutMs = 20_000,
63
+ takeoverTimeoutMs = 10_000,
64
+ log = () => {},
65
+ }) {
66
+ // (a) Confirm the freshly-spawned headless worker booted the new code.
67
+ let ready;
68
+ try {
69
+ newWorker.send({ type: MSG_PROBE_READY });
70
+ ready = await newWorker.waitFor([MSG_READY, MSG_FAILED], readyTimeoutMs);
71
+ } catch (err) {
72
+ log(`reload: readiness wait failed (${err.message}); rolling back`);
73
+ return RELOAD_ROLLED_BACK;
74
+ }
75
+ // (b) New worker did not come up cleanly → roll back fully, old stays primary.
76
+ if (!ready || ready.type === MSG_FAILED) {
77
+ log(`reload: new worker reported failed (${ready?.reason || 'no ready'}); rolling back`);
78
+ return RELOAD_ROLLED_BACK;
79
+ }
80
+
81
+ // (c) Old worker releases: stop accepting NEW, keep in-flight, stop writing,
82
+ // flush config+state ONE final time, drop the TUI. After this point the
83
+ // old worker holds no lease and no acceptor — and (single-writer) no
84
+ // refresh/prober write is still in flight (it awaits drainRefreshes).
85
+ try {
86
+ oldWorker.send({ type: MSG_RELEASE });
87
+ await oldWorker.waitFor([MSG_RELEASED], releaseTimeoutMs);
88
+ } catch (err) {
89
+ // Old worker didn't ack release. We must NOT hand the socket to the new
90
+ // worker (would risk two acceptors) — fall back to the abrupt path.
91
+ log(`reload: old worker did not release (${err.message}); falling back`);
92
+ return RELOAD_FALLBACK;
93
+ }
94
+
95
+ // The old worker has now freed the listening fd. Re-arm the supervisor's own
96
+ // acceptor over the cutover gap (covers new conns in the OS backlog) and get a
97
+ // fresh re-sendable handle for the new worker.
98
+ let liveHandle = handle;
99
+ try {
100
+ liveHandle = await prepareHandle(handle);
101
+ } catch (err) {
102
+ log(`reload: could not re-arm listening socket (${err.message}); falling back`);
103
+ return RELOAD_FALLBACK;
104
+ }
105
+
106
+ // (d) New worker takes over: it becomes the SOLE acceptor (old already closed)
107
+ // and ACQUIRES the writer lease (refresh/probe/persist re-enabled) + TUI.
108
+ try {
109
+ newWorker.send({ type: MSG_TAKEOVER }, liveHandle);
110
+ await newWorker.waitFor([MSG_PRIMARY], takeoverTimeoutMs);
111
+ } catch (err) {
112
+ // The new worker failed to start accepting after the old already released.
113
+ // Neither is accepting now — fall back so the supervisor force-restarts and
114
+ // a worker comes back up on the supervisor-owned socket (queued conns drain).
115
+ log(`reload: new worker did not take over (${err.message}); falling back`);
116
+ return RELOAD_FALLBACK;
117
+ }
118
+
119
+ log('reload: cutover complete; new worker is primary');
120
+ return RELOAD_SWAPPED;
121
+ }
package/src/server.js CHANGED
@@ -51,8 +51,25 @@ export function createProxyServer(accountManager, config, hooks = {}) {
51
51
  mkdir(logDir, { recursive: true }).catch(() => {});
52
52
  }
53
53
 
54
+ // Reload drain flag. When the worker releases the baton it sets this so every
55
+ // remaining response carries `Connection: close`, retiring the client's
56
+ // keep-alive socket instead of letting it pipeline a NEW request onto a worker
57
+ // that's shutting down. `connection` is hop-by-hop so it's stripped from the
58
+ // upstream response headers — a setHeader here survives the later writeHead.
59
+ let draining = false;
60
+
61
+ // Identifies the WORKER process that served a response — proves the supervisor
62
+ // (which holds the socket but does not serve) never swallowed the request. Set
63
+ // before any writeHead; `x-maxpool-*` is informative-only and stripped from
64
+ // upstream-bound request headers elsewhere.
65
+ const workerStamp = String(process.pid);
66
+
54
67
  const server = http.createServer(async (req, res) => {
55
68
  try {
69
+ try { res.setHeader('x-maxpool-worker', workerStamp); } catch { /* headers sent */ }
70
+ if (draining) {
71
+ try { res.setHeader('Connection', 'close'); } catch { /* headers may be sent */ }
72
+ }
56
73
  // Auth check — skip for localhost connections
57
74
  const clientKey = req.headers['x-api-key'];
58
75
  const remoteAddr = req.socket.remoteAddress;
@@ -187,6 +204,11 @@ export function createProxyServer(accountManager, config, hooks = {}) {
187
204
  }
188
205
  });
189
206
 
207
+ // Begin reload drain: every subsequent response gets `Connection: close` so
208
+ // keep-alive clients retire their socket and don't pipeline a new request onto
209
+ // this releasing worker.
210
+ server.maxpoolBeginDrain = () => { draining = true; };
211
+
190
212
  return server;
191
213
  }
192
214
 
@@ -930,7 +952,7 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
930
952
  return `All ${n} accounts exhausted. Retry in ${retryAfter}s.`;
931
953
  }
932
954
 
933
- export const __serverTest = { unavailableMessage, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat };
955
+ export const __serverTest = { unavailableMessage, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, describeRequest };
934
956
 
935
957
  async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
936
958
  if (!upstreamRes.body) return '';
@@ -1375,6 +1397,11 @@ function describeRequest(req, body) {
1375
1397
  if (requiresAnthropicThinkingIntegrity(json)) {
1376
1398
  info.requiresAnthropicThinkingIntegrity = true;
1377
1399
  }
1400
+ // We fully scanned this body for signed-thinking content. Only a successfully
1401
+ // scanned, thinking-free body is safe to migrate to another account (session
1402
+ // rebalancing); an unparsed body leaves this false → treated as NOT safe
1403
+ // (fail-closed) so we never replay a signed thinking block to a new account.
1404
+ info.bodyThinkingScanned = true;
1378
1405
  } catch {
1379
1406
  // Non-JSON requests are rare; body size still gives a useful load signal.
1380
1407
  }