maxpool 1.4.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/account-manager.js +37 -1
- package/src/config.js +77 -5
- package/src/index.js +628 -59
- package/src/oauth.js +4 -1
- package/src/prober.js +19 -10
- package/src/reload-protocol.js +121 -0
- package/src/server.js +22 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "maxpool",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.5.0",
|
|
4
4
|
"description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
],
|
|
14
14
|
"scripts": {
|
|
15
15
|
"start": "node src/index.js",
|
|
16
|
-
"test": "
|
|
16
|
+
"test": "bash scripts/run-tests.sh",
|
|
17
17
|
"lint": "eslint src/ test/",
|
|
18
18
|
"release": "bash scripts/release.sh"
|
|
19
19
|
},
|
package/src/account-manager.js
CHANGED
|
@@ -201,6 +201,21 @@ export class AccountManager {
|
|
|
201
201
|
bytes: 0, // aggregate buffered body bytes across all held requests
|
|
202
202
|
};
|
|
203
203
|
this.admissionPaused = false;
|
|
204
|
+
// Single-writer baton: only the lease holder may rotate OAuth refresh tokens
|
|
205
|
+
// (refresh tokens are single-use; two refreshers brick the account). A worker
|
|
206
|
+
// booted headless during a reload starts WITHOUT the lease and refreshes
|
|
207
|
+
// nothing until it acquires the baton. Default true so the standalone /
|
|
208
|
+
// direct-listen (non-supervised, headless service) path is unchanged.
|
|
209
|
+
this.writerLease = true;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* Acquire/release the single-writer baton. While released, ensureTokenFresh is
|
|
214
|
+
* a no-op (the worker serves on its existing access tokens but never rotates a
|
|
215
|
+
* single-use refresh token — that's the lease holder's job).
|
|
216
|
+
*/
|
|
217
|
+
setWriterLease(held) {
|
|
218
|
+
this.writerLease = Boolean(held);
|
|
204
219
|
}
|
|
205
220
|
|
|
206
221
|
/**
|
|
@@ -556,7 +571,7 @@ export class AccountManager {
|
|
|
556
571
|
console.log(`[Maxpool] Anthropic recovery probe failed; retrying in ${retryAfter}s (${reason})`);
|
|
557
572
|
}
|
|
558
573
|
|
|
559
|
-
noteAmbiguousRateLimit(accountIndex, fingerprint,
|
|
574
|
+
noteAmbiguousRateLimit(accountIndex, fingerprint, _retryAfterSeconds) {
|
|
560
575
|
if (!fingerprint) return false;
|
|
561
576
|
const now = Date.now();
|
|
562
577
|
const windowMs = 30_000;
|
|
@@ -1717,6 +1732,12 @@ export class AccountManager {
|
|
|
1717
1732
|
const account = this.accounts[accountIndex];
|
|
1718
1733
|
if (!account || account.type !== 'oauth' || !account.refreshToken) return true;
|
|
1719
1734
|
|
|
1735
|
+
// Single-writer baton: a worker without the lease NEVER rotates a single-use
|
|
1736
|
+
// refresh token (doing so would invalidate the lease holder's token →
|
|
1737
|
+
// invalid_grant → bricked account). It serves on its existing access token
|
|
1738
|
+
// for the bounded drain; the lease holder owns all rotation.
|
|
1739
|
+
if (!this.writerLease) return true;
|
|
1740
|
+
|
|
1720
1741
|
if (!force && !isTokenExpiringSoon(account.expiresAt)) return true;
|
|
1721
1742
|
|
|
1722
1743
|
// Coalesce concurrent refreshes
|
|
@@ -1724,6 +1745,9 @@ export class AccountManager {
|
|
|
1724
1745
|
|
|
1725
1746
|
account._refreshPromise = (async () => {
|
|
1726
1747
|
console.log(`[Maxpool] Refreshing token for account "${account.name}"...`);
|
|
1748
|
+
// Record the token we're rotating FROM so the persistence layer's
|
|
1749
|
+
// generation guard can detect another writer having already advanced it.
|
|
1750
|
+
account._refreshedFrom = account.refreshToken;
|
|
1727
1751
|
try {
|
|
1728
1752
|
const newTokens = await this._refreshAccessToken(account.refreshToken);
|
|
1729
1753
|
account.credential = newTokens.accessToken;
|
|
@@ -1755,6 +1779,18 @@ export class AccountManager {
|
|
|
1755
1779
|
return account._refreshPromise;
|
|
1756
1780
|
}
|
|
1757
1781
|
|
|
1782
|
+
/**
|
|
1783
|
+
* Await every in-flight OAuth token refresh to settle. The single-writer baton
|
|
1784
|
+
* uses this on RELEASE: a refresh that passed the `if(!writerLease) return` gate
|
|
1785
|
+
* BEFORE the lease was dropped is still awaiting its OAuth POST; the new worker
|
|
1786
|
+
* must not acquire the lease and rotate the SAME single-use token until these
|
|
1787
|
+
* settle, or the upstream invalidates one token → invalid_grant → bricked.
|
|
1788
|
+
*/
|
|
1789
|
+
async drainRefreshes() {
|
|
1790
|
+
const pending = this.accounts.map(a => a._refreshPromise).filter(Boolean);
|
|
1791
|
+
if (pending.length) await Promise.allSettled(pending);
|
|
1792
|
+
}
|
|
1793
|
+
|
|
1758
1794
|
/**
|
|
1759
1795
|
* Set a callback to persist refreshed tokens to config.
|
|
1760
1796
|
*/
|
package/src/config.js
CHANGED
|
@@ -171,28 +171,100 @@ export async function loadState() {
|
|
|
171
171
|
}
|
|
172
172
|
}
|
|
173
173
|
|
|
174
|
-
|
|
175
|
-
|
|
174
|
+
/**
|
|
175
|
+
* Read just the `_generation` integer of an on-disk JSON file (config or state)
|
|
176
|
+
* without parsing it as a typed object. Returns 0 when the file is missing,
|
|
177
|
+
* unreadable, or carries no generation (legacy files), so a first write always
|
|
178
|
+
* advances to >=1. Never throws.
|
|
179
|
+
*/
|
|
180
|
+
export async function readGeneration(path) {
|
|
181
|
+
try {
|
|
182
|
+
const parsed = JSON.parse(await readFile(path, 'utf-8'));
|
|
183
|
+
const g = Number(parsed?._generation);
|
|
184
|
+
return Number.isFinite(g) && g > 0 ? g : 0;
|
|
185
|
+
} catch {
|
|
186
|
+
return 0;
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* Generation-guarded state write (defense-in-depth single-writer enforcement).
|
|
192
|
+
*
|
|
193
|
+
* `state.quota` carries the learned quota. We stamp a monotonic `_generation`
|
|
194
|
+
* and refuse to clobber a file whose on-disk generation is NEWER than the one
|
|
195
|
+
* we last observed — that means another writer (a stale ex-lease worker) raced
|
|
196
|
+
* us, and overwriting would revert fresher quota. The baton sequencing already
|
|
197
|
+
* guarantees a single writer; this guard catches a sequencing bug.
|
|
198
|
+
*
|
|
199
|
+
* Pass `{ expectedGeneration }` to assert the on-disk file is still at the
|
|
200
|
+
* generation you read; omit it to read-then-bump unconditionally (single-owner
|
|
201
|
+
* fast path). Returns the generation written, or null when refused.
|
|
202
|
+
*/
|
|
203
|
+
let _stateWriteChain = Promise.resolve();
|
|
204
|
+
|
|
205
|
+
export function saveState(state, { expectedGeneration = null } = {}) {
|
|
206
|
+
// Serialize state writes (a 60s interval flush can otherwise race the final
|
|
207
|
+
// baton flush, read the same on-disk generation, and double-write).
|
|
208
|
+
const run = async () => {
|
|
209
|
+
const path = getStatePath();
|
|
210
|
+
const onDisk = await readGeneration(path);
|
|
211
|
+
if (expectedGeneration != null && onDisk > expectedGeneration) {
|
|
212
|
+
// A newer writer advanced the file under us — refuse the stale flush.
|
|
213
|
+
return null;
|
|
214
|
+
}
|
|
215
|
+
const next = onDisk + 1;
|
|
216
|
+
await atomicWrite(path, JSON.stringify({ ...state, _generation: next }, null, 2) + '\n');
|
|
217
|
+
return next;
|
|
218
|
+
};
|
|
219
|
+
const result = _stateWriteChain.then(run, run);
|
|
220
|
+
_stateWriteChain = result.then(() => {}, () => {});
|
|
221
|
+
return result;
|
|
176
222
|
}
|
|
177
223
|
|
|
224
|
+
/** Barrier: resolves once every queued state write has settled. */
|
|
225
|
+
export const flushStateWrites = () => _stateWriteChain;
|
|
226
|
+
|
|
178
227
|
let _configWriteChain = Promise.resolve();
|
|
179
228
|
|
|
229
|
+
/** Barrier: resolves once every queued config write (e.g. a fire-and-forget
|
|
230
|
+
* token-refresh persist) has settled. Awaited on baton release so a rotated
|
|
231
|
+
* token can't be left unwritten when the new worker boots from disk (M3). */
|
|
232
|
+
export const flushConfigWrites = () => _configWriteChain;
|
|
233
|
+
|
|
180
234
|
/**
|
|
181
235
|
* Serialize an in-process read-modify-write of the config so concurrent updates
|
|
182
236
|
* cannot lose each other's writes — e.g. a background OAuth token refresh
|
|
183
237
|
* (fire-and-forget) racing a TUI account add/delete. Each update re-reads the
|
|
184
238
|
* latest config, applies updater(config), then saves atomically.
|
|
185
239
|
*
|
|
240
|
+
* Defense-in-depth single-writer guard: the config carries a monotonic
|
|
241
|
+
* `_generation`. If `guardGeneration` is supplied and the on-disk generation is
|
|
242
|
+
* already NEWER than it, the write is REFUSED (a stale ex-lease worker raced the
|
|
243
|
+
* current writer). The updater additionally receives the on-disk generation as
|
|
244
|
+
* `config._generation` so a refresh updater can SKIP a token rotation it sees a
|
|
245
|
+
* fresher writer already performed. The generation is bumped on every write.
|
|
246
|
+
*
|
|
186
247
|
* NOTE: this serializes writes within THIS process only. A separate
|
|
187
248
|
* `maxpool import`/`login` process writing concurrently is not coordinated
|
|
188
249
|
* (that would require a lockfile) — but those are short, rare, human-driven.
|
|
189
250
|
*/
|
|
190
|
-
export function atomicConfigUpdate(updater) {
|
|
251
|
+
export function atomicConfigUpdate(updater, { guardGeneration = null } = {}) {
|
|
191
252
|
const run = async () => {
|
|
192
253
|
const config = await loadConfig() || createDefaultConfig();
|
|
193
|
-
|
|
254
|
+
const onDisk = Number(config._generation) || 0;
|
|
255
|
+
if (guardGeneration != null && onDisk > guardGeneration) {
|
|
256
|
+
// Refuse: a newer writer already advanced the config. Surface the live
|
|
257
|
+
// config so the caller can reconcile, but write nothing.
|
|
258
|
+
const refused = new Error('config generation advanced; stale write refused');
|
|
259
|
+
refused.code = 'STALE_GENERATION';
|
|
260
|
+
refused.onDiskGeneration = onDisk;
|
|
261
|
+
refused.config = config;
|
|
262
|
+
throw refused;
|
|
263
|
+
}
|
|
264
|
+
const result = await updater(config);
|
|
265
|
+
config._generation = onDisk + 1;
|
|
194
266
|
await saveConfig(config);
|
|
195
|
-
return config;
|
|
267
|
+
return result === undefined ? config : result;
|
|
196
268
|
};
|
|
197
269
|
// Run after any in-flight update regardless of whether it resolved or
|
|
198
270
|
// rejected, so one failure doesn't poison the chain; the caller still sees
|
package/src/index.js
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
import { spawn, spawnSync } from 'node:child_process';
|
|
4
4
|
import { createInterface } from 'node:readline';
|
|
5
|
-
import { loadOrCreateConfig, loadConfig, saveConfig, atomicConfigUpdate, getConfigPath, loadState, saveState } from './config.js';
|
|
5
|
+
import { loadOrCreateConfig, loadConfig, saveConfig, atomicConfigUpdate, getConfigPath, loadState, saveState, getStatePath, readGeneration, flushConfigWrites, flushStateWrites } from './config.js';
|
|
6
6
|
import { AccountManager } from './account-manager.js';
|
|
7
7
|
import { createProxyServer } from './server.js';
|
|
8
8
|
import { Prober } from './prober.js';
|
|
@@ -11,11 +11,20 @@ import { TUI } from './tui.js';
|
|
|
11
11
|
import { RestartController } from './restart-controller.js';
|
|
12
12
|
import { resolveAccounts } from './account-config.js';
|
|
13
13
|
import { maybeCheckForUpdate } from './updater.js';
|
|
14
|
+
import {
|
|
15
|
+
runReloadBaton,
|
|
16
|
+
RELOAD_SWAPPED, RELOAD_ROLLED_BACK,
|
|
17
|
+
MSG_LISTEN, MSG_RELEASE, MSG_TAKEOVER, MSG_PROBE_READY,
|
|
18
|
+
MSG_RELOAD_REQUEST, MSG_READY, MSG_FAILED, MSG_RELEASED, MSG_PRIMARY,
|
|
19
|
+
} from './reload-protocol.js';
|
|
14
20
|
|
|
15
21
|
const args = process.argv.slice(2);
|
|
16
22
|
const command = args[0];
|
|
17
23
|
const SERVER_RESTART_EXIT_CODE = 75;
|
|
18
24
|
const SERVER_WORKER_ENV = 'MAXPOOL_SERVER_WORKER';
|
|
25
|
+
// Set by the supervisor when it spawns a worker for a seamless reload: that
|
|
26
|
+
// worker boots HEADLESS (plain logs, no writer lease) and waits for the baton.
|
|
27
|
+
const SERVER_RELOAD_WORKER_ENV = 'MAXPOOL_RELOAD_WORKER';
|
|
19
28
|
|
|
20
29
|
switch (command) {
|
|
21
30
|
case 'server':
|
|
@@ -71,43 +80,329 @@ switch (command) {
|
|
|
71
80
|
// ── server ──────────────────────────────────────────────────
|
|
72
81
|
|
|
73
82
|
async function serverCommand() {
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
!process.stdout.isTTY ||
|
|
77
|
-
!process.stdin.isTTY
|
|
78
|
-
) {
|
|
83
|
+
// A spawned worker (env flag set by the supervisor) runs the proxy itself.
|
|
84
|
+
if (process.env[SERVER_WORKER_ENV] === '1') {
|
|
79
85
|
return serverWorkerCommand();
|
|
80
86
|
}
|
|
81
87
|
|
|
82
|
-
//
|
|
83
|
-
//
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
88
|
+
// Non-TTY (e.g. `maxpool server` as a background service): keep the existing
|
|
89
|
+
// tested direct-listen path. The seamless-reload feature is only active under
|
|
90
|
+
// the TTY supervisor; a service manager already handles restart/respawn.
|
|
91
|
+
// MAXPOOL_FORCE_SUPERVISOR=1 forces the supervisor path without a TTY (used by
|
|
92
|
+
// the reload integration tests; the worker still runs plain-log without a TTY).
|
|
93
|
+
const forceSupervisor = process.env.MAXPOOL_FORCE_SUPERVISOR === '1';
|
|
94
|
+
if (!forceSupervisor && (!process.stdout.isTTY || !process.stdin.isTTY)) {
|
|
95
|
+
return serverWorkerCommand();
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
return supervisorCommand();
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
// ── supervisor (TTY) ─────────────────────────────────────────
|
|
102
|
+
//
|
|
103
|
+
// Owns the listening socket for its whole life (never closes it) and hands the
|
|
104
|
+
// socket HANDLE to exactly one worker at a time over IPC. A worker requests a
|
|
105
|
+
// seamless reload; the supervisor spawns a fresh headless worker, runs the
|
|
106
|
+
// single-writer baton, then swaps. Any failure falls back to the tested abrupt
|
|
107
|
+
// exit-75 respawn. A crash-loop degrades to loud single-worker failure via an
|
|
108
|
+
// exponential backoff, never a tight fork loop.
|
|
109
|
+
|
|
110
|
+
async function supervisorCommand() {
|
|
111
|
+
const { createServer } = await import('node:net');
|
|
112
|
+
const config = await loadOrCreateConfig();
|
|
113
|
+
const port = config.proxy.port;
|
|
114
|
+
const host = config.proxy.host || '127.0.0.1';
|
|
115
|
+
|
|
116
|
+
// Bind once. EADDRINUSE / EACCES here is a cold-start failure → exit(1) is
|
|
117
|
+
// correct (there's no worker to keep alive yet).
|
|
118
|
+
let masterServer = createServer();
|
|
119
|
+
// We DROP any connection the supervisor accidentally accepts while a worker is
|
|
120
|
+
// also accepting on the shared handle — but the design avoids that: the
|
|
121
|
+
// supervisor's acceptor is only LIVE during the brief cutover gap (it stops
|
|
122
|
+
// once a worker confirms it is the sole acceptor). A bare handler is required
|
|
123
|
+
// so the rare gap-accepted socket isn't left dangling.
|
|
124
|
+
const relistenMaster = () => new Promise((resolve, reject) => {
|
|
125
|
+
if (masterServer.listening) { resolve(); return; }
|
|
126
|
+
const onErr = err => { masterServer.removeListener('listening', onListen); reject(err); };
|
|
127
|
+
const onListen = () => { masterServer.removeListener('error', onErr); resolve(); };
|
|
128
|
+
masterServer.once('error', onErr);
|
|
129
|
+
masterServer.once('listening', onListen);
|
|
130
|
+
masterServer.listen(port, host);
|
|
131
|
+
});
|
|
132
|
+
const closeMasterAccept = () => new Promise(resolve => {
|
|
133
|
+
if (!masterServer.listening) { resolve(); return; }
|
|
134
|
+
masterServer.close(() => resolve());
|
|
135
|
+
});
|
|
136
|
+
try {
|
|
137
|
+
await relistenMaster();
|
|
138
|
+
} catch (err) {
|
|
139
|
+
handleServerListenError(err, host, port);
|
|
140
|
+
return;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// Keep the supervisor attached to the shell. The worker shares the supervisor's
|
|
144
|
+
// process group, so a terminal Ctrl-C (SIGINT/SIGTERM) is delivered by the TTY
|
|
145
|
+
// to BOTH already — the supervisor must NOT forward those or the worker gets a
|
|
146
|
+
// doubled signal (the "second Ctrl-C force-quits" footgun). The supervisor
|
|
147
|
+
// ignores SIGINT/SIGTERM itself (the worker drains + exits, ending the turn).
|
|
148
|
+
let activeWorker = null;
|
|
149
|
+
const forwardSignal = sig => { try { activeWorker?.child.kill(sig); } catch { /* ignore */ } };
|
|
150
|
+
process.on('SIGINT', () => { /* delivered to the worker by the TTY group */ });
|
|
151
|
+
process.on('SIGTERM', () => { /* delivered to the worker by the TTY group */ });
|
|
152
|
+
// SIGHUP (from `kill -HUP <supervisor-pid>`) reaches ONLY the supervisor →
|
|
153
|
+
// forward it so the worker requests a seamless reload.
|
|
154
|
+
process.on('SIGHUP', () => forwardSignal('SIGHUP'));
|
|
155
|
+
// A spawn failure / stray rejection must NOT kill the supervisor (it would
|
|
156
|
+
// wedge the port and drop the service). Log and let the supervision loop or
|
|
157
|
+
// the reload's own error handling recover.
|
|
158
|
+
process.on('uncaughtException', err => {
|
|
159
|
+
console.error(`[Maxpool] Supervisor uncaughtException (continuing): ${err?.stack || err}`);
|
|
160
|
+
});
|
|
161
|
+
process.on('unhandledRejection', reason => {
|
|
162
|
+
console.error(`[Maxpool] Supervisor unhandledRejection (continuing): ${reason}`);
|
|
163
|
+
});
|
|
164
|
+
|
|
165
|
+
// After SIGKILLing a worker that may have owned the TUI, the worker had no
|
|
166
|
+
// chance to restore the terminal — the supervisor emits the restore itself
|
|
167
|
+
// (exit alt-screen + show cursor + raw off) so the user's shell is clean.
|
|
168
|
+
const restoreTerminalFromSupervisor = () => {
|
|
169
|
+
try {
|
|
170
|
+
if (process.stdout.isTTY) process.stdout.write('\x1b[?25h\x1b[?1049l');
|
|
171
|
+
if (process.stdin.isTTY) { try { process.stdin.setRawMode(false); } catch { /* ignore */ } }
|
|
172
|
+
} catch { /* never throw */ }
|
|
173
|
+
};
|
|
174
|
+
process.on('exit', restoreTerminalFromSupervisor);
|
|
175
|
+
|
|
176
|
+
// Crash-loop guard: count rapid consecutive non-restart exits and back off so
|
|
177
|
+
// a worker that crashes on boot doesn't spin the CPU forking. A clean run for
|
|
178
|
+
// a while resets the counter.
|
|
179
|
+
let crashCount = 0;
|
|
180
|
+
const CRASH_WINDOW_MS = 10_000;
|
|
181
|
+
const MAX_BACKOFF_MS = 8_000;
|
|
182
|
+
|
|
183
|
+
let reloadInFlight = false;
|
|
184
|
+
// Resolver for the CURRENT supervision turn. A swap re-points monitoring to
|
|
185
|
+
// the new worker WITHOUT ending the turn; only an exit/fallback resolves it.
|
|
186
|
+
let endTurn = null;
|
|
187
|
+
|
|
188
|
+
// Spawn a worker in the SAME process group as the supervisor. A terminal
|
|
189
|
+
// Ctrl-C (SIGINT/SIGTERM) is delivered by the TTY to the whole foreground
|
|
190
|
+
// group, so BOTH already receive it — the supervisor therefore does NOT
|
|
191
|
+
// forward those (that would double-deliver). Same-group also means a group
|
|
192
|
+
// SIGKILL of the supervisor reaps the worker (no orphan holding the port).
|
|
193
|
+
const spawnWorker = ({ reload = false } = {}) => {
|
|
194
|
+
const env = { ...process.env, [SERVER_WORKER_ENV]: '1' };
|
|
195
|
+
if (reload) env[SERVER_RELOAD_WORKER_ENV] = '1';
|
|
196
|
+
const child = spawn(process.execPath, process.argv.slice(1), {
|
|
197
|
+
cwd: process.cwd(),
|
|
198
|
+
env,
|
|
199
|
+
stdio: ['inherit', 'inherit', 'inherit', 'ipc'],
|
|
200
|
+
});
|
|
201
|
+
const worker = makeWorkerChannel(child);
|
|
202
|
+
if (!reload) {
|
|
203
|
+
// Cold start: hand the socket and tell it to go primary immediately.
|
|
204
|
+
worker.send({ type: MSG_LISTEN }, masterServer);
|
|
205
|
+
}
|
|
206
|
+
return worker;
|
|
207
|
+
};
|
|
208
|
+
|
|
209
|
+
// Wire a worker as the active primary: its RELOAD_REQUEST triggers the baton;
|
|
210
|
+
// its MSG_PRIMARY means it is now the sole acceptor (so the supervisor stops
|
|
211
|
+
// its own competing accept loop); its exit/spawn-error ends the turn.
|
|
212
|
+
const monitorAsActive = worker => {
|
|
213
|
+
activeWorker = worker;
|
|
214
|
+
worker.child.removeAllListeners('message');
|
|
215
|
+
worker.child.on('message', msg => {
|
|
216
|
+
if (msg?.type === MSG_RELOAD_REQUEST) orchestrateReload().catch(() => {});
|
|
217
|
+
else if (msg?.type === MSG_PRIMARY && activeWorker === worker) {
|
|
218
|
+
// The worker is now sole acceptor on the handle → stop the supervisor
|
|
219
|
+
// racing it for accepts (the steady-state ~78%-hang bug). The supervisor
|
|
220
|
+
// still HOLDS the socket via the worker's fd; it re-arms only at reload.
|
|
221
|
+
closeMasterAccept().catch(() => {});
|
|
222
|
+
}
|
|
223
|
+
});
|
|
224
|
+
worker.child.once('exit', (code, signal) => {
|
|
225
|
+
// Only the worker that is STILL active when it exits ends the turn. A
|
|
226
|
+
// reaped old worker (already swapped out) exiting must be ignored here.
|
|
227
|
+
if (activeWorker === worker) endTurn?.({ code, signal });
|
|
228
|
+
});
|
|
229
|
+
// A spawn-time failure (EMFILE/EAGAIN under load) surfaces as 'error', not
|
|
230
|
+
// 'exit' — treat it as a crashed turn so backoff handles it (M5).
|
|
231
|
+
worker.child.once('error', err => {
|
|
232
|
+
if (activeWorker === worker) endTurn?.({ code: 1, signal: null, spawnError: err.message });
|
|
233
|
+
});
|
|
234
|
+
};
|
|
235
|
+
|
|
236
|
+
// Reload orchestration: spawn a fresh headless worker, run the baton, swap.
|
|
237
|
+
const orchestrateReload = async () => {
|
|
238
|
+
if (reloadInFlight) return;
|
|
239
|
+
reloadInFlight = true;
|
|
240
|
+
const oldWorker = activeWorker;
|
|
241
|
+
let newWorker = null;
|
|
242
|
+
try {
|
|
243
|
+
newWorker = spawnWorker({ reload: true });
|
|
244
|
+
// Hold the new worker's spawn-error so a fork failure mid-baton becomes a
|
|
245
|
+
// clean ROLLED_BACK/FALLBACK instead of crashing the supervisor (M5).
|
|
246
|
+
newWorker.child.once('error', () => { /* surfaced via waitFor reject */ });
|
|
247
|
+
const outcome = await runReloadBaton({
|
|
248
|
+
oldWorker,
|
|
249
|
+
newWorker,
|
|
250
|
+
handle: masterServer,
|
|
251
|
+
// After the OLD worker has released its fd, re-arm the supervisor's own
|
|
252
|
+
// listener to cover the cutover gap, then hand THAT live handle to the
|
|
253
|
+
// new worker. The new worker's MSG_PRIMARY then closes it again.
|
|
254
|
+
prepareHandle: async () => { await relistenMaster(); return masterServer; },
|
|
255
|
+
log: msg => console.log(`[Maxpool] ${msg}`),
|
|
256
|
+
});
|
|
257
|
+
|
|
258
|
+
if (outcome === RELOAD_SWAPPED) {
|
|
259
|
+
// Re-point monitoring to the new worker (it's now primary), THEN reap the
|
|
260
|
+
// old one. Order matters: monitorAsActive sets activeWorker=new so the
|
|
261
|
+
// old worker's pending exit handler no-ops. monitorAsActive also handles
|
|
262
|
+
// the new worker's already-sent MSG_PRIMARY is moot here — the new worker
|
|
263
|
+
// sent PRIMARY during the baton; re-assert the master-accept close.
|
|
264
|
+
monitorAsActive(newWorker);
|
|
265
|
+
closeMasterAccept().catch(() => {});
|
|
266
|
+
reapOldWorker(oldWorker, config);
|
|
267
|
+
return;
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
if (outcome === RELOAD_ROLLED_BACK) {
|
|
271
|
+
// Old worker never released — it's still fully primary. Kill the new
|
|
272
|
+
// headless worker; nothing else changed. ZERO disruption.
|
|
273
|
+
try { newWorker.child.kill('SIGKILL'); } catch { /* ignore */ }
|
|
274
|
+
return;
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
// FALLBACK: the old worker may have released; neither is reliably primary.
|
|
278
|
+
// Kill both and let the respawn loop bring a fresh primary up on the
|
|
279
|
+
// supervisor-owned socket. Queued conns sit in the OS backlog meanwhile.
|
|
280
|
+
try { newWorker.child.kill('SIGKILL'); } catch { /* ignore */ }
|
|
281
|
+
try { oldWorker.child.kill('SIGKILL'); } catch { /* ignore */ }
|
|
282
|
+
restoreTerminalFromSupervisor(); // SIGKILLed workers can't restore the TUI
|
|
283
|
+
activeWorker = null;
|
|
284
|
+
endTurn?.({ code: null, signal: 'SIGKILL', fallback: true });
|
|
285
|
+
} catch (err) {
|
|
286
|
+
console.error(`[Maxpool] Reload error: ${err.message}; falling back to abrupt restart`);
|
|
287
|
+
try { newWorker?.child.kill('SIGKILL'); } catch { /* ignore */ }
|
|
288
|
+
try { oldWorker.child.kill('SIGKILL'); } catch { /* ignore */ }
|
|
289
|
+
restoreTerminalFromSupervisor();
|
|
290
|
+
activeWorker = null;
|
|
291
|
+
endTurn?.({ code: null, signal: 'SIGKILL', fallback: true });
|
|
292
|
+
} finally {
|
|
293
|
+
reloadInFlight = false;
|
|
294
|
+
}
|
|
295
|
+
};
|
|
296
|
+
|
|
297
|
+
// One supervision turn: monitor `worker` until it exits (or a fallback forces a
|
|
298
|
+
// respawn). Swaps re-point monitoring without resolving. Resolves exit info.
|
|
299
|
+
const superviseTurn = worker => new Promise(resolve => {
|
|
300
|
+
endTurn = info => { endTurn = null; resolve(info); };
|
|
301
|
+
monitorAsActive(worker);
|
|
302
|
+
});
|
|
303
|
+
|
|
87
304
|
try {
|
|
88
305
|
while (true) {
|
|
89
|
-
const
|
|
90
|
-
|
|
306
|
+
const worker = spawnWorker({ reload: false });
|
|
307
|
+
const startedAt = Date.now();
|
|
308
|
+
const result = await superviseTurn(worker);
|
|
309
|
+
|
|
310
|
+
if (result.fallback) {
|
|
311
|
+
// Fallback swap killed both workers; bring a fresh primary straight back.
|
|
312
|
+
crashCount = 0;
|
|
313
|
+
continue;
|
|
314
|
+
}
|
|
315
|
+
if (result.code === SERVER_RESTART_EXIT_CODE) {
|
|
316
|
+
// Abrupt self-restart (worker-initiated exit-75 fallback path).
|
|
317
|
+
crashCount = 0;
|
|
318
|
+
continue;
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// Non-restart exit. If it died fast, it's likely crash-looping on boot.
|
|
322
|
+
const ranFor = Date.now() - startedAt;
|
|
323
|
+
if (ranFor < CRASH_WINDOW_MS && (result.code ?? 1) !== 0) {
|
|
324
|
+
crashCount++;
|
|
325
|
+
const backoff = Math.min(MAX_BACKOFF_MS, 250 * 2 ** (crashCount - 1));
|
|
326
|
+
console.error(`[Maxpool] Worker exited (code ${result.code}, signal ${result.signal}) after ${ranFor}ms — crash #${crashCount}. Backing off ${backoff}ms before respawn.`);
|
|
327
|
+
await delay(backoff);
|
|
328
|
+
continue;
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
// Clean shutdown (q / signal). Propagate the exit code and stop.
|
|
91
332
|
if (result.signal) process.exitCode = 1;
|
|
92
333
|
else process.exitCode = result.code ?? 1;
|
|
93
334
|
return;
|
|
94
335
|
}
|
|
95
336
|
} finally {
|
|
96
|
-
|
|
97
|
-
process.off('SIGTERM', ignoreTerminalSignal);
|
|
337
|
+
try { masterServer.close(); } catch { /* ignore */ }
|
|
98
338
|
}
|
|
99
339
|
}
|
|
100
340
|
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
341
|
+
// Drain + reap a released worker. The worker exits itself once its bounded
|
|
342
|
+
// in-flight finishes; the supervisor SIGKILLs it if it outlives the drain cap.
|
|
343
|
+
function reapOldWorker(worker, config) {
|
|
344
|
+
if (!worker) return;
|
|
345
|
+
const drainTimeoutMs = Math.max(1000, Number(config.shutdown?.drainTimeoutMs) || 15_000);
|
|
346
|
+
let reaped = false;
|
|
347
|
+
const finish = () => { if (reaped) return; reaped = true; clearTimeout(timer); };
|
|
348
|
+
const timer = setTimeout(() => {
|
|
349
|
+
if (reaped) return;
|
|
350
|
+
console.error(`[Maxpool] Old worker outlived ${Math.ceil(drainTimeoutMs / 1000)}s drain cap; SIGKILL.`);
|
|
351
|
+
try { worker.child.kill('SIGKILL'); } catch { /* ignore */ }
|
|
352
|
+
finish();
|
|
353
|
+
}, drainTimeoutMs);
|
|
354
|
+
timer.unref?.();
|
|
355
|
+
worker.child.once('exit', finish);
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/**
|
|
359
|
+
* Wrap a child process in a small IPC channel: `.send(msg[,handle])` and an
|
|
360
|
+
* async `.waitFor([types], timeoutMs)` that resolves with the first matching
|
|
361
|
+
* message, or rejects on timeout / premature child exit.
|
|
362
|
+
*/
|
|
363
|
+
function makeWorkerChannel(child) {
|
|
364
|
+
return {
|
|
365
|
+
child,
|
|
366
|
+
send(msg, handle) {
|
|
367
|
+
try {
|
|
368
|
+
if (handle) child.send(msg, handle);
|
|
369
|
+
else child.send(msg);
|
|
370
|
+
} catch { /* IPC may be torn down mid-swap; baton timeouts cover it */ }
|
|
371
|
+
},
|
|
372
|
+
waitFor(types, timeoutMs) {
|
|
373
|
+
const want = Array.isArray(types) ? types : [types];
|
|
374
|
+
return new Promise((resolve, reject) => {
|
|
375
|
+
const cleanup = () => {
|
|
376
|
+
clearTimeout(timer);
|
|
377
|
+
child.removeListener('message', onMsg);
|
|
378
|
+
child.removeListener('exit', onExit);
|
|
379
|
+
child.removeListener('error', onError);
|
|
380
|
+
};
|
|
381
|
+
const onMsg = msg => {
|
|
382
|
+
if (msg && want.includes(msg.type)) { cleanup(); resolve(msg); }
|
|
383
|
+
};
|
|
384
|
+
const onExit = (code, signal) => {
|
|
385
|
+
cleanup();
|
|
386
|
+
reject(new Error(`worker exited (code ${code}, signal ${signal}) before ${want.join('/')}`));
|
|
387
|
+
};
|
|
388
|
+
// A spawn failure (EMFILE/EAGAIN) surfaces as 'error', not 'exit' — the
|
|
389
|
+
// baton must see it as a failed step (→ ROLLED_BACK/FALLBACK), not hang.
|
|
390
|
+
const onError = err => { cleanup(); reject(new Error(`worker spawn error before ${want.join('/')}: ${err.message}`)); };
|
|
391
|
+
const timer = setTimeout(() => {
|
|
392
|
+
cleanup();
|
|
393
|
+
reject(new Error(`timed out waiting for ${want.join('/')}`));
|
|
394
|
+
}, timeoutMs);
|
|
395
|
+
timer.unref?.();
|
|
396
|
+
child.on('message', onMsg);
|
|
397
|
+
child.once('exit', onExit);
|
|
398
|
+
child.once('error', onError);
|
|
399
|
+
});
|
|
400
|
+
},
|
|
401
|
+
};
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
function delay(ms) {
|
|
405
|
+
return new Promise(resolve => { const t = setTimeout(resolve, ms); t.unref?.(); });
|
|
111
406
|
}
|
|
112
407
|
|
|
113
408
|
async function serverWorkerCommand() {
|
|
@@ -138,23 +433,67 @@ async function serverWorkerCommand() {
|
|
|
138
433
|
config.routing?.preferredAccount,
|
|
139
434
|
);
|
|
140
435
|
|
|
436
|
+
// Supervised = spawned by the TTY supervisor over IPC (handle-based listen +
|
|
437
|
+
// baton). A reload worker boots HEADLESS without the writer lease and waits
|
|
438
|
+
// for the baton; a cold-start worker takes the lease on MSG_LISTEN.
|
|
439
|
+
const supervised = typeof process.send === 'function';
|
|
440
|
+
const isReloadWorker = process.env[SERVER_RELOAD_WORKER_ENV] === '1';
|
|
441
|
+
|
|
442
|
+
// M6: a reload worker boots WITHOUT the writer lease so the AM-level brick
|
|
443
|
+
// guard (ensureTokenFresh no-op) is CLOSED for the entire headless window —
|
|
444
|
+
// before any code path could trigger a refresh. acquireLease() flips it true
|
|
445
|
+
// at takeover. (writerLease defaults true for the standalone/direct path.)
|
|
446
|
+
if (isReloadWorker) accountManager.setWriterLease(false);
|
|
447
|
+
|
|
141
448
|
// Restore quota observed in a previous run so a restart doesn't lose routing
|
|
142
449
|
// accuracy and re-probe from scratch. Stale windows clear on first use.
|
|
450
|
+
// A reload worker restores quota IN-MEMORY only (state file handed via the
|
|
451
|
+
// lease holder); a cold/direct worker reads the on-disk state file.
|
|
143
452
|
const savedState = await loadState();
|
|
144
453
|
if (savedState?.quota) accountManager.restoreQuotaState(savedState.quota);
|
|
145
|
-
|
|
146
|
-
|
|
454
|
+
// Track the state-file generation we last observed so a stale flush is refused.
|
|
455
|
+
let stateGeneration = Number(savedState?._generation) || 0;
|
|
456
|
+
|
|
457
|
+
// ── single-writer baton: refresh / probe / persistence gated by the lease ──
|
|
458
|
+
// A worker without the lease writes NOTHING (no token rotation, no config
|
|
459
|
+
// write, no state write, no probe). Only the lease holder may write.
|
|
460
|
+
let hasLease = false;
|
|
461
|
+
// `force` performs the FINAL flush during a baton release, AFTER releaseLease()
|
|
462
|
+
// has already flipped hasLease=false (and cleared the periodic interval, so no
|
|
463
|
+
// write races this one). Without force, this no-ops post-release and the reload's
|
|
464
|
+
// learned-quota flush is silently dropped — the new worker would boot from up-to-
|
|
465
|
+
// 60s-stale quota and the state generation would never advance across a reload.
|
|
466
|
+
const persistQuotaState = (force = false) => {
|
|
467
|
+
if (!hasLease && !force) return Promise.resolve();
|
|
468
|
+
// The forced final flush (baton release / shutdown) is the SOLE writer at that
|
|
469
|
+
// point — releaseLease() already cleared the periodic interval and the next
|
|
470
|
+
// worker hasn't acquired the lease yet — so it bypasses the cross-worker
|
|
471
|
+
// stale-generation guard. Without this, a 60s-interval write that fired just
|
|
472
|
+
// before releaseLease could bump the on-disk generation and get the forced
|
|
473
|
+
// flush REFUSED (dropping the final quota snapshot). The guard only exists to
|
|
474
|
+
// serialize CROSS-worker writes; the final flush is provably intra-worker.
|
|
475
|
+
const expectedGeneration = force ? null : stateGeneration;
|
|
476
|
+
return saveState({ quota: accountManager.exportQuotaState() }, { expectedGeneration })
|
|
477
|
+
.then(written => { if (written != null) stateGeneration = written; })
|
|
478
|
+
.catch(() => {});
|
|
479
|
+
};
|
|
147
480
|
// Persist quota every minute; unref so it never keeps the process alive.
|
|
148
|
-
|
|
149
|
-
quotaSaveInterval.unref?.();
|
|
481
|
+
let quotaSaveInterval = null;
|
|
150
482
|
|
|
151
483
|
// Opt-in background quota probe (config.quotaProbeSeconds, default 0 = off).
|
|
152
484
|
const prober = new Prober(accountManager, { intervalMs: (config.quotaProbeSeconds || 0) * 1000 });
|
|
153
|
-
prober.start();
|
|
154
485
|
|
|
155
|
-
// Persist refreshed tokens back to config
|
|
156
|
-
//
|
|
157
|
-
|
|
486
|
+
// Persist refreshed tokens back to config. Defense-in-depth: the updater reads
|
|
487
|
+
// the on-disk refresh token and SKIPS the rotation if a fresher writer already
|
|
488
|
+
// advanced it (generation guard), so a stale write can't double-spend a token.
|
|
489
|
+
//
|
|
490
|
+
// We do NOT gate this on `hasLease`: a refresh that already STARTED (it passed
|
|
491
|
+
// the lease gate in ensureTokenFresh) MUST persist its rotated single-use token
|
|
492
|
+
// even if the lease was dropped while its POST was in flight — otherwise the
|
|
493
|
+
// baton hands off and the new worker boots from the now-invalidated on-disk
|
|
494
|
+
// token (B1/M3). New refreshes can't start without the lease (ensureTokenFresh
|
|
495
|
+
// no-ops), so every callback here is from a legitimate lease-era refresh.
|
|
496
|
+
const persistTokenRefresh = (idx, newTokens) => {
|
|
158
497
|
const account = accountManager.accounts[idx];
|
|
159
498
|
if (!account) return;
|
|
160
499
|
// Keep config.accounts in sync so TUI saveConfig doesn't clobber fresh tokens
|
|
@@ -178,21 +517,58 @@ async function serverWorkerCommand() {
|
|
|
178
517
|
// Match by UUID first, then by name — index may have shifted
|
|
179
518
|
const cfgIdx = findConfigAccount(diskConfig, account);
|
|
180
519
|
if (cfgIdx >= 0) {
|
|
181
|
-
diskConfig.accounts[cfgIdx]
|
|
182
|
-
|
|
183
|
-
|
|
520
|
+
const onDisk = diskConfig.accounts[cfgIdx];
|
|
521
|
+
// Generation guard: if the on-disk refresh token already advanced past
|
|
522
|
+
// the token we rotated FROM, another writer beat us — skip the write so
|
|
523
|
+
// we don't revert a fresher single-use token (the brick-the-account case).
|
|
524
|
+
if (onDisk.refreshToken && onDisk.refreshToken !== account._refreshedFrom &&
|
|
525
|
+
onDisk.refreshToken !== newTokens.refreshToken) {
|
|
526
|
+
return;
|
|
527
|
+
}
|
|
528
|
+
onDisk.accessToken = newTokens.accessToken;
|
|
529
|
+
onDisk.refreshToken = newTokens.refreshToken;
|
|
530
|
+
onDisk.expiresAt = newTokens.expiresAt;
|
|
184
531
|
}
|
|
185
|
-
}).catch(err =>
|
|
186
|
-
|
|
532
|
+
}).catch(err => {
|
|
533
|
+
if (err?.code === 'STALE_GENERATION') return; // another writer advanced; benign
|
|
534
|
+
console.error(`[Maxpool] Failed to save refreshed token: ${err.message}`);
|
|
535
|
+
});
|
|
536
|
+
};
|
|
537
|
+
accountManager.onTokenRefresh(persistTokenRefresh);
|
|
538
|
+
|
|
187
539
|
const port = config.proxy.port;
|
|
188
540
|
const host = config.proxy.host || '127.0.0.1';
|
|
189
|
-
|
|
541
|
+
// A headless reload worker NEVER drives the TUI (single-owner terminal); it
|
|
542
|
+
// takes the TUI only on baton takeover. Cold/direct workers use it if on a TTY.
|
|
543
|
+
const useTUI = process.stdout.isTTY && process.stdin.isTTY && !isReloadWorker;
|
|
190
544
|
|
|
191
545
|
let tui = null;
|
|
192
546
|
let server = null;
|
|
193
547
|
let syncTimer = null;
|
|
194
548
|
let draining = false;
|
|
195
549
|
let restartController = null;
|
|
550
|
+
|
|
551
|
+
// Best-effort terminal restore on ANY abnormal exit path (uncaughtException,
|
|
552
|
+
// a bare process.exit, a crash) so the user's shell is never left in raw mode
|
|
553
|
+
// or the alt-screen. Idempotent and safe even when no TUI was running.
|
|
554
|
+
const restoreTerminal = () => {
|
|
555
|
+
try {
|
|
556
|
+
if (tui?.running) { tui.stop(); return; }
|
|
557
|
+
if (process.stdout.isTTY) process.stdout.write('\x1b[?25h\x1b[?1049l');
|
|
558
|
+
if (process.stdin.isTTY) { try { process.stdin.setRawMode(false); } catch { /* ignore */ } }
|
|
559
|
+
} catch { /* never throw from a restore */ }
|
|
560
|
+
};
|
|
561
|
+
process.on('exit', restoreTerminal);
|
|
562
|
+
process.on('uncaughtException', err => {
|
|
563
|
+
console.error(`[Maxpool] Worker uncaughtException: ${err?.stack || err}`);
|
|
564
|
+
restoreTerminal();
|
|
565
|
+
// A reload worker must NEVER exit(1) (escapes the supervisor exit-75 loop).
|
|
566
|
+
// Stay alive so the supervisor's baton timeouts roll us back cleanly.
|
|
567
|
+
if (!isReloadWorker) process.exit(SERVER_RESTART_EXIT_CODE);
|
|
568
|
+
});
|
|
569
|
+
process.on('unhandledRejection', reason => {
|
|
570
|
+
console.error(`[Maxpool] Worker unhandledRejection: ${reason}`);
|
|
571
|
+
});
|
|
196
572
|
// Quit drains in-flight requests, then force-exits. Kept short so a single
|
|
197
573
|
// 'q' / Ctrl-C / SIGTERM actually quits under a continuous request flood
|
|
198
574
|
// (where there are always active requests) instead of waiting indefinitely.
|
|
@@ -214,22 +590,74 @@ async function serverWorkerCommand() {
|
|
|
214
590
|
},
|
|
215
591
|
};
|
|
216
592
|
|
|
593
|
+
// ── writer lease (single-writer baton) ──
|
|
594
|
+
// Acquiring the lease turns ON token rotation, the quota-save interval, and the
|
|
595
|
+
// prober. Releasing turns them all OFF and flushes once. Exactly one worker
|
|
596
|
+
// holds the lease at a time — enforced by the supervisor's baton sequencing.
|
|
597
|
+
const acquireLease = async () => {
|
|
598
|
+
if (hasLease) return;
|
|
599
|
+
// M4: re-read the on-disk state generation NOW. A reload's old worker bumped
|
|
600
|
+
// it during its final flush; without this re-sync the new primary's
|
|
601
|
+
// saveState(expectedGeneration=<boot N>) would be refused for its whole
|
|
602
|
+
// tenure (quota persistence wedged forever). As sole writer it safely adopts
|
|
603
|
+
// the current on-disk generation.
|
|
604
|
+
try { stateGeneration = await readGeneration(getStatePath()); } catch { /* keep prior */ }
|
|
605
|
+
hasLease = true;
|
|
606
|
+
accountManager.setWriterLease(true);
|
|
607
|
+
if (!quotaSaveInterval) {
|
|
608
|
+
quotaSaveInterval = setInterval(() => { persistQuotaState(); }, 60_000);
|
|
609
|
+
quotaSaveInterval.unref?.();
|
|
610
|
+
}
|
|
611
|
+
prober.start();
|
|
612
|
+
};
|
|
613
|
+
// Stop scheduling writes and flip the lease. Returns a promise that settles
|
|
614
|
+
// once any in-flight prober cycle has finished (so no probe-driven token
|
|
615
|
+
// rotation is still pending). Token-refresh draining is awaited separately in
|
|
616
|
+
// the baton release (drainRefreshes) before the lease is handed off.
|
|
617
|
+
const releaseLease = async () => {
|
|
618
|
+
if (!hasLease) return;
|
|
619
|
+
hasLease = false;
|
|
620
|
+
if (quotaSaveInterval) { clearInterval(quotaSaveInterval); quotaSaveInterval = null; }
|
|
621
|
+
accountManager.setWriterLease(false);
|
|
622
|
+
await prober.stop(); // awaits any in-flight probe cycle (B1)
|
|
623
|
+
};
|
|
624
|
+
|
|
625
|
+
// Abrupt self-restart — the tested fallback path. Cuts in-flight connections;
|
|
626
|
+
// clients retry ~2s. Used when NOT supervised, or when a seamless reload can't
|
|
627
|
+
// be requested. NEVER exit(1) here (that escapes the exit-75 supervisor loop).
|
|
217
628
|
const restartWorkerNow = () => {
|
|
218
629
|
if (draining) return;
|
|
219
630
|
draining = true;
|
|
220
631
|
if (syncTimer) clearInterval(syncTimer);
|
|
221
|
-
|
|
222
|
-
persistQuotaState(); // flush learned quota so the restart restores it
|
|
223
|
-
prober.stop();
|
|
224
|
-
if (tui?.running) tui.stop();
|
|
632
|
+
if (tui?.running) { tui.stop(); }
|
|
225
633
|
console.log('\n[Maxpool] Restarting server now; queued requests will reconnect automatically.');
|
|
226
634
|
server.closeAllConnections?.();
|
|
227
|
-
|
|
635
|
+
// Best-effort: flush quota + settle in-flight refreshes/prober before exit so
|
|
636
|
+
// the respawned worker doesn't boot from a half-written state. Bounded so the
|
|
637
|
+
// abrupt path stays fast even if a write hangs.
|
|
638
|
+
Promise.race([
|
|
639
|
+
(async () => { await persistQuotaState(); await releaseLease(); await flushConfigWrites(); await flushStateWrites(); })(),
|
|
640
|
+
delay(2000),
|
|
641
|
+
]).finally(() => process.exit(SERVER_RESTART_EXIT_CODE));
|
|
642
|
+
};
|
|
643
|
+
|
|
644
|
+
// Seamless reload entry point. When supervised, ask the supervisor to run the
|
|
645
|
+
// baton (spawn new worker, hand off socket + lease). If anything is off
|
|
646
|
+
// (not supervised, no IPC), fall back to the abrupt restart.
|
|
647
|
+
const requestReload = () => {
|
|
648
|
+
if (draining) return;
|
|
649
|
+
if (supervised) {
|
|
650
|
+
try {
|
|
651
|
+
process.send({ type: MSG_RELOAD_REQUEST });
|
|
652
|
+
return;
|
|
653
|
+
} catch { /* IPC gone — fall through to abrupt restart */ }
|
|
654
|
+
}
|
|
655
|
+
restartWorkerNow();
|
|
228
656
|
};
|
|
229
657
|
|
|
230
658
|
restartController = new RestartController({
|
|
231
659
|
pauseAdmission: () => accountManager.setAdmissionPaused(true),
|
|
232
|
-
restartNow:
|
|
660
|
+
restartNow: requestReload,
|
|
233
661
|
});
|
|
234
662
|
|
|
235
663
|
const shutdownGracefully = (reason, options = {}) => {
|
|
@@ -240,10 +668,16 @@ async function serverWorkerCommand() {
|
|
|
240
668
|
|
|
241
669
|
draining = true;
|
|
242
670
|
if (syncTimer) clearInterval(syncTimer);
|
|
243
|
-
clearInterval(quotaSaveInterval);
|
|
244
|
-
persistQuotaState(); // best-effort final flush of learned quota
|
|
245
|
-
prober.stop();
|
|
246
671
|
if (tui?.running) tui.stop();
|
|
672
|
+
// Best-effort final flush + settle writes (fire-and-forget; the drain timeout
|
|
673
|
+
// below bounds total quit time, and write barriers protect the on-disk state).
|
|
674
|
+
(async () => {
|
|
675
|
+
await releaseLease();
|
|
676
|
+
await accountManager.drainRefreshes();
|
|
677
|
+
await persistQuotaState(true); // force: lease already dropped above (else no-op)
|
|
678
|
+
await flushConfigWrites();
|
|
679
|
+
await flushStateWrites();
|
|
680
|
+
})().catch(() => {});
|
|
247
681
|
|
|
248
682
|
console.log(`\n[Maxpool] Draining shutdown (${reason}).`);
|
|
249
683
|
console.log(`[Maxpool] Stopped accepting new requests; waiting up to ${Math.ceil(drainTimeoutMs / 1000)}s for ${restartController.activeRequests.size} active request(s), then forcing exit. Press Ctrl-C again to force now.`);
|
|
@@ -345,12 +779,12 @@ async function serverWorkerCommand() {
|
|
|
345
779
|
}
|
|
346
780
|
}, syncIntervalMs);
|
|
347
781
|
syncTimer.unref();
|
|
348
|
-
const onListenError = err => handleServerListenError(err, host, port);
|
|
349
|
-
server.once('error', onListenError);
|
|
350
782
|
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
783
|
+
// Become the live primary: start serving UI/logs, take the writer lease, run
|
|
784
|
+
// the update check. `viaTakeover` true means we acquired the socket through the
|
|
785
|
+
// baton (a reload) — freeze the update check so a reload doesn't re-probe npm
|
|
786
|
+
// (only a cold supervisor start self-updates).
|
|
787
|
+
const becomePrimary = async ({ viaTakeover }) => {
|
|
354
788
|
if (tui) {
|
|
355
789
|
if (tui.start()) {
|
|
356
790
|
console.log(`Listening on ${host}:${port} with ${accounts.length} account(s)`);
|
|
@@ -361,15 +795,150 @@ async function serverWorkerCommand() {
|
|
|
361
795
|
} else {
|
|
362
796
|
logPlainServerStart({ host, port, accounts, threshold, config });
|
|
363
797
|
}
|
|
798
|
+
// On a reload TAKEOVER, adopt the latest tokens from disk. The headless
|
|
799
|
+
// reload worker loaded config at spawn time; the OLD worker may have rotated
|
|
800
|
+
// a single-use refresh token DURING the handoff (and flushed it before
|
|
801
|
+
// MSG_RELEASED). Without this re-sync the new worker would present the now-
|
|
802
|
+
// invalidated token on its first refresh → invalid_grant → bricked account.
|
|
803
|
+
if (viaTakeover) {
|
|
804
|
+
try {
|
|
805
|
+
const diskConfig = await loadConfig();
|
|
806
|
+
if (diskConfig) await syncAccountsFromDisk(diskConfig, config, accountManager);
|
|
807
|
+
} catch (err) {
|
|
808
|
+
console.error(`[Maxpool] Token re-sync at takeover failed: ${err.message}`);
|
|
809
|
+
}
|
|
810
|
+
}
|
|
364
811
|
|
|
365
|
-
//
|
|
366
|
-
//
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
812
|
+
// Acquire the lease (re-syncs state generation + enables writes) BEFORE we
|
|
813
|
+
// announce primacy, so the moment the supervisor reaps the old worker we are
|
|
814
|
+
// a fully-functional sole writer.
|
|
815
|
+
await acquireLease();
|
|
816
|
+
|
|
817
|
+
// Tell the supervisor we are the sole acceptor now → it stops its own accept
|
|
818
|
+
// loop (the steady-state race fix). Cold start AND takeover both signal this.
|
|
819
|
+
if (supervised) { try { process.send({ type: MSG_PRIMARY }); } catch { /* ignore */ } }
|
|
820
|
+
|
|
821
|
+
// Reload-storm guard: only a cold (non-reload) lease holder probes npm.
|
|
822
|
+
// A reload-spawned/takeover worker must NEVER re-probe (1x not 2x traffic).
|
|
823
|
+
if (!viaTakeover && !isReloadWorker) {
|
|
824
|
+
if (process.env.MAXPOOL_TEST_LOG_UPDATE_CHECK === '1') console.log('[Maxpool] UPDATE_CHECK_FIRED');
|
|
825
|
+
const notify = msg => (tui?._addLog ? tui._addLog(msg) : console.log(`[Maxpool] ${msg}`));
|
|
826
|
+
maybeCheckForUpdate(config, notify).catch(() => {});
|
|
827
|
+
} else if (process.env.MAXPOOL_TEST_LOG_UPDATE_CHECK === '1') {
|
|
828
|
+
console.log('[Maxpool] UPDATE_CHECK_SKIPPED (reload)');
|
|
829
|
+
}
|
|
830
|
+
};
|
|
831
|
+
|
|
832
|
+
// ── start accepting ──
|
|
833
|
+
if (supervised) {
|
|
834
|
+
// Supervised: NEVER listen(port) directly — the supervisor owns the socket.
|
|
835
|
+
// Wait for the baton over IPC. A cold worker gets MSG_LISTEN; a reload worker
|
|
836
|
+
// boots headless (already done above) and waits for MSG_PROBE_READY/TAKEOVER.
|
|
837
|
+
server.on('error', err => console.error(`[Maxpool] Server error: ${err.message}`));
|
|
838
|
+
|
|
839
|
+
const listenOnHandle = handle => new Promise((resolve, reject) => {
|
|
840
|
+
server.once('error', reject);
|
|
841
|
+
server.listen(handle, () => { server.removeListener('error', reject); resolve(); });
|
|
842
|
+
});
|
|
843
|
+
|
|
844
|
+
// Baton release: the old primary gives up acceptance + the writer lease.
|
|
845
|
+
// MSG_RELEASED MUST mean "no write is in flight" — not merely "flag flipped"
|
|
846
|
+
// — so the new worker can acquire the lease and rotate the single-use refresh
|
|
847
|
+
// token WITHOUT racing a still-pending rotation here (B1) and boots from a
|
|
848
|
+
// config that already has any rotated token persisted (M3).
|
|
849
|
+
const releaseBatonAndDrain = async () => {
|
|
850
|
+
if (draining) { try { process.send({ type: MSG_RELEASED }); } catch { /* ignore */ } return; }
|
|
851
|
+
draining = true;
|
|
852
|
+
if (syncTimer) clearInterval(syncTimer);
|
|
853
|
+
// Stop accepting NEW connections; KEEP in-flight requests alive.
|
|
854
|
+
server.maxpoolBeginDrain?.();
|
|
855
|
+
server.close(() => {});
|
|
856
|
+
server.closeIdleConnections?.(); // retire idle keep-alive sockets now
|
|
857
|
+
|
|
858
|
+
// 1) Stop scheduling writes + flip the lease + await any in-flight prober
|
|
859
|
+
// cycle (releaseLease awaits prober.stop()).
|
|
860
|
+
await releaseLease();
|
|
861
|
+
// 2) Await every in-flight OAuth token refresh that started before the lease
|
|
862
|
+
// dropped — the headline brick hazard (B1).
|
|
863
|
+
await accountManager.drainRefreshes();
|
|
864
|
+
// 3) Final flush of learned quota. FORCE it: releaseLease() above already
|
|
865
|
+
// flipped hasLease=false, and persistQuotaState no-ops without the lease —
|
|
866
|
+
// so an unforced call here silently drops the reload's final quota flush.
|
|
867
|
+
await persistQuotaState(true);
|
|
868
|
+
// 4) Barrier: ensure every queued config write (a rotated-token persist is
|
|
869
|
+
// fire-and-forget) AND state write has actually hit disk before we hand
|
|
870
|
+
// off (M3 — otherwise the new worker boots from the invalidated token).
|
|
871
|
+
await flushConfigWrites();
|
|
872
|
+
await flushStateWrites();
|
|
873
|
+
|
|
874
|
+
if (tui?.running) tui.stop(); // restore terminal before the new worker takes it
|
|
875
|
+
try { process.send({ type: MSG_RELEASED }); } catch { /* ignore */ }
|
|
876
|
+
|
|
877
|
+
// Drain bounded in-flight on EXISTING access tokens (no refresh needed),
|
|
878
|
+
// then exit(0). The supervisor SIGKILLs us if we outlive its drain cap.
|
|
879
|
+
const waitForDrain = () => {
|
|
880
|
+
if (restartController.activeRequests.size === 0) { process.exit(0); return; }
|
|
881
|
+
};
|
|
882
|
+
const drainPoll = setInterval(waitForDrain, 200);
|
|
883
|
+
drainPoll.unref?.();
|
|
884
|
+
const hardCap = setTimeout(() => {
|
|
885
|
+
console.error(`[Maxpool] Released worker drain cap reached with ${restartController.activeRequests.size} active; exiting.`);
|
|
886
|
+
process.exit(0);
|
|
887
|
+
}, drainTimeoutMs);
|
|
888
|
+
hardCap.unref?.();
|
|
889
|
+
waitForDrain();
|
|
890
|
+
};
|
|
891
|
+
|
|
892
|
+
process.on('message', async (msg, handle) => {
|
|
893
|
+
try {
|
|
894
|
+
if (msg?.type === MSG_LISTEN && handle) {
|
|
895
|
+
// Cold start: take the socket and go primary immediately.
|
|
896
|
+
await listenOnHandle(handle);
|
|
897
|
+
await becomePrimary({ viaTakeover: false });
|
|
898
|
+
} else if (msg?.type === MSG_PROBE_READY) {
|
|
899
|
+
// Headless reload worker: we've booted the new code and restored quota
|
|
900
|
+
// in memory. Confirm readiness (we are NOT yet accepting / writing).
|
|
901
|
+
// Test hook: simulate a new-version that fails to boot → forces the
|
|
902
|
+
// supervisor's rollback (old worker stays primary, zero disruption).
|
|
903
|
+
if (process.env.MAXPOOL_TEST_FAIL_RELOAD_WORKER === '1') {
|
|
904
|
+
process.send({ type: MSG_FAILED, reason: 'test-forced failure' });
|
|
905
|
+
} else {
|
|
906
|
+
process.send({ type: MSG_READY });
|
|
907
|
+
}
|
|
908
|
+
} else if (msg?.type === MSG_TAKEOVER && handle) {
|
|
909
|
+
// Baton acquire: the old worker already released, so we're the SOLE
|
|
910
|
+
// acceptor. Start accepting + take the writer lease + TUI. becomePrimary
|
|
911
|
+
// sends MSG_PRIMARY (the baton waits on it).
|
|
912
|
+
await listenOnHandle(handle);
|
|
913
|
+
await becomePrimary({ viaTakeover: true });
|
|
914
|
+
} else if (msg?.type === MSG_RELEASE) {
|
|
915
|
+
// Baton release: stop accepting NEW (keep in-flight), retire keep-alive
|
|
916
|
+
// sockets with Connection: close, stop writing, flush once, drop TUI.
|
|
917
|
+
await releaseBatonAndDrain();
|
|
918
|
+
}
|
|
919
|
+
} catch (err) {
|
|
920
|
+
// A failed reload worker must NOT exit(1) (kills the supervisor loop).
|
|
921
|
+
// Report failure so the supervisor rolls back; stay alive harmlessly.
|
|
922
|
+
console.error(`[Maxpool] Reload worker error: ${err.message}`);
|
|
923
|
+
try { process.send({ type: MSG_FAILED, reason: err.message }); } catch { /* ignore */ }
|
|
924
|
+
}
|
|
925
|
+
});
|
|
926
|
+
} else {
|
|
927
|
+
// Direct (non-TTY service) path: bind the port ourselves, exactly as before.
|
|
928
|
+
const onListenError = err => handleServerListenError(err, host, port);
|
|
929
|
+
server.once('error', onListenError);
|
|
930
|
+
server.listen(port, host, () => {
|
|
931
|
+
server.removeListener('error', onListenError);
|
|
932
|
+
server.on('error', err => console.error(`[Maxpool] Server error: ${err.message}`));
|
|
933
|
+
becomePrimary({ viaTakeover: false }).catch(err => console.error(`[Maxpool] ${err.message}`));
|
|
934
|
+
});
|
|
935
|
+
}
|
|
370
936
|
|
|
371
937
|
process.on('SIGINT', () => shutdownGracefully('SIGINT'));
|
|
372
938
|
process.on('SIGTERM', () => shutdownGracefully('SIGTERM'));
|
|
939
|
+
// SIGHUP requests a seamless reload (the conventional "reload" signal). Under
|
|
940
|
+
// the supervisor this runs the baton; otherwise it falls back to exit-75.
|
|
941
|
+
process.on('SIGHUP', () => requestReload());
|
|
373
942
|
}
|
|
374
943
|
|
|
375
944
|
function logPlainServerStart({ host, port, accounts, threshold, config }) {
|
|
@@ -607,7 +1176,7 @@ async function accountsCommand() {
|
|
|
607
1176
|
a.refreshToken = newTokens.refreshToken;
|
|
608
1177
|
a.expiresAt = newTokens.expiresAt;
|
|
609
1178
|
configDirty = true;
|
|
610
|
-
} catch
|
|
1179
|
+
} catch {
|
|
611
1180
|
// refresh failed — fetchProfile will report the specific error
|
|
612
1181
|
}
|
|
613
1182
|
}));
|
package/src/oauth.js
CHANGED
|
@@ -4,7 +4,10 @@ import { createInterface } from 'node:readline';
|
|
|
4
4
|
import http from 'node:http';
|
|
5
5
|
|
|
6
6
|
const PROFILE_URL = 'https://api.anthropic.com/api/oauth/profile';
|
|
7
|
-
|
|
7
|
+
// Token endpoint is overridable via env so a stub OAuth server can be pointed at
|
|
8
|
+
// for integration tests (e.g. the single-use-rotating-token reload torture test).
|
|
9
|
+
const DEFAULT_TOKEN_ENDPOINT = process.env.MAXPOOL_OAUTH_TOKEN_ENDPOINT
|
|
10
|
+
|| 'https://platform.claude.com/v1/oauth/token';
|
|
8
11
|
const DEFAULT_CLIENT_ID = '9d1c250a-e61b-44d9-88ed-5944d1962f5e';
|
|
9
12
|
|
|
10
13
|
/**
|
package/src/prober.js
CHANGED
|
@@ -40,20 +40,29 @@ export class Prober {
|
|
|
40
40
|
}
|
|
41
41
|
}
|
|
42
42
|
|
|
43
|
-
|
|
43
|
+
/** Stop scheduling AND await any probe cycle already in flight, so a caller
|
|
44
|
+
* (the baton release) can be sure no probe-driven token rotation is pending
|
|
45
|
+
* before it hands the writer lease to another worker. */
|
|
46
|
+
async stop() {
|
|
44
47
|
if (this.timer) { clearInterval(this.timer); this.timer = null; }
|
|
48
|
+
if (this._inflight) { try { await this._inflight; } catch { /* swallow */ } }
|
|
45
49
|
}
|
|
46
50
|
|
|
47
|
-
/** Probe every OAuth account once. Overlapping cycles are skipped.
|
|
48
|
-
|
|
49
|
-
|
|
51
|
+
/** Probe every OAuth account once. Overlapping cycles are skipped. The active
|
|
52
|
+
* cycle is tracked on `_inflight` so stop() can await it. */
|
|
53
|
+
probeAll() {
|
|
54
|
+
if (this._running) return this._inflight || Promise.resolve();
|
|
50
55
|
this._running = true;
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
56
|
+
this._inflight = (async () => {
|
|
57
|
+
try {
|
|
58
|
+
const accounts = this.am.accounts.filter(a => a.type === 'oauth' && a.credential);
|
|
59
|
+
await Promise.all(accounts.map(a => this.probeOne(a)));
|
|
60
|
+
} finally {
|
|
61
|
+
this._running = false;
|
|
62
|
+
this._inflight = null;
|
|
63
|
+
}
|
|
64
|
+
})();
|
|
65
|
+
return this._inflight;
|
|
57
66
|
}
|
|
58
67
|
|
|
59
68
|
async probeOne(account) {
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
// IPC message vocabulary + the supervisor-side baton orchestration for the
|
|
2
|
+
// near-zero-downtime reload (issue #6). Pure logic, no process side-effects, so
|
|
3
|
+
// the baton sequence is unit-testable against fake workers.
|
|
4
|
+
//
|
|
5
|
+
// Roles:
|
|
6
|
+
// Supervisor — owns the listening socket for life, never closes it, spawns
|
|
7
|
+
// workers and hands the socket HANDLE to exactly one acceptor at a time.
|
|
8
|
+
// Worker — accepts on the shared handle; exactly one worker holds the
|
|
9
|
+
// WRITER LEASE (refresh/probe/persist) at any instant.
|
|
10
|
+
//
|
|
11
|
+
// The baton (single-writer guarantee): old worker RELEASES (stops accepting +
|
|
12
|
+
// stops writing, flushes once) BEFORE the new worker ACQUIRES. Reads overlap;
|
|
13
|
+
// writes never do.
|
|
14
|
+
|
|
15
|
+
// Supervisor → worker
|
|
16
|
+
export const MSG_LISTEN = 'listen'; // (with handle) accept on this socket + take TUI/lease
|
|
17
|
+
export const MSG_RELEASE = 'release'; // stop accepting, stop writing, flush, give up TUI
|
|
18
|
+
export const MSG_TAKEOVER = 'takeover'; // (with handle) start accepting + acquire writer lease + TUI
|
|
19
|
+
export const MSG_PROBE_READY = 'probe-ready'; // ask a headless worker to confirm it booted OK
|
|
20
|
+
|
|
21
|
+
// Worker → supervisor
|
|
22
|
+
export const MSG_RELOAD_REQUEST = 'reload-request'; // primary worker asks for a seamless reload
|
|
23
|
+
export const MSG_READY = 'ready'; // headless worker booted new code successfully
|
|
24
|
+
export const MSG_FAILED = 'failed'; // worker failed to boot/bind/restore
|
|
25
|
+
export const MSG_RELEASED = 'released'; // old worker stopped accepting + writing, flushed
|
|
26
|
+
export const MSG_PRIMARY = 'primary'; // new worker is now sole acceptor + lease holder
|
|
27
|
+
export const MSG_QUOTA_STATE = 'quota-state'; // old worker hands its in-memory quota to supervisor
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Outcome codes for a reload attempt, surfaced to the supervisor so it can log
|
|
31
|
+
* and decide fallback. SWAPPED = fully cut over; ROLLED_BACK = old worker stays
|
|
32
|
+
* primary (no disruption); FALLBACK = abrupt exit-75 restart path taken.
|
|
33
|
+
*/
|
|
34
|
+
export const RELOAD_SWAPPED = 'swapped';
|
|
35
|
+
export const RELOAD_ROLLED_BACK = 'rolled-back';
|
|
36
|
+
export const RELOAD_FALLBACK = 'fallback';
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Drive the baton handshake over two worker "channels" (anything with
|
|
40
|
+
* .send(msg[,handle]) and an async waitFor(type, timeoutMs) that resolves with
|
|
41
|
+
* the message payload or rejects on timeout/exit). Returns one of the RELOAD_*
|
|
42
|
+
* outcomes. The caller (supervisor) owns spawning, killing, reaping and the
|
|
43
|
+
* actual fallback restart; this function only sequences the protocol and tells
|
|
44
|
+
* the caller what happened.
|
|
45
|
+
*
|
|
46
|
+
* Sequence (matches design step 2a–2e):
|
|
47
|
+
* a. new worker already spawned headless (no lease); we ask it to confirm READY
|
|
48
|
+
* b. READY fails → ROLLED_BACK: caller kills new, old stays primary
|
|
49
|
+
* c. old RELEASE → waits RELEASED (old stopped accepting + writing, flushed)
|
|
50
|
+
* d. new TAKEOVER → waits PRIMARY (new is sole acceptor + lease holder)
|
|
51
|
+
* e. caller drains+reaps old
|
|
52
|
+
*
|
|
53
|
+
* Any throw → FALLBACK (caller does the tested abrupt restart). All-or-nothing:
|
|
54
|
+
* we never leave both accepting or both writing.
|
|
55
|
+
*/
|
|
56
|
+
export async function runReloadBaton({
|
|
57
|
+
oldWorker,
|
|
58
|
+
newWorker,
|
|
59
|
+
handle,
|
|
60
|
+
prepareHandle = async h => h,
|
|
61
|
+
readyTimeoutMs = 10_000,
|
|
62
|
+
releaseTimeoutMs = 20_000,
|
|
63
|
+
takeoverTimeoutMs = 10_000,
|
|
64
|
+
log = () => {},
|
|
65
|
+
}) {
|
|
66
|
+
// (a) Confirm the freshly-spawned headless worker booted the new code.
|
|
67
|
+
let ready;
|
|
68
|
+
try {
|
|
69
|
+
newWorker.send({ type: MSG_PROBE_READY });
|
|
70
|
+
ready = await newWorker.waitFor([MSG_READY, MSG_FAILED], readyTimeoutMs);
|
|
71
|
+
} catch (err) {
|
|
72
|
+
log(`reload: readiness wait failed (${err.message}); rolling back`);
|
|
73
|
+
return RELOAD_ROLLED_BACK;
|
|
74
|
+
}
|
|
75
|
+
// (b) New worker did not come up cleanly → roll back fully, old stays primary.
|
|
76
|
+
if (!ready || ready.type === MSG_FAILED) {
|
|
77
|
+
log(`reload: new worker reported failed (${ready?.reason || 'no ready'}); rolling back`);
|
|
78
|
+
return RELOAD_ROLLED_BACK;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
// (c) Old worker releases: stop accepting NEW, keep in-flight, stop writing,
|
|
82
|
+
// flush config+state ONE final time, drop the TUI. After this point the
|
|
83
|
+
// old worker holds no lease and no acceptor — and (single-writer) no
|
|
84
|
+
// refresh/prober write is still in flight (it awaits drainRefreshes).
|
|
85
|
+
try {
|
|
86
|
+
oldWorker.send({ type: MSG_RELEASE });
|
|
87
|
+
await oldWorker.waitFor([MSG_RELEASED], releaseTimeoutMs);
|
|
88
|
+
} catch (err) {
|
|
89
|
+
// Old worker didn't ack release. We must NOT hand the socket to the new
|
|
90
|
+
// worker (would risk two acceptors) — fall back to the abrupt path.
|
|
91
|
+
log(`reload: old worker did not release (${err.message}); falling back`);
|
|
92
|
+
return RELOAD_FALLBACK;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// The old worker has now freed the listening fd. Re-arm the supervisor's own
|
|
96
|
+
// acceptor over the cutover gap (covers new conns in the OS backlog) and get a
|
|
97
|
+
// fresh re-sendable handle for the new worker.
|
|
98
|
+
let liveHandle = handle;
|
|
99
|
+
try {
|
|
100
|
+
liveHandle = await prepareHandle(handle);
|
|
101
|
+
} catch (err) {
|
|
102
|
+
log(`reload: could not re-arm listening socket (${err.message}); falling back`);
|
|
103
|
+
return RELOAD_FALLBACK;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// (d) New worker takes over: it becomes the SOLE acceptor (old already closed)
|
|
107
|
+
// and ACQUIRES the writer lease (refresh/probe/persist re-enabled) + TUI.
|
|
108
|
+
try {
|
|
109
|
+
newWorker.send({ type: MSG_TAKEOVER }, liveHandle);
|
|
110
|
+
await newWorker.waitFor([MSG_PRIMARY], takeoverTimeoutMs);
|
|
111
|
+
} catch (err) {
|
|
112
|
+
// The new worker failed to start accepting after the old already released.
|
|
113
|
+
// Neither is accepting now — fall back so the supervisor force-restarts and
|
|
114
|
+
// a worker comes back up on the supervisor-owned socket (queued conns drain).
|
|
115
|
+
log(`reload: new worker did not take over (${err.message}); falling back`);
|
|
116
|
+
return RELOAD_FALLBACK;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
log('reload: cutover complete; new worker is primary');
|
|
120
|
+
return RELOAD_SWAPPED;
|
|
121
|
+
}
|
package/src/server.js
CHANGED
|
@@ -51,8 +51,25 @@ export function createProxyServer(accountManager, config, hooks = {}) {
|
|
|
51
51
|
mkdir(logDir, { recursive: true }).catch(() => {});
|
|
52
52
|
}
|
|
53
53
|
|
|
54
|
+
// Reload drain flag. When the worker releases the baton it sets this so every
|
|
55
|
+
// remaining response carries `Connection: close`, retiring the client's
|
|
56
|
+
// keep-alive socket instead of letting it pipeline a NEW request onto a worker
|
|
57
|
+
// that's shutting down. `connection` is hop-by-hop so it's stripped from the
|
|
58
|
+
// upstream response headers — a setHeader here survives the later writeHead.
|
|
59
|
+
let draining = false;
|
|
60
|
+
|
|
61
|
+
// Identifies the WORKER process that served a response — proves the supervisor
|
|
62
|
+
// (which holds the socket but does not serve) never swallowed the request. Set
|
|
63
|
+
// before any writeHead; `x-maxpool-*` is informative-only and stripped from
|
|
64
|
+
// upstream-bound request headers elsewhere.
|
|
65
|
+
const workerStamp = String(process.pid);
|
|
66
|
+
|
|
54
67
|
const server = http.createServer(async (req, res) => {
|
|
55
68
|
try {
|
|
69
|
+
try { res.setHeader('x-maxpool-worker', workerStamp); } catch { /* headers sent */ }
|
|
70
|
+
if (draining) {
|
|
71
|
+
try { res.setHeader('Connection', 'close'); } catch { /* headers may be sent */ }
|
|
72
|
+
}
|
|
56
73
|
// Auth check — skip for localhost connections
|
|
57
74
|
const clientKey = req.headers['x-api-key'];
|
|
58
75
|
const remoteAddr = req.socket.remoteAddress;
|
|
@@ -187,6 +204,11 @@ export function createProxyServer(accountManager, config, hooks = {}) {
|
|
|
187
204
|
}
|
|
188
205
|
});
|
|
189
206
|
|
|
207
|
+
// Begin reload drain: every subsequent response gets `Connection: close` so
|
|
208
|
+
// keep-alive clients retire their socket and don't pipeline a new request onto
|
|
209
|
+
// this releasing worker.
|
|
210
|
+
server.maxpoolBeginDrain = () => { draining = true; };
|
|
211
|
+
|
|
190
212
|
return server;
|
|
191
213
|
}
|
|
192
214
|
|