maxpool 1.3.1 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.js CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  import { spawn, spawnSync } from 'node:child_process';
4
4
  import { createInterface } from 'node:readline';
5
- import { loadOrCreateConfig, loadConfig, saveConfig, atomicConfigUpdate, getConfigPath, loadState, saveState } from './config.js';
5
+ import { loadOrCreateConfig, loadConfig, saveConfig, atomicConfigUpdate, getConfigPath, loadState, saveState, getStatePath, readGeneration, flushConfigWrites, flushStateWrites } from './config.js';
6
6
  import { AccountManager } from './account-manager.js';
7
7
  import { createProxyServer } from './server.js';
8
8
  import { Prober } from './prober.js';
@@ -11,11 +11,20 @@ import { TUI } from './tui.js';
11
11
  import { RestartController } from './restart-controller.js';
12
12
  import { resolveAccounts } from './account-config.js';
13
13
  import { maybeCheckForUpdate } from './updater.js';
14
+ import {
15
+ runReloadBaton,
16
+ RELOAD_SWAPPED, RELOAD_ROLLED_BACK,
17
+ MSG_LISTEN, MSG_RELEASE, MSG_TAKEOVER, MSG_PROBE_READY,
18
+ MSG_RELOAD_REQUEST, MSG_READY, MSG_FAILED, MSG_RELEASED, MSG_PRIMARY,
19
+ } from './reload-protocol.js';
14
20
 
15
21
  const args = process.argv.slice(2);
16
22
  const command = args[0];
17
23
  const SERVER_RESTART_EXIT_CODE = 75;
18
24
  const SERVER_WORKER_ENV = 'MAXPOOL_SERVER_WORKER';
25
+ // Set by the supervisor when it spawns a worker for a seamless reload: that
26
+ // worker boots HEADLESS (plain logs, no writer lease) and waits for the baton.
27
+ const SERVER_RELOAD_WORKER_ENV = 'MAXPOOL_RELOAD_WORKER';
19
28
 
20
29
  switch (command) {
21
30
  case 'server':
@@ -71,43 +80,329 @@ switch (command) {
71
80
  // ── server ──────────────────────────────────────────────────
72
81
 
73
82
  async function serverCommand() {
74
- if (
75
- process.env[SERVER_WORKER_ENV] === '1' ||
76
- !process.stdout.isTTY ||
77
- !process.stdin.isTTY
78
- ) {
83
+ // A spawned worker (env flag set by the supervisor) runs the proxy itself.
84
+ if (process.env[SERVER_WORKER_ENV] === '1') {
79
85
  return serverWorkerCommand();
80
86
  }
81
87
 
82
- // Keep a stable foreground process attached to the shell. The worker can
83
- // then request a restart without orphaning its replacement or losing TTY IO.
84
- const ignoreTerminalSignal = () => {};
85
- process.on('SIGINT', ignoreTerminalSignal);
86
- process.on('SIGTERM', ignoreTerminalSignal);
88
+ // Non-TTY (e.g. `maxpool server` as a background service): keep the existing
89
+ // tested direct-listen path. The seamless-reload feature is only active under
90
+ // the TTY supervisor; a service manager already handles restart/respawn.
91
+ // MAXPOOL_FORCE_SUPERVISOR=1 forces the supervisor path without a TTY (used by
92
+ // the reload integration tests; the worker still runs plain-log without a TTY).
93
+ const forceSupervisor = process.env.MAXPOOL_FORCE_SUPERVISOR === '1';
94
+ if (!forceSupervisor && (!process.stdout.isTTY || !process.stdin.isTTY)) {
95
+ return serverWorkerCommand();
96
+ }
97
+
98
+ return supervisorCommand();
99
+ }
100
+
101
+ // ── supervisor (TTY) ─────────────────────────────────────────
102
+ //
103
+ // Owns the listening socket for its whole life (never closes it) and hands the
104
+ // socket HANDLE to exactly one worker at a time over IPC. A worker requests a
105
+ // seamless reload; the supervisor spawns a fresh headless worker, runs the
106
+ // single-writer baton, then swaps. Any failure falls back to the tested abrupt
107
+ // exit-75 respawn. A crash-loop degrades to loud single-worker failure via an
108
+ // exponential backoff, never a tight fork loop.
109
+
110
+ async function supervisorCommand() {
111
+ const { createServer } = await import('node:net');
112
+ const config = await loadOrCreateConfig();
113
+ const port = config.proxy.port;
114
+ const host = config.proxy.host || '127.0.0.1';
115
+
116
+ // Bind once. EADDRINUSE / EACCES here is a cold-start failure → exit(1) is
117
+ // correct (there's no worker to keep alive yet).
118
+ let masterServer = createServer();
119
+ // We DROP any connection the supervisor accidentally accepts while a worker is
120
+ // also accepting on the shared handle — but the design avoids that: the
121
+ // supervisor's acceptor is only LIVE during the brief cutover gap (it stops
122
+ // once a worker confirms it is the sole acceptor). A bare handler is required
123
+ // so the rare gap-accepted socket isn't left dangling.
124
+ const relistenMaster = () => new Promise((resolve, reject) => {
125
+ if (masterServer.listening) { resolve(); return; }
126
+ const onErr = err => { masterServer.removeListener('listening', onListen); reject(err); };
127
+ const onListen = () => { masterServer.removeListener('error', onErr); resolve(); };
128
+ masterServer.once('error', onErr);
129
+ masterServer.once('listening', onListen);
130
+ masterServer.listen(port, host);
131
+ });
132
+ const closeMasterAccept = () => new Promise(resolve => {
133
+ if (!masterServer.listening) { resolve(); return; }
134
+ masterServer.close(() => resolve());
135
+ });
136
+ try {
137
+ await relistenMaster();
138
+ } catch (err) {
139
+ handleServerListenError(err, host, port);
140
+ return;
141
+ }
142
+
143
+ // Keep the supervisor attached to the shell. The worker shares the supervisor's
144
+ // process group, so a terminal Ctrl-C (SIGINT/SIGTERM) is delivered by the TTY
145
+ // to BOTH already — the supervisor must NOT forward those or the worker gets a
146
+ // doubled signal (the "second Ctrl-C force-quits" footgun). The supervisor
147
+ // ignores SIGINT/SIGTERM itself (the worker drains + exits, ending the turn).
148
+ let activeWorker = null;
149
+ const forwardSignal = sig => { try { activeWorker?.child.kill(sig); } catch { /* ignore */ } };
150
+ process.on('SIGINT', () => { /* delivered to the worker by the TTY group */ });
151
+ process.on('SIGTERM', () => { /* delivered to the worker by the TTY group */ });
152
+ // SIGHUP (from `kill -HUP <supervisor-pid>`) reaches ONLY the supervisor →
153
+ // forward it so the worker requests a seamless reload.
154
+ process.on('SIGHUP', () => forwardSignal('SIGHUP'));
155
+ // A spawn failure / stray rejection must NOT kill the supervisor (it would
156
+ // wedge the port and drop the service). Log and let the supervision loop or
157
+ // the reload's own error handling recover.
158
+ process.on('uncaughtException', err => {
159
+ console.error(`[Maxpool] Supervisor uncaughtException (continuing): ${err?.stack || err}`);
160
+ });
161
+ process.on('unhandledRejection', reason => {
162
+ console.error(`[Maxpool] Supervisor unhandledRejection (continuing): ${reason}`);
163
+ });
164
+
165
+ // After SIGKILLing a worker that may have owned the TUI, the worker had no
166
+ // chance to restore the terminal — the supervisor emits the restore itself
167
+ // (exit alt-screen + show cursor + raw off) so the user's shell is clean.
168
+ const restoreTerminalFromSupervisor = () => {
169
+ try {
170
+ if (process.stdout.isTTY) process.stdout.write('\x1b[?25h\x1b[?1049l');
171
+ if (process.stdin.isTTY) { try { process.stdin.setRawMode(false); } catch { /* ignore */ } }
172
+ } catch { /* never throw */ }
173
+ };
174
+ process.on('exit', restoreTerminalFromSupervisor);
175
+
176
+ // Crash-loop guard: count rapid consecutive non-restart exits and back off so
177
+ // a worker that crashes on boot doesn't spin the CPU forking. A clean run for
178
+ // a while resets the counter.
179
+ let crashCount = 0;
180
+ const CRASH_WINDOW_MS = 10_000;
181
+ const MAX_BACKOFF_MS = 8_000;
182
+
183
+ let reloadInFlight = false;
184
+ // Resolver for the CURRENT supervision turn. A swap re-points monitoring to
185
+ // the new worker WITHOUT ending the turn; only an exit/fallback resolves it.
186
+ let endTurn = null;
187
+
188
+ // Spawn a worker in the SAME process group as the supervisor. A terminal
189
+ // Ctrl-C (SIGINT/SIGTERM) is delivered by the TTY to the whole foreground
190
+ // group, so BOTH already receive it — the supervisor therefore does NOT
191
+ // forward those (that would double-deliver). Same-group also means a group
192
+ // SIGKILL of the supervisor reaps the worker (no orphan holding the port).
193
+ const spawnWorker = ({ reload = false } = {}) => {
194
+ const env = { ...process.env, [SERVER_WORKER_ENV]: '1' };
195
+ if (reload) env[SERVER_RELOAD_WORKER_ENV] = '1';
196
+ const child = spawn(process.execPath, process.argv.slice(1), {
197
+ cwd: process.cwd(),
198
+ env,
199
+ stdio: ['inherit', 'inherit', 'inherit', 'ipc'],
200
+ });
201
+ const worker = makeWorkerChannel(child);
202
+ if (!reload) {
203
+ // Cold start: hand the socket and tell it to go primary immediately.
204
+ worker.send({ type: MSG_LISTEN }, masterServer);
205
+ }
206
+ return worker;
207
+ };
208
+
209
+ // Wire a worker as the active primary: its RELOAD_REQUEST triggers the baton;
210
+ // its MSG_PRIMARY means it is now the sole acceptor (so the supervisor stops
211
+ // its own competing accept loop); its exit/spawn-error ends the turn.
212
+ const monitorAsActive = worker => {
213
+ activeWorker = worker;
214
+ worker.child.removeAllListeners('message');
215
+ worker.child.on('message', msg => {
216
+ if (msg?.type === MSG_RELOAD_REQUEST) orchestrateReload().catch(() => {});
217
+ else if (msg?.type === MSG_PRIMARY && activeWorker === worker) {
218
+ // The worker is now sole acceptor on the handle → stop the supervisor
219
+ // racing it for accepts (the steady-state ~78%-hang bug). The supervisor
220
+ // still HOLDS the socket via the worker's fd; it re-arms only at reload.
221
+ closeMasterAccept().catch(() => {});
222
+ }
223
+ });
224
+ worker.child.once('exit', (code, signal) => {
225
+ // Only the worker that is STILL active when it exits ends the turn. A
226
+ // reaped old worker (already swapped out) exiting must be ignored here.
227
+ if (activeWorker === worker) endTurn?.({ code, signal });
228
+ });
229
+ // A spawn-time failure (EMFILE/EAGAIN under load) surfaces as 'error', not
230
+ // 'exit' — treat it as a crashed turn so backoff handles it (M5).
231
+ worker.child.once('error', err => {
232
+ if (activeWorker === worker) endTurn?.({ code: 1, signal: null, spawnError: err.message });
233
+ });
234
+ };
235
+
236
+ // Reload orchestration: spawn a fresh headless worker, run the baton, swap.
237
+ const orchestrateReload = async () => {
238
+ if (reloadInFlight) return;
239
+ reloadInFlight = true;
240
+ const oldWorker = activeWorker;
241
+ let newWorker = null;
242
+ try {
243
+ newWorker = spawnWorker({ reload: true });
244
+ // Hold the new worker's spawn-error so a fork failure mid-baton becomes a
245
+ // clean ROLLED_BACK/FALLBACK instead of crashing the supervisor (M5).
246
+ newWorker.child.once('error', () => { /* surfaced via waitFor reject */ });
247
+ const outcome = await runReloadBaton({
248
+ oldWorker,
249
+ newWorker,
250
+ handle: masterServer,
251
+ // After the OLD worker has released its fd, re-arm the supervisor's own
252
+ // listener to cover the cutover gap, then hand THAT live handle to the
253
+ // new worker. The new worker's MSG_PRIMARY then closes it again.
254
+ prepareHandle: async () => { await relistenMaster(); return masterServer; },
255
+ log: msg => console.log(`[Maxpool] ${msg}`),
256
+ });
257
+
258
+ if (outcome === RELOAD_SWAPPED) {
259
+ // Re-point monitoring to the new worker (it's now primary), THEN reap the
260
+ // old one. Order matters: monitorAsActive sets activeWorker=new so the
261
+ // old worker's pending exit handler no-ops. monitorAsActive also handles
262
+ // the new worker's already-sent MSG_PRIMARY is moot here — the new worker
263
+ // sent PRIMARY during the baton; re-assert the master-accept close.
264
+ monitorAsActive(newWorker);
265
+ closeMasterAccept().catch(() => {});
266
+ reapOldWorker(oldWorker, config);
267
+ return;
268
+ }
269
+
270
+ if (outcome === RELOAD_ROLLED_BACK) {
271
+ // Old worker never released — it's still fully primary. Kill the new
272
+ // headless worker; nothing else changed. ZERO disruption.
273
+ try { newWorker.child.kill('SIGKILL'); } catch { /* ignore */ }
274
+ return;
275
+ }
276
+
277
+ // FALLBACK: the old worker may have released; neither is reliably primary.
278
+ // Kill both and let the respawn loop bring a fresh primary up on the
279
+ // supervisor-owned socket. Queued conns sit in the OS backlog meanwhile.
280
+ try { newWorker.child.kill('SIGKILL'); } catch { /* ignore */ }
281
+ try { oldWorker.child.kill('SIGKILL'); } catch { /* ignore */ }
282
+ restoreTerminalFromSupervisor(); // SIGKILLed workers can't restore the TUI
283
+ activeWorker = null;
284
+ endTurn?.({ code: null, signal: 'SIGKILL', fallback: true });
285
+ } catch (err) {
286
+ console.error(`[Maxpool] Reload error: ${err.message}; falling back to abrupt restart`);
287
+ try { newWorker?.child.kill('SIGKILL'); } catch { /* ignore */ }
288
+ try { oldWorker.child.kill('SIGKILL'); } catch { /* ignore */ }
289
+ restoreTerminalFromSupervisor();
290
+ activeWorker = null;
291
+ endTurn?.({ code: null, signal: 'SIGKILL', fallback: true });
292
+ } finally {
293
+ reloadInFlight = false;
294
+ }
295
+ };
296
+
297
+ // One supervision turn: monitor `worker` until it exits (or a fallback forces a
298
+ // respawn). Swaps re-point monitoring without resolving. Resolves exit info.
299
+ const superviseTurn = worker => new Promise(resolve => {
300
+ endTurn = info => { endTurn = null; resolve(info); };
301
+ monitorAsActive(worker);
302
+ });
303
+
87
304
  try {
88
305
  while (true) {
89
- const result = await runServerWorker();
90
- if (result.code === SERVER_RESTART_EXIT_CODE) continue;
306
+ const worker = spawnWorker({ reload: false });
307
+ const startedAt = Date.now();
308
+ const result = await superviseTurn(worker);
309
+
310
+ if (result.fallback) {
311
+ // Fallback swap killed both workers; bring a fresh primary straight back.
312
+ crashCount = 0;
313
+ continue;
314
+ }
315
+ if (result.code === SERVER_RESTART_EXIT_CODE) {
316
+ // Abrupt self-restart (worker-initiated exit-75 fallback path).
317
+ crashCount = 0;
318
+ continue;
319
+ }
320
+
321
+ // Non-restart exit. If it died fast, it's likely crash-looping on boot.
322
+ const ranFor = Date.now() - startedAt;
323
+ if (ranFor < CRASH_WINDOW_MS && (result.code ?? 1) !== 0) {
324
+ crashCount++;
325
+ const backoff = Math.min(MAX_BACKOFF_MS, 250 * 2 ** (crashCount - 1));
326
+ console.error(`[Maxpool] Worker exited (code ${result.code}, signal ${result.signal}) after ${ranFor}ms — crash #${crashCount}. Backing off ${backoff}ms before respawn.`);
327
+ await delay(backoff);
328
+ continue;
329
+ }
330
+
331
+ // Clean shutdown (q / signal). Propagate the exit code and stop.
91
332
  if (result.signal) process.exitCode = 1;
92
333
  else process.exitCode = result.code ?? 1;
93
334
  return;
94
335
  }
95
336
  } finally {
96
- process.off('SIGINT', ignoreTerminalSignal);
97
- process.off('SIGTERM', ignoreTerminalSignal);
337
+ try { masterServer.close(); } catch { /* ignore */ }
98
338
  }
99
339
  }
100
340
 
101
- function runServerWorker() {
102
- return new Promise((resolve, reject) => {
103
- const child = spawn(process.execPath, process.argv.slice(1), {
104
- cwd: process.cwd(),
105
- env: { ...process.env, [SERVER_WORKER_ENV]: '1' },
106
- stdio: 'inherit',
107
- });
108
- child.once('error', reject);
109
- child.once('exit', (code, signal) => resolve({ code, signal }));
110
- });
341
+ // Drain + reap a released worker. The worker exits itself once its bounded
342
+ // in-flight finishes; the supervisor SIGKILLs it if it outlives the drain cap.
343
+ function reapOldWorker(worker, config) {
344
+ if (!worker) return;
345
+ const drainTimeoutMs = Math.max(1000, Number(config.shutdown?.drainTimeoutMs) || 15_000);
346
+ let reaped = false;
347
+ const finish = () => { if (reaped) return; reaped = true; clearTimeout(timer); };
348
+ const timer = setTimeout(() => {
349
+ if (reaped) return;
350
+ console.error(`[Maxpool] Old worker outlived ${Math.ceil(drainTimeoutMs / 1000)}s drain cap; SIGKILL.`);
351
+ try { worker.child.kill('SIGKILL'); } catch { /* ignore */ }
352
+ finish();
353
+ }, drainTimeoutMs);
354
+ timer.unref?.();
355
+ worker.child.once('exit', finish);
356
+ }
357
+
358
+ /**
359
+ * Wrap a child process in a small IPC channel: `.send(msg[,handle])` and an
360
+ * async `.waitFor([types], timeoutMs)` that resolves with the first matching
361
+ * message, or rejects on timeout / premature child exit.
362
+ */
363
+ function makeWorkerChannel(child) {
364
+ return {
365
+ child,
366
+ send(msg, handle) {
367
+ try {
368
+ if (handle) child.send(msg, handle);
369
+ else child.send(msg);
370
+ } catch { /* IPC may be torn down mid-swap; baton timeouts cover it */ }
371
+ },
372
+ waitFor(types, timeoutMs) {
373
+ const want = Array.isArray(types) ? types : [types];
374
+ return new Promise((resolve, reject) => {
375
+ const cleanup = () => {
376
+ clearTimeout(timer);
377
+ child.removeListener('message', onMsg);
378
+ child.removeListener('exit', onExit);
379
+ child.removeListener('error', onError);
380
+ };
381
+ const onMsg = msg => {
382
+ if (msg && want.includes(msg.type)) { cleanup(); resolve(msg); }
383
+ };
384
+ const onExit = (code, signal) => {
385
+ cleanup();
386
+ reject(new Error(`worker exited (code ${code}, signal ${signal}) before ${want.join('/')}`));
387
+ };
388
+ // A spawn failure (EMFILE/EAGAIN) surfaces as 'error', not 'exit' — the
389
+ // baton must see it as a failed step (→ ROLLED_BACK/FALLBACK), not hang.
390
+ const onError = err => { cleanup(); reject(new Error(`worker spawn error before ${want.join('/')}: ${err.message}`)); };
391
+ const timer = setTimeout(() => {
392
+ cleanup();
393
+ reject(new Error(`timed out waiting for ${want.join('/')}`));
394
+ }, timeoutMs);
395
+ timer.unref?.();
396
+ child.on('message', onMsg);
397
+ child.once('exit', onExit);
398
+ child.once('error', onError);
399
+ });
400
+ },
401
+ };
402
+ }
403
+
404
+ function delay(ms) {
405
+ return new Promise(resolve => { const t = setTimeout(resolve, ms); t.unref?.(); });
111
406
  }
112
407
 
113
408
  async function serverWorkerCommand() {
@@ -138,23 +433,67 @@ async function serverWorkerCommand() {
138
433
  config.routing?.preferredAccount,
139
434
  );
140
435
 
436
+ // Supervised = spawned by the TTY supervisor over IPC (handle-based listen +
437
+ // baton). A reload worker boots HEADLESS without the writer lease and waits
438
+ // for the baton; a cold-start worker takes the lease on MSG_LISTEN.
439
+ const supervised = typeof process.send === 'function';
440
+ const isReloadWorker = process.env[SERVER_RELOAD_WORKER_ENV] === '1';
441
+
442
+ // M6: a reload worker boots WITHOUT the writer lease so the AM-level brick
443
+ // guard (ensureTokenFresh no-op) is CLOSED for the entire headless window —
444
+ // before any code path could trigger a refresh. acquireLease() flips it true
445
+ // at takeover. (writerLease defaults true for the standalone/direct path.)
446
+ if (isReloadWorker) accountManager.setWriterLease(false);
447
+
141
448
  // Restore quota observed in a previous run so a restart doesn't lose routing
142
449
  // accuracy and re-probe from scratch. Stale windows clear on first use.
450
+ // A reload worker restores quota IN-MEMORY only (state file handed via the
451
+ // lease holder); a cold/direct worker reads the on-disk state file.
143
452
  const savedState = await loadState();
144
453
  if (savedState?.quota) accountManager.restoreQuotaState(savedState.quota);
145
- const persistQuotaState = () =>
146
- saveState({ quota: accountManager.exportQuotaState() }).catch(() => {});
454
+ // Track the state-file generation we last observed so a stale flush is refused.
455
+ let stateGeneration = Number(savedState?._generation) || 0;
456
+
457
+ // ── single-writer baton: refresh / probe / persistence gated by the lease ──
458
+ // A worker without the lease writes NOTHING (no token rotation, no config
459
+ // write, no state write, no probe). Only the lease holder may write.
460
+ let hasLease = false;
461
+ // `force` performs the FINAL flush during a baton release, AFTER releaseLease()
462
+ // has already flipped hasLease=false (and cleared the periodic interval, so no
463
+ // write races this one). Without force, this no-ops post-release and the reload's
464
+ // learned-quota flush is silently dropped — the new worker would boot from up-to-
465
+ // 60s-stale quota and the state generation would never advance across a reload.
466
+ const persistQuotaState = (force = false) => {
467
+ if (!hasLease && !force) return Promise.resolve();
468
+ // The forced final flush (baton release / shutdown) is the SOLE writer at that
469
+ // point — releaseLease() already cleared the periodic interval and the next
470
+ // worker hasn't acquired the lease yet — so it bypasses the cross-worker
471
+ // stale-generation guard. Without this, a 60s-interval write that fired just
472
+ // before releaseLease could bump the on-disk generation and get the forced
473
+ // flush REFUSED (dropping the final quota snapshot). The guard only exists to
474
+ // serialize CROSS-worker writes; the final flush is provably intra-worker.
475
+ const expectedGeneration = force ? null : stateGeneration;
476
+ return saveState({ quota: accountManager.exportQuotaState() }, { expectedGeneration })
477
+ .then(written => { if (written != null) stateGeneration = written; })
478
+ .catch(() => {});
479
+ };
147
480
  // Persist quota every minute; unref so it never keeps the process alive.
148
- const quotaSaveInterval = setInterval(persistQuotaState, 60_000);
149
- quotaSaveInterval.unref?.();
481
+ let quotaSaveInterval = null;
150
482
 
151
483
  // Opt-in background quota probe (config.quotaProbeSeconds, default 0 = off).
152
484
  const prober = new Prober(accountManager, { intervalMs: (config.quotaProbeSeconds || 0) * 1000 });
153
- prober.start();
154
485
 
155
- // Persist refreshed tokens back to config (re-read from disk to avoid clobbering
156
- // accounts added externally, e.g. by `maxpool import` while server is running)
157
- accountManager.onTokenRefresh((idx, newTokens) => {
486
+ // Persist refreshed tokens back to config. Defense-in-depth: the updater reads
487
+ // the on-disk refresh token and SKIPS the rotation if a fresher writer already
488
+ // advanced it (generation guard), so a stale write can't double-spend a token.
489
+ //
490
+ // We do NOT gate this on `hasLease`: a refresh that already STARTED (it passed
491
+ // the lease gate in ensureTokenFresh) MUST persist its rotated single-use token
492
+ // even if the lease was dropped while its POST was in flight — otherwise the
493
+ // baton hands off and the new worker boots from the now-invalidated on-disk
494
+ // token (B1/M3). New refreshes can't start without the lease (ensureTokenFresh
495
+ // no-ops), so every callback here is from a legitimate lease-era refresh.
496
+ const persistTokenRefresh = (idx, newTokens) => {
158
497
  const account = accountManager.accounts[idx];
159
498
  if (!account) return;
160
499
  // Keep config.accounts in sync so TUI saveConfig doesn't clobber fresh tokens
@@ -178,21 +517,58 @@ async function serverWorkerCommand() {
178
517
  // Match by UUID first, then by name — index may have shifted
179
518
  const cfgIdx = findConfigAccount(diskConfig, account);
180
519
  if (cfgIdx >= 0) {
181
- diskConfig.accounts[cfgIdx].accessToken = newTokens.accessToken;
182
- diskConfig.accounts[cfgIdx].refreshToken = newTokens.refreshToken;
183
- diskConfig.accounts[cfgIdx].expiresAt = newTokens.expiresAt;
520
+ const onDisk = diskConfig.accounts[cfgIdx];
521
+ // Generation guard: if the on-disk refresh token already advanced past
522
+ // the token we rotated FROM, another writer beat us — skip the write so
523
+ // we don't revert a fresher single-use token (the brick-the-account case).
524
+ if (onDisk.refreshToken && onDisk.refreshToken !== account._refreshedFrom &&
525
+ onDisk.refreshToken !== newTokens.refreshToken) {
526
+ return;
527
+ }
528
+ onDisk.accessToken = newTokens.accessToken;
529
+ onDisk.refreshToken = newTokens.refreshToken;
530
+ onDisk.expiresAt = newTokens.expiresAt;
184
531
  }
185
- }).catch(err => console.error(`[Maxpool] Failed to save refreshed token: ${err.message}`));
186
- });
532
+ }).catch(err => {
533
+ if (err?.code === 'STALE_GENERATION') return; // another writer advanced; benign
534
+ console.error(`[Maxpool] Failed to save refreshed token: ${err.message}`);
535
+ });
536
+ };
537
+ accountManager.onTokenRefresh(persistTokenRefresh);
538
+
187
539
  const port = config.proxy.port;
188
540
  const host = config.proxy.host || '127.0.0.1';
189
- const useTUI = process.stdout.isTTY && process.stdin.isTTY;
541
+ // A headless reload worker NEVER drives the TUI (single-owner terminal); it
542
+ // takes the TUI only on baton takeover. Cold/direct workers use it if on a TTY.
543
+ const useTUI = process.stdout.isTTY && process.stdin.isTTY && !isReloadWorker;
190
544
 
191
545
  let tui = null;
192
546
  let server = null;
193
547
  let syncTimer = null;
194
548
  let draining = false;
195
549
  let restartController = null;
550
+
551
+ // Best-effort terminal restore on ANY abnormal exit path (uncaughtException,
552
+ // a bare process.exit, a crash) so the user's shell is never left in raw mode
553
+ // or the alt-screen. Idempotent and safe even when no TUI was running.
554
+ const restoreTerminal = () => {
555
+ try {
556
+ if (tui?.running) { tui.stop(); return; }
557
+ if (process.stdout.isTTY) process.stdout.write('\x1b[?25h\x1b[?1049l');
558
+ if (process.stdin.isTTY) { try { process.stdin.setRawMode(false); } catch { /* ignore */ } }
559
+ } catch { /* never throw from a restore */ }
560
+ };
561
+ process.on('exit', restoreTerminal);
562
+ process.on('uncaughtException', err => {
563
+ console.error(`[Maxpool] Worker uncaughtException: ${err?.stack || err}`);
564
+ restoreTerminal();
565
+ // A reload worker must NEVER exit(1) (escapes the supervisor exit-75 loop).
566
+ // Stay alive so the supervisor's baton timeouts roll us back cleanly.
567
+ if (!isReloadWorker) process.exit(SERVER_RESTART_EXIT_CODE);
568
+ });
569
+ process.on('unhandledRejection', reason => {
570
+ console.error(`[Maxpool] Worker unhandledRejection: ${reason}`);
571
+ });
196
572
  // Quit drains in-flight requests, then force-exits. Kept short so a single
197
573
  // 'q' / Ctrl-C / SIGTERM actually quits under a continuous request flood
198
574
  // (where there are always active requests) instead of waiting indefinitely.
@@ -214,22 +590,74 @@ async function serverWorkerCommand() {
214
590
  },
215
591
  };
216
592
 
593
+ // ── writer lease (single-writer baton) ──
594
+ // Acquiring the lease turns ON token rotation, the quota-save interval, and the
595
+ // prober. Releasing turns them all OFF and flushes once. Exactly one worker
596
+ // holds the lease at a time — enforced by the supervisor's baton sequencing.
597
+ const acquireLease = async () => {
598
+ if (hasLease) return;
599
+ // M4: re-read the on-disk state generation NOW. A reload's old worker bumped
600
+ // it during its final flush; without this re-sync the new primary's
601
+ // saveState(expectedGeneration=<boot N>) would be refused for its whole
602
+ // tenure (quota persistence wedged forever). As sole writer it safely adopts
603
+ // the current on-disk generation.
604
+ try { stateGeneration = await readGeneration(getStatePath()); } catch { /* keep prior */ }
605
+ hasLease = true;
606
+ accountManager.setWriterLease(true);
607
+ if (!quotaSaveInterval) {
608
+ quotaSaveInterval = setInterval(() => { persistQuotaState(); }, 60_000);
609
+ quotaSaveInterval.unref?.();
610
+ }
611
+ prober.start();
612
+ };
613
+ // Stop scheduling writes and flip the lease. Returns a promise that settles
614
+ // once any in-flight prober cycle has finished (so no probe-driven token
615
+ // rotation is still pending). Token-refresh draining is awaited separately in
616
+ // the baton release (drainRefreshes) before the lease is handed off.
617
+ const releaseLease = async () => {
618
+ if (!hasLease) return;
619
+ hasLease = false;
620
+ if (quotaSaveInterval) { clearInterval(quotaSaveInterval); quotaSaveInterval = null; }
621
+ accountManager.setWriterLease(false);
622
+ await prober.stop(); // awaits any in-flight probe cycle (B1)
623
+ };
624
+
625
+ // Abrupt self-restart — the tested fallback path. Cuts in-flight connections;
626
+ // clients retry ~2s. Used when NOT supervised, or when a seamless reload can't
627
+ // be requested. NEVER exit(1) here (that escapes the exit-75 supervisor loop).
217
628
  const restartWorkerNow = () => {
218
629
  if (draining) return;
219
630
  draining = true;
220
631
  if (syncTimer) clearInterval(syncTimer);
221
- clearInterval(quotaSaveInterval);
222
- persistQuotaState(); // flush learned quota so the restart restores it
223
- prober.stop();
224
- if (tui?.running) tui.stop();
632
+ if (tui?.running) { tui.stop(); }
225
633
  console.log('\n[Maxpool] Restarting server now; queued requests will reconnect automatically.');
226
634
  server.closeAllConnections?.();
227
- process.exit(SERVER_RESTART_EXIT_CODE);
635
+ // Best-effort: flush quota + settle in-flight refreshes/prober before exit so
636
+ // the respawned worker doesn't boot from a half-written state. Bounded so the
637
+ // abrupt path stays fast even if a write hangs.
638
+ Promise.race([
639
+ (async () => { await persistQuotaState(); await releaseLease(); await flushConfigWrites(); await flushStateWrites(); })(),
640
+ delay(2000),
641
+ ]).finally(() => process.exit(SERVER_RESTART_EXIT_CODE));
642
+ };
643
+
644
+ // Seamless reload entry point. When supervised, ask the supervisor to run the
645
+ // baton (spawn new worker, hand off socket + lease). If anything is off
646
+ // (not supervised, no IPC), fall back to the abrupt restart.
647
+ const requestReload = () => {
648
+ if (draining) return;
649
+ if (supervised) {
650
+ try {
651
+ process.send({ type: MSG_RELOAD_REQUEST });
652
+ return;
653
+ } catch { /* IPC gone — fall through to abrupt restart */ }
654
+ }
655
+ restartWorkerNow();
228
656
  };
229
657
 
230
658
  restartController = new RestartController({
231
659
  pauseAdmission: () => accountManager.setAdmissionPaused(true),
232
- restartNow: restartWorkerNow,
660
+ restartNow: requestReload,
233
661
  });
234
662
 
235
663
  const shutdownGracefully = (reason, options = {}) => {
@@ -240,10 +668,16 @@ async function serverWorkerCommand() {
240
668
 
241
669
  draining = true;
242
670
  if (syncTimer) clearInterval(syncTimer);
243
- clearInterval(quotaSaveInterval);
244
- persistQuotaState(); // best-effort final flush of learned quota
245
- prober.stop();
246
671
  if (tui?.running) tui.stop();
672
+ // Best-effort final flush + settle writes (fire-and-forget; the drain timeout
673
+ // below bounds total quit time, and write barriers protect the on-disk state).
674
+ (async () => {
675
+ await releaseLease();
676
+ await accountManager.drainRefreshes();
677
+ await persistQuotaState(true); // force: lease already dropped above (else no-op)
678
+ await flushConfigWrites();
679
+ await flushStateWrites();
680
+ })().catch(() => {});
247
681
 
248
682
  console.log(`\n[Maxpool] Draining shutdown (${reason}).`);
249
683
  console.log(`[Maxpool] Stopped accepting new requests; waiting up to ${Math.ceil(drainTimeoutMs / 1000)}s for ${restartController.activeRequests.size} active request(s), then forcing exit. Press Ctrl-C again to force now.`);
@@ -345,12 +779,12 @@ async function serverWorkerCommand() {
345
779
  }
346
780
  }, syncIntervalMs);
347
781
  syncTimer.unref();
348
- const onListenError = err => handleServerListenError(err, host, port);
349
- server.once('error', onListenError);
350
782
 
351
- server.listen(port, host, () => {
352
- server.removeListener('error', onListenError);
353
- server.on('error', err => console.error(`[Maxpool] Server error: ${err.message}`));
783
+ // Become the live primary: start serving UI/logs, take the writer lease, run
784
+ // the update check. `viaTakeover` true means we acquired the socket through the
785
+ // baton (a reload) — freeze the update check so a reload doesn't re-probe npm
786
+ // (only a cold supervisor start self-updates).
787
+ const becomePrimary = async ({ viaTakeover }) => {
354
788
  if (tui) {
355
789
  if (tui.start()) {
356
790
  console.log(`Listening on ${host}:${port} with ${accounts.length} account(s)`);
@@ -361,15 +795,150 @@ async function serverWorkerCommand() {
361
795
  } else {
362
796
  logPlainServerStart({ host, port, accounts, threshold, config });
363
797
  }
798
+ // On a reload TAKEOVER, adopt the latest tokens from disk. The headless
799
+ // reload worker loaded config at spawn time; the OLD worker may have rotated
800
+ // a single-use refresh token DURING the handoff (and flushed it before
801
+ // MSG_RELEASED). Without this re-sync the new worker would present the now-
802
+ // invalidated token on its first refresh → invalid_grant → bricked account.
803
+ if (viaTakeover) {
804
+ try {
805
+ const diskConfig = await loadConfig();
806
+ if (diskConfig) await syncAccountsFromDisk(diskConfig, config, accountManager);
807
+ } catch (err) {
808
+ console.error(`[Maxpool] Token re-sync at takeover failed: ${err.message}`);
809
+ }
810
+ }
364
811
 
365
- // Non-blocking update check. Notifies (or self-updates if config.autoUpdate);
366
- // never interrupts the running proxy. Failures are swallowed.
367
- const notify = msg => (tui?._addLog ? tui._addLog(msg) : console.log(`[Maxpool] ${msg}`));
368
- maybeCheckForUpdate(config, notify).catch(() => {});
369
- });
812
+ // Acquire the lease (re-syncs state generation + enables writes) BEFORE we
813
+ // announce primacy, so the moment the supervisor reaps the old worker we are
814
+ // a fully-functional sole writer.
815
+ await acquireLease();
816
+
817
+ // Tell the supervisor we are the sole acceptor now → it stops its own accept
818
+ // loop (the steady-state race fix). Cold start AND takeover both signal this.
819
+ if (supervised) { try { process.send({ type: MSG_PRIMARY }); } catch { /* ignore */ } }
820
+
821
+ // Reload-storm guard: only a cold (non-reload) lease holder probes npm.
822
+ // A reload-spawned/takeover worker must NEVER re-probe (1x not 2x traffic).
823
+ if (!viaTakeover && !isReloadWorker) {
824
+ if (process.env.MAXPOOL_TEST_LOG_UPDATE_CHECK === '1') console.log('[Maxpool] UPDATE_CHECK_FIRED');
825
+ const notify = msg => (tui?._addLog ? tui._addLog(msg) : console.log(`[Maxpool] ${msg}`));
826
+ maybeCheckForUpdate(config, notify).catch(() => {});
827
+ } else if (process.env.MAXPOOL_TEST_LOG_UPDATE_CHECK === '1') {
828
+ console.log('[Maxpool] UPDATE_CHECK_SKIPPED (reload)');
829
+ }
830
+ };
831
+
832
+ // ── start accepting ──
833
+ if (supervised) {
834
+ // Supervised: NEVER listen(port) directly — the supervisor owns the socket.
835
+ // Wait for the baton over IPC. A cold worker gets MSG_LISTEN; a reload worker
836
+ // boots headless (already done above) and waits for MSG_PROBE_READY/TAKEOVER.
837
+ server.on('error', err => console.error(`[Maxpool] Server error: ${err.message}`));
838
+
839
+ const listenOnHandle = handle => new Promise((resolve, reject) => {
840
+ server.once('error', reject);
841
+ server.listen(handle, () => { server.removeListener('error', reject); resolve(); });
842
+ });
843
+
844
+ // Baton release: the old primary gives up acceptance + the writer lease.
845
+ // MSG_RELEASED MUST mean "no write is in flight" — not merely "flag flipped"
846
+ // — so the new worker can acquire the lease and rotate the single-use refresh
847
+ // token WITHOUT racing a still-pending rotation here (B1) and boots from a
848
+ // config that already has any rotated token persisted (M3).
849
+ const releaseBatonAndDrain = async () => {
850
+ if (draining) { try { process.send({ type: MSG_RELEASED }); } catch { /* ignore */ } return; }
851
+ draining = true;
852
+ if (syncTimer) clearInterval(syncTimer);
853
+ // Stop accepting NEW connections; KEEP in-flight requests alive.
854
+ server.maxpoolBeginDrain?.();
855
+ server.close(() => {});
856
+ server.closeIdleConnections?.(); // retire idle keep-alive sockets now
857
+
858
+ // 1) Stop scheduling writes + flip the lease + await any in-flight prober
859
+ // cycle (releaseLease awaits prober.stop()).
860
+ await releaseLease();
861
+ // 2) Await every in-flight OAuth token refresh that started before the lease
862
+ // dropped — the headline brick hazard (B1).
863
+ await accountManager.drainRefreshes();
864
+ // 3) Final flush of learned quota. FORCE it: releaseLease() above already
865
+ // flipped hasLease=false, and persistQuotaState no-ops without the lease —
866
+ // so an unforced call here silently drops the reload's final quota flush.
867
+ await persistQuotaState(true);
868
+ // 4) Barrier: ensure every queued config write (a rotated-token persist is
869
+ // fire-and-forget) AND state write has actually hit disk before we hand
870
+ // off (M3 — otherwise the new worker boots from the invalidated token).
871
+ await flushConfigWrites();
872
+ await flushStateWrites();
873
+
874
+ if (tui?.running) tui.stop(); // restore terminal before the new worker takes it
875
+ try { process.send({ type: MSG_RELEASED }); } catch { /* ignore */ }
876
+
877
+ // Drain bounded in-flight on EXISTING access tokens (no refresh needed),
878
+ // then exit(0). The supervisor SIGKILLs us if we outlive its drain cap.
879
+ const waitForDrain = () => {
880
+ if (restartController.activeRequests.size === 0) { process.exit(0); return; }
881
+ };
882
+ const drainPoll = setInterval(waitForDrain, 200);
883
+ drainPoll.unref?.();
884
+ const hardCap = setTimeout(() => {
885
+ console.error(`[Maxpool] Released worker drain cap reached with ${restartController.activeRequests.size} active; exiting.`);
886
+ process.exit(0);
887
+ }, drainTimeoutMs);
888
+ hardCap.unref?.();
889
+ waitForDrain();
890
+ };
891
+
892
+ process.on('message', async (msg, handle) => {
893
+ try {
894
+ if (msg?.type === MSG_LISTEN && handle) {
895
+ // Cold start: take the socket and go primary immediately.
896
+ await listenOnHandle(handle);
897
+ await becomePrimary({ viaTakeover: false });
898
+ } else if (msg?.type === MSG_PROBE_READY) {
899
+ // Headless reload worker: we've booted the new code and restored quota
900
+ // in memory. Confirm readiness (we are NOT yet accepting / writing).
901
+ // Test hook: simulate a new-version that fails to boot → forces the
902
+ // supervisor's rollback (old worker stays primary, zero disruption).
903
+ if (process.env.MAXPOOL_TEST_FAIL_RELOAD_WORKER === '1') {
904
+ process.send({ type: MSG_FAILED, reason: 'test-forced failure' });
905
+ } else {
906
+ process.send({ type: MSG_READY });
907
+ }
908
+ } else if (msg?.type === MSG_TAKEOVER && handle) {
909
+ // Baton acquire: the old worker already released, so we're the SOLE
910
+ // acceptor. Start accepting + take the writer lease + TUI. becomePrimary
911
+ // sends MSG_PRIMARY (the baton waits on it).
912
+ await listenOnHandle(handle);
913
+ await becomePrimary({ viaTakeover: true });
914
+ } else if (msg?.type === MSG_RELEASE) {
915
+ // Baton release: stop accepting NEW (keep in-flight), retire keep-alive
916
+ // sockets with Connection: close, stop writing, flush once, drop TUI.
917
+ await releaseBatonAndDrain();
918
+ }
919
+ } catch (err) {
920
+ // A failed reload worker must NOT exit(1) (kills the supervisor loop).
921
+ // Report failure so the supervisor rolls back; stay alive harmlessly.
922
+ console.error(`[Maxpool] Reload worker error: ${err.message}`);
923
+ try { process.send({ type: MSG_FAILED, reason: err.message }); } catch { /* ignore */ }
924
+ }
925
+ });
926
+ } else {
927
+ // Direct (non-TTY service) path: bind the port ourselves, exactly as before.
928
+ const onListenError = err => handleServerListenError(err, host, port);
929
+ server.once('error', onListenError);
930
+ server.listen(port, host, () => {
931
+ server.removeListener('error', onListenError);
932
+ server.on('error', err => console.error(`[Maxpool] Server error: ${err.message}`));
933
+ becomePrimary({ viaTakeover: false }).catch(err => console.error(`[Maxpool] ${err.message}`));
934
+ });
935
+ }
370
936
 
371
937
  process.on('SIGINT', () => shutdownGracefully('SIGINT'));
372
938
  process.on('SIGTERM', () => shutdownGracefully('SIGTERM'));
939
+ // SIGHUP requests a seamless reload (the conventional "reload" signal). Under
940
+ // the supervisor this runs the baton; otherwise it falls back to exit-75.
941
+ process.on('SIGHUP', () => requestReload());
373
942
  }
374
943
 
375
944
  function logPlainServerStart({ host, port, accounts, threshold, config }) {
@@ -607,7 +1176,7 @@ async function accountsCommand() {
607
1176
  a.refreshToken = newTokens.refreshToken;
608
1177
  a.expiresAt = newTokens.expiresAt;
609
1178
  configDirty = true;
610
- } catch (err) {
1179
+ } catch {
611
1180
  // refresh failed — fetchProfile will report the specific error
612
1181
  }
613
1182
  }));