@zgeoff/atc 0.1.4 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -121,3 +121,8 @@ live fleet (name, cwd, Claude session id) to its SQLite store. If the daemon its
121
121
  SIGKILL, reboot — the child claude processes die with it, but every session's transcript is already
122
122
  on disk. Start atc and press `R`: the whole fleet respawns via `claude --resume`. Only deliberate
123
123
  kills (`K`, `Y` eject) remove entries from the fleet, so it stays restorable.
124
+
125
+ Restoring revives one session at a time: the next `claude --resume` starts only once the previous
126
+ one has reported it is up (its `SessionStart` hook), so bringing back a dozen sessions no longer
127
+ launches a dozen Claude processes at the same instant and pins the machine. The list fills in live
128
+ as each comes back.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@zgeoff/atc",
3
- "version": "0.1.4",
3
+ "version": "0.1.5",
4
4
  "description": "Terminal control tower for Claude Code sessions",
5
5
  "homepage": "https://github.com/zgeoff/atc#readme",
6
6
  "bugs": "https://github.com/zgeoff/atc/issues",
package/src/cli.ts CHANGED
@@ -50,16 +50,26 @@ const main = defineCommand({
50
50
  // Test harnesses shrink the outbound queue to force overflow
51
51
  // deterministically; unset means the production default.
52
52
  const queueBytes = Number(process.env['ATC_QUEUE_BYTES']);
53
+ const cfg = config.loadConfig();
54
+
55
+ // Cap on how long a fleet restore waits for one revived session to
56
+ // report it has booted before moving to the next. Tests pin it to
57
+ // keep timing deterministic; unset means the production default.
58
+ const capOverride = Number(process.env['ATC_RESTORE_BOOT_TIMEOUT_MS']);
59
+
60
+ const restoreBootTimeoutMs =
61
+ Number.isFinite(capOverride) && capOverride >= 0 ? capOverride : 15_000;
53
62
 
54
63
  const handle = daemon.startDaemon({
55
64
  headlessRunner: (runOpts, hooks) => headless.startHeadlessRun(runOpts, hooks),
56
65
  socketPath: config.daemonSocketPath,
57
66
  reporterSocketPath: config.socketPath,
58
67
  build: getBuild(),
59
- adapter: new claude.ClaudeAdapter(config.loadConfig()),
68
+ adapter: new claude.ClaudeAdapter(cfg),
60
69
  dbPath: config.dbFile,
61
70
  legacyFleetPath: config.legacyFleetFile,
62
71
  pidPath: config.daemonPidFile,
72
+ restoreBootTimeoutMs,
63
73
  ...(Number.isFinite(queueBytes) && queueBytes > 0 ? { queueBytes } : {}),
64
74
  onQuit: () => process.exit(0),
65
75
  });
package/src/daemon.ts CHANGED
@@ -10,7 +10,7 @@ import { MAX_CHUNK, PROTOCOL_V } from './protocol';
10
10
  import type { EventMsg } from './protocol';
11
11
  import { ScreenModel } from './screen-model';
12
12
  import { SessionManager } from './sessions';
13
- import type { SessionDescriptor, SessionState } from './sessions';
13
+ import type { FleetEntry, Session, SessionDescriptor, SessionState } from './sessions';
14
14
  import { StateStore } from './state-store';
15
15
 
16
16
  export interface DaemonOptions {
@@ -44,6 +44,13 @@ export interface DaemonOptions {
44
44
  // before starting the headless run anyway.
45
45
  readonly ejectSettleMs?: number;
46
46
 
47
+ // A fleet-wide restore revives one session at a time, waiting for each to
48
+ // report it has booted before starting the next so the machine is not
49
+ // buried under a dozen simultaneous agent boots. This caps how long a
50
+ // single revive waits for that signal before moving on regardless, so a
51
+ // session that never reports cannot stall the rest. Zero waits forever.
52
+ readonly restoreBootTimeoutMs?: number;
53
+
47
54
  // Called after a client-requested quit has stopped the daemon; the real
48
55
  // entrypoint exits the process, tests leave it unset.
49
56
  readonly onQuit?: () => void;
@@ -126,6 +133,37 @@ export function startDaemon(opts: DaemonOptions): DaemonHandle {
126
133
  // (or a settle timeout) before the headless run starts.
127
134
  const pendingEjects = new Map<string, () => void>();
128
135
 
136
+ // A staggered fleet restore parks a resolver here while it waits for the
137
+ // session it just revived to report it has booted; the reporter fires it on
138
+ // SessionStart, and a dying revive fires it too so a failed resume does not
139
+ // hold up the rest. Resolving a settled promise again is a no-op.
140
+ const bootWaiters = new Map<string, () => void>();
141
+
142
+ // Blocks until the session reports SessionStart, it dies, or the cap
143
+ // elapses (a positive cap only; zero waits on the signal alone).
144
+ const waitForBoot = (sessionID: string, capMs: number): Promise<void> => {
145
+ const settled = Promise.withResolvers<void>();
146
+ const timer = capMs > 0 ? setTimeout(settled.resolve, capMs) : undefined;
147
+
148
+ bootWaiters.set(sessionID, settled.resolve);
149
+
150
+ if (timer !== undefined) {
151
+ restoreTimers.add(timer);
152
+ }
153
+
154
+ return (async () => {
155
+ await settled.promise;
156
+
157
+ bootWaiters.delete(sessionID);
158
+
159
+ if (timer !== undefined) {
160
+ clearTimeout(timer);
161
+
162
+ restoreTimers.delete(timer);
163
+ }
164
+ })();
165
+ };
166
+
129
167
  const startHeadlessTurn = (sessionID: string, prompt: string): boolean => {
130
168
  const runner = opts.headlessRunner;
131
169
  const s = mgr.sessions.find((x) => x.id === sessionID);
@@ -172,6 +210,7 @@ export function startDaemon(opts: DaemonOptions): DaemonHandle {
172
210
  const screens = new Map<string, ScreenModel>();
173
211
  const resizeTimers = new Map<string, ReturnType<typeof setTimeout>>();
174
212
  const detectTimers = new Map<string, ReturnType<typeof setTimeout>>();
213
+ const restoreTimers = new Set<ReturnType<typeof setTimeout>>();
175
214
 
176
215
  // The screen tier of the detector stack: once a session's output has
177
216
  // quiesced, judge the serialized screen and flip running/needs_you.
@@ -345,6 +384,12 @@ export function startDaemon(opts: DaemonOptions): DaemonHandle {
345
384
  mgr.onEvent = (kind, s) => {
346
385
  recordAttention(kind, s);
347
386
 
387
+ // A revive that dies before it ever announces itself must still release
388
+ // the staggered restore, or a failed resume would hold up the fleet.
389
+ if (kind === 'removed' || (kind === 'state' && s.pty === null)) {
390
+ bootWaiters.get(s.id)?.();
391
+ }
392
+
348
393
  if (kind === 'removed') {
349
394
  attachments.removeSession(s.id);
350
395
  seqs.delete(s.id);
@@ -397,6 +442,12 @@ export function startDaemon(opts: DaemonOptions): DaemonHandle {
397
442
  pendingEjects.get(e.atcId)?.();
398
443
  }
399
444
 
445
+ // A revived session announcing itself is the cue a staggered restore
446
+ // waits on before booting the next one.
447
+ if (e.event === 'SessionStart') {
448
+ bootWaiters.get(e.atcId)?.();
449
+ }
450
+
400
451
  mgr.applyHook(e);
401
452
  }, opts.reporterSocketPath);
402
453
 
@@ -563,13 +614,12 @@ export function startDaemon(opts: DaemonOptions): DaemonHandle {
563
614
  getEffectiveDims: (sessionID) =>
564
615
  attachments.findEffectiveDims(sessionID) ?? ptyDims.get(sessionID) ?? { cols: 80, rows: 24 },
565
616
  restoreFleet: (cols, rows) => {
566
- let restored = 0;
617
+ const isLive = (claudeId: string | undefined) =>
618
+ mgr.sessions.some((s) => s.pty !== null && s.claudeId === claudeId);
567
619
 
568
- for (const entry of store.loadFleet()) {
569
- const live = mgr.sessions.some((s) => s.pty !== null && s.claudeId === entry.claudeId);
570
-
571
- if (live) {
572
- continue;
620
+ const restoreEntry = (entry: FleetEntry): Session | null => {
621
+ if (isLive(entry.claudeId)) {
622
+ return null;
573
623
  }
574
624
 
575
625
  const s = mgr.spawn(
@@ -586,10 +636,37 @@ export function startDaemon(opts: DaemonOptions): DaemonHandle {
586
636
  ptyDims.set(s.id, { cols, rows });
587
637
  screens.set(s.id, new ScreenModel(cols, rows));
588
638
 
589
- restored++;
639
+ return s;
640
+ };
641
+
642
+ const [first, ...rest] = store.loadFleet().filter((entry) => !isLive(entry.claudeId));
643
+
644
+ if (first === undefined) {
645
+ return 0;
590
646
  }
591
647
 
592
- return restored;
648
+ // The first revive is synchronous so a caller can attach at once; each
649
+ // later revive waits for the previous session to report it has booted,
650
+ // so a heavy fleet comes up one process at a time instead of all at
651
+ // once. A per-session cap keeps a session that never reports from
652
+ // stalling the rest.
653
+ const cap = opts.restoreBootTimeoutMs ?? 0;
654
+
655
+ const restoreRest = async (previous: Session | null) => {
656
+ let prev = previous;
657
+
658
+ for (const entry of rest) {
659
+ if (prev !== null) {
660
+ await waitForBoot(prev.id, cap);
661
+ }
662
+
663
+ prev = restoreEntry(entry);
664
+ }
665
+ };
666
+
667
+ void restoreRest(restoreEntry(first));
668
+
669
+ return 1 + rest.length;
593
670
  },
594
671
  };
595
672
 
@@ -628,6 +705,10 @@ export function startDaemon(opts: DaemonOptions): DaemonHandle {
628
705
  clearTimeout(timer);
629
706
  }
630
707
 
708
+ for (const timer of restoreTimers) {
709
+ clearTimeout(timer);
710
+ }
711
+
631
712
  server.stop(true);
632
713
  reporter.stop(true);
633
714
  mgr.killAll();