@north-light/crouter 0.3.167 → 0.3.169

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -85,6 +85,7 @@ const daemonRestart = defineLeaf({
85
85
  outputKind: 'object',
86
86
  effects: [
87
87
  'The daemon acknowledges immediately, then after a short grace tears its whole broker fleet down, releases its claim, spawns a successor on the runtime generation currently selected, and exits.',
88
+ 'Each torn-down broker hands its foreground bash commands to the file-backed background job system; they keep running and report completion by urgent inbox message.',
88
89
  'Every node it tore down — including the caller — is resumed by the successor through the ordinary startup recovery sweep.',
89
90
  'The calling process is killed as part of that teardown, AFTER this result is returned.',
90
91
  ],
@@ -34,7 +34,6 @@ import { tmpdir } from 'node:os';
34
34
  import { dirname, join } from 'node:path';
35
35
  import { fileURLToPath } from 'node:url';
36
36
  import { createAgentSessionServices, createAgentSessionFromServices, SessionManager, VERSION, } from '../runtime/broker-sdk.js';
37
- import { CANVAS_EXTENSIONS } from '../runtime/launch.js';
38
37
  import { buildBrokerSession } from '../runtime/broker.js';
39
38
  // M-9 (broker.ts fault-and-die model gate): buildBrokerSession with no cfg.model
40
39
  // now calls the REAL SDK's modelRegistry.getAvailable() and faults if it's empty
@@ -83,7 +82,7 @@ test('D2 broker fork branch: buildBrokerSession(cfg.forkFrom) yields a NEW id +
83
82
  // if broker.ts:~1106 reverts to that throw, this rejects and the test goes RED.
84
83
  const cfg = {
85
84
  cwd: dir,
86
- extensionPaths: [...CANVAS_EXTENSIONS, C3_EXT],
85
+ extensionPaths: [C3_EXT],
87
86
  forkFrom: srcFile,
88
87
  model: 'c3prov/c3model',
89
88
  };
@@ -84,7 +84,7 @@ function cfg(cwd, extra = {}) {
84
84
  test('C3 — services path registers an extension model provider; the broker session gets it', async () => {
85
85
  const cwd = mkdtempSync(join(tmpdir(), 'crtr-c3-'));
86
86
  try {
87
- const { session, services } = await buildBrokerSession(realEngine, cfg(cwd, { model: 'c3prov/c3model' }));
87
+ const { session, services } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [C3_EXT], model: 'c3prov/c3model' }));
88
88
  try {
89
89
  // The extension's provider was registered into the SERVICES model runtime (this
90
90
  // is the registerProvider step plain createAgentSession skips).
@@ -107,7 +107,7 @@ test('C3 — services path registers an extension model provider; the broker ses
107
107
  test('C3b — broker splits model thinking suffix before SDK registry lookup', async () => {
108
108
  const cwd = mkdtempSync(join(tmpdir(), 'crtr-c3b-'));
109
109
  try {
110
- const { session } = await buildBrokerSession(realEngine, cfg(cwd, { model: 'c3prov/c3model:high' }));
110
+ const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [C3_EXT], model: 'c3prov/c3model:high' }));
111
111
  try {
112
112
  assert.equal(session.model?.id, 'c3model', 'C3b: suffixed model specs resolve by model id instead of falling back to SDK default');
113
113
  }
@@ -420,7 +420,7 @@ test('C5 — isLeadingEngineCommand against a REAL session (registered extension
420
420
  // proven against a plain object shaped like the real accessors.
421
421
  const cwd = mkdtempSync(join(tmpdir(), 'crtr-c5-cmd-'));
422
422
  try {
423
- const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [...brokerExts(), C5_EXT], model: 'c3prov/c3model' }));
423
+ const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [C3_EXT, C5_EXT], model: 'c3prov/c3model' }));
424
424
  try {
425
425
  assert.equal(isLeadingEngineCommand('/devcmd', session), true, 'a real registered extension command matches');
426
426
  assert.equal(isLeadingEngineCommand('/devcmd extra args', session), true, 'space-separated still matches against a real session');
@@ -744,7 +744,7 @@ test('Major 6 (broker half) — engineLeadingCommandTokens degrades to empty per
744
744
  test('Major 6 (broker half) — engineLeadingCommandTokens against a REAL session enumerates the real registered extension command', async () => {
745
745
  const cwd = mkdtempSync(join(tmpdir(), 'crtr-m6-cmd-'));
746
746
  try {
747
- const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [...brokerExts(), C5_EXT], model: 'c3prov/c3model' }));
747
+ const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [C3_EXT, C5_EXT], model: 'c3prov/c3model' }));
748
748
  try {
749
749
  const tokens = engineLeadingCommandTokens(session);
750
750
  assert.ok(tokens.includes('devcmd'), 'the real registered extension command name is present in the enumeration');
@@ -35,6 +35,7 @@ import { fileURLToPath } from 'node:url';
35
35
  import { spawn, spawnSync } from 'node:child_process';
36
36
  import { createNode, getNode, subscribe, updateNode, clearPid } from '../canvas/canvas.js';
37
37
  import { armFollowUpWork } from '../canvas/human-work-outbox.js';
38
+ import { armCron, hasPendingCancelOnWakeCron } from '../canvas/crons.js';
38
39
  import { readInboxSince } from '../feed/inbox.js';
39
40
  import { closeDb } from '../canvas/db.js';
40
41
  import { buildPiArgv, CANVAS_EXTENSIONS } from '../runtime/launch.js';
@@ -346,6 +347,53 @@ test('reviveNode PROCEEDS when the node has NO fleet entry — cycle counter bum
346
347
  await h.dispose();
347
348
  }
348
349
  });
350
+ // ---------------------------------------------------------------------------
351
+ // BUG LOCKED — a recovery relaunch must not eat the node's armed deadline.
352
+ // A node armed `crtr node wait deadline` and dormed; its broker then exited and
353
+ // the daemon respawned it (fleet.#respawn → reviveNode). reviveNode used to
354
+ // cancel every cancel-on-wake cron unconditionally, so the deadline vanished
355
+ // with nothing having woken the node and the wait could never settle. A wake
356
+ // (inbox delivery, attach, explicit revive) still consumes it.
357
+ // ---------------------------------------------------------------------------
358
+ function armDeadline(anchor, cronId) {
359
+ armCron({
360
+ cron_id: cronId,
361
+ name: `deadline:${anchor}`,
362
+ created_by: anchor,
363
+ command: 'true',
364
+ fire_at: new Date(Date.now() + 3_600_000).toISOString(),
365
+ recur: null,
366
+ tz: null,
367
+ expires_at: null,
368
+ anchor_node: anchor,
369
+ cancel_on_wake: true,
370
+ cwd: home,
371
+ env_json: null,
372
+ profile: null,
373
+ scope: 'profile',
374
+ run_timeout_s: 300,
375
+ overlap: 'skip',
376
+ on_output: 'on-failure',
377
+ sink: '',
378
+ tier: 'normal',
379
+ });
380
+ }
381
+ test('a recovery relaunch keeps the armed deadline; a wake revive consumes it', async () => {
382
+ const h = await createHarness({ headless: true, sessionPrefix: 'crtr-revive-deadline' });
383
+ try {
384
+ const recovered = h.fabricateBrokerNode({ status: 'active', intent: null, pi_pid: deadPid(), pi_session_id: 'uuid-1' });
385
+ armDeadline(recovered, 'cron-recovery');
386
+ reviveNode(recovered, { resume: true, recovery: true });
387
+ assert.equal(hasPendingCancelOnWakeCron(recovered), true, 'the daemon replacing a dead broker is not a wake — the deadline survives so the wait can still settle');
388
+ const woken = h.fabricateBrokerNode({ status: 'active', intent: null, pi_pid: deadPid(), pi_session_id: 'uuid-2' });
389
+ armDeadline(woken, 'cron-wake');
390
+ reviveNode(woken, { resume: true });
391
+ assert.equal(hasPendingCancelOnWakeCron(woken), false, 'a wake revive won the race and cancels the deadline');
392
+ }
393
+ finally {
394
+ await h.dispose();
395
+ }
396
+ });
349
397
  test('reviveNode fresh-fallback clears the stale session identity (Major-3: no phantom stranded-relaunch)', async () => {
350
398
  const h = await createHarness({ headless: true, sessionPrefix: 'crtr-revive-freshfb' });
351
399
  try {
@@ -4,7 +4,9 @@ export interface BashJobPaths {
4
4
  cmdSh: string;
5
5
  jobLog: string;
6
6
  jobExit: string;
7
+ jobRun: string;
7
8
  jobBg: string;
9
+ jobDone: string;
8
10
  /** Process-group id of the job's detached supervisor, written at spawn. It is
9
11
  * what makes a job stoppable from outside the agent that started it (the
10
12
  * Inspector's cancel) — without it on disk, only the agent's own handoff
@@ -42,25 +44,22 @@ export interface BashJobStatus {
42
44
  logPath: string;
43
45
  }
44
46
  /** Every live backgrounded bash job for this node — job.bg present, job.exit
45
- * absent — whether handed off by the menu or auto-backgrounded at the valve's
46
- * deadline. Read-only; never mutates. Distinct from (and shares no code path
47
- * with) the foreground-only `runningBashJobs` above: that scanner deliberately
48
- * EXCLUDES job.bg, this one REQUIRES it, so reusing one for the other would
49
- * corrupt foreground-admission semantics. Skips malformed/partially-written
47
+ * absent — whether handed off by the menu, the valve deadline, or broker
48
+ * termination. Read-only; never mutates. Skips malformed/partially-written
50
49
  * directories. Returns oldest-first by startedAtMs. */
51
50
  export declare function activeBackgroundBashJobs(contextDir: string): BashJobStatus[];
52
- /** The start time (ms epoch) of the OLDEST still-foreground bash command for
51
+ export type BackgroundBashJobResult = 'backgrounded' | 'already-backgrounded' | 'finished' | 'missing';
52
+ /** Atomically hand one foreground job to its detached supervisor. The rename
53
+ * is the ownership decision: exactly one caller can move job.run to job.bg,
54
+ * while the supervisor races to move the same source to job.done. */
55
+ export declare function backgroundBashJob(paths: BashJobPaths): BackgroundBashJobResult;
56
+ /** The start time (ms epoch) of the oldest still-foreground bash command for
53
57
  * this node, or undefined when none is running. This is the authoritative
54
- * source for the attach viewer's R4 "background it" footer hint: it reads the
55
- * exact same on-disk contract the valve writes (a job directory exists with
56
- * no job.bg and no job.exit while the command is running in the foreground,
57
- * and the valve deletes the directory the moment the command finishes or is
58
- * aborted — see canvas-bash-valve.ts), so a fresh or reconnected attach (which
59
- * has no memory of a broker-relayed tool-execution event) sees exactly what's
60
- * really running, and a viewer never needs its own call-id tracking that could
61
- * go stale across a broker/session replacement. */
58
+ * source for the attach viewer's background hint: a fresh or reconnected
59
+ * attach reads the same job.run state that backgrounding and completion race
60
+ * to claim, without viewer-local call tracking that can go stale. */
62
61
  export declare function oldestRunningBashJobStartedAtMs(contextDir: string): number | undefined;
63
- /** Request immediate handoff for every bash command still waiting on pi. The
64
- * valve observes job.bg within one poll, resolves the tool call, and leaves its
65
- * detached supervisor to notify the node when the real command exits. */
62
+ /** Request immediate handoff for every bash command still waiting on pi. Only
63
+ * the caller that wins the job.run → job.bg rename reports the job as handed
64
+ * off; another backgrounder or the foreground-completion path may win the race. */
66
65
  export declare function backgroundRunningBashJobs(contextDir: string): BashJobPaths[];
@@ -1,9 +1,9 @@
1
- // File-backed control plane for bash commands that have been handed off by the
2
- // canvas bash valve. The valve starts every command in one of these directories;
3
- // the prefix menu requests backgrounding by creating job.bg. Keeping that signal
4
- // on disk means the command can outlive its pi process and still report its exit.
1
+ // File-backed control plane for bash commands started by the canvas bash valve.
2
+ // Every foreground command owns job.run; backgrounding atomically renames it to
3
+ // job.bg, while foreground completion renames it to job.done. The detached
4
+ // supervisor can therefore outlive pi and still decide whether to report exit.
5
5
  import { randomBytes } from 'node:crypto';
6
- import { closeSync, existsSync, openSync, readdirSync, readFileSync, readSync, statSync, writeFileSync } from 'node:fs';
6
+ import { closeSync, existsSync, openSync, readdirSync, readFileSync, readSync, renameSync, statSync } from 'node:fs';
7
7
  import { join } from 'node:path';
8
8
  export function bashJobsDir(contextDir) {
9
9
  return join(contextDir, 'jobs');
@@ -16,7 +16,9 @@ export function bashJobPaths(contextDir, jobId) {
16
16
  cmdSh: join(dir, 'cmd.sh'),
17
17
  jobLog: join(dir, 'job.log'),
18
18
  jobExit: join(dir, 'job.exit'),
19
+ jobRun: join(dir, 'job.run'),
19
20
  jobBg: join(dir, 'job.bg'),
21
+ jobDone: join(dir, 'job.done'),
20
22
  jobPgid: join(dir, 'job.pgid'),
21
23
  };
22
24
  }
@@ -94,11 +96,8 @@ export function formatBashElapsed(ms) {
94
96
  return `${hours}h ${String(minutes).padStart(2, '0')}m`;
95
97
  }
96
98
  /** Every live backgrounded bash job for this node — job.bg present, job.exit
97
- * absent — whether handed off by the menu or auto-backgrounded at the valve's
98
- * deadline. Read-only; never mutates. Distinct from (and shares no code path
99
- * with) the foreground-only `runningBashJobs` above: that scanner deliberately
100
- * EXCLUDES job.bg, this one REQUIRES it, so reusing one for the other would
101
- * corrupt foreground-admission semantics. Skips malformed/partially-written
99
+ * absent — whether handed off by the menu, the valve deadline, or broker
100
+ * termination. Read-only; never mutates. Skips malformed/partially-written
102
101
  * directories. Returns oldest-first by startedAtMs. */
103
102
  export function activeBackgroundBashJobs(contextDir) {
104
103
  let entries;
@@ -127,9 +126,27 @@ export function activeBackgroundBashJobs(contextDir) {
127
126
  }
128
127
  return jobs.sort((a, b) => a.startedAtMs - b.startedAtMs);
129
128
  }
130
- /** Every still-foreground bash command for this node. A completed command has
131
- * job.exit; an already-backgrounded one has job.bg and must not be selected a
132
- * second time. Invalid or concurrently-removed directories are ignored. */
129
+ /** Atomically hand one foreground job to its detached supervisor. The rename
130
+ * is the ownership decision: exactly one caller can move job.run to job.bg,
131
+ * while the supervisor races to move the same source to job.done. */
132
+ export function backgroundBashJob(paths) {
133
+ try {
134
+ renameSync(paths.jobRun, paths.jobBg);
135
+ return 'backgrounded';
136
+ }
137
+ catch (err) {
138
+ if (err.code !== 'ENOENT')
139
+ throw err;
140
+ if (existsSync(paths.jobBg))
141
+ return 'already-backgrounded';
142
+ if (existsSync(paths.jobDone))
143
+ return 'finished';
144
+ return 'missing';
145
+ }
146
+ }
147
+ /** Every still-foreground bash command for this node. job.run is the sole live
148
+ * foreground state; a handoff or foreground completion atomically removes it.
149
+ * Invalid or concurrently-removed directories are ignored. */
133
150
  function runningBashJobs(contextDir) {
134
151
  let entries;
135
152
  try {
@@ -140,21 +157,16 @@ function runningBashJobs(contextDir) {
140
157
  }
141
158
  return entries.flatMap((jobId) => {
142
159
  const paths = bashJobPaths(contextDir, jobId);
143
- return existsSync(paths.cmdSh) && existsSync(paths.jobLog) && !existsSync(paths.jobExit) && !existsSync(paths.jobBg)
160
+ return existsSync(paths.cmdSh) && existsSync(paths.jobLog) && existsSync(paths.jobRun)
144
161
  ? [paths]
145
162
  : [];
146
163
  });
147
164
  }
148
- /** The start time (ms epoch) of the OLDEST still-foreground bash command for
165
+ /** The start time (ms epoch) of the oldest still-foreground bash command for
149
166
  * this node, or undefined when none is running. This is the authoritative
150
- * source for the attach viewer's R4 "background it" footer hint: it reads the
151
- * exact same on-disk contract the valve writes (a job directory exists with
152
- * no job.bg and no job.exit while the command is running in the foreground,
153
- * and the valve deletes the directory the moment the command finishes or is
154
- * aborted — see canvas-bash-valve.ts), so a fresh or reconnected attach (which
155
- * has no memory of a broker-relayed tool-execution event) sees exactly what's
156
- * really running, and a viewer never needs its own call-id tracking that could
157
- * go stale across a broker/session replacement. */
167
+ * source for the attach viewer's background hint: a fresh or reconnected
168
+ * attach reads the same job.run state that backgrounding and completion race
169
+ * to claim, without viewer-local call tracking that can go stale. */
158
170
  export function oldestRunningBashJobStartedAtMs(contextDir) {
159
171
  let oldest;
160
172
  for (const paths of runningBashJobs(contextDir)) {
@@ -166,25 +178,14 @@ export function oldestRunningBashJobStartedAtMs(contextDir) {
166
178
  }
167
179
  return oldest;
168
180
  }
169
- /** Request immediate handoff for every bash command still waiting on pi. The
170
- * valve observes job.bg within one poll, resolves the tool call, and leaves its
171
- * detached supervisor to notify the node when the real command exits. */
181
+ /** Request immediate handoff for every bash command still waiting on pi. Only
182
+ * the caller that wins the job.run → job.bg rename reports the job as handed
183
+ * off; another backgrounder or the foreground-completion path may win the race. */
172
184
  export function backgroundRunningBashJobs(contextDir) {
173
185
  const backgrounded = [];
174
186
  for (const paths of runningBashJobs(contextDir)) {
175
- try {
176
- // A job may exit or be claimed by another prefix press between the scan
177
- // and this write. Recheck the two terminal/control files so that stale
178
- // candidates never masquerade as a successful handoff.
179
- if (existsSync(paths.jobExit) || existsSync(paths.jobBg))
180
- continue;
181
- writeFileSync(paths.jobBg, '');
187
+ if (backgroundBashJob(paths) === 'backgrounded')
182
188
  backgrounded.push(paths);
183
- }
184
- catch {
185
- // The valve owns cleanup; a concurrently removed directory is simply no
186
- // longer a running command to hand off.
187
- }
188
189
  }
189
190
  return backgrounded;
190
191
  }
@@ -102,6 +102,31 @@ test('dashboardRowsAllFromSource shows a node watched through its live broker',
102
102
  const rows = await dashboardRowsAllFromSource(source);
103
103
  assert.equal(rows[0]?.viewed, true);
104
104
  });
105
+ test('enrichRowsFromSource serializes a full dashboard source read', async () => {
106
+ const nodeRows = ['a', 'b', 'c'].map(row);
107
+ let inFlight = 0;
108
+ let maxInFlight = 0;
109
+ const source = {
110
+ getNode: async (id) => {
111
+ inFlight++;
112
+ maxInFlight = Math.max(maxInFlight, inFlight);
113
+ await new Promise((resolve) => setImmediate(resolve));
114
+ inFlight--;
115
+ return meta(id);
116
+ },
117
+ getRow: async () => null,
118
+ listNodes: async () => nodeRows,
119
+ subscriptionsOf: async () => [],
120
+ subscribersOf: async () => [],
121
+ view: async () => [],
122
+ ticketCountsForView: async () => ({}),
123
+ hasActiveLiveSubscription: async () => false,
124
+ };
125
+ const rows = await dashboardRowsAllFromSource(source);
126
+ await enrichRowsFromSource(source, rows);
127
+ assert.equal(maxInFlight, 1);
128
+ assert.deepEqual(rows.map((entry) => entry.name), ['a', 'b', 'c']);
129
+ });
105
130
  let home;
106
131
  let localCwd;
107
132
  let sessionFile;
@@ -160,8 +160,9 @@ export declare function setCronLastOutputHash(cron_id: string, hash: string): vo
160
160
  * terminal node that stopped without finishing. */
161
161
  export declare function hasPendingCancelOnWakeCron(anchor_node: string): boolean;
162
162
  /** DELETE every cancel-on-wake cron anchored to this node — the deadline rule.
163
- * Two seams call it: reviveNode on every revive-for-any-reason (the wake won
164
- * the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
163
+ * Two seams call it: reviveNode on a WAKE — every revive except a recovery
164
+ * relaunch, which only replaces a dead broker instance and leaves the wait
165
+ * open (the wake won the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
165
166
  * with the node, so its deadline must not fire against a finalized or closed
166
167
  * row). Standing declarative crons are untouched — only cancel-on-wake rows.
167
168
  * Kills any in-flight
@@ -219,8 +219,9 @@ export function hasPendingCancelOnWakeCron(anchor_node) {
219
219
  .get(anchor_node) !== undefined);
220
220
  }
221
221
  /** DELETE every cancel-on-wake cron anchored to this node — the deadline rule.
222
- * Two seams call it: reviveNode on every revive-for-any-reason (the wake won
223
- * the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
222
+ * Two seams call it: reviveNode on a WAKE — every revive except a recovery
223
+ * relaunch, which only replaces a dead broker instance and leaves the wait
224
+ * open (the wake won the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
224
225
  * with the node, so its deadline must not fire against a finalized or closed
225
226
  * row). Standing declarative crons are untouched — only cancel-on-wake rows.
226
227
  * Kills any in-flight
@@ -1,17 +1,8 @@
1
- /** True if a process with `pid` is currently alive (signal-0 probe) AND not a
2
- * zombie. `kill(pid, 0)` throws ESRCH when the process is gone; EPERM means
3
- * it exists but isn't ours — still alive (a foreign process is never OUR
4
- * zombie, so the `isZombie` check is skipped for that branch — see below). A
5
- * null/undefined pid (legacy / never-booted) reads dead.
6
- *
7
- * The zombie check matters because EVERY broker this module supervises is
8
- * spawned `detached: true` by crtr itself (`headlessBrokerHost.launch`) — a
9
- * dead broker whose spawning process hasn't reaped it yet is a zombie, and
10
- * `kill(pid, 0)` alone reports it as alive, wedging the daemon's revive guard
11
- * (`isPidAlive(pi_pid)`, `revive.ts`) forever: the daemon never revives an
12
- * already-live-looking row, so a refreshed/crashed node whose broker zombied
13
- * out under a long-lived spawning process (see `isZombie`'s doc) never comes
14
- * back on its own. */
1
+ /** True if a process with `pid` is currently alive. `kill(pid, 0)` provides the
2
+ * initial existence check; EPERM means the foreign process exists. For our
3
+ * processes, the later `ps` read is authoritative because it also distinguishes
4
+ * zombies and processes reaped between the two probes. Only an unavailable
5
+ * `ps` probe fails open to alive. A null/undefined pid reads dead. */
15
6
  export declare function isPidAlive(pid: number | null | undefined): boolean;
16
7
  /** THE single precision-aware identity compare, used EVERYWHERE two
17
8
  * `composeIdentity` fingerprints are compared (this module's own
@@ -10,46 +10,31 @@
10
10
  import { spawnSync } from 'node:child_process';
11
11
  import { readFileSync } from 'node:fs';
12
12
  import { readKernelBootId } from './boot-id.js';
13
- /** Is `pid` a ZOMBIE (exited, awaiting reap by its parent)? `kill(pid, 0)`
14
- * reports a zombie as a perfectly normal alive process — its PID is still
15
- * occupied in the process table — so callers relying on that signal-0 probe
16
- * alone (see `isPidAlive`) misread a dead broker as alive forever whenever
17
- * its spawning process never reaps it (crouton-labs zombie-broker bug: the
18
- * front door's foreground attach session parks its OWN event loop for the
19
- * whole session — see `bootRoot` in runtime/boot-root.ts — so it can't process
20
- * the SIGCHLD that would otherwise reap its detached broker child in real
21
- * time). `ps -o stat=` reports a leading `Z` for a zombie on BOTH BSD ps
22
- * (macOS) and procps-ng (Linux), so one portable probe covers both platforms
23
- * without a Linux-only `/proc` special case. Best-effort: any probe failure
24
- * (spawn error, no matching row, non-zero exit) reads as NOT a zombie —
25
- * fail-open, matching this module's other guards; a probe failure must never
26
- * be misread as proof of death (a live pid IS alive; only a positive `Z`
27
- * read says otherwise). */
28
- function isZombie(pid) {
13
+ /** Classify the current `ps` row for a pid. BSD ps and procps-ng both report a
14
+ * leading `Z` for zombies and exit silently with empty output when no process
15
+ * matches. Only a probe that could not run is unknown; that is the sole
16
+ * fail-open result. */
17
+ function psProcessState(pid) {
29
18
  try {
30
19
  const r = spawnSync('ps', ['-o', 'stat=', '-p', String(pid)], { encoding: 'utf8', timeout: 2000 });
31
- if (r.status !== 0 || typeof r.stdout !== 'string')
32
- return false;
33
- return r.stdout.trim().charAt(0) === 'Z';
20
+ if (r.error != null || r.signal != null || typeof r.stdout !== 'string')
21
+ return 'unknown';
22
+ if (typeof r.stderr === 'string' && r.stderr.trim() !== '')
23
+ return 'unknown';
24
+ const stat = r.stdout.trim();
25
+ if (stat === '')
26
+ return 'gone';
27
+ return stat.charAt(0) === 'Z' ? 'zombie' : 'live';
34
28
  }
35
29
  catch {
36
- return false;
30
+ return 'unknown';
37
31
  }
38
32
  }
39
- /** True if a process with `pid` is currently alive (signal-0 probe) AND not a
40
- * zombie. `kill(pid, 0)` throws ESRCH when the process is gone; EPERM means
41
- * it exists but isn't ours — still alive (a foreign process is never OUR
42
- * zombie, so the `isZombie` check is skipped for that branch — see below). A
43
- * null/undefined pid (legacy / never-booted) reads dead.
44
- *
45
- * The zombie check matters because EVERY broker this module supervises is
46
- * spawned `detached: true` by crtr itself (`headlessBrokerHost.launch`) — a
47
- * dead broker whose spawning process hasn't reaped it yet is a zombie, and
48
- * `kill(pid, 0)` alone reports it as alive, wedging the daemon's revive guard
49
- * (`isPidAlive(pi_pid)`, `revive.ts`) forever: the daemon never revives an
50
- * already-live-looking row, so a refreshed/crashed node whose broker zombied
51
- * out under a long-lived spawning process (see `isZombie`'s doc) never comes
52
- * back on its own. */
33
+ /** True if a process with `pid` is currently alive. `kill(pid, 0)` provides the
34
+ * initial existence check; EPERM means the foreign process exists. For our
35
+ * processes, the later `ps` read is authoritative because it also distinguishes
36
+ * zombies and processes reaped between the two probes. Only an unavailable
37
+ * `ps` probe fails open to alive. A null/undefined pid reads dead. */
53
38
  export function isPidAlive(pid) {
54
39
  if (pid == null)
55
40
  return false;
@@ -59,7 +44,8 @@ export function isPidAlive(pid) {
59
44
  catch (e) {
60
45
  return e.code === 'EPERM';
61
46
  }
62
- return !isZombie(pid);
47
+ const state = psProcessState(pid);
48
+ return state !== 'zombie' && state !== 'gone';
63
49
  }
64
50
  /** Matches the kernel `boot_id` UUID shape (`/proc/sys/kernel/random/boot_id`),
65
51
  * used to tell a NEW-format identity base (a per-boot UUID) apart from a
@@ -430,14 +430,18 @@ export async function enrichRowsFromSource(source, rows, asks) {
430
430
  return;
431
431
  const remote = source instanceof RemoteCanvasSource;
432
432
  const askMap = asks ?? {};
433
- await Promise.all(todo.map(async (row) => {
433
+ // A full dashboard may contain thousands of rows. Keep source reads serial:
434
+ // ApiCanvasSource turns each into a unix-socket request, and an unbounded
435
+ // Promise.all can overflow crtrd's accept backlog then spuriously trigger
436
+ // the client's cold-daemon path even though the daemon is still running.
437
+ for (const row of todo) {
434
438
  const meta = await source.getNode(row.node_id);
435
439
  if (meta !== null)
436
440
  row.name = fullName(meta);
437
441
  row.ctx_tokens = remote ? 0 : (readNodeTelemetry(row.node_id).tokens_in ?? 0);
438
442
  row.asks = askMap[row.node_id] ?? 0;
439
443
  row.enriched = true;
440
- }));
444
+ }
441
445
  }
442
446
  /** goal (initial-prompt.md) and session parts are local disk reads — suppressed
443
447
  * to undefined for a remote source. */
@@ -95,7 +95,9 @@ export function reviveAll() {
95
95
  const result = { revived: [], failed: [] };
96
96
  for (const meta of listDisconnected()) {
97
97
  try {
98
- reviveNode(meta.node_id, { resume: true });
98
+ // Reconnecting a disconnected broker is recovery, not a wake — a node
99
+ // waiting on an armed deadline keeps it across the sweep.
100
+ reviveNode(meta.node_id, { resume: true, recovery: true });
99
101
  result.revived.push(meta.node_id);
100
102
  }
101
103
  catch (err) {
@@ -49,10 +49,17 @@ export interface ReviveResult {
49
49
  /** Relaunch `nodeId`'s broker engine from its persisted recipe and update canvas
50
50
  * meta. Opens no viewer (engine-only).
51
51
  *
52
+ * `recovery: true` marks a launch that only replaces a broker instance which
53
+ * died or was torn down — the daemon's respawn after a broker exit, and the
54
+ * `reviveAll` sweep over disconnected nodes. Nothing external arrived for the
55
+ * node, so such a launch is NOT a wake and does not consume the node's armed
56
+ * deadline (see the cancelCronsOnWake call below).
57
+ *
52
58
  * Throws if the node does not exist. All other failures propagate as-is —
53
59
  * callers (daemon, command) decide how to handle.
54
60
  */
55
61
  export declare function reviveNode(nodeId: string, opts: {
56
62
  resume: boolean;
57
63
  wakeReason?: ReviveWakeReason;
64
+ recovery?: boolean;
58
65
  }): ReviveResult;
@@ -92,6 +92,12 @@ function isUnstartedBirth(nodeId) {
92
92
  /** Relaunch `nodeId`'s broker engine from its persisted recipe and update canvas
93
93
  * meta. Opens no viewer (engine-only).
94
94
  *
95
+ * `recovery: true` marks a launch that only replaces a broker instance which
96
+ * died or was torn down — the daemon's respawn after a broker exit, and the
97
+ * `reviveAll` sweep over disconnected nodes. Nothing external arrived for the
98
+ * node, so such a launch is NOT a wake and does not consume the node's armed
99
+ * deadline (see the cancelCronsOnWake call below).
100
+ *
95
101
  * Throws if the node does not exist. All other failures propagate as-is —
96
102
  * callers (daemon, command) decide how to handle.
97
103
  */
@@ -309,9 +315,15 @@ export function reviveNode(nodeId, opts) {
309
315
  // unchecked result, is what scopes a later diagnostic read to this attempt.
310
316
  clearFault(nodeId);
311
317
  const bootFault = beginBootFaultAttempt(nodeId);
312
- // A cron armed --cancel-on-wake belongs to the dormancy being left, so any
313
- // revive of its anchor deletes it.
314
- cancelCronsOnWake(nodeId);
318
+ // A cron armed --cancel-on-wake belongs to the dormancy being left, so a WAKE
319
+ // of its anchor deletes it — an inbox delivery, an attach, an explicit
320
+ // revive. A recovery relaunch is not a wake: the broker instance died or was
321
+ // torn down and the daemon is putting the same waiting node back on its feet,
322
+ // with nothing external having arrived. The wait the deadline bounds is still
323
+ // open, so the deadline survives — the same rule lifecycle.ts applies to the
324
+ // `crash` transition (a deadline MUST survive instance death).
325
+ if (opts.recovery !== true)
326
+ cancelCronsOnWake(nodeId);
315
327
  let launched;
316
328
  try {
317
329
  launched = headlessBrokerHost.launch(nodeId, inv, {
@@ -352,6 +352,10 @@ export class DaemonFleet {
352
352
  reviveNode(nodeId, {
353
353
  resume: action === 'respawn-resume',
354
354
  wakeReason: cleanAbort ? 'runtime-restart-abort' : undefined,
355
+ // Replacing a dead broker instance is recovery, not a wake: nothing
356
+ // external arrived for this node, so an armed deadline survives the
357
+ // relaunch instead of being consumed by it.
358
+ recovery: true,
355
359
  });
356
360
  }
357
361
  catch (err) {