@north-light/crouter 0.3.168 → 0.3.169

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -85,6 +85,7 @@ const daemonRestart = defineLeaf({
85
85
  outputKind: 'object',
86
86
  effects: [
87
87
  'The daemon acknowledges immediately, then after a short grace tears its whole broker fleet down, releases its claim, spawns a successor on the runtime generation currently selected, and exits.',
88
+ 'Each torn-down broker hands its foreground bash commands to the file-backed background job system; they keep running and report completion by urgent inbox message.',
88
89
  'Every node it tore down — including the caller — is resumed by the successor through the ordinary startup recovery sweep.',
89
90
  'The calling process is killed as part of that teardown, AFTER this result is returned.',
90
91
  ],
@@ -35,6 +35,7 @@ import { fileURLToPath } from 'node:url';
35
35
  import { spawn, spawnSync } from 'node:child_process';
36
36
  import { createNode, getNode, subscribe, updateNode, clearPid } from '../canvas/canvas.js';
37
37
  import { armFollowUpWork } from '../canvas/human-work-outbox.js';
38
+ import { armCron, hasPendingCancelOnWakeCron } from '../canvas/crons.js';
38
39
  import { readInboxSince } from '../feed/inbox.js';
39
40
  import { closeDb } from '../canvas/db.js';
40
41
  import { buildPiArgv, CANVAS_EXTENSIONS } from '../runtime/launch.js';
@@ -346,6 +347,53 @@ test('reviveNode PROCEEDS when the node has NO fleet entry — cycle counter bum
346
347
  await h.dispose();
347
348
  }
348
349
  });
350
+ // ---------------------------------------------------------------------------
351
+ // BUG LOCKED — a recovery relaunch must not eat the node's armed deadline.
352
+ // A node armed `crtr node wait deadline` and dormed; its broker then exited and
353
+ // the daemon respawned it (fleet.#respawn → reviveNode). reviveNode used to
354
+ // cancel every cancel-on-wake cron unconditionally, so the deadline vanished
355
+ // with nothing having woken the node and the wait could never settle. A wake
356
+ // (inbox delivery, attach, explicit revive) still consumes it.
357
+ // ---------------------------------------------------------------------------
358
+ function armDeadline(anchor, cronId) {
359
+ armCron({
360
+ cron_id: cronId,
361
+ name: `deadline:${anchor}`,
362
+ created_by: anchor,
363
+ command: 'true',
364
+ fire_at: new Date(Date.now() + 3_600_000).toISOString(),
365
+ recur: null,
366
+ tz: null,
367
+ expires_at: null,
368
+ anchor_node: anchor,
369
+ cancel_on_wake: true,
370
+ cwd: home,
371
+ env_json: null,
372
+ profile: null,
373
+ scope: 'profile',
374
+ run_timeout_s: 300,
375
+ overlap: 'skip',
376
+ on_output: 'on-failure',
377
+ sink: '',
378
+ tier: 'normal',
379
+ });
380
+ }
381
+ test('a recovery relaunch keeps the armed deadline; a wake revive consumes it', async () => {
382
+ const h = await createHarness({ headless: true, sessionPrefix: 'crtr-revive-deadline' });
383
+ try {
384
+ const recovered = h.fabricateBrokerNode({ status: 'active', intent: null, pi_pid: deadPid(), pi_session_id: 'uuid-1' });
385
+ armDeadline(recovered, 'cron-recovery');
386
+ reviveNode(recovered, { resume: true, recovery: true });
387
+ assert.equal(hasPendingCancelOnWakeCron(recovered), true, 'the daemon replacing a dead broker is not a wake — the deadline survives so the wait can still settle');
388
+ const woken = h.fabricateBrokerNode({ status: 'active', intent: null, pi_pid: deadPid(), pi_session_id: 'uuid-2' });
389
+ armDeadline(woken, 'cron-wake');
390
+ reviveNode(woken, { resume: true });
391
+ assert.equal(hasPendingCancelOnWakeCron(woken), false, 'a wake revive won the race and cancels the deadline');
392
+ }
393
+ finally {
394
+ await h.dispose();
395
+ }
396
+ });
349
397
  test('reviveNode fresh-fallback clears the stale session identity (Major-3: no phantom stranded-relaunch)', async () => {
350
398
  const h = await createHarness({ headless: true, sessionPrefix: 'crtr-revive-freshfb' });
351
399
  try {
@@ -4,7 +4,9 @@ export interface BashJobPaths {
4
4
  cmdSh: string;
5
5
  jobLog: string;
6
6
  jobExit: string;
7
+ jobRun: string;
7
8
  jobBg: string;
9
+ jobDone: string;
8
10
  /** Process-group id of the job's detached supervisor, written at spawn. It is
9
11
  * what makes a job stoppable from outside the agent that started it (the
10
12
  * Inspector's cancel) — without it on disk, only the agent's own handoff
@@ -42,25 +44,22 @@ export interface BashJobStatus {
42
44
  logPath: string;
43
45
  }
44
46
  /** Every live backgrounded bash job for this node — job.bg present, job.exit
45
- * absent — whether handed off by the menu or auto-backgrounded at the valve's
46
- * deadline. Read-only; never mutates. Distinct from (and shares no code path
47
- * with) the foreground-only `runningBashJobs` above: that scanner deliberately
48
- * EXCLUDES job.bg, this one REQUIRES it, so reusing one for the other would
49
- * corrupt foreground-admission semantics. Skips malformed/partially-written
47
+ * absent — whether handed off by the menu, the valve deadline, or broker
48
+ * termination. Read-only; never mutates. Skips malformed/partially-written
50
49
  * directories. Returns oldest-first by startedAtMs. */
51
50
  export declare function activeBackgroundBashJobs(contextDir: string): BashJobStatus[];
52
- /** The start time (ms epoch) of the OLDEST still-foreground bash command for
51
+ export type BackgroundBashJobResult = 'backgrounded' | 'already-backgrounded' | 'finished' | 'missing';
52
+ /** Atomically hand one foreground job to its detached supervisor. The rename
53
+ * is the ownership decision: exactly one caller can move job.run to job.bg,
54
+ * while the supervisor races to move the same source to job.done. */
55
+ export declare function backgroundBashJob(paths: BashJobPaths): BackgroundBashJobResult;
56
+ /** The start time (ms epoch) of the oldest still-foreground bash command for
53
57
  * this node, or undefined when none is running. This is the authoritative
54
- * source for the attach viewer's R4 "background it" footer hint: it reads the
55
- * exact same on-disk contract the valve writes (a job directory exists with
56
- * no job.bg and no job.exit while the command is running in the foreground,
57
- * and the valve deletes the directory the moment the command finishes or is
58
- * aborted — see canvas-bash-valve.ts), so a fresh or reconnected attach (which
59
- * has no memory of a broker-relayed tool-execution event) sees exactly what's
60
- * really running, and a viewer never needs its own call-id tracking that could
61
- * go stale across a broker/session replacement. */
58
+ * source for the attach viewer's background hint: a fresh or reconnected
59
+ * attach reads the same job.run state that backgrounding and completion race
60
+ * to claim, without viewer-local call tracking that can go stale. */
62
61
  export declare function oldestRunningBashJobStartedAtMs(contextDir: string): number | undefined;
63
- /** Request immediate handoff for every bash command still waiting on pi. The
64
- * valve observes job.bg within one poll, resolves the tool call, and leaves its
65
- * detached supervisor to notify the node when the real command exits. */
62
+ /** Request immediate handoff for every bash command still waiting on pi. Only
63
+ * the caller that wins the job.run → job.bg rename reports the job as handed
64
+ * off; another backgrounder or the foreground-completion path may win the race. */
66
65
  export declare function backgroundRunningBashJobs(contextDir: string): BashJobPaths[];
@@ -1,9 +1,9 @@
1
- // File-backed control plane for bash commands that have been handed off by the
2
- // canvas bash valve. The valve starts every command in one of these directories;
3
- // the prefix menu requests backgrounding by creating job.bg. Keeping that signal
4
- // on disk means the command can outlive its pi process and still report its exit.
1
+ // File-backed control plane for bash commands started by the canvas bash valve.
2
+ // Every foreground command owns job.run; backgrounding atomically renames it to
3
+ // job.bg, while foreground completion renames it to job.done. The detached
4
+ // supervisor can therefore outlive pi and still decide whether to report exit.
5
5
  import { randomBytes } from 'node:crypto';
6
- import { closeSync, existsSync, openSync, readdirSync, readFileSync, readSync, statSync, writeFileSync } from 'node:fs';
6
+ import { closeSync, existsSync, openSync, readdirSync, readFileSync, readSync, renameSync, statSync } from 'node:fs';
7
7
  import { join } from 'node:path';
8
8
  export function bashJobsDir(contextDir) {
9
9
  return join(contextDir, 'jobs');
@@ -16,7 +16,9 @@ export function bashJobPaths(contextDir, jobId) {
16
16
  cmdSh: join(dir, 'cmd.sh'),
17
17
  jobLog: join(dir, 'job.log'),
18
18
  jobExit: join(dir, 'job.exit'),
19
+ jobRun: join(dir, 'job.run'),
19
20
  jobBg: join(dir, 'job.bg'),
21
+ jobDone: join(dir, 'job.done'),
20
22
  jobPgid: join(dir, 'job.pgid'),
21
23
  };
22
24
  }
@@ -94,11 +96,8 @@ export function formatBashElapsed(ms) {
94
96
  return `${hours}h ${String(minutes).padStart(2, '0')}m`;
95
97
  }
96
98
  /** Every live backgrounded bash job for this node — job.bg present, job.exit
97
- * absent — whether handed off by the menu or auto-backgrounded at the valve's
98
- * deadline. Read-only; never mutates. Distinct from (and shares no code path
99
- * with) the foreground-only `runningBashJobs` above: that scanner deliberately
100
- * EXCLUDES job.bg, this one REQUIRES it, so reusing one for the other would
101
- * corrupt foreground-admission semantics. Skips malformed/partially-written
99
+ * absent — whether handed off by the menu, the valve deadline, or broker
100
+ * termination. Read-only; never mutates. Skips malformed/partially-written
102
101
  * directories. Returns oldest-first by startedAtMs. */
103
102
  export function activeBackgroundBashJobs(contextDir) {
104
103
  let entries;
@@ -127,9 +126,27 @@ export function activeBackgroundBashJobs(contextDir) {
127
126
  }
128
127
  return jobs.sort((a, b) => a.startedAtMs - b.startedAtMs);
129
128
  }
130
- /** Every still-foreground bash command for this node. A completed command has
131
- * job.exit; an already-backgrounded one has job.bg and must not be selected a
132
- * second time. Invalid or concurrently-removed directories are ignored. */
129
+ /** Atomically hand one foreground job to its detached supervisor. The rename
130
+ * is the ownership decision: exactly one caller can move job.run to job.bg,
131
+ * while the supervisor races to move the same source to job.done. */
132
+ export function backgroundBashJob(paths) {
133
+ try {
134
+ renameSync(paths.jobRun, paths.jobBg);
135
+ return 'backgrounded';
136
+ }
137
+ catch (err) {
138
+ if (err.code !== 'ENOENT')
139
+ throw err;
140
+ if (existsSync(paths.jobBg))
141
+ return 'already-backgrounded';
142
+ if (existsSync(paths.jobDone))
143
+ return 'finished';
144
+ return 'missing';
145
+ }
146
+ }
147
+ /** Every still-foreground bash command for this node. job.run is the sole live
148
+ * foreground state; a handoff or foreground completion atomically removes it.
149
+ * Invalid or concurrently-removed directories are ignored. */
133
150
  function runningBashJobs(contextDir) {
134
151
  let entries;
135
152
  try {
@@ -140,21 +157,16 @@ function runningBashJobs(contextDir) {
140
157
  }
141
158
  return entries.flatMap((jobId) => {
142
159
  const paths = bashJobPaths(contextDir, jobId);
143
- return existsSync(paths.cmdSh) && existsSync(paths.jobLog) && !existsSync(paths.jobExit) && !existsSync(paths.jobBg)
160
+ return existsSync(paths.cmdSh) && existsSync(paths.jobLog) && existsSync(paths.jobRun)
144
161
  ? [paths]
145
162
  : [];
146
163
  });
147
164
  }
148
- /** The start time (ms epoch) of the OLDEST still-foreground bash command for
165
+ /** The start time (ms epoch) of the oldest still-foreground bash command for
149
166
  * this node, or undefined when none is running. This is the authoritative
150
- * source for the attach viewer's R4 "background it" footer hint: it reads the
151
- * exact same on-disk contract the valve writes (a job directory exists with
152
- * no job.bg and no job.exit while the command is running in the foreground,
153
- * and the valve deletes the directory the moment the command finishes or is
154
- * aborted — see canvas-bash-valve.ts), so a fresh or reconnected attach (which
155
- * has no memory of a broker-relayed tool-execution event) sees exactly what's
156
- * really running, and a viewer never needs its own call-id tracking that could
157
- * go stale across a broker/session replacement. */
167
+ * source for the attach viewer's background hint: a fresh or reconnected
168
+ * attach reads the same job.run state that backgrounding and completion race
169
+ * to claim, without viewer-local call tracking that can go stale. */
158
170
  export function oldestRunningBashJobStartedAtMs(contextDir) {
159
171
  let oldest;
160
172
  for (const paths of runningBashJobs(contextDir)) {
@@ -166,25 +178,14 @@ export function oldestRunningBashJobStartedAtMs(contextDir) {
166
178
  }
167
179
  return oldest;
168
180
  }
169
- /** Request immediate handoff for every bash command still waiting on pi. The
170
- * valve observes job.bg within one poll, resolves the tool call, and leaves its
171
- * detached supervisor to notify the node when the real command exits. */
181
+ /** Request immediate handoff for every bash command still waiting on pi. Only
182
+ * the caller that wins the job.run → job.bg rename reports the job as handed
183
+ * off; another backgrounder or the foreground-completion path may win the race. */
172
184
  export function backgroundRunningBashJobs(contextDir) {
173
185
  const backgrounded = [];
174
186
  for (const paths of runningBashJobs(contextDir)) {
175
- try {
176
- // A job may exit or be claimed by another prefix press between the scan
177
- // and this write. Recheck the two terminal/control files so that stale
178
- // candidates never masquerade as a successful handoff.
179
- if (existsSync(paths.jobExit) || existsSync(paths.jobBg))
180
- continue;
181
- writeFileSync(paths.jobBg, '');
187
+ if (backgroundBashJob(paths) === 'backgrounded')
182
188
  backgrounded.push(paths);
183
- }
184
- catch {
185
- // The valve owns cleanup; a concurrently removed directory is simply no
186
- // longer a running command to hand off.
187
- }
188
189
  }
189
190
  return backgrounded;
190
191
  }
@@ -102,6 +102,31 @@ test('dashboardRowsAllFromSource shows a node watched through its live broker',
102
102
  const rows = await dashboardRowsAllFromSource(source);
103
103
  assert.equal(rows[0]?.viewed, true);
104
104
  });
105
+ test('enrichRowsFromSource serializes a full dashboard source read', async () => {
106
+ const nodeRows = ['a', 'b', 'c'].map(row);
107
+ let inFlight = 0;
108
+ let maxInFlight = 0;
109
+ const source = {
110
+ getNode: async (id) => {
111
+ inFlight++;
112
+ maxInFlight = Math.max(maxInFlight, inFlight);
113
+ await new Promise((resolve) => setImmediate(resolve));
114
+ inFlight--;
115
+ return meta(id);
116
+ },
117
+ getRow: async () => null,
118
+ listNodes: async () => nodeRows,
119
+ subscriptionsOf: async () => [],
120
+ subscribersOf: async () => [],
121
+ view: async () => [],
122
+ ticketCountsForView: async () => ({}),
123
+ hasActiveLiveSubscription: async () => false,
124
+ };
125
+ const rows = await dashboardRowsAllFromSource(source);
126
+ await enrichRowsFromSource(source, rows);
127
+ assert.equal(maxInFlight, 1);
128
+ assert.deepEqual(rows.map((entry) => entry.name), ['a', 'b', 'c']);
129
+ });
105
130
  let home;
106
131
  let localCwd;
107
132
  let sessionFile;
@@ -160,8 +160,9 @@ export declare function setCronLastOutputHash(cron_id: string, hash: string): vo
160
160
  * terminal node that stopped without finishing. */
161
161
  export declare function hasPendingCancelOnWakeCron(anchor_node: string): boolean;
162
162
  /** DELETE every cancel-on-wake cron anchored to this node — the deadline rule.
163
- * Two seams call it: reviveNode on every revive-for-any-reason (the wake won
164
- * the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
163
+ * Two seams call it: reviveNode on a WAKE — every revive except a recovery
164
+ * relaunch, which only replaces a dead broker instance and leaves the wait
165
+ * open (the wake won the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
165
166
  * with the node, so its deadline must not fire against a finalized or closed
166
167
  * row). Standing declarative crons are untouched — only cancel-on-wake rows.
167
168
  * Kills any in-flight
@@ -219,8 +219,9 @@ export function hasPendingCancelOnWakeCron(anchor_node) {
219
219
  .get(anchor_node) !== undefined);
220
220
  }
221
221
  /** DELETE every cancel-on-wake cron anchored to this node — the deadline rule.
222
- * Two seams call it: reviveNode on every revive-for-any-reason (the wake won
223
- * the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
222
+ * Two seams call it: reviveNode on a WAKE — every revive except a recovery
223
+ * relaunch, which only replaces a dead broker instance and leaves the wait
224
+ * open (the wake won the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
224
225
  * with the node, so its deadline must not fire against a finalized or closed
225
226
  * row). Standing declarative crons are untouched — only cancel-on-wake rows.
226
227
  * Kills any in-flight
@@ -430,14 +430,18 @@ export async function enrichRowsFromSource(source, rows, asks) {
430
430
  return;
431
431
  const remote = source instanceof RemoteCanvasSource;
432
432
  const askMap = asks ?? {};
433
- await Promise.all(todo.map(async (row) => {
433
+ // A full dashboard may contain thousands of rows. Keep source reads serial:
434
+ // ApiCanvasSource turns each into a unix-socket request, and an unbounded
435
+ // Promise.all can overflow crtrd's accept backlog then spuriously trigger
436
+ // the client's cold-daemon path even though the daemon is still running.
437
+ for (const row of todo) {
434
438
  const meta = await source.getNode(row.node_id);
435
439
  if (meta !== null)
436
440
  row.name = fullName(meta);
437
441
  row.ctx_tokens = remote ? 0 : (readNodeTelemetry(row.node_id).tokens_in ?? 0);
438
442
  row.asks = askMap[row.node_id] ?? 0;
439
443
  row.enriched = true;
440
- }));
444
+ }
441
445
  }
442
446
  /** goal (initial-prompt.md) and session parts are local disk reads — suppressed
443
447
  * to undefined for a remote source. */
@@ -95,7 +95,9 @@ export function reviveAll() {
95
95
  const result = { revived: [], failed: [] };
96
96
  for (const meta of listDisconnected()) {
97
97
  try {
98
- reviveNode(meta.node_id, { resume: true });
98
+ // Reconnecting a disconnected broker is recovery, not a wake — a node
99
+ // waiting on an armed deadline keeps it across the sweep.
100
+ reviveNode(meta.node_id, { resume: true, recovery: true });
99
101
  result.revived.push(meta.node_id);
100
102
  }
101
103
  catch (err) {
@@ -49,10 +49,17 @@ export interface ReviveResult {
49
49
  /** Relaunch `nodeId`'s broker engine from its persisted recipe and update canvas
50
50
  * meta. Opens no viewer (engine-only).
51
51
  *
52
+ * `recovery: true` marks a launch that only replaces a broker instance which
53
+ * died or was torn down — the daemon's respawn after a broker exit, and the
54
+ * `reviveAll` sweep over disconnected nodes. Nothing external arrived for the
55
+ * node, so such a launch is NOT a wake and does not consume the node's armed
56
+ * deadline (see the cancelCronsOnWake call below).
57
+ *
52
58
  * Throws if the node does not exist. All other failures propagate as-is —
53
59
  * callers (daemon, command) decide how to handle.
54
60
  */
55
61
  export declare function reviveNode(nodeId: string, opts: {
56
62
  resume: boolean;
57
63
  wakeReason?: ReviveWakeReason;
64
+ recovery?: boolean;
58
65
  }): ReviveResult;
@@ -92,6 +92,12 @@ function isUnstartedBirth(nodeId) {
92
92
  /** Relaunch `nodeId`'s broker engine from its persisted recipe and update canvas
93
93
  * meta. Opens no viewer (engine-only).
94
94
  *
95
+ * `recovery: true` marks a launch that only replaces a broker instance which
96
+ * died or was torn down — the daemon's respawn after a broker exit, and the
97
+ * `reviveAll` sweep over disconnected nodes. Nothing external arrived for the
98
+ * node, so such a launch is NOT a wake and does not consume the node's armed
99
+ * deadline (see the cancelCronsOnWake call below).
100
+ *
95
101
  * Throws if the node does not exist. All other failures propagate as-is —
96
102
  * callers (daemon, command) decide how to handle.
97
103
  */
@@ -309,9 +315,15 @@ export function reviveNode(nodeId, opts) {
309
315
  // unchecked result, is what scopes a later diagnostic read to this attempt.
310
316
  clearFault(nodeId);
311
317
  const bootFault = beginBootFaultAttempt(nodeId);
312
- // A cron armed --cancel-on-wake belongs to the dormancy being left, so any
313
- // revive of its anchor deletes it.
314
- cancelCronsOnWake(nodeId);
318
+ // A cron armed --cancel-on-wake belongs to the dormancy being left, so a WAKE
319
+ // of its anchor deletes it — an inbox delivery, an attach, an explicit
320
+ // revive. A recovery relaunch is not a wake: the broker instance died or was
321
+ // torn down and the daemon is putting the same waiting node back on its feet,
322
+ // with nothing external having arrived. The wait the deadline bounds is still
323
+ // open, so the deadline survives — the same rule lifecycle.ts applies to the
324
+ // `crash` transition (a deadline MUST survive instance death).
325
+ if (opts.recovery !== true)
326
+ cancelCronsOnWake(nodeId);
315
327
  let launched;
316
328
  try {
317
329
  launched = headlessBrokerHost.launch(nodeId, inv, {
@@ -352,6 +352,10 @@ export class DaemonFleet {
352
352
  reviveNode(nodeId, {
353
353
  resume: action === 'respawn-resume',
354
354
  wakeReason: cleanAbort ? 'runtime-restart-abort' : undefined,
355
+ // Replacing a dead broker instance is recovery, not a wake: nothing
356
+ // external arrived for this node, so an armed deadline survives the
357
+ // relaunch instead of being consumed by it.
358
+ recovery: true,
355
359
  });
356
360
  }
357
361
  catch (err) {
@@ -27,7 +27,7 @@
27
27
  import { spawn } from 'node:child_process';
28
28
  import { closeSync, existsSync, mkdirSync, openSync, readFileSync, readSync, rmSync, statSync, writeFileSync, } from 'node:fs';
29
29
  import { homedir } from 'node:os';
30
- import { bashJobPaths, formatBashElapsed, newBashJobId } from '../core/bash-jobs.js';
30
+ import { backgroundBashJob, bashJobPaths, formatBashElapsed, newBashJobId } from '../core/bash-jobs.js';
31
31
  import { createBashToolDefinition } from '@earendil-works/pi-coding-agent';
32
32
  // ---------------------------------------------------------------------------
33
33
  // Deadline — 5 minutes by default, overridable via CRTR_BASH_VALVE_MS
@@ -49,7 +49,8 @@ function resolveValveMs() {
49
49
  //
50
50
  // Positional argv (never interpolated into the script — the command text
51
51
  // lives only in cmd.sh):
52
- // $1 cmd.sh $2 job.log $3 job.exit $4 job.bg $5 jobId $6 nodeId
52
+ // $1 cmd.sh $2 job.log $3 job.exit $4 job.run
53
+ // $5 job.done $6 job.bg $7 jobId $8 nodeId
53
54
  //
54
55
  // `--` before the paths is a dummy $0 (a standard `bash -c script -- args`
55
56
  // idiom) so the real payload lands at $1 onward.
@@ -58,13 +59,38 @@ const SUPERVISOR_SCRIPT = [
58
59
  'bash "$1" > "$2" 2>&1',
59
60
  'ec=$?',
60
61
  'echo "$ec" > "$3"',
61
- 'if [ -f "$4" ]; then',
62
- ' crtr node message send "Background bash job $5 finished (exit $ec). Log: $2 Command: $1" --to "$6" --tier urgent || true',
62
+ 'if mv "$4" "$5" 2>/dev/null; then',
63
+ ' :',
64
+ 'elif [ -f "$6" ]; then',
65
+ ' crtr node message send "Background bash job $7 finished (exit $ec). Log: $2 Command: $1" --to "$8" --tier urgent || true',
63
66
  'fi',
64
67
  ].join('\n');
65
68
  function makeJobPaths(contextDir) {
66
69
  return bashJobPaths(contextDir, newBashJobId());
67
70
  }
71
+ const liveForegroundJobs = new Set();
72
+ let terminationHandoffInstalled = false;
73
+ /** The broker is about to exit. Persist every foreground handoff synchronously
74
+ * before pi's own termination listener aborts its active tool calls. */
75
+ function backgroundLiveJobsForTermination() {
76
+ for (const paths of [...liveForegroundJobs]) {
77
+ try {
78
+ backgroundBashJob(paths);
79
+ liveForegroundJobs.delete(paths);
80
+ }
81
+ catch (err) {
82
+ process.stderr.write(`crtr: failed to background bash job ${paths.jobId} during broker termination: ${err.message}\n`);
83
+ }
84
+ }
85
+ }
86
+ function installTerminationHandoff() {
87
+ if (terminationHandoffInstalled)
88
+ return;
89
+ terminationHandoffInstalled = true;
90
+ process.prependListener('SIGTERM', backgroundLiveJobsForTermination);
91
+ if (process.platform !== 'win32')
92
+ process.prependListener('SIGHUP', backgroundLiveJobsForTermination);
93
+ }
68
94
  function killGroup(pgid) {
69
95
  try {
70
96
  process.kill(-pgid, 'SIGTERM');
@@ -194,6 +220,7 @@ export function createValveOperations(nodeId, contextDir) {
194
220
  mkdirSync(paths.dir, { recursive: true });
195
221
  writeFileSync(paths.cmdSh, command);
196
222
  writeFileSync(paths.jobLog, '');
223
+ writeFileSync(paths.jobRun, '');
197
224
  let offset = 0;
198
225
  let settled = false;
199
226
  let pgid;
@@ -205,6 +232,7 @@ export function createValveOperations(nodeId, contextDir) {
205
232
  clearInterval(interval);
206
233
  if (signal)
207
234
  signal.removeEventListener('abort', onAbort);
235
+ liveForegroundJobs.delete(paths);
208
236
  };
209
237
  const finish = (act) => {
210
238
  if (settled)
@@ -213,7 +241,7 @@ export function createValveOperations(nodeId, contextDir) {
213
241
  teardown();
214
242
  act();
215
243
  };
216
- function background(auto = false) {
244
+ function finishBackground(auto = false) {
217
245
  const elapsed = formatBashElapsed(Date.now() - startedAt);
218
246
  const notice = `\n[backgrounded as job ${paths.jobId} after ${elapsed} in the foreground${auto ? ' (auto)' : ''}. ` +
219
247
  `It will interrupt you with a message when it exits. ` +
@@ -222,8 +250,35 @@ export function createValveOperations(nodeId, contextDir) {
222
250
  finish(() => resolve({ exitCode: 0 }));
223
251
  // Backgrounded — job dir persists (inspectable); no cleanup here.
224
252
  }
253
+ function finishForeground() {
254
+ offset = tailOnce(paths.jobLog, offset, onData);
255
+ const exitCode = readExitCode(paths.jobExit);
256
+ finish(() => resolve({ exitCode }));
257
+ cleanupJobDir(paths.dir);
258
+ }
259
+ function requestBackground(auto = false) {
260
+ const state = backgroundBashJob(paths);
261
+ if (state === 'backgrounded' || state === 'already-backgrounded') {
262
+ finishBackground(auto);
263
+ return;
264
+ }
265
+ if (state === 'finished') {
266
+ finishForeground();
267
+ return;
268
+ }
269
+ finish(() => reject(new Error(`bash job ${paths.jobId} disappeared before it could be backgrounded`)));
270
+ cleanupJobDir(paths.dir);
271
+ }
225
272
  function onAbort() {
226
- tailOnce(paths.jobLog, offset, onData);
273
+ offset = tailOnce(paths.jobLog, offset, onData);
274
+ if (existsSync(paths.jobBg)) {
275
+ finishBackground();
276
+ return;
277
+ }
278
+ // Once the command has written its exit code, let the supervisor
279
+ // finish the job.run → job.done transition before cleanup.
280
+ if (existsSync(paths.jobExit))
281
+ return;
227
282
  if (pgid)
228
283
  killGroup(pgid);
229
284
  finish(() => reject(new Error('aborted')));
@@ -234,7 +289,7 @@ export function createValveOperations(nodeId, contextDir) {
234
289
  const execution = resolveExecutionCwd(cwd);
235
290
  if (execution.warning)
236
291
  onData(Buffer.from(execution.warning));
237
- const child = spawn('bash', ['-c', SUPERVISOR_SCRIPT, '--', paths.cmdSh, paths.jobLog, paths.jobExit, paths.jobBg, paths.jobId, nodeId], {
292
+ const child = spawn('bash', ['-c', SUPERVISOR_SCRIPT, '--', paths.cmdSh, paths.jobLog, paths.jobExit, paths.jobRun, paths.jobDone, paths.jobBg, paths.jobId, nodeId], {
238
293
  cwd: execution.cwd,
239
294
  detached: true,
240
295
  stdio: 'ignore',
@@ -242,27 +297,31 @@ export function createValveOperations(nodeId, contextDir) {
242
297
  });
243
298
  child.on('error', (err) => {
244
299
  finish(() => reject(err));
300
+ cleanupJobDir(paths.dir);
245
301
  });
246
302
  pgid = child.pid;
247
303
  // The detached supervisor IS the group leader, so its pid is the pgid.
248
304
  // Persisting it is what lets a human stop the job from the Inspector
249
305
  // later; the agent's own handoff notice quotes the same number.
250
- if (pgid !== undefined)
306
+ if (pgid !== undefined) {
251
307
  writeFileSync(paths.jobPgid, String(pgid));
308
+ liveForegroundJobs.add(paths);
309
+ }
252
310
  child.unref();
253
311
  interval = setInterval(() => {
254
312
  offset = tailOnce(paths.jobLog, offset, onData);
255
- if (existsSync(paths.jobExit)) {
256
- offset = tailOnce(paths.jobLog, offset, onData);
257
- const exitCode = readExitCode(paths.jobExit);
258
- finish(() => resolve({ exitCode }));
259
- cleanupJobDir(paths.dir);
313
+ if (existsSync(paths.jobDone)) {
314
+ finishForeground();
260
315
  return;
261
316
  }
262
317
  if (existsSync(paths.jobBg)) {
263
- background();
318
+ finishBackground();
264
319
  return;
265
320
  }
321
+ // job.exit is written before the supervisor claims job.done. Do not
322
+ // time out or delete the directory during that in-progress exit.
323
+ if (existsSync(paths.jobExit))
324
+ return;
266
325
  const elapsedMs = Date.now() - startedAt;
267
326
  if (timeout !== undefined && timeout > 0) {
268
327
  if (elapsedMs >= timeout * 1000) {
@@ -273,10 +332,8 @@ export function createValveOperations(nodeId, contextDir) {
273
332
  }
274
333
  return;
275
334
  }
276
- if (elapsedMs >= valveMs) {
277
- writeFileSync(paths.jobBg, '');
278
- background(true);
279
- }
335
+ if (elapsedMs >= valveMs)
336
+ requestBackground(true);
280
337
  }, POLL_MS);
281
338
  });
282
339
  },
@@ -292,6 +349,7 @@ export default function (pi) {
292
349
  const contextDir = process.env['CRTR_CONTEXT_DIR'];
293
350
  if (contextDir === undefined || contextDir.trim() === '')
294
351
  return; // defensive; always set alongside CRTR_NODE_ID
352
+ installTerminationHandoff();
295
353
  // Registered on session_start (fires for startup, new, resume, fork, and
296
354
  // /reload) so we get ctx.cwd — the same session cwd the builtin bash tool
297
355
  // is built against. registerTool replaces the builtin by name; re-firing on
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@north-light/crouter",
3
- "version": "0.3.168",
3
+ "version": "0.3.169",
4
4
  "description": "crtr — agent runtime with memory, plugins, and marketplaces",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
package/runtime.lock.json CHANGED
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "@north-light/crouter",
3
- "version": "0.3.168",
3
+ "version": "0.3.169",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "@north-light/crouter",
9
- "version": "0.3.168",
9
+ "version": "0.3.169",
10
10
  "hasInstallScript": true,
11
11
  "license": "MIT",
12
12
  "dependencies": {