@north-light/crouter 0.3.168 → 0.3.169
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +4 -0
- package/dist/clients/attach/viewer.js +236 -236
- package/dist/commands/sys/daemon.js +1 -0
- package/dist/core/__tests__/revive.test.js +48 -0
- package/dist/core/bash-jobs.d.ts +16 -17
- package/dist/core/bash-jobs.js +39 -38
- package/dist/core/canvas/__tests__/render-remote.test.js +25 -0
- package/dist/core/canvas/crons.d.ts +3 -2
- package/dist/core/canvas/crons.js +3 -2
- package/dist/core/canvas/render-source.js +6 -2
- package/dist/core/runtime/revive-all.js +3 -1
- package/dist/core/runtime/revive.d.ts +7 -0
- package/dist/core/runtime/revive.js +15 -3
- package/dist/daemon/fleet.js +4 -0
- package/dist/pi-extensions/canvas-bash-valve.js +76 -18
- package/package.json +1 -1
- package/runtime.lock.json +2 -2
- package/scripts/install-runtime.mjs +36 -13
|
@@ -85,6 +85,7 @@ const daemonRestart = defineLeaf({
|
|
|
85
85
|
outputKind: 'object',
|
|
86
86
|
effects: [
|
|
87
87
|
'The daemon acknowledges immediately, then after a short grace tears its whole broker fleet down, releases its claim, spawns a successor on the runtime generation currently selected, and exits.',
|
|
88
|
+
'Each torn-down broker hands its foreground bash commands to the file-backed background job system; they keep running and report completion by urgent inbox message.',
|
|
88
89
|
'Every node it tore down — including the caller — is resumed by the successor through the ordinary startup recovery sweep.',
|
|
89
90
|
'The calling process is killed as part of that teardown, AFTER this result is returned.',
|
|
90
91
|
],
|
|
@@ -35,6 +35,7 @@ import { fileURLToPath } from 'node:url';
|
|
|
35
35
|
import { spawn, spawnSync } from 'node:child_process';
|
|
36
36
|
import { createNode, getNode, subscribe, updateNode, clearPid } from '../canvas/canvas.js';
|
|
37
37
|
import { armFollowUpWork } from '../canvas/human-work-outbox.js';
|
|
38
|
+
import { armCron, hasPendingCancelOnWakeCron } from '../canvas/crons.js';
|
|
38
39
|
import { readInboxSince } from '../feed/inbox.js';
|
|
39
40
|
import { closeDb } from '../canvas/db.js';
|
|
40
41
|
import { buildPiArgv, CANVAS_EXTENSIONS } from '../runtime/launch.js';
|
|
@@ -346,6 +347,53 @@ test('reviveNode PROCEEDS when the node has NO fleet entry — cycle counter bum
|
|
|
346
347
|
await h.dispose();
|
|
347
348
|
}
|
|
348
349
|
});
|
|
350
|
+
// ---------------------------------------------------------------------------
|
|
351
|
+
// BUG LOCKED — a recovery relaunch must not eat the node's armed deadline.
|
|
352
|
+
// A node armed `crtr node wait deadline` and dormed; its broker then exited and
|
|
353
|
+
// the daemon respawned it (fleet.#respawn → reviveNode). reviveNode used to
|
|
354
|
+
// cancel every cancel-on-wake cron unconditionally, so the deadline vanished
|
|
355
|
+
// with nothing having woken the node and the wait could never settle. A wake
|
|
356
|
+
// (inbox delivery, attach, explicit revive) still consumes it.
|
|
357
|
+
// ---------------------------------------------------------------------------
|
|
358
|
+
function armDeadline(anchor, cronId) {
|
|
359
|
+
armCron({
|
|
360
|
+
cron_id: cronId,
|
|
361
|
+
name: `deadline:${anchor}`,
|
|
362
|
+
created_by: anchor,
|
|
363
|
+
command: 'true',
|
|
364
|
+
fire_at: new Date(Date.now() + 3_600_000).toISOString(),
|
|
365
|
+
recur: null,
|
|
366
|
+
tz: null,
|
|
367
|
+
expires_at: null,
|
|
368
|
+
anchor_node: anchor,
|
|
369
|
+
cancel_on_wake: true,
|
|
370
|
+
cwd: home,
|
|
371
|
+
env_json: null,
|
|
372
|
+
profile: null,
|
|
373
|
+
scope: 'profile',
|
|
374
|
+
run_timeout_s: 300,
|
|
375
|
+
overlap: 'skip',
|
|
376
|
+
on_output: 'on-failure',
|
|
377
|
+
sink: '',
|
|
378
|
+
tier: 'normal',
|
|
379
|
+
});
|
|
380
|
+
}
|
|
381
|
+
test('a recovery relaunch keeps the armed deadline; a wake revive consumes it', async () => {
|
|
382
|
+
const h = await createHarness({ headless: true, sessionPrefix: 'crtr-revive-deadline' });
|
|
383
|
+
try {
|
|
384
|
+
const recovered = h.fabricateBrokerNode({ status: 'active', intent: null, pi_pid: deadPid(), pi_session_id: 'uuid-1' });
|
|
385
|
+
armDeadline(recovered, 'cron-recovery');
|
|
386
|
+
reviveNode(recovered, { resume: true, recovery: true });
|
|
387
|
+
assert.equal(hasPendingCancelOnWakeCron(recovered), true, 'the daemon replacing a dead broker is not a wake — the deadline survives so the wait can still settle');
|
|
388
|
+
const woken = h.fabricateBrokerNode({ status: 'active', intent: null, pi_pid: deadPid(), pi_session_id: 'uuid-2' });
|
|
389
|
+
armDeadline(woken, 'cron-wake');
|
|
390
|
+
reviveNode(woken, { resume: true });
|
|
391
|
+
assert.equal(hasPendingCancelOnWakeCron(woken), false, 'a wake revive won the race and cancels the deadline');
|
|
392
|
+
}
|
|
393
|
+
finally {
|
|
394
|
+
await h.dispose();
|
|
395
|
+
}
|
|
396
|
+
});
|
|
349
397
|
test('reviveNode fresh-fallback clears the stale session identity (Major-3: no phantom stranded-relaunch)', async () => {
|
|
350
398
|
const h = await createHarness({ headless: true, sessionPrefix: 'crtr-revive-freshfb' });
|
|
351
399
|
try {
|
package/dist/core/bash-jobs.d.ts
CHANGED
|
@@ -4,7 +4,9 @@ export interface BashJobPaths {
|
|
|
4
4
|
cmdSh: string;
|
|
5
5
|
jobLog: string;
|
|
6
6
|
jobExit: string;
|
|
7
|
+
jobRun: string;
|
|
7
8
|
jobBg: string;
|
|
9
|
+
jobDone: string;
|
|
8
10
|
/** Process-group id of the job's detached supervisor, written at spawn. It is
|
|
9
11
|
* what makes a job stoppable from outside the agent that started it (the
|
|
10
12
|
* Inspector's cancel) — without it on disk, only the agent's own handoff
|
|
@@ -42,25 +44,22 @@ export interface BashJobStatus {
|
|
|
42
44
|
logPath: string;
|
|
43
45
|
}
|
|
44
46
|
/** Every live backgrounded bash job for this node — job.bg present, job.exit
|
|
45
|
-
* absent — whether handed off by the menu
|
|
46
|
-
*
|
|
47
|
-
* with) the foreground-only `runningBashJobs` above: that scanner deliberately
|
|
48
|
-
* EXCLUDES job.bg, this one REQUIRES it, so reusing one for the other would
|
|
49
|
-
* corrupt foreground-admission semantics. Skips malformed/partially-written
|
|
47
|
+
* absent — whether handed off by the menu, the valve deadline, or broker
|
|
48
|
+
* termination. Read-only; never mutates. Skips malformed/partially-written
|
|
50
49
|
* directories. Returns oldest-first by startedAtMs. */
|
|
51
50
|
export declare function activeBackgroundBashJobs(contextDir: string): BashJobStatus[];
|
|
52
|
-
|
|
51
|
+
export type BackgroundBashJobResult = 'backgrounded' | 'already-backgrounded' | 'finished' | 'missing';
|
|
52
|
+
/** Atomically hand one foreground job to its detached supervisor. The rename
|
|
53
|
+
* is the ownership decision: exactly one caller can move job.run to job.bg,
|
|
54
|
+
* while the supervisor races to move the same source to job.done. */
|
|
55
|
+
export declare function backgroundBashJob(paths: BashJobPaths): BackgroundBashJobResult;
|
|
56
|
+
/** The start time (ms epoch) of the oldest still-foreground bash command for
|
|
53
57
|
* this node, or undefined when none is running. This is the authoritative
|
|
54
|
-
* source for the attach viewer's
|
|
55
|
-
*
|
|
56
|
-
*
|
|
57
|
-
* and the valve deletes the directory the moment the command finishes or is
|
|
58
|
-
* aborted — see canvas-bash-valve.ts), so a fresh or reconnected attach (which
|
|
59
|
-
* has no memory of a broker-relayed tool-execution event) sees exactly what's
|
|
60
|
-
* really running, and a viewer never needs its own call-id tracking that could
|
|
61
|
-
* go stale across a broker/session replacement. */
|
|
58
|
+
* source for the attach viewer's background hint: a fresh or reconnected
|
|
59
|
+
* attach reads the same job.run state that backgrounding and completion race
|
|
60
|
+
* to claim, without viewer-local call tracking that can go stale. */
|
|
62
61
|
export declare function oldestRunningBashJobStartedAtMs(contextDir: string): number | undefined;
|
|
63
|
-
/** Request immediate handoff for every bash command still waiting on pi.
|
|
64
|
-
*
|
|
65
|
-
*
|
|
62
|
+
/** Request immediate handoff for every bash command still waiting on pi. Only
|
|
63
|
+
* the caller that wins the job.run → job.bg rename reports the job as handed
|
|
64
|
+
* off; another backgrounder or the foreground-completion path may win the race. */
|
|
66
65
|
export declare function backgroundRunningBashJobs(contextDir: string): BashJobPaths[];
|
package/dist/core/bash-jobs.js
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
// File-backed control plane for bash commands
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
1
|
+
// File-backed control plane for bash commands started by the canvas bash valve.
|
|
2
|
+
// Every foreground command owns job.run; backgrounding atomically renames it to
|
|
3
|
+
// job.bg, while foreground completion renames it to job.done. The detached
|
|
4
|
+
// supervisor can therefore outlive pi and still decide whether to report exit.
|
|
5
5
|
import { randomBytes } from 'node:crypto';
|
|
6
|
-
import { closeSync, existsSync, openSync, readdirSync, readFileSync, readSync,
|
|
6
|
+
import { closeSync, existsSync, openSync, readdirSync, readFileSync, readSync, renameSync, statSync } from 'node:fs';
|
|
7
7
|
import { join } from 'node:path';
|
|
8
8
|
export function bashJobsDir(contextDir) {
|
|
9
9
|
return join(contextDir, 'jobs');
|
|
@@ -16,7 +16,9 @@ export function bashJobPaths(contextDir, jobId) {
|
|
|
16
16
|
cmdSh: join(dir, 'cmd.sh'),
|
|
17
17
|
jobLog: join(dir, 'job.log'),
|
|
18
18
|
jobExit: join(dir, 'job.exit'),
|
|
19
|
+
jobRun: join(dir, 'job.run'),
|
|
19
20
|
jobBg: join(dir, 'job.bg'),
|
|
21
|
+
jobDone: join(dir, 'job.done'),
|
|
20
22
|
jobPgid: join(dir, 'job.pgid'),
|
|
21
23
|
};
|
|
22
24
|
}
|
|
@@ -94,11 +96,8 @@ export function formatBashElapsed(ms) {
|
|
|
94
96
|
return `${hours}h ${String(minutes).padStart(2, '0')}m`;
|
|
95
97
|
}
|
|
96
98
|
/** Every live backgrounded bash job for this node — job.bg present, job.exit
|
|
97
|
-
* absent — whether handed off by the menu
|
|
98
|
-
*
|
|
99
|
-
* with) the foreground-only `runningBashJobs` above: that scanner deliberately
|
|
100
|
-
* EXCLUDES job.bg, this one REQUIRES it, so reusing one for the other would
|
|
101
|
-
* corrupt foreground-admission semantics. Skips malformed/partially-written
|
|
99
|
+
* absent — whether handed off by the menu, the valve deadline, or broker
|
|
100
|
+
* termination. Read-only; never mutates. Skips malformed/partially-written
|
|
102
101
|
* directories. Returns oldest-first by startedAtMs. */
|
|
103
102
|
export function activeBackgroundBashJobs(contextDir) {
|
|
104
103
|
let entries;
|
|
@@ -127,9 +126,27 @@ export function activeBackgroundBashJobs(contextDir) {
|
|
|
127
126
|
}
|
|
128
127
|
return jobs.sort((a, b) => a.startedAtMs - b.startedAtMs);
|
|
129
128
|
}
|
|
130
|
-
/**
|
|
131
|
-
*
|
|
132
|
-
*
|
|
129
|
+
/** Atomically hand one foreground job to its detached supervisor. The rename
|
|
130
|
+
* is the ownership decision: exactly one caller can move job.run to job.bg,
|
|
131
|
+
* while the supervisor races to move the same source to job.done. */
|
|
132
|
+
export function backgroundBashJob(paths) {
|
|
133
|
+
try {
|
|
134
|
+
renameSync(paths.jobRun, paths.jobBg);
|
|
135
|
+
return 'backgrounded';
|
|
136
|
+
}
|
|
137
|
+
catch (err) {
|
|
138
|
+
if (err.code !== 'ENOENT')
|
|
139
|
+
throw err;
|
|
140
|
+
if (existsSync(paths.jobBg))
|
|
141
|
+
return 'already-backgrounded';
|
|
142
|
+
if (existsSync(paths.jobDone))
|
|
143
|
+
return 'finished';
|
|
144
|
+
return 'missing';
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
/** Every still-foreground bash command for this node. job.run is the sole live
|
|
148
|
+
* foreground state; a handoff or foreground completion atomically removes it.
|
|
149
|
+
* Invalid or concurrently-removed directories are ignored. */
|
|
133
150
|
function runningBashJobs(contextDir) {
|
|
134
151
|
let entries;
|
|
135
152
|
try {
|
|
@@ -140,21 +157,16 @@ function runningBashJobs(contextDir) {
|
|
|
140
157
|
}
|
|
141
158
|
return entries.flatMap((jobId) => {
|
|
142
159
|
const paths = bashJobPaths(contextDir, jobId);
|
|
143
|
-
return existsSync(paths.cmdSh) && existsSync(paths.jobLog) &&
|
|
160
|
+
return existsSync(paths.cmdSh) && existsSync(paths.jobLog) && existsSync(paths.jobRun)
|
|
144
161
|
? [paths]
|
|
145
162
|
: [];
|
|
146
163
|
});
|
|
147
164
|
}
|
|
148
|
-
/** The start time (ms epoch) of the
|
|
165
|
+
/** The start time (ms epoch) of the oldest still-foreground bash command for
|
|
149
166
|
* this node, or undefined when none is running. This is the authoritative
|
|
150
|
-
* source for the attach viewer's
|
|
151
|
-
*
|
|
152
|
-
*
|
|
153
|
-
* and the valve deletes the directory the moment the command finishes or is
|
|
154
|
-
* aborted — see canvas-bash-valve.ts), so a fresh or reconnected attach (which
|
|
155
|
-
* has no memory of a broker-relayed tool-execution event) sees exactly what's
|
|
156
|
-
* really running, and a viewer never needs its own call-id tracking that could
|
|
157
|
-
* go stale across a broker/session replacement. */
|
|
167
|
+
* source for the attach viewer's background hint: a fresh or reconnected
|
|
168
|
+
* attach reads the same job.run state that backgrounding and completion race
|
|
169
|
+
* to claim, without viewer-local call tracking that can go stale. */
|
|
158
170
|
export function oldestRunningBashJobStartedAtMs(contextDir) {
|
|
159
171
|
let oldest;
|
|
160
172
|
for (const paths of runningBashJobs(contextDir)) {
|
|
@@ -166,25 +178,14 @@ export function oldestRunningBashJobStartedAtMs(contextDir) {
|
|
|
166
178
|
}
|
|
167
179
|
return oldest;
|
|
168
180
|
}
|
|
169
|
-
/** Request immediate handoff for every bash command still waiting on pi.
|
|
170
|
-
*
|
|
171
|
-
*
|
|
181
|
+
/** Request immediate handoff for every bash command still waiting on pi. Only
|
|
182
|
+
* the caller that wins the job.run → job.bg rename reports the job as handed
|
|
183
|
+
* off; another backgrounder or the foreground-completion path may win the race. */
|
|
172
184
|
export function backgroundRunningBashJobs(contextDir) {
|
|
173
185
|
const backgrounded = [];
|
|
174
186
|
for (const paths of runningBashJobs(contextDir)) {
|
|
175
|
-
|
|
176
|
-
// A job may exit or be claimed by another prefix press between the scan
|
|
177
|
-
// and this write. Recheck the two terminal/control files so that stale
|
|
178
|
-
// candidates never masquerade as a successful handoff.
|
|
179
|
-
if (existsSync(paths.jobExit) || existsSync(paths.jobBg))
|
|
180
|
-
continue;
|
|
181
|
-
writeFileSync(paths.jobBg, '');
|
|
187
|
+
if (backgroundBashJob(paths) === 'backgrounded')
|
|
182
188
|
backgrounded.push(paths);
|
|
183
|
-
}
|
|
184
|
-
catch {
|
|
185
|
-
// The valve owns cleanup; a concurrently removed directory is simply no
|
|
186
|
-
// longer a running command to hand off.
|
|
187
|
-
}
|
|
188
189
|
}
|
|
189
190
|
return backgrounded;
|
|
190
191
|
}
|
|
@@ -102,6 +102,31 @@ test('dashboardRowsAllFromSource shows a node watched through its live broker',
|
|
|
102
102
|
const rows = await dashboardRowsAllFromSource(source);
|
|
103
103
|
assert.equal(rows[0]?.viewed, true);
|
|
104
104
|
});
|
|
105
|
+
test('enrichRowsFromSource serializes a full dashboard source read', async () => {
|
|
106
|
+
const nodeRows = ['a', 'b', 'c'].map(row);
|
|
107
|
+
let inFlight = 0;
|
|
108
|
+
let maxInFlight = 0;
|
|
109
|
+
const source = {
|
|
110
|
+
getNode: async (id) => {
|
|
111
|
+
inFlight++;
|
|
112
|
+
maxInFlight = Math.max(maxInFlight, inFlight);
|
|
113
|
+
await new Promise((resolve) => setImmediate(resolve));
|
|
114
|
+
inFlight--;
|
|
115
|
+
return meta(id);
|
|
116
|
+
},
|
|
117
|
+
getRow: async () => null,
|
|
118
|
+
listNodes: async () => nodeRows,
|
|
119
|
+
subscriptionsOf: async () => [],
|
|
120
|
+
subscribersOf: async () => [],
|
|
121
|
+
view: async () => [],
|
|
122
|
+
ticketCountsForView: async () => ({}),
|
|
123
|
+
hasActiveLiveSubscription: async () => false,
|
|
124
|
+
};
|
|
125
|
+
const rows = await dashboardRowsAllFromSource(source);
|
|
126
|
+
await enrichRowsFromSource(source, rows);
|
|
127
|
+
assert.equal(maxInFlight, 1);
|
|
128
|
+
assert.deepEqual(rows.map((entry) => entry.name), ['a', 'b', 'c']);
|
|
129
|
+
});
|
|
105
130
|
let home;
|
|
106
131
|
let localCwd;
|
|
107
132
|
let sessionFile;
|
|
@@ -160,8 +160,9 @@ export declare function setCronLastOutputHash(cron_id: string, hash: string): vo
|
|
|
160
160
|
* terminal node that stopped without finishing. */
|
|
161
161
|
export declare function hasPendingCancelOnWakeCron(anchor_node: string): boolean;
|
|
162
162
|
/** DELETE every cancel-on-wake cron anchored to this node — the deadline rule.
|
|
163
|
-
* Two seams call it: reviveNode on every revive
|
|
164
|
-
*
|
|
163
|
+
* Two seams call it: reviveNode on a WAKE — every revive except a recovery
|
|
164
|
+
* relaunch, which only replaces a dead broker instance and leaves the wait
|
|
165
|
+
* open (the wake won the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
|
|
165
166
|
* with the node, so its deadline must not fire against a finalized or closed
|
|
166
167
|
* row). Standing declarative crons are untouched — only cancel-on-wake rows.
|
|
167
168
|
* Kills any in-flight
|
|
@@ -219,8 +219,9 @@ export function hasPendingCancelOnWakeCron(anchor_node) {
|
|
|
219
219
|
.get(anchor_node) !== undefined);
|
|
220
220
|
}
|
|
221
221
|
/** DELETE every cancel-on-wake cron anchored to this node — the deadline rule.
|
|
222
|
-
* Two seams call it: reviveNode on every revive
|
|
223
|
-
*
|
|
222
|
+
* Two seams call it: reviveNode on a WAKE — every revive except a recovery
|
|
223
|
+
* relaunch, which only replaces a dead broker instance and leaves the wait
|
|
224
|
+
* open (the wake won the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
|
|
224
225
|
* with the node, so its deadline must not fire against a finalized or closed
|
|
225
226
|
* row). Standing declarative crons are untouched — only cancel-on-wake rows.
|
|
226
227
|
* Kills any in-flight
|
|
@@ -430,14 +430,18 @@ export async function enrichRowsFromSource(source, rows, asks) {
|
|
|
430
430
|
return;
|
|
431
431
|
const remote = source instanceof RemoteCanvasSource;
|
|
432
432
|
const askMap = asks ?? {};
|
|
433
|
-
|
|
433
|
+
// A full dashboard may contain thousands of rows. Keep source reads serial:
|
|
434
|
+
// ApiCanvasSource turns each into a unix-socket request, and an unbounded
|
|
435
|
+
// Promise.all can overflow crtrd's accept backlog then spuriously trigger
|
|
436
|
+
// the client's cold-daemon path even though the daemon is still running.
|
|
437
|
+
for (const row of todo) {
|
|
434
438
|
const meta = await source.getNode(row.node_id);
|
|
435
439
|
if (meta !== null)
|
|
436
440
|
row.name = fullName(meta);
|
|
437
441
|
row.ctx_tokens = remote ? 0 : (readNodeTelemetry(row.node_id).tokens_in ?? 0);
|
|
438
442
|
row.asks = askMap[row.node_id] ?? 0;
|
|
439
443
|
row.enriched = true;
|
|
440
|
-
}
|
|
444
|
+
}
|
|
441
445
|
}
|
|
442
446
|
/** goal (initial-prompt.md) and session parts are local disk reads — suppressed
|
|
443
447
|
* to undefined for a remote source. */
|
|
@@ -95,7 +95,9 @@ export function reviveAll() {
|
|
|
95
95
|
const result = { revived: [], failed: [] };
|
|
96
96
|
for (const meta of listDisconnected()) {
|
|
97
97
|
try {
|
|
98
|
-
|
|
98
|
+
// Reconnecting a disconnected broker is recovery, not a wake — a node
|
|
99
|
+
// waiting on an armed deadline keeps it across the sweep.
|
|
100
|
+
reviveNode(meta.node_id, { resume: true, recovery: true });
|
|
99
101
|
result.revived.push(meta.node_id);
|
|
100
102
|
}
|
|
101
103
|
catch (err) {
|
|
@@ -49,10 +49,17 @@ export interface ReviveResult {
|
|
|
49
49
|
/** Relaunch `nodeId`'s broker engine from its persisted recipe and update canvas
|
|
50
50
|
* meta. Opens no viewer (engine-only).
|
|
51
51
|
*
|
|
52
|
+
* `recovery: true` marks a launch that only replaces a broker instance which
|
|
53
|
+
* died or was torn down — the daemon's respawn after a broker exit, and the
|
|
54
|
+
* `reviveAll` sweep over disconnected nodes. Nothing external arrived for the
|
|
55
|
+
* node, so such a launch is NOT a wake and does not consume the node's armed
|
|
56
|
+
* deadline (see the cancelCronsOnWake call below).
|
|
57
|
+
*
|
|
52
58
|
* Throws if the node does not exist. All other failures propagate as-is —
|
|
53
59
|
* callers (daemon, command) decide how to handle.
|
|
54
60
|
*/
|
|
55
61
|
export declare function reviveNode(nodeId: string, opts: {
|
|
56
62
|
resume: boolean;
|
|
57
63
|
wakeReason?: ReviveWakeReason;
|
|
64
|
+
recovery?: boolean;
|
|
58
65
|
}): ReviveResult;
|
|
@@ -92,6 +92,12 @@ function isUnstartedBirth(nodeId) {
|
|
|
92
92
|
/** Relaunch `nodeId`'s broker engine from its persisted recipe and update canvas
|
|
93
93
|
* meta. Opens no viewer (engine-only).
|
|
94
94
|
*
|
|
95
|
+
* `recovery: true` marks a launch that only replaces a broker instance which
|
|
96
|
+
* died or was torn down — the daemon's respawn after a broker exit, and the
|
|
97
|
+
* `reviveAll` sweep over disconnected nodes. Nothing external arrived for the
|
|
98
|
+
* node, so such a launch is NOT a wake and does not consume the node's armed
|
|
99
|
+
* deadline (see the cancelCronsOnWake call below).
|
|
100
|
+
*
|
|
95
101
|
* Throws if the node does not exist. All other failures propagate as-is —
|
|
96
102
|
* callers (daemon, command) decide how to handle.
|
|
97
103
|
*/
|
|
@@ -309,9 +315,15 @@ export function reviveNode(nodeId, opts) {
|
|
|
309
315
|
// unchecked result, is what scopes a later diagnostic read to this attempt.
|
|
310
316
|
clearFault(nodeId);
|
|
311
317
|
const bootFault = beginBootFaultAttempt(nodeId);
|
|
312
|
-
// A cron armed --cancel-on-wake belongs to the dormancy being left, so
|
|
313
|
-
//
|
|
314
|
-
|
|
318
|
+
// A cron armed --cancel-on-wake belongs to the dormancy being left, so a WAKE
|
|
319
|
+
// of its anchor deletes it — an inbox delivery, an attach, an explicit
|
|
320
|
+
// revive. A recovery relaunch is not a wake: the broker instance died or was
|
|
321
|
+
// torn down and the daemon is putting the same waiting node back on its feet,
|
|
322
|
+
// with nothing external having arrived. The wait the deadline bounds is still
|
|
323
|
+
// open, so the deadline survives — the same rule lifecycle.ts applies to the
|
|
324
|
+
// `crash` transition (a deadline MUST survive instance death).
|
|
325
|
+
if (opts.recovery !== true)
|
|
326
|
+
cancelCronsOnWake(nodeId);
|
|
315
327
|
let launched;
|
|
316
328
|
try {
|
|
317
329
|
launched = headlessBrokerHost.launch(nodeId, inv, {
|
package/dist/daemon/fleet.js
CHANGED
|
@@ -352,6 +352,10 @@ export class DaemonFleet {
|
|
|
352
352
|
reviveNode(nodeId, {
|
|
353
353
|
resume: action === 'respawn-resume',
|
|
354
354
|
wakeReason: cleanAbort ? 'runtime-restart-abort' : undefined,
|
|
355
|
+
// Replacing a dead broker instance is recovery, not a wake: nothing
|
|
356
|
+
// external arrived for this node, so an armed deadline survives the
|
|
357
|
+
// relaunch instead of being consumed by it.
|
|
358
|
+
recovery: true,
|
|
355
359
|
});
|
|
356
360
|
}
|
|
357
361
|
catch (err) {
|
|
@@ -27,7 +27,7 @@
|
|
|
27
27
|
import { spawn } from 'node:child_process';
|
|
28
28
|
import { closeSync, existsSync, mkdirSync, openSync, readFileSync, readSync, rmSync, statSync, writeFileSync, } from 'node:fs';
|
|
29
29
|
import { homedir } from 'node:os';
|
|
30
|
-
import { bashJobPaths, formatBashElapsed, newBashJobId } from '../core/bash-jobs.js';
|
|
30
|
+
import { backgroundBashJob, bashJobPaths, formatBashElapsed, newBashJobId } from '../core/bash-jobs.js';
|
|
31
31
|
import { createBashToolDefinition } from '@earendil-works/pi-coding-agent';
|
|
32
32
|
// ---------------------------------------------------------------------------
|
|
33
33
|
// Deadline — 5 minutes by default, overridable via CRTR_BASH_VALVE_MS
|
|
@@ -49,7 +49,8 @@ function resolveValveMs() {
|
|
|
49
49
|
//
|
|
50
50
|
// Positional argv (never interpolated into the script — the command text
|
|
51
51
|
// lives only in cmd.sh):
|
|
52
|
-
// $1 cmd.sh
|
|
52
|
+
// $1 cmd.sh $2 job.log $3 job.exit $4 job.run
|
|
53
|
+
// $5 job.done $6 job.bg $7 jobId $8 nodeId
|
|
53
54
|
//
|
|
54
55
|
// `--` before the paths is a dummy $0 (a standard `bash -c script -- args`
|
|
55
56
|
// idiom) so the real payload lands at $1 onward.
|
|
@@ -58,13 +59,38 @@ const SUPERVISOR_SCRIPT = [
|
|
|
58
59
|
'bash "$1" > "$2" 2>&1',
|
|
59
60
|
'ec=$?',
|
|
60
61
|
'echo "$ec" > "$3"',
|
|
61
|
-
'if
|
|
62
|
-
'
|
|
62
|
+
'if mv "$4" "$5" 2>/dev/null; then',
|
|
63
|
+
' :',
|
|
64
|
+
'elif [ -f "$6" ]; then',
|
|
65
|
+
' crtr node message send "Background bash job $7 finished (exit $ec). Log: $2 Command: $1" --to "$8" --tier urgent || true',
|
|
63
66
|
'fi',
|
|
64
67
|
].join('\n');
|
|
65
68
|
function makeJobPaths(contextDir) {
|
|
66
69
|
return bashJobPaths(contextDir, newBashJobId());
|
|
67
70
|
}
|
|
71
|
+
const liveForegroundJobs = new Set();
|
|
72
|
+
let terminationHandoffInstalled = false;
|
|
73
|
+
/** The broker is about to exit. Persist every foreground handoff synchronously
|
|
74
|
+
* before pi's own termination listener aborts its active tool calls. */
|
|
75
|
+
function backgroundLiveJobsForTermination() {
|
|
76
|
+
for (const paths of [...liveForegroundJobs]) {
|
|
77
|
+
try {
|
|
78
|
+
backgroundBashJob(paths);
|
|
79
|
+
liveForegroundJobs.delete(paths);
|
|
80
|
+
}
|
|
81
|
+
catch (err) {
|
|
82
|
+
process.stderr.write(`crtr: failed to background bash job ${paths.jobId} during broker termination: ${err.message}\n`);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
function installTerminationHandoff() {
|
|
87
|
+
if (terminationHandoffInstalled)
|
|
88
|
+
return;
|
|
89
|
+
terminationHandoffInstalled = true;
|
|
90
|
+
process.prependListener('SIGTERM', backgroundLiveJobsForTermination);
|
|
91
|
+
if (process.platform !== 'win32')
|
|
92
|
+
process.prependListener('SIGHUP', backgroundLiveJobsForTermination);
|
|
93
|
+
}
|
|
68
94
|
function killGroup(pgid) {
|
|
69
95
|
try {
|
|
70
96
|
process.kill(-pgid, 'SIGTERM');
|
|
@@ -194,6 +220,7 @@ export function createValveOperations(nodeId, contextDir) {
|
|
|
194
220
|
mkdirSync(paths.dir, { recursive: true });
|
|
195
221
|
writeFileSync(paths.cmdSh, command);
|
|
196
222
|
writeFileSync(paths.jobLog, '');
|
|
223
|
+
writeFileSync(paths.jobRun, '');
|
|
197
224
|
let offset = 0;
|
|
198
225
|
let settled = false;
|
|
199
226
|
let pgid;
|
|
@@ -205,6 +232,7 @@ export function createValveOperations(nodeId, contextDir) {
|
|
|
205
232
|
clearInterval(interval);
|
|
206
233
|
if (signal)
|
|
207
234
|
signal.removeEventListener('abort', onAbort);
|
|
235
|
+
liveForegroundJobs.delete(paths);
|
|
208
236
|
};
|
|
209
237
|
const finish = (act) => {
|
|
210
238
|
if (settled)
|
|
@@ -213,7 +241,7 @@ export function createValveOperations(nodeId, contextDir) {
|
|
|
213
241
|
teardown();
|
|
214
242
|
act();
|
|
215
243
|
};
|
|
216
|
-
function
|
|
244
|
+
function finishBackground(auto = false) {
|
|
217
245
|
const elapsed = formatBashElapsed(Date.now() - startedAt);
|
|
218
246
|
const notice = `\n[backgrounded as job ${paths.jobId} after ${elapsed} in the foreground${auto ? ' (auto)' : ''}. ` +
|
|
219
247
|
`It will interrupt you with a message when it exits. ` +
|
|
@@ -222,8 +250,35 @@ export function createValveOperations(nodeId, contextDir) {
|
|
|
222
250
|
finish(() => resolve({ exitCode: 0 }));
|
|
223
251
|
// Backgrounded — job dir persists (inspectable); no cleanup here.
|
|
224
252
|
}
|
|
253
|
+
function finishForeground() {
|
|
254
|
+
offset = tailOnce(paths.jobLog, offset, onData);
|
|
255
|
+
const exitCode = readExitCode(paths.jobExit);
|
|
256
|
+
finish(() => resolve({ exitCode }));
|
|
257
|
+
cleanupJobDir(paths.dir);
|
|
258
|
+
}
|
|
259
|
+
function requestBackground(auto = false) {
|
|
260
|
+
const state = backgroundBashJob(paths);
|
|
261
|
+
if (state === 'backgrounded' || state === 'already-backgrounded') {
|
|
262
|
+
finishBackground(auto);
|
|
263
|
+
return;
|
|
264
|
+
}
|
|
265
|
+
if (state === 'finished') {
|
|
266
|
+
finishForeground();
|
|
267
|
+
return;
|
|
268
|
+
}
|
|
269
|
+
finish(() => reject(new Error(`bash job ${paths.jobId} disappeared before it could be backgrounded`)));
|
|
270
|
+
cleanupJobDir(paths.dir);
|
|
271
|
+
}
|
|
225
272
|
function onAbort() {
|
|
226
|
-
tailOnce(paths.jobLog, offset, onData);
|
|
273
|
+
offset = tailOnce(paths.jobLog, offset, onData);
|
|
274
|
+
if (existsSync(paths.jobBg)) {
|
|
275
|
+
finishBackground();
|
|
276
|
+
return;
|
|
277
|
+
}
|
|
278
|
+
// Once the command has written its exit code, let the supervisor
|
|
279
|
+
// finish the job.run → job.done transition before cleanup.
|
|
280
|
+
if (existsSync(paths.jobExit))
|
|
281
|
+
return;
|
|
227
282
|
if (pgid)
|
|
228
283
|
killGroup(pgid);
|
|
229
284
|
finish(() => reject(new Error('aborted')));
|
|
@@ -234,7 +289,7 @@ export function createValveOperations(nodeId, contextDir) {
|
|
|
234
289
|
const execution = resolveExecutionCwd(cwd);
|
|
235
290
|
if (execution.warning)
|
|
236
291
|
onData(Buffer.from(execution.warning));
|
|
237
|
-
const child = spawn('bash', ['-c', SUPERVISOR_SCRIPT, '--', paths.cmdSh, paths.jobLog, paths.jobExit, paths.jobBg, paths.jobId, nodeId], {
|
|
292
|
+
const child = spawn('bash', ['-c', SUPERVISOR_SCRIPT, '--', paths.cmdSh, paths.jobLog, paths.jobExit, paths.jobRun, paths.jobDone, paths.jobBg, paths.jobId, nodeId], {
|
|
238
293
|
cwd: execution.cwd,
|
|
239
294
|
detached: true,
|
|
240
295
|
stdio: 'ignore',
|
|
@@ -242,27 +297,31 @@ export function createValveOperations(nodeId, contextDir) {
|
|
|
242
297
|
});
|
|
243
298
|
child.on('error', (err) => {
|
|
244
299
|
finish(() => reject(err));
|
|
300
|
+
cleanupJobDir(paths.dir);
|
|
245
301
|
});
|
|
246
302
|
pgid = child.pid;
|
|
247
303
|
// The detached supervisor IS the group leader, so its pid is the pgid.
|
|
248
304
|
// Persisting it is what lets a human stop the job from the Inspector
|
|
249
305
|
// later; the agent's own handoff notice quotes the same number.
|
|
250
|
-
if (pgid !== undefined)
|
|
306
|
+
if (pgid !== undefined) {
|
|
251
307
|
writeFileSync(paths.jobPgid, String(pgid));
|
|
308
|
+
liveForegroundJobs.add(paths);
|
|
309
|
+
}
|
|
252
310
|
child.unref();
|
|
253
311
|
interval = setInterval(() => {
|
|
254
312
|
offset = tailOnce(paths.jobLog, offset, onData);
|
|
255
|
-
if (existsSync(paths.
|
|
256
|
-
|
|
257
|
-
const exitCode = readExitCode(paths.jobExit);
|
|
258
|
-
finish(() => resolve({ exitCode }));
|
|
259
|
-
cleanupJobDir(paths.dir);
|
|
313
|
+
if (existsSync(paths.jobDone)) {
|
|
314
|
+
finishForeground();
|
|
260
315
|
return;
|
|
261
316
|
}
|
|
262
317
|
if (existsSync(paths.jobBg)) {
|
|
263
|
-
|
|
318
|
+
finishBackground();
|
|
264
319
|
return;
|
|
265
320
|
}
|
|
321
|
+
// job.exit is written before the supervisor claims job.done. Do not
|
|
322
|
+
// time out or delete the directory during that in-progress exit.
|
|
323
|
+
if (existsSync(paths.jobExit))
|
|
324
|
+
return;
|
|
266
325
|
const elapsedMs = Date.now() - startedAt;
|
|
267
326
|
if (timeout !== undefined && timeout > 0) {
|
|
268
327
|
if (elapsedMs >= timeout * 1000) {
|
|
@@ -273,10 +332,8 @@ export function createValveOperations(nodeId, contextDir) {
|
|
|
273
332
|
}
|
|
274
333
|
return;
|
|
275
334
|
}
|
|
276
|
-
if (elapsedMs >= valveMs)
|
|
277
|
-
|
|
278
|
-
background(true);
|
|
279
|
-
}
|
|
335
|
+
if (elapsedMs >= valveMs)
|
|
336
|
+
requestBackground(true);
|
|
280
337
|
}, POLL_MS);
|
|
281
338
|
});
|
|
282
339
|
},
|
|
@@ -292,6 +349,7 @@ export default function (pi) {
|
|
|
292
349
|
const contextDir = process.env['CRTR_CONTEXT_DIR'];
|
|
293
350
|
if (contextDir === undefined || contextDir.trim() === '')
|
|
294
351
|
return; // defensive; always set alongside CRTR_NODE_ID
|
|
352
|
+
installTerminationHandoff();
|
|
295
353
|
// Registered on session_start (fires for startup, new, resume, fork, and
|
|
296
354
|
// /reload) so we get ctx.cwd — the same session cwd the builtin bash tool
|
|
297
355
|
// is built against. registerTool replaces the builtin by name; re-firing on
|
package/package.json
CHANGED
package/runtime.lock.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@north-light/crouter",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.169",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "@north-light/crouter",
|
|
9
|
-
"version": "0.3.
|
|
9
|
+
"version": "0.3.169",
|
|
10
10
|
"hasInstallScript": true,
|
|
11
11
|
"license": "MIT",
|
|
12
12
|
"dependencies": {
|