@north-light/crouter 0.3.167 → 0.3.169
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +4 -0
- package/dist/clients/attach/viewer.js +314 -314
- package/dist/commands/sys/daemon.js +1 -0
- package/dist/core/__tests__/broker-fork-seam.test.js +1 -2
- package/dist/core/__tests__/broker-sdk-wiring.test.js +4 -4
- package/dist/core/__tests__/revive.test.js +48 -0
- package/dist/core/bash-jobs.d.ts +16 -17
- package/dist/core/bash-jobs.js +39 -38
- package/dist/core/canvas/__tests__/render-remote.test.js +25 -0
- package/dist/core/canvas/crons.d.ts +3 -2
- package/dist/core/canvas/crons.js +3 -2
- package/dist/core/canvas/pid.d.ts +5 -14
- package/dist/core/canvas/pid.js +21 -35
- package/dist/core/canvas/render-source.js +6 -2
- package/dist/core/runtime/revive-all.js +3 -1
- package/dist/core/runtime/revive.d.ts +7 -0
- package/dist/core/runtime/revive.js +15 -3
- package/dist/daemon/fleet.js +4 -0
- package/dist/pi-extensions/canvas-bash-valve.js +76 -18
- package/dist/web-client/assets/index-U_NZ66VE.js +79 -0
- package/dist/web-client/assets/{index-BgLGlZ3D.css → index-aaeu4adv.css} +1 -1
- package/dist/web-client/index.html +2 -2
- package/dist/web-client/sw.js +1 -1
- package/package.json +1 -1
- package/runtime.lock.json +6 -6
- package/scripts/install-runtime.mjs +36 -13
- package/dist/web-client/assets/index-CmoNqcCv.js +0 -79
|
@@ -85,6 +85,7 @@ const daemonRestart = defineLeaf({
|
|
|
85
85
|
outputKind: 'object',
|
|
86
86
|
effects: [
|
|
87
87
|
'The daemon acknowledges immediately, then after a short grace tears its whole broker fleet down, releases its claim, spawns a successor on the runtime generation currently selected, and exits.',
|
|
88
|
+
'Each torn-down broker hands its foreground bash commands to the file-backed background job system; they keep running and report completion by urgent inbox message.',
|
|
88
89
|
'Every node it tore down — including the caller — is resumed by the successor through the ordinary startup recovery sweep.',
|
|
89
90
|
'The calling process is killed as part of that teardown, AFTER this result is returned.',
|
|
90
91
|
],
|
|
@@ -34,7 +34,6 @@ import { tmpdir } from 'node:os';
|
|
|
34
34
|
import { dirname, join } from 'node:path';
|
|
35
35
|
import { fileURLToPath } from 'node:url';
|
|
36
36
|
import { createAgentSessionServices, createAgentSessionFromServices, SessionManager, VERSION, } from '../runtime/broker-sdk.js';
|
|
37
|
-
import { CANVAS_EXTENSIONS } from '../runtime/launch.js';
|
|
38
37
|
import { buildBrokerSession } from '../runtime/broker.js';
|
|
39
38
|
// M-9 (broker.ts fault-and-die model gate): buildBrokerSession with no cfg.model
|
|
40
39
|
// now calls the REAL SDK's modelRegistry.getAvailable() and faults if it's empty
|
|
@@ -83,7 +82,7 @@ test('D2 broker fork branch: buildBrokerSession(cfg.forkFrom) yields a NEW id +
|
|
|
83
82
|
// if broker.ts:~1106 reverts to that throw, this rejects and the test goes RED.
|
|
84
83
|
const cfg = {
|
|
85
84
|
cwd: dir,
|
|
86
|
-
extensionPaths: [
|
|
85
|
+
extensionPaths: [C3_EXT],
|
|
87
86
|
forkFrom: srcFile,
|
|
88
87
|
model: 'c3prov/c3model',
|
|
89
88
|
};
|
|
@@ -84,7 +84,7 @@ function cfg(cwd, extra = {}) {
|
|
|
84
84
|
test('C3 — services path registers an extension model provider; the broker session gets it', async () => {
|
|
85
85
|
const cwd = mkdtempSync(join(tmpdir(), 'crtr-c3-'));
|
|
86
86
|
try {
|
|
87
|
-
const { session, services } = await buildBrokerSession(realEngine, cfg(cwd, { model: 'c3prov/c3model' }));
|
|
87
|
+
const { session, services } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [C3_EXT], model: 'c3prov/c3model' }));
|
|
88
88
|
try {
|
|
89
89
|
// The extension's provider was registered into the SERVICES model runtime (this
|
|
90
90
|
// is the registerProvider step plain createAgentSession skips).
|
|
@@ -107,7 +107,7 @@ test('C3 — services path registers an extension model provider; the broker ses
|
|
|
107
107
|
test('C3b — broker splits model thinking suffix before SDK registry lookup', async () => {
|
|
108
108
|
const cwd = mkdtempSync(join(tmpdir(), 'crtr-c3b-'));
|
|
109
109
|
try {
|
|
110
|
-
const { session } = await buildBrokerSession(realEngine, cfg(cwd, { model: 'c3prov/c3model:high' }));
|
|
110
|
+
const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [C3_EXT], model: 'c3prov/c3model:high' }));
|
|
111
111
|
try {
|
|
112
112
|
assert.equal(session.model?.id, 'c3model', 'C3b: suffixed model specs resolve by model id instead of falling back to SDK default');
|
|
113
113
|
}
|
|
@@ -420,7 +420,7 @@ test('C5 — isLeadingEngineCommand against a REAL session (registered extension
|
|
|
420
420
|
// proven against a plain object shaped like the real accessors.
|
|
421
421
|
const cwd = mkdtempSync(join(tmpdir(), 'crtr-c5-cmd-'));
|
|
422
422
|
try {
|
|
423
|
-
const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [
|
|
423
|
+
const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [C3_EXT, C5_EXT], model: 'c3prov/c3model' }));
|
|
424
424
|
try {
|
|
425
425
|
assert.equal(isLeadingEngineCommand('/devcmd', session), true, 'a real registered extension command matches');
|
|
426
426
|
assert.equal(isLeadingEngineCommand('/devcmd extra args', session), true, 'space-separated still matches against a real session');
|
|
@@ -744,7 +744,7 @@ test('Major 6 (broker half) — engineLeadingCommandTokens degrades to empty per
|
|
|
744
744
|
test('Major 6 (broker half) — engineLeadingCommandTokens against a REAL session enumerates the real registered extension command', async () => {
|
|
745
745
|
const cwd = mkdtempSync(join(tmpdir(), 'crtr-m6-cmd-'));
|
|
746
746
|
try {
|
|
747
|
-
const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [
|
|
747
|
+
const { session } = await buildBrokerSession(realEngine, cfg(cwd, { extensionPaths: [C3_EXT, C5_EXT], model: 'c3prov/c3model' }));
|
|
748
748
|
try {
|
|
749
749
|
const tokens = engineLeadingCommandTokens(session);
|
|
750
750
|
assert.ok(tokens.includes('devcmd'), 'the real registered extension command name is present in the enumeration');
|
|
@@ -35,6 +35,7 @@ import { fileURLToPath } from 'node:url';
|
|
|
35
35
|
import { spawn, spawnSync } from 'node:child_process';
|
|
36
36
|
import { createNode, getNode, subscribe, updateNode, clearPid } from '../canvas/canvas.js';
|
|
37
37
|
import { armFollowUpWork } from '../canvas/human-work-outbox.js';
|
|
38
|
+
import { armCron, hasPendingCancelOnWakeCron } from '../canvas/crons.js';
|
|
38
39
|
import { readInboxSince } from '../feed/inbox.js';
|
|
39
40
|
import { closeDb } from '../canvas/db.js';
|
|
40
41
|
import { buildPiArgv, CANVAS_EXTENSIONS } from '../runtime/launch.js';
|
|
@@ -346,6 +347,53 @@ test('reviveNode PROCEEDS when the node has NO fleet entry — cycle counter bum
|
|
|
346
347
|
await h.dispose();
|
|
347
348
|
}
|
|
348
349
|
});
|
|
350
|
+
// ---------------------------------------------------------------------------
|
|
351
|
+
// BUG LOCKED — a recovery relaunch must not eat the node's armed deadline.
|
|
352
|
+
// A node armed `crtr node wait deadline` and dormed; its broker then exited and
|
|
353
|
+
// the daemon respawned it (fleet.#respawn → reviveNode). reviveNode used to
|
|
354
|
+
// cancel every cancel-on-wake cron unconditionally, so the deadline vanished
|
|
355
|
+
// with nothing having woken the node and the wait could never settle. A wake
|
|
356
|
+
// (inbox delivery, attach, explicit revive) still consumes it.
|
|
357
|
+
// ---------------------------------------------------------------------------
|
|
358
|
+
function armDeadline(anchor, cronId) {
|
|
359
|
+
armCron({
|
|
360
|
+
cron_id: cronId,
|
|
361
|
+
name: `deadline:${anchor}`,
|
|
362
|
+
created_by: anchor,
|
|
363
|
+
command: 'true',
|
|
364
|
+
fire_at: new Date(Date.now() + 3_600_000).toISOString(),
|
|
365
|
+
recur: null,
|
|
366
|
+
tz: null,
|
|
367
|
+
expires_at: null,
|
|
368
|
+
anchor_node: anchor,
|
|
369
|
+
cancel_on_wake: true,
|
|
370
|
+
cwd: home,
|
|
371
|
+
env_json: null,
|
|
372
|
+
profile: null,
|
|
373
|
+
scope: 'profile',
|
|
374
|
+
run_timeout_s: 300,
|
|
375
|
+
overlap: 'skip',
|
|
376
|
+
on_output: 'on-failure',
|
|
377
|
+
sink: '',
|
|
378
|
+
tier: 'normal',
|
|
379
|
+
});
|
|
380
|
+
}
|
|
381
|
+
test('a recovery relaunch keeps the armed deadline; a wake revive consumes it', async () => {
|
|
382
|
+
const h = await createHarness({ headless: true, sessionPrefix: 'crtr-revive-deadline' });
|
|
383
|
+
try {
|
|
384
|
+
const recovered = h.fabricateBrokerNode({ status: 'active', intent: null, pi_pid: deadPid(), pi_session_id: 'uuid-1' });
|
|
385
|
+
armDeadline(recovered, 'cron-recovery');
|
|
386
|
+
reviveNode(recovered, { resume: true, recovery: true });
|
|
387
|
+
assert.equal(hasPendingCancelOnWakeCron(recovered), true, 'the daemon replacing a dead broker is not a wake — the deadline survives so the wait can still settle');
|
|
388
|
+
const woken = h.fabricateBrokerNode({ status: 'active', intent: null, pi_pid: deadPid(), pi_session_id: 'uuid-2' });
|
|
389
|
+
armDeadline(woken, 'cron-wake');
|
|
390
|
+
reviveNode(woken, { resume: true });
|
|
391
|
+
assert.equal(hasPendingCancelOnWakeCron(woken), false, 'a wake revive won the race and cancels the deadline');
|
|
392
|
+
}
|
|
393
|
+
finally {
|
|
394
|
+
await h.dispose();
|
|
395
|
+
}
|
|
396
|
+
});
|
|
349
397
|
test('reviveNode fresh-fallback clears the stale session identity (Major-3: no phantom stranded-relaunch)', async () => {
|
|
350
398
|
const h = await createHarness({ headless: true, sessionPrefix: 'crtr-revive-freshfb' });
|
|
351
399
|
try {
|
package/dist/core/bash-jobs.d.ts
CHANGED
|
@@ -4,7 +4,9 @@ export interface BashJobPaths {
|
|
|
4
4
|
cmdSh: string;
|
|
5
5
|
jobLog: string;
|
|
6
6
|
jobExit: string;
|
|
7
|
+
jobRun: string;
|
|
7
8
|
jobBg: string;
|
|
9
|
+
jobDone: string;
|
|
8
10
|
/** Process-group id of the job's detached supervisor, written at spawn. It is
|
|
9
11
|
* what makes a job stoppable from outside the agent that started it (the
|
|
10
12
|
* Inspector's cancel) — without it on disk, only the agent's own handoff
|
|
@@ -42,25 +44,22 @@ export interface BashJobStatus {
|
|
|
42
44
|
logPath: string;
|
|
43
45
|
}
|
|
44
46
|
/** Every live backgrounded bash job for this node — job.bg present, job.exit
|
|
45
|
-
* absent — whether handed off by the menu
|
|
46
|
-
*
|
|
47
|
-
* with) the foreground-only `runningBashJobs` above: that scanner deliberately
|
|
48
|
-
* EXCLUDES job.bg, this one REQUIRES it, so reusing one for the other would
|
|
49
|
-
* corrupt foreground-admission semantics. Skips malformed/partially-written
|
|
47
|
+
* absent — whether handed off by the menu, the valve deadline, or broker
|
|
48
|
+
* termination. Read-only; never mutates. Skips malformed/partially-written
|
|
50
49
|
* directories. Returns oldest-first by startedAtMs. */
|
|
51
50
|
export declare function activeBackgroundBashJobs(contextDir: string): BashJobStatus[];
|
|
52
|
-
|
|
51
|
+
export type BackgroundBashJobResult = 'backgrounded' | 'already-backgrounded' | 'finished' | 'missing';
|
|
52
|
+
/** Atomically hand one foreground job to its detached supervisor. The rename
|
|
53
|
+
* is the ownership decision: exactly one caller can move job.run to job.bg,
|
|
54
|
+
* while the supervisor races to move the same source to job.done. */
|
|
55
|
+
export declare function backgroundBashJob(paths: BashJobPaths): BackgroundBashJobResult;
|
|
56
|
+
/** The start time (ms epoch) of the oldest still-foreground bash command for
|
|
53
57
|
* this node, or undefined when none is running. This is the authoritative
|
|
54
|
-
* source for the attach viewer's
|
|
55
|
-
*
|
|
56
|
-
*
|
|
57
|
-
* and the valve deletes the directory the moment the command finishes or is
|
|
58
|
-
* aborted — see canvas-bash-valve.ts), so a fresh or reconnected attach (which
|
|
59
|
-
* has no memory of a broker-relayed tool-execution event) sees exactly what's
|
|
60
|
-
* really running, and a viewer never needs its own call-id tracking that could
|
|
61
|
-
* go stale across a broker/session replacement. */
|
|
58
|
+
* source for the attach viewer's background hint: a fresh or reconnected
|
|
59
|
+
* attach reads the same job.run state that backgrounding and completion race
|
|
60
|
+
* to claim, without viewer-local call tracking that can go stale. */
|
|
62
61
|
export declare function oldestRunningBashJobStartedAtMs(contextDir: string): number | undefined;
|
|
63
|
-
/** Request immediate handoff for every bash command still waiting on pi.
|
|
64
|
-
*
|
|
65
|
-
*
|
|
62
|
+
/** Request immediate handoff for every bash command still waiting on pi. Only
|
|
63
|
+
* the caller that wins the job.run → job.bg rename reports the job as handed
|
|
64
|
+
* off; another backgrounder or the foreground-completion path may win the race. */
|
|
66
65
|
export declare function backgroundRunningBashJobs(contextDir: string): BashJobPaths[];
|
package/dist/core/bash-jobs.js
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
// File-backed control plane for bash commands
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
1
|
+
// File-backed control plane for bash commands started by the canvas bash valve.
|
|
2
|
+
// Every foreground command owns job.run; backgrounding atomically renames it to
|
|
3
|
+
// job.bg, while foreground completion renames it to job.done. The detached
|
|
4
|
+
// supervisor can therefore outlive pi and still decide whether to report exit.
|
|
5
5
|
import { randomBytes } from 'node:crypto';
|
|
6
|
-
import { closeSync, existsSync, openSync, readdirSync, readFileSync, readSync,
|
|
6
|
+
import { closeSync, existsSync, openSync, readdirSync, readFileSync, readSync, renameSync, statSync } from 'node:fs';
|
|
7
7
|
import { join } from 'node:path';
|
|
8
8
|
export function bashJobsDir(contextDir) {
|
|
9
9
|
return join(contextDir, 'jobs');
|
|
@@ -16,7 +16,9 @@ export function bashJobPaths(contextDir, jobId) {
|
|
|
16
16
|
cmdSh: join(dir, 'cmd.sh'),
|
|
17
17
|
jobLog: join(dir, 'job.log'),
|
|
18
18
|
jobExit: join(dir, 'job.exit'),
|
|
19
|
+
jobRun: join(dir, 'job.run'),
|
|
19
20
|
jobBg: join(dir, 'job.bg'),
|
|
21
|
+
jobDone: join(dir, 'job.done'),
|
|
20
22
|
jobPgid: join(dir, 'job.pgid'),
|
|
21
23
|
};
|
|
22
24
|
}
|
|
@@ -94,11 +96,8 @@ export function formatBashElapsed(ms) {
|
|
|
94
96
|
return `${hours}h ${String(minutes).padStart(2, '0')}m`;
|
|
95
97
|
}
|
|
96
98
|
/** Every live backgrounded bash job for this node — job.bg present, job.exit
|
|
97
|
-
* absent — whether handed off by the menu
|
|
98
|
-
*
|
|
99
|
-
* with) the foreground-only `runningBashJobs` above: that scanner deliberately
|
|
100
|
-
* EXCLUDES job.bg, this one REQUIRES it, so reusing one for the other would
|
|
101
|
-
* corrupt foreground-admission semantics. Skips malformed/partially-written
|
|
99
|
+
* absent — whether handed off by the menu, the valve deadline, or broker
|
|
100
|
+
* termination. Read-only; never mutates. Skips malformed/partially-written
|
|
102
101
|
* directories. Returns oldest-first by startedAtMs. */
|
|
103
102
|
export function activeBackgroundBashJobs(contextDir) {
|
|
104
103
|
let entries;
|
|
@@ -127,9 +126,27 @@ export function activeBackgroundBashJobs(contextDir) {
|
|
|
127
126
|
}
|
|
128
127
|
return jobs.sort((a, b) => a.startedAtMs - b.startedAtMs);
|
|
129
128
|
}
|
|
130
|
-
/**
|
|
131
|
-
*
|
|
132
|
-
*
|
|
129
|
+
/** Atomically hand one foreground job to its detached supervisor. The rename
|
|
130
|
+
* is the ownership decision: exactly one caller can move job.run to job.bg,
|
|
131
|
+
* while the supervisor races to move the same source to job.done. */
|
|
132
|
+
export function backgroundBashJob(paths) {
|
|
133
|
+
try {
|
|
134
|
+
renameSync(paths.jobRun, paths.jobBg);
|
|
135
|
+
return 'backgrounded';
|
|
136
|
+
}
|
|
137
|
+
catch (err) {
|
|
138
|
+
if (err.code !== 'ENOENT')
|
|
139
|
+
throw err;
|
|
140
|
+
if (existsSync(paths.jobBg))
|
|
141
|
+
return 'already-backgrounded';
|
|
142
|
+
if (existsSync(paths.jobDone))
|
|
143
|
+
return 'finished';
|
|
144
|
+
return 'missing';
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
/** Every still-foreground bash command for this node. job.run is the sole live
|
|
148
|
+
* foreground state; a handoff or foreground completion atomically removes it.
|
|
149
|
+
* Invalid or concurrently-removed directories are ignored. */
|
|
133
150
|
function runningBashJobs(contextDir) {
|
|
134
151
|
let entries;
|
|
135
152
|
try {
|
|
@@ -140,21 +157,16 @@ function runningBashJobs(contextDir) {
|
|
|
140
157
|
}
|
|
141
158
|
return entries.flatMap((jobId) => {
|
|
142
159
|
const paths = bashJobPaths(contextDir, jobId);
|
|
143
|
-
return existsSync(paths.cmdSh) && existsSync(paths.jobLog) &&
|
|
160
|
+
return existsSync(paths.cmdSh) && existsSync(paths.jobLog) && existsSync(paths.jobRun)
|
|
144
161
|
? [paths]
|
|
145
162
|
: [];
|
|
146
163
|
});
|
|
147
164
|
}
|
|
148
|
-
/** The start time (ms epoch) of the
|
|
165
|
+
/** The start time (ms epoch) of the oldest still-foreground bash command for
|
|
149
166
|
* this node, or undefined when none is running. This is the authoritative
|
|
150
|
-
* source for the attach viewer's
|
|
151
|
-
*
|
|
152
|
-
*
|
|
153
|
-
* and the valve deletes the directory the moment the command finishes or is
|
|
154
|
-
* aborted — see canvas-bash-valve.ts), so a fresh or reconnected attach (which
|
|
155
|
-
* has no memory of a broker-relayed tool-execution event) sees exactly what's
|
|
156
|
-
* really running, and a viewer never needs its own call-id tracking that could
|
|
157
|
-
* go stale across a broker/session replacement. */
|
|
167
|
+
* source for the attach viewer's background hint: a fresh or reconnected
|
|
168
|
+
* attach reads the same job.run state that backgrounding and completion race
|
|
169
|
+
* to claim, without viewer-local call tracking that can go stale. */
|
|
158
170
|
export function oldestRunningBashJobStartedAtMs(contextDir) {
|
|
159
171
|
let oldest;
|
|
160
172
|
for (const paths of runningBashJobs(contextDir)) {
|
|
@@ -166,25 +178,14 @@ export function oldestRunningBashJobStartedAtMs(contextDir) {
|
|
|
166
178
|
}
|
|
167
179
|
return oldest;
|
|
168
180
|
}
|
|
169
|
-
/** Request immediate handoff for every bash command still waiting on pi.
|
|
170
|
-
*
|
|
171
|
-
*
|
|
181
|
+
/** Request immediate handoff for every bash command still waiting on pi. Only
|
|
182
|
+
* the caller that wins the job.run → job.bg rename reports the job as handed
|
|
183
|
+
* off; another backgrounder or the foreground-completion path may win the race. */
|
|
172
184
|
export function backgroundRunningBashJobs(contextDir) {
|
|
173
185
|
const backgrounded = [];
|
|
174
186
|
for (const paths of runningBashJobs(contextDir)) {
|
|
175
|
-
|
|
176
|
-
// A job may exit or be claimed by another prefix press between the scan
|
|
177
|
-
// and this write. Recheck the two terminal/control files so that stale
|
|
178
|
-
// candidates never masquerade as a successful handoff.
|
|
179
|
-
if (existsSync(paths.jobExit) || existsSync(paths.jobBg))
|
|
180
|
-
continue;
|
|
181
|
-
writeFileSync(paths.jobBg, '');
|
|
187
|
+
if (backgroundBashJob(paths) === 'backgrounded')
|
|
182
188
|
backgrounded.push(paths);
|
|
183
|
-
}
|
|
184
|
-
catch {
|
|
185
|
-
// The valve owns cleanup; a concurrently removed directory is simply no
|
|
186
|
-
// longer a running command to hand off.
|
|
187
|
-
}
|
|
188
189
|
}
|
|
189
190
|
return backgrounded;
|
|
190
191
|
}
|
|
@@ -102,6 +102,31 @@ test('dashboardRowsAllFromSource shows a node watched through its live broker',
|
|
|
102
102
|
const rows = await dashboardRowsAllFromSource(source);
|
|
103
103
|
assert.equal(rows[0]?.viewed, true);
|
|
104
104
|
});
|
|
105
|
+
test('enrichRowsFromSource serializes a full dashboard source read', async () => {
|
|
106
|
+
const nodeRows = ['a', 'b', 'c'].map(row);
|
|
107
|
+
let inFlight = 0;
|
|
108
|
+
let maxInFlight = 0;
|
|
109
|
+
const source = {
|
|
110
|
+
getNode: async (id) => {
|
|
111
|
+
inFlight++;
|
|
112
|
+
maxInFlight = Math.max(maxInFlight, inFlight);
|
|
113
|
+
await new Promise((resolve) => setImmediate(resolve));
|
|
114
|
+
inFlight--;
|
|
115
|
+
return meta(id);
|
|
116
|
+
},
|
|
117
|
+
getRow: async () => null,
|
|
118
|
+
listNodes: async () => nodeRows,
|
|
119
|
+
subscriptionsOf: async () => [],
|
|
120
|
+
subscribersOf: async () => [],
|
|
121
|
+
view: async () => [],
|
|
122
|
+
ticketCountsForView: async () => ({}),
|
|
123
|
+
hasActiveLiveSubscription: async () => false,
|
|
124
|
+
};
|
|
125
|
+
const rows = await dashboardRowsAllFromSource(source);
|
|
126
|
+
await enrichRowsFromSource(source, rows);
|
|
127
|
+
assert.equal(maxInFlight, 1);
|
|
128
|
+
assert.deepEqual(rows.map((entry) => entry.name), ['a', 'b', 'c']);
|
|
129
|
+
});
|
|
105
130
|
let home;
|
|
106
131
|
let localCwd;
|
|
107
132
|
let sessionFile;
|
|
@@ -160,8 +160,9 @@ export declare function setCronLastOutputHash(cron_id: string, hash: string): vo
|
|
|
160
160
|
* terminal node that stopped without finishing. */
|
|
161
161
|
export declare function hasPendingCancelOnWakeCron(anchor_node: string): boolean;
|
|
162
162
|
/** DELETE every cancel-on-wake cron anchored to this node — the deadline rule.
|
|
163
|
-
* Two seams call it: reviveNode on every revive
|
|
164
|
-
*
|
|
163
|
+
* Two seams call it: reviveNode on a WAKE — every revive except a recovery
|
|
164
|
+
* relaunch, which only replaces a dead broker instance and leaves the wait
|
|
165
|
+
* open (the wake won the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
|
|
165
166
|
* with the node, so its deadline must not fire against a finalized or closed
|
|
166
167
|
* row). Standing declarative crons are untouched — only cancel-on-wake rows.
|
|
167
168
|
* Kills any in-flight
|
|
@@ -219,8 +219,9 @@ export function hasPendingCancelOnWakeCron(anchor_node) {
|
|
|
219
219
|
.get(anchor_node) !== undefined);
|
|
220
220
|
}
|
|
221
221
|
/** DELETE every cancel-on-wake cron anchored to this node — the deadline rule.
|
|
222
|
-
* Two seams call it: reviveNode on every revive
|
|
223
|
-
*
|
|
222
|
+
* Two seams call it: reviveNode on a WAKE — every revive except a recovery
|
|
223
|
+
* relaunch, which only replaces a dead broker instance and leaves the wait
|
|
224
|
+
* open (the wake won the race), and the `finish`/`cancel` lifecycle transitions (the wait ended
|
|
224
225
|
* with the node, so its deadline must not fire against a finalized or closed
|
|
225
226
|
* row). Standing declarative crons are untouched — only cancel-on-wake rows.
|
|
226
227
|
* Kills any in-flight
|
|
@@ -1,17 +1,8 @@
|
|
|
1
|
-
/** True if a process with `pid` is currently alive (
|
|
2
|
-
*
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
* null/undefined pid
|
|
6
|
-
*
|
|
7
|
-
* The zombie check matters because EVERY broker this module supervises is
|
|
8
|
-
* spawned `detached: true` by crtr itself (`headlessBrokerHost.launch`) — a
|
|
9
|
-
* dead broker whose spawning process hasn't reaped it yet is a zombie, and
|
|
10
|
-
* `kill(pid, 0)` alone reports it as alive, wedging the daemon's revive guard
|
|
11
|
-
* (`isPidAlive(pi_pid)`, `revive.ts`) forever: the daemon never revives an
|
|
12
|
-
* already-live-looking row, so a refreshed/crashed node whose broker zombied
|
|
13
|
-
* out under a long-lived spawning process (see `isZombie`'s doc) never comes
|
|
14
|
-
* back on its own. */
|
|
1
|
+
/** True if a process with `pid` is currently alive. `kill(pid, 0)` provides the
|
|
2
|
+
* initial existence check; EPERM means the foreign process exists. For our
|
|
3
|
+
* processes, the later `ps` read is authoritative because it also distinguishes
|
|
4
|
+
* zombies and processes reaped between the two probes. Only an unavailable
|
|
5
|
+
* `ps` probe fails open to alive. A null/undefined pid reads dead. */
|
|
15
6
|
export declare function isPidAlive(pid: number | null | undefined): boolean;
|
|
16
7
|
/** THE single precision-aware identity compare, used EVERYWHERE two
|
|
17
8
|
* `composeIdentity` fingerprints are compared (this module's own
|
package/dist/core/canvas/pid.js
CHANGED
|
@@ -10,46 +10,31 @@
|
|
|
10
10
|
import { spawnSync } from 'node:child_process';
|
|
11
11
|
import { readFileSync } from 'node:fs';
|
|
12
12
|
import { readKernelBootId } from './boot-id.js';
|
|
13
|
-
/**
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
|
|
18
|
-
* front door's foreground attach session parks its OWN event loop for the
|
|
19
|
-
* whole session — see `bootRoot` in runtime/boot-root.ts — so it can't process
|
|
20
|
-
* the SIGCHLD that would otherwise reap its detached broker child in real
|
|
21
|
-
* time). `ps -o stat=` reports a leading `Z` for a zombie on BOTH BSD ps
|
|
22
|
-
* (macOS) and procps-ng (Linux), so one portable probe covers both platforms
|
|
23
|
-
* without a Linux-only `/proc` special case. Best-effort: any probe failure
|
|
24
|
-
* (spawn error, no matching row, non-zero exit) reads as NOT a zombie —
|
|
25
|
-
* fail-open, matching this module's other guards; a probe failure must never
|
|
26
|
-
* be misread as proof of death (a live pid IS alive; only a positive `Z`
|
|
27
|
-
* read says otherwise). */
|
|
28
|
-
function isZombie(pid) {
|
|
13
|
+
/** Classify the current `ps` row for a pid. BSD ps and procps-ng both report a
|
|
14
|
+
* leading `Z` for zombies and exit silently with empty output when no process
|
|
15
|
+
* matches. Only a probe that could not run is unknown; that is the sole
|
|
16
|
+
* fail-open result. */
|
|
17
|
+
function psProcessState(pid) {
|
|
29
18
|
try {
|
|
30
19
|
const r = spawnSync('ps', ['-o', 'stat=', '-p', String(pid)], { encoding: 'utf8', timeout: 2000 });
|
|
31
|
-
if (r.
|
|
32
|
-
return
|
|
33
|
-
|
|
20
|
+
if (r.error != null || r.signal != null || typeof r.stdout !== 'string')
|
|
21
|
+
return 'unknown';
|
|
22
|
+
if (typeof r.stderr === 'string' && r.stderr.trim() !== '')
|
|
23
|
+
return 'unknown';
|
|
24
|
+
const stat = r.stdout.trim();
|
|
25
|
+
if (stat === '')
|
|
26
|
+
return 'gone';
|
|
27
|
+
return stat.charAt(0) === 'Z' ? 'zombie' : 'live';
|
|
34
28
|
}
|
|
35
29
|
catch {
|
|
36
|
-
return
|
|
30
|
+
return 'unknown';
|
|
37
31
|
}
|
|
38
32
|
}
|
|
39
|
-
/** True if a process with `pid` is currently alive (
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
* null/undefined pid
|
|
44
|
-
*
|
|
45
|
-
* The zombie check matters because EVERY broker this module supervises is
|
|
46
|
-
* spawned `detached: true` by crtr itself (`headlessBrokerHost.launch`) — a
|
|
47
|
-
* dead broker whose spawning process hasn't reaped it yet is a zombie, and
|
|
48
|
-
* `kill(pid, 0)` alone reports it as alive, wedging the daemon's revive guard
|
|
49
|
-
* (`isPidAlive(pi_pid)`, `revive.ts`) forever: the daemon never revives an
|
|
50
|
-
* already-live-looking row, so a refreshed/crashed node whose broker zombied
|
|
51
|
-
* out under a long-lived spawning process (see `isZombie`'s doc) never comes
|
|
52
|
-
* back on its own. */
|
|
33
|
+
/** True if a process with `pid` is currently alive. `kill(pid, 0)` provides the
|
|
34
|
+
* initial existence check; EPERM means the foreign process exists. For our
|
|
35
|
+
* processes, the later `ps` read is authoritative because it also distinguishes
|
|
36
|
+
* zombies and processes reaped between the two probes. Only an unavailable
|
|
37
|
+
* `ps` probe fails open to alive. A null/undefined pid reads dead. */
|
|
53
38
|
export function isPidAlive(pid) {
|
|
54
39
|
if (pid == null)
|
|
55
40
|
return false;
|
|
@@ -59,7 +44,8 @@ export function isPidAlive(pid) {
|
|
|
59
44
|
catch (e) {
|
|
60
45
|
return e.code === 'EPERM';
|
|
61
46
|
}
|
|
62
|
-
|
|
47
|
+
const state = psProcessState(pid);
|
|
48
|
+
return state !== 'zombie' && state !== 'gone';
|
|
63
49
|
}
|
|
64
50
|
/** Matches the kernel `boot_id` UUID shape (`/proc/sys/kernel/random/boot_id`),
|
|
65
51
|
* used to tell a NEW-format identity base (a per-boot UUID) apart from a
|
|
@@ -430,14 +430,18 @@ export async function enrichRowsFromSource(source, rows, asks) {
|
|
|
430
430
|
return;
|
|
431
431
|
const remote = source instanceof RemoteCanvasSource;
|
|
432
432
|
const askMap = asks ?? {};
|
|
433
|
-
|
|
433
|
+
// A full dashboard may contain thousands of rows. Keep source reads serial:
|
|
434
|
+
// ApiCanvasSource turns each into a unix-socket request, and an unbounded
|
|
435
|
+
// Promise.all can overflow crtrd's accept backlog then spuriously trigger
|
|
436
|
+
// the client's cold-daemon path even though the daemon is still running.
|
|
437
|
+
for (const row of todo) {
|
|
434
438
|
const meta = await source.getNode(row.node_id);
|
|
435
439
|
if (meta !== null)
|
|
436
440
|
row.name = fullName(meta);
|
|
437
441
|
row.ctx_tokens = remote ? 0 : (readNodeTelemetry(row.node_id).tokens_in ?? 0);
|
|
438
442
|
row.asks = askMap[row.node_id] ?? 0;
|
|
439
443
|
row.enriched = true;
|
|
440
|
-
}
|
|
444
|
+
}
|
|
441
445
|
}
|
|
442
446
|
/** goal (initial-prompt.md) and session parts are local disk reads — suppressed
|
|
443
447
|
* to undefined for a remote source. */
|
|
@@ -95,7 +95,9 @@ export function reviveAll() {
|
|
|
95
95
|
const result = { revived: [], failed: [] };
|
|
96
96
|
for (const meta of listDisconnected()) {
|
|
97
97
|
try {
|
|
98
|
-
|
|
98
|
+
// Reconnecting a disconnected broker is recovery, not a wake — a node
|
|
99
|
+
// waiting on an armed deadline keeps it across the sweep.
|
|
100
|
+
reviveNode(meta.node_id, { resume: true, recovery: true });
|
|
99
101
|
result.revived.push(meta.node_id);
|
|
100
102
|
}
|
|
101
103
|
catch (err) {
|
|
@@ -49,10 +49,17 @@ export interface ReviveResult {
|
|
|
49
49
|
/** Relaunch `nodeId`'s broker engine from its persisted recipe and update canvas
|
|
50
50
|
* meta. Opens no viewer (engine-only).
|
|
51
51
|
*
|
|
52
|
+
* `recovery: true` marks a launch that only replaces a broker instance which
|
|
53
|
+
* died or was torn down — the daemon's respawn after a broker exit, and the
|
|
54
|
+
* `reviveAll` sweep over disconnected nodes. Nothing external arrived for the
|
|
55
|
+
* node, so such a launch is NOT a wake and does not consume the node's armed
|
|
56
|
+
* deadline (see the cancelCronsOnWake call below).
|
|
57
|
+
*
|
|
52
58
|
* Throws if the node does not exist. All other failures propagate as-is —
|
|
53
59
|
* callers (daemon, command) decide how to handle.
|
|
54
60
|
*/
|
|
55
61
|
export declare function reviveNode(nodeId: string, opts: {
|
|
56
62
|
resume: boolean;
|
|
57
63
|
wakeReason?: ReviveWakeReason;
|
|
64
|
+
recovery?: boolean;
|
|
58
65
|
}): ReviveResult;
|
|
@@ -92,6 +92,12 @@ function isUnstartedBirth(nodeId) {
|
|
|
92
92
|
/** Relaunch `nodeId`'s broker engine from its persisted recipe and update canvas
|
|
93
93
|
* meta. Opens no viewer (engine-only).
|
|
94
94
|
*
|
|
95
|
+
* `recovery: true` marks a launch that only replaces a broker instance which
|
|
96
|
+
* died or was torn down — the daemon's respawn after a broker exit, and the
|
|
97
|
+
* `reviveAll` sweep over disconnected nodes. Nothing external arrived for the
|
|
98
|
+
* node, so such a launch is NOT a wake and does not consume the node's armed
|
|
99
|
+
* deadline (see the cancelCronsOnWake call below).
|
|
100
|
+
*
|
|
95
101
|
* Throws if the node does not exist. All other failures propagate as-is —
|
|
96
102
|
* callers (daemon, command) decide how to handle.
|
|
97
103
|
*/
|
|
@@ -309,9 +315,15 @@ export function reviveNode(nodeId, opts) {
|
|
|
309
315
|
// unchecked result, is what scopes a later diagnostic read to this attempt.
|
|
310
316
|
clearFault(nodeId);
|
|
311
317
|
const bootFault = beginBootFaultAttempt(nodeId);
|
|
312
|
-
// A cron armed --cancel-on-wake belongs to the dormancy being left, so
|
|
313
|
-
//
|
|
314
|
-
|
|
318
|
+
// A cron armed --cancel-on-wake belongs to the dormancy being left, so a WAKE
|
|
319
|
+
// of its anchor deletes it — an inbox delivery, an attach, an explicit
|
|
320
|
+
// revive. A recovery relaunch is not a wake: the broker instance died or was
|
|
321
|
+
// torn down and the daemon is putting the same waiting node back on its feet,
|
|
322
|
+
// with nothing external having arrived. The wait the deadline bounds is still
|
|
323
|
+
// open, so the deadline survives — the same rule lifecycle.ts applies to the
|
|
324
|
+
// `crash` transition (a deadline MUST survive instance death).
|
|
325
|
+
if (opts.recovery !== true)
|
|
326
|
+
cancelCronsOnWake(nodeId);
|
|
315
327
|
let launched;
|
|
316
328
|
try {
|
|
317
329
|
launched = headlessBrokerHost.launch(nodeId, inv, {
|
package/dist/daemon/fleet.js
CHANGED
|
@@ -352,6 +352,10 @@ export class DaemonFleet {
|
|
|
352
352
|
reviveNode(nodeId, {
|
|
353
353
|
resume: action === 'respawn-resume',
|
|
354
354
|
wakeReason: cleanAbort ? 'runtime-restart-abort' : undefined,
|
|
355
|
+
// Replacing a dead broker instance is recovery, not a wake: nothing
|
|
356
|
+
// external arrived for this node, so an armed deadline survives the
|
|
357
|
+
// relaunch instead of being consumed by it.
|
|
358
|
+
recovery: true,
|
|
355
359
|
});
|
|
356
360
|
}
|
|
357
361
|
catch (err) {
|