@zhuxixi/pi-agent-board 0.6.2 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/docs/superpowers/plans/2026-09-08-issue-11-attach-runtime-desync-heal.md +917 -0
- package/docs/superpowers/plans/2026-09-09-harden-runner-architecture.md +603 -0
- package/docs/superpowers/plans/2026-09-09-single-writer-completion.md +252 -0
- package/docs/superpowers/specs/2026-09-07-issue-11-attach-runtime-desync-heal-design.md +130 -0
- package/docs/superpowers/specs/2026-09-09-harden-runner-architecture-design.md +298 -0
- package/package.json +1 -1
- package/runner/job-runner-legacy.mjs +68 -0
- package/runner/job-runner.mjs +370 -67
- package/runner/pty-runner-legacy.mjs +50 -0
- package/runner/pty-runner.mjs +69 -31
- package/runner/state-coordinator.mjs +403 -0
- package/runner/state-runner.mjs +89 -15
- package/src/commands/bg.ts +2 -1
- package/src/core/coordinator-client.mjs +313 -0
- package/src/core/coordinator-journal.mjs +282 -0
- package/src/core/coordinator-protocol.mjs +12 -0
- package/src/core/launch.mjs +15 -0
- package/src/core/paths.mjs +16 -0
- package/src/core/pty-attach-jiggle-controller.mjs +57 -4
- package/src/core/pty-attach-render.mjs +30 -0
- package/src/core/state-commands.mjs +617 -0
- package/src/core/types.mjs +2 -0
- package/src/runtime/service.mjs +416 -111
- package/src/ui/dashboard.ts +77 -110
- package/src/ui/pty-attach.ts +63 -1
package/src/commands/bg.ts
CHANGED
|
@@ -51,7 +51,8 @@ async function handleBgCommand(args: string, ctx: ExtensionCommandContext, opts:
|
|
|
51
51
|
},
|
|
52
52
|
});
|
|
53
53
|
const model = modelRef(ctx.model as any);
|
|
54
|
-
|
|
54
|
+
// adoptSession is async (routes through the view-state coordinator, issue #91).
|
|
55
|
+
const adopted = await service.adoptSession({
|
|
55
56
|
sessionFile,
|
|
56
57
|
cwd: ctx.cwd,
|
|
57
58
|
model,
|
|
@@ -0,0 +1,313 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Client for the detached View State Coordinator (issue #91, spec D3).
|
|
3
|
+
*
|
|
4
|
+
* Every semantic-state writer (dashboard service, job-runner, state-runner,
|
|
5
|
+
* CLI) submits `{type:"state_command"}` envelopes through `sendStateCommand`
|
|
6
|
+
* instead of writing state.json/status.json directly. The client resolves the
|
|
7
|
+
* coordinator endpoint, ensures a coordinator is live (spawning one if needed
|
|
8
|
+
* — idempotent via the coordinator's own lease), and waits for the matching
|
|
9
|
+
* `state_command_result`.
|
|
10
|
+
*
|
|
11
|
+
* Ambiguity contract (binding, from the Task 4 review): the coordinator can
|
|
12
|
+
* crash mid-command (e.g. an fs error) WITHOUT replying. The client then sees
|
|
13
|
+
* a connection reset or a timeout. Both mean the command MAY already be
|
|
14
|
+
* journaled and will replay on coordinator restart, so these outcomes are
|
|
15
|
+
* AMBIGUOUS: they resolve `{status:"rejected", reason:"timeout"|
|
|
16
|
+
* "connection_reset", materializedRevision:0}` and the client NEVER retries
|
|
17
|
+
* with a fresh commandId. Retrying with the SAME commandId after reconnect is
|
|
18
|
+
* safe (the coordinator dedupes) but is the caller's decision — which is why
|
|
19
|
+
* the caller may pin `command.commandId`.
|
|
20
|
+
*
|
|
21
|
+
* `materializedRevision: 0` on client-side outcomes means "no revision was
|
|
22
|
+
* observed" — it is not a store revision. Real coordinator replies always
|
|
23
|
+
* carry the actual revision (≥ 1 after any applied command; decided
|
|
24
|
+
* rejections carry the current revision).
|
|
25
|
+
*/
|
|
26
|
+
import { createConnection } from "node:net";
|
|
27
|
+
import { existsSync, readFileSync } from "node:fs";
|
|
28
|
+
import { fileURLToPath } from "node:url";
|
|
29
|
+
import { newRunId } from "./ids.mjs";
|
|
30
|
+
import { launchCoordinator } from "./launch.mjs";
|
|
31
|
+
import * as P from "./paths.mjs";
|
|
32
|
+
import { COORDINATOR_PROTOCOL_VERSION } from "./coordinator-protocol.mjs";
|
|
33
|
+
|
|
34
|
+
const COORDINATOR_SCRIPT = fileURLToPath(
|
|
35
|
+
new URL("../../runner/state-coordinator.mjs", import.meta.url),
|
|
36
|
+
);
|
|
37
|
+
|
|
38
|
+
const PROBE_TIMEOUT_MS = 1_000;
|
|
39
|
+
const ENSURE_WINDOW_MS = 10_000;
|
|
40
|
+
const ENSURE_POLL_MS = 100;
|
|
41
|
+
const COMMAND_TIMEOUT_MS = 5_000;
|
|
42
|
+
|
|
43
|
+
/** @typedef {{ status: "applied"|"rejected", reason: string|null, materializedRevision: number }} StateCommandResult */
|
|
44
|
+
/** @typedef {{ ok: boolean, instanceId?: string, pid?: number|null, error?: "coordinator_disabled"|"coordinator_unavailable"|"coordinator_stale_protocol" }} EnsureResult */
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Whether the coordinator is switched off for this process (tests/legacy
|
|
48
|
+
* escape hatch). `AGENT_BOARD_COORDINATOR=off` makes every command resolve
|
|
49
|
+
* `coordinator_disabled` so callers fall back to the pre-coordinator path.
|
|
50
|
+
* @returns {boolean}
|
|
51
|
+
*/
|
|
52
|
+
export function coordinatorDisabled() {
|
|
53
|
+
return /^(0|false|off|no)$/i.test(String(process.env.AGENT_BOARD_COORDINATOR ?? "").trim());
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* One probe round-trip: connect, ping, wait for the pong. The coordinator
|
|
58
|
+
* binds its socket only after boot replay, so a pong proves full readiness.
|
|
59
|
+
* @param {string} socketPath
|
|
60
|
+
* @param {(path: string) => import("node:net").Socket} connect
|
|
61
|
+
* @param {number} timeoutMs
|
|
62
|
+
* @returns {Promise<{ instanceId: string, protocolVersion: number }|null>} null on timeout/error;
|
|
63
|
+
* a pong without a numeric protocolVersion counts as version 1 (pre-#107 baseline)
|
|
64
|
+
*/
|
|
65
|
+
function probeOnce(socketPath, connect, timeoutMs) {
|
|
66
|
+
return new Promise((resolve) => {
|
|
67
|
+
let settled = false;
|
|
68
|
+
let buffer = "";
|
|
69
|
+
/** @type {import("node:net").Socket|null} */
|
|
70
|
+
let socket = null;
|
|
71
|
+
const finish = (instanceId) => {
|
|
72
|
+
if (settled) return;
|
|
73
|
+
settled = true;
|
|
74
|
+
clearTimeout(timer);
|
|
75
|
+
try { socket?.destroy(); } catch { /* best effort */ }
|
|
76
|
+
resolve(instanceId);
|
|
77
|
+
};
|
|
78
|
+
const timer = setTimeout(() => finish(null), timeoutMs);
|
|
79
|
+
timer.unref?.();
|
|
80
|
+
try {
|
|
81
|
+
socket = connect(socketPath);
|
|
82
|
+
} catch {
|
|
83
|
+
finish(null);
|
|
84
|
+
return;
|
|
85
|
+
}
|
|
86
|
+
// Missing filesystem sockets surface as async errors, not throw-on-connect
|
|
87
|
+
// (and Windows pipes never exist as files — never gate on existsSync).
|
|
88
|
+
socket.on("error", () => finish(null));
|
|
89
|
+
socket.on("connect", () => {
|
|
90
|
+
try {
|
|
91
|
+
socket?.write(JSON.stringify({ type: "ping" }) + "\n");
|
|
92
|
+
} catch {
|
|
93
|
+
finish(null);
|
|
94
|
+
}
|
|
95
|
+
});
|
|
96
|
+
socket.on("data", (chunk) => {
|
|
97
|
+
buffer += chunk.toString("utf8");
|
|
98
|
+
const lines = buffer.split("\n");
|
|
99
|
+
buffer = lines.pop() ?? "";
|
|
100
|
+
for (const line of lines) {
|
|
101
|
+
if (!line.trim()) continue;
|
|
102
|
+
try {
|
|
103
|
+
const msg = JSON.parse(line);
|
|
104
|
+
if (msg?.type === "pong" && typeof msg.instanceId === "string") {
|
|
105
|
+
finish({
|
|
106
|
+
instanceId: msg.instanceId,
|
|
107
|
+
protocolVersion: typeof msg.protocolVersion === "number" ? msg.protocolVersion : 1,
|
|
108
|
+
});
|
|
109
|
+
return;
|
|
110
|
+
}
|
|
111
|
+
} catch {
|
|
112
|
+
// malformed line — keep waiting for the pong
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
});
|
|
116
|
+
});
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Read the coordinator lease's owner.json and return the owning pid. The
|
|
121
|
+
* lease is written by whichever process owns the "_coordinator" lock, so it
|
|
122
|
+
* identifies the live coordinator even when its pong predates the pid field.
|
|
123
|
+
* @param {string} root
|
|
124
|
+
* @returns {number|null}
|
|
125
|
+
*/
|
|
126
|
+
function readCoordinatorLeasePid(root) {
|
|
127
|
+
try {
|
|
128
|
+
const lockPath = P.viewLockPath(root, "_coordinator", "state-coordinator");
|
|
129
|
+
const owner = JSON.parse(readFileSync(`${lockPath}/owner.json`, "utf8"));
|
|
130
|
+
const pid = Number(owner?.identity?.pid ?? owner?.pid ?? 0);
|
|
131
|
+
return Number.isFinite(pid) && pid > 0 ? pid : null;
|
|
132
|
+
} catch {
|
|
133
|
+
return null;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/** Grace window for a SIGTERMed stale coordinator to unlink its socket. */
|
|
138
|
+
const REPLACE_TIMEOUT_MS = 3_000;
|
|
139
|
+
const REPLACE_POLL_MS = 50;
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Terminate a stale-protocol coordinator (issue #108) and wait for its socket
|
|
143
|
+
* to disappear so a fresh instance can bind. SIGTERM triggers the
|
|
144
|
+
* coordinator's graceful shutdown (socket unlink + lease release); its crash
|
|
145
|
+
* safety (fsync-before-materialize journal + boot replay) makes the kill safe
|
|
146
|
+
* even mid-command — an in-flight reply is the documented ambiguous outcome.
|
|
147
|
+
* @param {string} root
|
|
148
|
+
* @param {string} socketPath
|
|
149
|
+
* @returns {Promise<boolean>} whether the socket is gone (replacement can proceed)
|
|
150
|
+
*/
|
|
151
|
+
async function replaceStaleCoordinator(root, socketPath) {
|
|
152
|
+
const pid = readCoordinatorLeasePid(root);
|
|
153
|
+
if (pid != null) {
|
|
154
|
+
try { process.kill(pid, "SIGTERM"); } catch { /* already gone */ }
|
|
155
|
+
}
|
|
156
|
+
const deadline = Date.now() + REPLACE_TIMEOUT_MS;
|
|
157
|
+
while (Date.now() < deadline) {
|
|
158
|
+
if (!existsSync(socketPath)) return true;
|
|
159
|
+
await new Promise((r) => setTimeout(r, REPLACE_POLL_MS));
|
|
160
|
+
}
|
|
161
|
+
return false;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Make sure a coordinator is live for this board root: probe the endpoint and,
|
|
166
|
+
* on failure, spawn one and poll until it answers (the lease guarantees a
|
|
167
|
+
* single owner even under concurrent spawns — the loser exits silently).
|
|
168
|
+
* Repeated/concurrent calls are safe and cheap once a coordinator is up.
|
|
169
|
+
*
|
|
170
|
+
* Protocol gate (issue #108): a live coordinator reporting a protocol version
|
|
171
|
+
* older than this client's build is TERMINATED and respawned — otherwise a
|
|
172
|
+
* detached old-build instance survives every extension update and rejects
|
|
173
|
+
* every new command kind (`unknown_kind`), silently stranding state writes.
|
|
174
|
+
* If the stale instance cannot be replaced (unkillable, socket stuck), report
|
|
175
|
+
* `coordinator_stale_protocol` instead of pretending it is healthy.
|
|
176
|
+
*
|
|
177
|
+
* @param {string} root
|
|
178
|
+
* @param {{ runnerScript?: string, node?: string, probeTimeoutMs?: number, ensureWindowMs?: number, pollMs?: number, connect?: (path: string) => import("node:net").Socket }} [opts]
|
|
179
|
+
* @returns {Promise<EnsureResult>}
|
|
180
|
+
*/
|
|
181
|
+
export async function ensureCoordinator(root, opts = {}) {
|
|
182
|
+
if (coordinatorDisabled()) return { ok: false, error: "coordinator_disabled" };
|
|
183
|
+
const connect = opts.connect ?? createConnection;
|
|
184
|
+
const probeTimeoutMs = opts.probeTimeoutMs ?? PROBE_TIMEOUT_MS;
|
|
185
|
+
const windowMs = opts.ensureWindowMs ?? ENSURE_WINDOW_MS;
|
|
186
|
+
const pollMs = opts.pollMs ?? ENSURE_POLL_MS;
|
|
187
|
+
const socketPath = P.coordinatorEndpointPathFor(process.platform, root);
|
|
188
|
+
|
|
189
|
+
const first = await probeOnce(socketPath, connect, probeTimeoutMs);
|
|
190
|
+
if (first) {
|
|
191
|
+
if (first.protocolVersion >= COORDINATOR_PROTOCOL_VERSION) return { ok: true, instanceId: first.instanceId };
|
|
192
|
+
if (!await replaceStaleCoordinator(root, socketPath)) {
|
|
193
|
+
return { ok: false, error: "coordinator_stale_protocol" };
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
const launched = launchCoordinator(root, { runnerScript: opts.runnerScript ?? COORDINATOR_SCRIPT, node: opts.node });
|
|
198
|
+
const deadline = Date.now() + windowMs;
|
|
199
|
+
while (Date.now() < deadline) {
|
|
200
|
+
await new Promise((r) => setTimeout(r, pollMs));
|
|
201
|
+
const probe = await probeOnce(socketPath, connect, probeTimeoutMs);
|
|
202
|
+
if (probe && probe.protocolVersion >= COORDINATOR_PROTOCOL_VERSION) {
|
|
203
|
+
return { ok: true, instanceId: probe.instanceId, pid: launched.pid };
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
return { ok: false, error: "coordinator_unavailable", pid: launched.pid };
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Submit one state command to the coordinator and wait for its result.
|
|
211
|
+
* Never throws; every failure path resolves a rejected result (see the
|
|
212
|
+
* ambiguity contract in the module doc).
|
|
213
|
+
* @param {string} root
|
|
214
|
+
* @param {object} command Command fields (`kind`, `viewId`, `runId`, `source`,
|
|
215
|
+
* `expectedRevision`, `payload`); an explicit `commandId` pins the identity
|
|
216
|
+
* for safe same-id retries, otherwise one is generated via `newRunId()`.
|
|
217
|
+
* @param {{ timeoutMs?: number, runnerScript?: string, node?: string, socketPath?: string, connect?: (path: string) => import("node:net").Socket, ensure?: typeof ensureCoordinator }} [opts]
|
|
218
|
+
* @returns {Promise<StateCommandResult>}
|
|
219
|
+
*/
|
|
220
|
+
export async function sendStateCommand(root, command, opts = {}) {
|
|
221
|
+
if (coordinatorDisabled()) {
|
|
222
|
+
return { status: "rejected", reason: "coordinator_disabled", materializedRevision: 0 };
|
|
223
|
+
}
|
|
224
|
+
const ensure = opts.ensure ?? ensureCoordinator;
|
|
225
|
+
const ensured = await ensure(root, {
|
|
226
|
+
runnerScript: opts.runnerScript,
|
|
227
|
+
node: opts.node,
|
|
228
|
+
connect: opts.connect,
|
|
229
|
+
});
|
|
230
|
+
if (!ensured.ok) {
|
|
231
|
+
return { status: "rejected", reason: ensured.error ?? "coordinator_unavailable", materializedRevision: 0 };
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
const timeoutMs = opts.timeoutMs ?? COMMAND_TIMEOUT_MS;
|
|
235
|
+
const connect = opts.connect ?? createConnection;
|
|
236
|
+
const socketPath = opts.socketPath ?? P.coordinatorEndpointPathFor(process.platform, root);
|
|
237
|
+
const commandId = typeof command.commandId === "string" && command.commandId
|
|
238
|
+
? command.commandId
|
|
239
|
+
: newRunId();
|
|
240
|
+
const envelope = { ...command, type: "state_command", commandId };
|
|
241
|
+
|
|
242
|
+
return await new Promise((resolve) => {
|
|
243
|
+
let settled = false;
|
|
244
|
+
let wrote = false;
|
|
245
|
+
let buffer = "";
|
|
246
|
+
/** @type {import("node:net").Socket|null} */
|
|
247
|
+
let socket = null;
|
|
248
|
+
const finish = (result) => {
|
|
249
|
+
if (settled) return;
|
|
250
|
+
settled = true;
|
|
251
|
+
clearTimeout(timer);
|
|
252
|
+
try { socket?.destroy(); } catch { /* best effort */ }
|
|
253
|
+
resolve(result);
|
|
254
|
+
};
|
|
255
|
+
const timer = setTimeout(() => finish({ status: "rejected", reason: "timeout", materializedRevision: 0 }), timeoutMs);
|
|
256
|
+
timer.unref?.();
|
|
257
|
+
try {
|
|
258
|
+
socket = connect(socketPath);
|
|
259
|
+
} catch {
|
|
260
|
+
// synchronous connect failure — nothing was sent, so not ambiguous
|
|
261
|
+
finish({ status: "rejected", reason: "connection_failed", materializedRevision: 0 });
|
|
262
|
+
return;
|
|
263
|
+
}
|
|
264
|
+
socket.on("error", () => {
|
|
265
|
+
// A reset AFTER the envelope was written is ambiguous (the command may
|
|
266
|
+
// be journaled); a failure before that is a plain delivery failure.
|
|
267
|
+
finish(wrote
|
|
268
|
+
? { status: "rejected", reason: "connection_reset", materializedRevision: 0 }
|
|
269
|
+
: { status: "rejected", reason: "connection_failed", materializedRevision: 0 });
|
|
270
|
+
});
|
|
271
|
+
socket.on("close", () => {
|
|
272
|
+
finish(wrote
|
|
273
|
+
? { status: "rejected", reason: "connection_reset", materializedRevision: 0 }
|
|
274
|
+
: { status: "rejected", reason: "connection_failed", materializedRevision: 0 });
|
|
275
|
+
});
|
|
276
|
+
socket.on("connect", () => {
|
|
277
|
+
try {
|
|
278
|
+
socket?.write(JSON.stringify(envelope) + "\n");
|
|
279
|
+
wrote = true;
|
|
280
|
+
} catch {
|
|
281
|
+
finish({ status: "rejected", reason: "connection_failed", materializedRevision: 0 });
|
|
282
|
+
}
|
|
283
|
+
});
|
|
284
|
+
socket.on("data", (chunk) => {
|
|
285
|
+
buffer += chunk.toString("utf8");
|
|
286
|
+
const lines = buffer.split("\n");
|
|
287
|
+
buffer = lines.pop() ?? "";
|
|
288
|
+
for (const line of lines) {
|
|
289
|
+
if (!line.trim()) continue;
|
|
290
|
+
let msg;
|
|
291
|
+
try {
|
|
292
|
+
msg = JSON.parse(line);
|
|
293
|
+
} catch {
|
|
294
|
+
continue; // malformed line — keep waiting for the result
|
|
295
|
+
}
|
|
296
|
+
if (msg?.type === "state_command_result" && msg.commandId === commandId) {
|
|
297
|
+
finish({
|
|
298
|
+
status: msg.status === "applied" ? "applied" : "rejected",
|
|
299
|
+
reason: typeof msg.reason === "string" ? msg.reason : null,
|
|
300
|
+
materializedRevision: typeof msg.materializedRevision === "number" ? msg.materializedRevision : 0,
|
|
301
|
+
});
|
|
302
|
+
return;
|
|
303
|
+
}
|
|
304
|
+
if (msg?.type === "error") {
|
|
305
|
+
// Protocol misuse on our side — a definitive, non-journaled rejection.
|
|
306
|
+
finish({ status: "rejected", reason: `coordinator_error:${msg.message ?? "unknown"}`, materializedRevision: 0 });
|
|
307
|
+
return;
|
|
308
|
+
}
|
|
309
|
+
// pong/other lines are ignored — keep waiting for the matching result.
|
|
310
|
+
}
|
|
311
|
+
});
|
|
312
|
+
});
|
|
313
|
+
}
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Durable command journal for the View State Coordinator (issue #91, spec D3).
|
|
3
|
+
*
|
|
4
|
+
* Write order contract: the coordinator appends a command record here and
|
|
5
|
+
* fsyncs BEFORE materializing state.json/status.json, so a crash between the
|
|
6
|
+
* two can be repaired by replaying the journal on restart. The checkpoint
|
|
7
|
+
* records how much of the journal (`journalBytes`) is already reflected in
|
|
8
|
+
* materialized state; GC may only drop that prefix after the checkpoint write
|
|
9
|
+
* itself succeeded. Full-coverage GC truncates the journal to empty: the
|
|
10
|
+
* covered commands leave the journal, so post-GC idempotency dedupe is owned
|
|
11
|
+
* by the coordinator's in-memory ring of recent commandIds (older commandIds
|
|
12
|
+
* fall back to current-state re-decision, whose apply paths are idempotent).
|
|
13
|
+
* `screen.log`'s fs-injection and temp+rename patterns are
|
|
14
|
+
* reused so every fs call is injectable in tests.
|
|
15
|
+
*/
|
|
16
|
+
import {
|
|
17
|
+
closeSync,
|
|
18
|
+
existsSync,
|
|
19
|
+
fstatSync,
|
|
20
|
+
fsyncSync,
|
|
21
|
+
mkdirSync,
|
|
22
|
+
openSync,
|
|
23
|
+
readSync,
|
|
24
|
+
renameSync,
|
|
25
|
+
statSync,
|
|
26
|
+
unlinkSync,
|
|
27
|
+
writeSync,
|
|
28
|
+
} from "node:fs";
|
|
29
|
+
import { dirname, join } from "node:path";
|
|
30
|
+
|
|
31
|
+
export const JOURNAL_FILE_NAME = "state-journal.jsonl";
|
|
32
|
+
export const CHECKPOINT_FILE_NAME = "state-journal.checkpoint.json";
|
|
33
|
+
|
|
34
|
+
export const defaultJournalFs = Object.freeze({
|
|
35
|
+
closeSync,
|
|
36
|
+
existsSync,
|
|
37
|
+
fstatSync,
|
|
38
|
+
fsyncSync,
|
|
39
|
+
mkdirSync,
|
|
40
|
+
openSync,
|
|
41
|
+
readSync,
|
|
42
|
+
renameSync,
|
|
43
|
+
statSync,
|
|
44
|
+
unlinkSync,
|
|
45
|
+
writeSync,
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
/** @param {string} root @returns {string} */
|
|
49
|
+
export function journalPath(root) {
|
|
50
|
+
return join(root, JOURNAL_FILE_NAME);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/** @param {string} root @returns {string} */
|
|
54
|
+
export function checkpointPath(root) {
|
|
55
|
+
return join(root, CHECKPOINT_FILE_NAME);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Append one processed-command record and fsync it before returning.
|
|
60
|
+
* @param {string} root
|
|
61
|
+
* @param {{ command: object, result: { status: string, reason: string|null }, materializedRevision: number, at: number }} record
|
|
62
|
+
* @param {typeof defaultJournalFs} [fs]
|
|
63
|
+
* @returns {number} journal size in bytes after the append (usable as the next
|
|
64
|
+
* checkpoint's `journalBytes`).
|
|
65
|
+
*/
|
|
66
|
+
export function appendCommand(root, record, fs = defaultJournalFs) {
|
|
67
|
+
const file = journalPath(root);
|
|
68
|
+
fs.mkdirSync(dirname(file), { recursive: true });
|
|
69
|
+
const payload = Buffer.from(`${JSON.stringify(record)}\n`, "utf8");
|
|
70
|
+
let fd;
|
|
71
|
+
try {
|
|
72
|
+
fd = fs.openSync(file, "a");
|
|
73
|
+
let offset = 0;
|
|
74
|
+
while (offset < payload.length) offset += fs.writeSync(fd, payload, offset, payload.length - offset);
|
|
75
|
+
fs.fsyncSync(fd);
|
|
76
|
+
} finally {
|
|
77
|
+
if (fd !== undefined) {
|
|
78
|
+
try { fs.closeSync(fd); } catch { /* already closed */ }
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
return fileSize(file, 0, fs);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Read every parseable journal record. A corrupt line (crash mid-append) is
|
|
86
|
+
* skipped — same semantics as atomic.mjs readJsonl, but with injectable fs.
|
|
87
|
+
* @param {string} root
|
|
88
|
+
* @param {typeof defaultJournalFs} [fs]
|
|
89
|
+
* @returns {Array<{ command: object, result: { status: string, reason: string|null }, materializedRevision: number, at: number }>}
|
|
90
|
+
*/
|
|
91
|
+
export function readJournal(root, fs = defaultJournalFs) {
|
|
92
|
+
const data = readAllBytes(journalPath(root), fs);
|
|
93
|
+
const entries = [];
|
|
94
|
+
for (const line of data.toString("utf8").split("\n")) {
|
|
95
|
+
const trimmed = line.trim();
|
|
96
|
+
if (!trimmed) continue;
|
|
97
|
+
try {
|
|
98
|
+
entries.push(JSON.parse(trimmed));
|
|
99
|
+
} catch {
|
|
100
|
+
/* skip corrupt line */
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
return entries;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Idempotency lookup: the recorded result of an already-processed commandId,
|
|
108
|
+
* or null. On the (never-intended) duplicate, the FIRST entry is the original.
|
|
109
|
+
* @param {string} root
|
|
110
|
+
* @param {string} commandId
|
|
111
|
+
* @param {typeof defaultJournalFs} [fs]
|
|
112
|
+
* @returns {{ status: string, reason: string|null } | null}
|
|
113
|
+
*/
|
|
114
|
+
export function findProcessedCommand(root, commandId, fs = defaultJournalFs) {
|
|
115
|
+
for (const entry of readJournal(root, fs)) {
|
|
116
|
+
if (entry?.command?.commandId === commandId) return entry.result ?? null;
|
|
117
|
+
}
|
|
118
|
+
return null;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* @param {string} root
|
|
123
|
+
* @param {typeof defaultJournalFs} [fs]
|
|
124
|
+
* @returns {{ materializedRevision: number, journalBytes: number } | null}
|
|
125
|
+
*/
|
|
126
|
+
export function readCheckpoint(root, fs = defaultJournalFs) {
|
|
127
|
+
const file = checkpointPath(root);
|
|
128
|
+
if (!fs.existsSync(file)) return null;
|
|
129
|
+
let raw;
|
|
130
|
+
try {
|
|
131
|
+
raw = readAllBytes(file, fs).toString("utf8");
|
|
132
|
+
} catch {
|
|
133
|
+
return null;
|
|
134
|
+
}
|
|
135
|
+
if (!raw.trim()) return null;
|
|
136
|
+
try {
|
|
137
|
+
return JSON.parse(raw);
|
|
138
|
+
} catch {
|
|
139
|
+
return null;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Atomically replace the checkpoint (temp file + fsync + rename, screen-log
|
|
145
|
+
* pattern). Callers may only run gcJournal after this returns true.
|
|
146
|
+
* @param {string} root
|
|
147
|
+
* @param {{ materializedRevision: number, journalBytes: number }} checkpoint
|
|
148
|
+
* @param {typeof defaultJournalFs} [fs]
|
|
149
|
+
* @returns {boolean} whether the checkpoint is durably on disk
|
|
150
|
+
*/
|
|
151
|
+
export function writeCheckpoint(root, checkpoint, fs = defaultJournalFs) {
|
|
152
|
+
const file = checkpointPath(root);
|
|
153
|
+
fs.mkdirSync(dirname(file), { recursive: true });
|
|
154
|
+
return replaceFile(file, Buffer.from(`${JSON.stringify(checkpoint, null, 2)}\n`, "utf8"), fs);
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Drop the journal prefix already covered by a successful checkpoint. Without
|
|
159
|
+
* a checkpoint this is a no-op (nothing proves the prefix is materialized).
|
|
160
|
+
* A checkpoint covering the whole journal truncates the file to empty — every
|
|
161
|
+
* covered record is materialized, and keeping it would grow the journal (and
|
|
162
|
+
* every findProcessedCommand/boot-replay scan) without bound.
|
|
163
|
+
* @param {string} root
|
|
164
|
+
* @param {typeof defaultJournalFs} [fs]
|
|
165
|
+
* @returns {number} journal size in bytes after the call
|
|
166
|
+
*/
|
|
167
|
+
export function gcJournal(root, fs = defaultJournalFs) {
|
|
168
|
+
const file = journalPath(root);
|
|
169
|
+
const size = fileSize(file, 0, fs);
|
|
170
|
+
if (size === 0) return 0;
|
|
171
|
+
const checkpoint = readCheckpoint(root, fs);
|
|
172
|
+
if (!checkpoint || typeof checkpoint.journalBytes !== "number") return size;
|
|
173
|
+
const journalBytes = Math.floor(checkpoint.journalBytes);
|
|
174
|
+
if (journalBytes <= 0) return size;
|
|
175
|
+
if (journalBytes >= size) {
|
|
176
|
+
// Full coverage: everything in the file is materialized per the
|
|
177
|
+
// checkpoint. Truncate to empty; dedupe for covered commandIds is the
|
|
178
|
+
// coordinator's in-memory ring's job now.
|
|
179
|
+
if (!replaceFile(file, Buffer.alloc(0), fs)) return fileSize(file, size, fs);
|
|
180
|
+
return 0;
|
|
181
|
+
}
|
|
182
|
+
// journalBytes < size means the file exists with an un-checkpointed tail.
|
|
183
|
+
const tail = readAllBytes(file, fs, journalBytes);
|
|
184
|
+
if (!replaceFile(file, tail, fs)) return fileSize(file, size, fs);
|
|
185
|
+
return fileSize(file, tail.length, fs);
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Truncate a crash-torn journal tail so the next append stays parseable.
|
|
190
|
+
*
|
|
191
|
+
* `appendCommand` writes `JSON+\n` in a partial-write loop; a process killed
|
|
192
|
+
* mid-append leaves bytes without a trailing newline. Without repair, the next
|
|
193
|
+
* append concatenates onto that torn line and the merged line is unparseable — the
|
|
194
|
+
* freshly fsynced+acked record becomes invisible to every readJournal /
|
|
195
|
+
* boot-replay / revision scan. The repair is byte-exact and cheap: scan
|
|
196
|
+
* backwards for the last newline and drop everything after it (a journal with
|
|
197
|
+
* no newline at all is torn from byte 0 and truncates to empty). Complete but
|
|
198
|
+
* corrupt lines are kept — readJournal already skips them.
|
|
199
|
+
* @param {string} root
|
|
200
|
+
* @param {typeof defaultJournalFs} [fs]
|
|
201
|
+
* @returns {number} journal size in bytes after the call
|
|
202
|
+
*/
|
|
203
|
+
export function repairJournalTail(root, fs = defaultJournalFs) {
|
|
204
|
+
const file = journalPath(root);
|
|
205
|
+
const size = fileSize(file, 0, fs);
|
|
206
|
+
if (size === 0) return 0;
|
|
207
|
+
const data = readAllBytes(file, fs);
|
|
208
|
+
const lastNewline = data.lastIndexOf("\n");
|
|
209
|
+
if (lastNewline === data.length - 1) return size;
|
|
210
|
+
const keep = lastNewline + 1;
|
|
211
|
+
if (!replaceFile(file, data.subarray(0, keep), fs)) return fileSize(file, size, fs);
|
|
212
|
+
return keep;
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/**
|
|
216
|
+
* Read the file from `position` (default 0) to EOF via injectable readSync.
|
|
217
|
+
* @param {string} file
|
|
218
|
+
* @param {typeof defaultJournalFs} fs
|
|
219
|
+
* @param {number} [position]
|
|
220
|
+
* @returns {Buffer}
|
|
221
|
+
*/
|
|
222
|
+
function readAllBytes(file, fs, position = 0) {
|
|
223
|
+
if (!fs.existsSync(file)) return Buffer.alloc(0);
|
|
224
|
+
let fd;
|
|
225
|
+
try {
|
|
226
|
+
fd = fs.openSync(file, "r");
|
|
227
|
+
const size = Math.max(0, fs.fstatSync(fd).size - position);
|
|
228
|
+
if (size === 0) return Buffer.alloc(0);
|
|
229
|
+
const output = Buffer.allocUnsafe(size);
|
|
230
|
+
let offset = 0;
|
|
231
|
+
while (offset < size) {
|
|
232
|
+
const read = fs.readSync(fd, output, offset, size - offset, position + offset);
|
|
233
|
+
if (read === 0) break;
|
|
234
|
+
offset += read;
|
|
235
|
+
}
|
|
236
|
+
return offset === size ? output : output.subarray(0, offset);
|
|
237
|
+
} finally {
|
|
238
|
+
if (fd !== undefined) {
|
|
239
|
+
try { fs.closeSync(fd); } catch { /* already closed */ }
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* Temp file + fsync + rename replacement (screen-log replaceScreenLog pattern).
|
|
246
|
+
* @param {string} file
|
|
247
|
+
* @param {Buffer} data
|
|
248
|
+
* @param {typeof defaultJournalFs} fs
|
|
249
|
+
* @returns {boolean}
|
|
250
|
+
*/
|
|
251
|
+
function replaceFile(file, data, fs) {
|
|
252
|
+
const temp = `${file}.${process.pid}.${tempSequence++}.tmp`;
|
|
253
|
+
let fd;
|
|
254
|
+
try {
|
|
255
|
+
fs.mkdirSync(dirname(file), { recursive: true });
|
|
256
|
+
fd = fs.openSync(temp, "w");
|
|
257
|
+
let offset = 0;
|
|
258
|
+
while (offset < data.length) offset += fs.writeSync(fd, data, offset, data.length - offset, offset);
|
|
259
|
+
fs.fsyncSync(fd);
|
|
260
|
+
fs.closeSync(fd);
|
|
261
|
+
fd = undefined;
|
|
262
|
+
fs.renameSync(temp, file);
|
|
263
|
+
return true;
|
|
264
|
+
} catch {
|
|
265
|
+
if (fd !== undefined) {
|
|
266
|
+
try { fs.closeSync(fd); } catch { /* already closed */ }
|
|
267
|
+
}
|
|
268
|
+
try { fs.unlinkSync(temp); } catch { /* best effort */ }
|
|
269
|
+
return false;
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/** @param {string} file @param {number} fallback @param {typeof defaultJournalFs} fs */
|
|
274
|
+
function fileSize(file, fallback, fs) {
|
|
275
|
+
try {
|
|
276
|
+
return fs.statSync(file).size;
|
|
277
|
+
} catch {
|
|
278
|
+
return fallback;
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
let tempSequence = 0;
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Protocol version of the View State Coordinator control socket (issue #108).
|
|
3
|
+
*
|
|
4
|
+
* Bumped whenever a coordinator release adds command kinds or changes the
|
|
5
|
+
* envelope contract. A client built against version N refuses to talk to a
|
|
6
|
+
* coordinator reporting < N — a pong WITHOUT the field counts as version 1
|
|
7
|
+
* (the pre-#107 baseline) — and replaces the stale instance with a fresh one
|
|
8
|
+
* (SIGTERM via the coordinator lease pid, then respawn) instead of feeding
|
|
9
|
+
* it commands it cannot understand (`sync_foreground rejected (unknown_kind)`
|
|
10
|
+
* used to strand every foreground state write on extension updates).
|
|
11
|
+
*/
|
|
12
|
+
export const COORDINATOR_PROTOCOL_VERSION = 2;
|
package/src/core/launch.mjs
CHANGED
|
@@ -120,3 +120,18 @@ export function launchAutoState(root, config, opts) {
|
|
|
120
120
|
|
|
121
121
|
return { pid: child.pid ?? null, configPath };
|
|
122
122
|
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Launch the detached view-state coordinator for a board root (issue #91, spec
|
|
126
|
+
* D3). No config file: the coordinator takes the root as its only argument.
|
|
127
|
+
* Idempotent by lease — a second instance loses the coordinator lease and
|
|
128
|
+
* exits silently, so callers may spawn freely on probe failure.
|
|
129
|
+
* @param {string} root
|
|
130
|
+
* @param {{ runnerScript: string, node?: string }} opts
|
|
131
|
+
* @returns {{ pid: number|null }}
|
|
132
|
+
*/
|
|
133
|
+
export function launchCoordinator(root, opts) {
|
|
134
|
+
const node = opts.node ?? resolveNode();
|
|
135
|
+
const child = spawnDetached(node, [opts.runnerScript, root], root);
|
|
136
|
+
return { pid: child.pid ?? null };
|
|
137
|
+
}
|
package/src/core/paths.mjs
CHANGED
|
@@ -89,6 +89,22 @@ export function hostEndpointPathFor(platform, root, viewId, instanceId) {
|
|
|
89
89
|
}
|
|
90
90
|
/** @param {string} root @param {string} viewId */
|
|
91
91
|
export const screenLogPath = (root, viewId) => path.join(viewDir(root, viewId), "screen.log");
|
|
92
|
+
/**
|
|
93
|
+
* Well-known endpoint for the board-root View State Coordinator (issue #91, spec D3).
|
|
94
|
+
* Exactly one coordinator may own a root (token-fenced lease), so one stable path
|
|
95
|
+
* suffices; a stale POSIX socket left by a crashed coordinator is unlinked by the
|
|
96
|
+
* new lease owner before bind. win32 pipe names embed a 16-hex hash of the root to
|
|
97
|
+
* stay under the 256-char limit and keep per-root isolation.
|
|
98
|
+
* @param {"win32"|"linux"|"darwin"} platform
|
|
99
|
+
* @param {string} root
|
|
100
|
+
*/
|
|
101
|
+
export function coordinatorEndpointPathFor(platform, root) {
|
|
102
|
+
if (platform === "win32") {
|
|
103
|
+
const hash = createHash("sha256").update(String(root)).digest("hex").slice(0, 16);
|
|
104
|
+
return `\\\\.\\pipe\\agent-board-coordinator-${hash}`;
|
|
105
|
+
}
|
|
106
|
+
return path.join(root, "coordinator.sock");
|
|
107
|
+
}
|
|
92
108
|
/** @param {string} root @param {string} viewId */
|
|
93
109
|
export const hostPidPath = (root, viewId) => path.join(viewDir(root, viewId), "host-pid.json");
|
|
94
110
|
/** @param {string} root @param {string} viewId */
|