@zhuxixi/pi-agent-board 0.6.2 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -51,7 +51,8 @@ async function handleBgCommand(args: string, ctx: ExtensionCommandContext, opts:
51
51
  },
52
52
  });
53
53
  const model = modelRef(ctx.model as any);
54
- const adopted = service.adoptSession({
54
+ // adoptSession is async (routes through the view-state coordinator, issue #91).
55
+ const adopted = await service.adoptSession({
55
56
  sessionFile,
56
57
  cwd: ctx.cwd,
57
58
  model,
@@ -0,0 +1,313 @@
1
+ /**
2
+ * Client for the detached View State Coordinator (issue #91, spec D3).
3
+ *
4
+ * Every semantic-state writer (dashboard service, job-runner, state-runner,
5
+ * CLI) submits `{type:"state_command"}` envelopes through `sendStateCommand`
6
+ * instead of writing state.json/status.json directly. The client resolves the
7
+ * coordinator endpoint, ensures a coordinator is live (spawning one if needed
8
+ * — idempotent via the coordinator's own lease), and waits for the matching
9
+ * `state_command_result`.
10
+ *
11
+ * Ambiguity contract (binding, from the Task 4 review): the coordinator can
12
+ * crash mid-command (e.g. an fs error) WITHOUT replying. The client then sees
13
+ * a connection reset or a timeout. Both mean the command MAY already be
14
+ * journaled and will replay on coordinator restart, so these outcomes are
15
+ * AMBIGUOUS: they resolve `{status:"rejected", reason:"timeout"|
16
+ * "connection_reset", materializedRevision:0}` and the client NEVER retries
17
+ * with a fresh commandId. Retrying with the SAME commandId after reconnect is
18
+ * safe (the coordinator dedupes) but is the caller's decision — which is why
19
+ * the caller may pin `command.commandId`.
20
+ *
21
+ * `materializedRevision: 0` on client-side outcomes means "no revision was
22
+ * observed" — it is not a store revision. Real coordinator replies always
23
+ * carry the actual revision (≥ 1 after any applied command; decided
24
+ * rejections carry the current revision).
25
+ */
26
+ import { createConnection } from "node:net";
27
+ import { existsSync, readFileSync } from "node:fs";
28
+ import { fileURLToPath } from "node:url";
29
+ import { newRunId } from "./ids.mjs";
30
+ import { launchCoordinator } from "./launch.mjs";
31
+ import * as P from "./paths.mjs";
32
+ import { COORDINATOR_PROTOCOL_VERSION } from "./coordinator-protocol.mjs";
33
+
34
+ const COORDINATOR_SCRIPT = fileURLToPath(
35
+ new URL("../../runner/state-coordinator.mjs", import.meta.url),
36
+ );
37
+
38
+ const PROBE_TIMEOUT_MS = 1_000;
39
+ const ENSURE_WINDOW_MS = 10_000;
40
+ const ENSURE_POLL_MS = 100;
41
+ const COMMAND_TIMEOUT_MS = 5_000;
42
+
43
+ /** @typedef {{ status: "applied"|"rejected", reason: string|null, materializedRevision: number }} StateCommandResult */
44
+ /** @typedef {{ ok: boolean, instanceId?: string, pid?: number|null, error?: "coordinator_disabled"|"coordinator_unavailable"|"coordinator_stale_protocol" }} EnsureResult */
45
+
46
+ /**
47
+ * Whether the coordinator is switched off for this process (tests/legacy
48
+ * escape hatch). `AGENT_BOARD_COORDINATOR=off` makes every command resolve
49
+ * `coordinator_disabled` so callers fall back to the pre-coordinator path.
50
+ * @returns {boolean}
51
+ */
52
+ export function coordinatorDisabled() {
53
+ return /^(0|false|off|no)$/i.test(String(process.env.AGENT_BOARD_COORDINATOR ?? "").trim());
54
+ }
55
+
56
+ /**
57
+ * One probe round-trip: connect, ping, wait for the pong. The coordinator
58
+ * binds its socket only after boot replay, so a pong proves full readiness.
59
+ * @param {string} socketPath
60
+ * @param {(path: string) => import("node:net").Socket} connect
61
+ * @param {number} timeoutMs
62
+ * @returns {Promise<{ instanceId: string, protocolVersion: number }|null>} null on timeout/error;
63
+ * a pong without a numeric protocolVersion counts as version 1 (pre-#107 baseline)
64
+ */
65
+ function probeOnce(socketPath, connect, timeoutMs) {
66
+ return new Promise((resolve) => {
67
+ let settled = false;
68
+ let buffer = "";
69
+ /** @type {import("node:net").Socket|null} */
70
+ let socket = null;
71
+ const finish = (instanceId) => {
72
+ if (settled) return;
73
+ settled = true;
74
+ clearTimeout(timer);
75
+ try { socket?.destroy(); } catch { /* best effort */ }
76
+ resolve(instanceId);
77
+ };
78
+ const timer = setTimeout(() => finish(null), timeoutMs);
79
+ timer.unref?.();
80
+ try {
81
+ socket = connect(socketPath);
82
+ } catch {
83
+ finish(null);
84
+ return;
85
+ }
86
+ // Missing filesystem sockets surface as async errors, not throw-on-connect
87
+ // (and Windows pipes never exist as files — never gate on existsSync).
88
+ socket.on("error", () => finish(null));
89
+ socket.on("connect", () => {
90
+ try {
91
+ socket?.write(JSON.stringify({ type: "ping" }) + "\n");
92
+ } catch {
93
+ finish(null);
94
+ }
95
+ });
96
+ socket.on("data", (chunk) => {
97
+ buffer += chunk.toString("utf8");
98
+ const lines = buffer.split("\n");
99
+ buffer = lines.pop() ?? "";
100
+ for (const line of lines) {
101
+ if (!line.trim()) continue;
102
+ try {
103
+ const msg = JSON.parse(line);
104
+ if (msg?.type === "pong" && typeof msg.instanceId === "string") {
105
+ finish({
106
+ instanceId: msg.instanceId,
107
+ protocolVersion: typeof msg.protocolVersion === "number" ? msg.protocolVersion : 1,
108
+ });
109
+ return;
110
+ }
111
+ } catch {
112
+ // malformed line — keep waiting for the pong
113
+ }
114
+ }
115
+ });
116
+ });
117
+ }
118
+
119
+ /**
120
+ * Read the coordinator lease's owner.json and return the owning pid. The
121
+ * lease is written by whichever process owns the "_coordinator" lock, so it
122
+ * identifies the live coordinator even when its pong predates the pid field.
123
+ * @param {string} root
124
+ * @returns {number|null}
125
+ */
126
+ function readCoordinatorLeasePid(root) {
127
+ try {
128
+ const lockPath = P.viewLockPath(root, "_coordinator", "state-coordinator");
129
+ const owner = JSON.parse(readFileSync(`${lockPath}/owner.json`, "utf8"));
130
+ const pid = Number(owner?.identity?.pid ?? owner?.pid ?? 0);
131
+ return Number.isFinite(pid) && pid > 0 ? pid : null;
132
+ } catch {
133
+ return null;
134
+ }
135
+ }
136
+
137
+ /** Grace window for a SIGTERMed stale coordinator to unlink its socket. */
138
+ const REPLACE_TIMEOUT_MS = 3_000;
139
+ const REPLACE_POLL_MS = 50;
140
+
141
+ /**
142
+ * Terminate a stale-protocol coordinator (issue #108) and wait for its socket
143
+ * to disappear so a fresh instance can bind. SIGTERM triggers the
144
+ * coordinator's graceful shutdown (socket unlink + lease release); its crash
145
+ * safety (fsync-before-materialize journal + boot replay) makes the kill safe
146
+ * even mid-command — an in-flight reply is the documented ambiguous outcome.
147
+ * @param {string} root
148
+ * @param {string} socketPath
149
+ * @returns {Promise<boolean>} whether the socket is gone (replacement can proceed)
150
+ */
151
+ async function replaceStaleCoordinator(root, socketPath) {
152
+ const pid = readCoordinatorLeasePid(root);
153
+ if (pid != null) {
154
+ try { process.kill(pid, "SIGTERM"); } catch { /* already gone */ }
155
+ }
156
+ const deadline = Date.now() + REPLACE_TIMEOUT_MS;
157
+ while (Date.now() < deadline) {
158
+ if (!existsSync(socketPath)) return true;
159
+ await new Promise((r) => setTimeout(r, REPLACE_POLL_MS));
160
+ }
161
+ return false;
162
+ }
163
+
164
+ /**
165
+ * Make sure a coordinator is live for this board root: probe the endpoint and,
166
+ * on failure, spawn one and poll until it answers (the lease guarantees a
167
+ * single owner even under concurrent spawns — the loser exits silently).
168
+ * Repeated/concurrent calls are safe and cheap once a coordinator is up.
169
+ *
170
+ * Protocol gate (issue #108): a live coordinator reporting a protocol version
171
+ * older than this client's build is TERMINATED and respawned — otherwise a
172
+ * detached old-build instance survives every extension update and rejects
173
+ * every new command kind (`unknown_kind`), silently stranding state writes.
174
+ * If the stale instance cannot be replaced (unkillable, socket stuck), report
175
+ * `coordinator_stale_protocol` instead of pretending it is healthy.
176
+ *
177
+ * @param {string} root
178
+ * @param {{ runnerScript?: string, node?: string, probeTimeoutMs?: number, ensureWindowMs?: number, pollMs?: number, connect?: (path: string) => import("node:net").Socket }} [opts]
179
+ * @returns {Promise<EnsureResult>}
180
+ */
181
+ export async function ensureCoordinator(root, opts = {}) {
182
+ if (coordinatorDisabled()) return { ok: false, error: "coordinator_disabled" };
183
+ const connect = opts.connect ?? createConnection;
184
+ const probeTimeoutMs = opts.probeTimeoutMs ?? PROBE_TIMEOUT_MS;
185
+ const windowMs = opts.ensureWindowMs ?? ENSURE_WINDOW_MS;
186
+ const pollMs = opts.pollMs ?? ENSURE_POLL_MS;
187
+ const socketPath = P.coordinatorEndpointPathFor(process.platform, root);
188
+
189
+ const first = await probeOnce(socketPath, connect, probeTimeoutMs);
190
+ if (first) {
191
+ if (first.protocolVersion >= COORDINATOR_PROTOCOL_VERSION) return { ok: true, instanceId: first.instanceId };
192
+ if (!await replaceStaleCoordinator(root, socketPath)) {
193
+ return { ok: false, error: "coordinator_stale_protocol" };
194
+ }
195
+ }
196
+
197
+ const launched = launchCoordinator(root, { runnerScript: opts.runnerScript ?? COORDINATOR_SCRIPT, node: opts.node });
198
+ const deadline = Date.now() + windowMs;
199
+ while (Date.now() < deadline) {
200
+ await new Promise((r) => setTimeout(r, pollMs));
201
+ const probe = await probeOnce(socketPath, connect, probeTimeoutMs);
202
+ if (probe && probe.protocolVersion >= COORDINATOR_PROTOCOL_VERSION) {
203
+ return { ok: true, instanceId: probe.instanceId, pid: launched.pid };
204
+ }
205
+ }
206
+ return { ok: false, error: "coordinator_unavailable", pid: launched.pid };
207
+ }
208
+
209
+ /**
210
+ * Submit one state command to the coordinator and wait for its result.
211
+ * Never throws; every failure path resolves a rejected result (see the
212
+ * ambiguity contract in the module doc).
213
+ * @param {string} root
214
+ * @param {object} command Command fields (`kind`, `viewId`, `runId`, `source`,
215
+ * `expectedRevision`, `payload`); an explicit `commandId` pins the identity
216
+ * for safe same-id retries, otherwise one is generated via `newRunId()`.
217
+ * @param {{ timeoutMs?: number, runnerScript?: string, node?: string, socketPath?: string, connect?: (path: string) => import("node:net").Socket, ensure?: typeof ensureCoordinator }} [opts]
218
+ * @returns {Promise<StateCommandResult>}
219
+ */
220
+ export async function sendStateCommand(root, command, opts = {}) {
221
+ if (coordinatorDisabled()) {
222
+ return { status: "rejected", reason: "coordinator_disabled", materializedRevision: 0 };
223
+ }
224
+ const ensure = opts.ensure ?? ensureCoordinator;
225
+ const ensured = await ensure(root, {
226
+ runnerScript: opts.runnerScript,
227
+ node: opts.node,
228
+ connect: opts.connect,
229
+ });
230
+ if (!ensured.ok) {
231
+ return { status: "rejected", reason: ensured.error ?? "coordinator_unavailable", materializedRevision: 0 };
232
+ }
233
+
234
+ const timeoutMs = opts.timeoutMs ?? COMMAND_TIMEOUT_MS;
235
+ const connect = opts.connect ?? createConnection;
236
+ const socketPath = opts.socketPath ?? P.coordinatorEndpointPathFor(process.platform, root);
237
+ const commandId = typeof command.commandId === "string" && command.commandId
238
+ ? command.commandId
239
+ : newRunId();
240
+ const envelope = { ...command, type: "state_command", commandId };
241
+
242
+ return await new Promise((resolve) => {
243
+ let settled = false;
244
+ let wrote = false;
245
+ let buffer = "";
246
+ /** @type {import("node:net").Socket|null} */
247
+ let socket = null;
248
+ const finish = (result) => {
249
+ if (settled) return;
250
+ settled = true;
251
+ clearTimeout(timer);
252
+ try { socket?.destroy(); } catch { /* best effort */ }
253
+ resolve(result);
254
+ };
255
+ const timer = setTimeout(() => finish({ status: "rejected", reason: "timeout", materializedRevision: 0 }), timeoutMs);
256
+ timer.unref?.();
257
+ try {
258
+ socket = connect(socketPath);
259
+ } catch {
260
+ // synchronous connect failure — nothing was sent, so not ambiguous
261
+ finish({ status: "rejected", reason: "connection_failed", materializedRevision: 0 });
262
+ return;
263
+ }
264
+ socket.on("error", () => {
265
+ // A reset AFTER the envelope was written is ambiguous (the command may
266
+ // be journaled); a failure before that is a plain delivery failure.
267
+ finish(wrote
268
+ ? { status: "rejected", reason: "connection_reset", materializedRevision: 0 }
269
+ : { status: "rejected", reason: "connection_failed", materializedRevision: 0 });
270
+ });
271
+ socket.on("close", () => {
272
+ finish(wrote
273
+ ? { status: "rejected", reason: "connection_reset", materializedRevision: 0 }
274
+ : { status: "rejected", reason: "connection_failed", materializedRevision: 0 });
275
+ });
276
+ socket.on("connect", () => {
277
+ try {
278
+ socket?.write(JSON.stringify(envelope) + "\n");
279
+ wrote = true;
280
+ } catch {
281
+ finish({ status: "rejected", reason: "connection_failed", materializedRevision: 0 });
282
+ }
283
+ });
284
+ socket.on("data", (chunk) => {
285
+ buffer += chunk.toString("utf8");
286
+ const lines = buffer.split("\n");
287
+ buffer = lines.pop() ?? "";
288
+ for (const line of lines) {
289
+ if (!line.trim()) continue;
290
+ let msg;
291
+ try {
292
+ msg = JSON.parse(line);
293
+ } catch {
294
+ continue; // malformed line — keep waiting for the result
295
+ }
296
+ if (msg?.type === "state_command_result" && msg.commandId === commandId) {
297
+ finish({
298
+ status: msg.status === "applied" ? "applied" : "rejected",
299
+ reason: typeof msg.reason === "string" ? msg.reason : null,
300
+ materializedRevision: typeof msg.materializedRevision === "number" ? msg.materializedRevision : 0,
301
+ });
302
+ return;
303
+ }
304
+ if (msg?.type === "error") {
305
+ // Protocol misuse on our side — a definitive, non-journaled rejection.
306
+ finish({ status: "rejected", reason: `coordinator_error:${msg.message ?? "unknown"}`, materializedRevision: 0 });
307
+ return;
308
+ }
309
+ // pong/other lines are ignored — keep waiting for the matching result.
310
+ }
311
+ });
312
+ });
313
+ }
@@ -0,0 +1,282 @@
1
+ /**
2
+ * Durable command journal for the View State Coordinator (issue #91, spec D3).
3
+ *
4
+ * Write order contract: the coordinator appends a command record here and
5
+ * fsyncs BEFORE materializing state.json/status.json, so a crash between the
6
+ * two can be repaired by replaying the journal on restart. The checkpoint
7
+ * records how much of the journal (`journalBytes`) is already reflected in
8
+ * materialized state; GC may only drop that prefix after the checkpoint write
9
+ * itself succeeded. Full-coverage GC truncates the journal to empty: the
10
+ * covered commands leave the journal, so post-GC idempotency dedupe is owned
11
+ * by the coordinator's in-memory ring of recent commandIds (older commandIds
12
+ * fall back to current-state re-decision, whose apply paths are idempotent).
13
+ * `screen.log`'s fs-injection and temp+rename patterns are
14
+ * reused so every fs call is injectable in tests.
15
+ */
16
+ import {
17
+ closeSync,
18
+ existsSync,
19
+ fstatSync,
20
+ fsyncSync,
21
+ mkdirSync,
22
+ openSync,
23
+ readSync,
24
+ renameSync,
25
+ statSync,
26
+ unlinkSync,
27
+ writeSync,
28
+ } from "node:fs";
29
+ import { dirname, join } from "node:path";
30
+
31
+ export const JOURNAL_FILE_NAME = "state-journal.jsonl";
32
+ export const CHECKPOINT_FILE_NAME = "state-journal.checkpoint.json";
33
+
34
+ export const defaultJournalFs = Object.freeze({
35
+ closeSync,
36
+ existsSync,
37
+ fstatSync,
38
+ fsyncSync,
39
+ mkdirSync,
40
+ openSync,
41
+ readSync,
42
+ renameSync,
43
+ statSync,
44
+ unlinkSync,
45
+ writeSync,
46
+ });
47
+
48
+ /** @param {string} root @returns {string} */
49
+ export function journalPath(root) {
50
+ return join(root, JOURNAL_FILE_NAME);
51
+ }
52
+
53
+ /** @param {string} root @returns {string} */
54
+ export function checkpointPath(root) {
55
+ return join(root, CHECKPOINT_FILE_NAME);
56
+ }
57
+
58
+ /**
59
+ * Append one processed-command record and fsync it before returning.
60
+ * @param {string} root
61
+ * @param {{ command: object, result: { status: string, reason: string|null }, materializedRevision: number, at: number }} record
62
+ * @param {typeof defaultJournalFs} [fs]
63
+ * @returns {number} journal size in bytes after the append (usable as the next
64
+ * checkpoint's `journalBytes`).
65
+ */
66
+ export function appendCommand(root, record, fs = defaultJournalFs) {
67
+ const file = journalPath(root);
68
+ fs.mkdirSync(dirname(file), { recursive: true });
69
+ const payload = Buffer.from(`${JSON.stringify(record)}\n`, "utf8");
70
+ let fd;
71
+ try {
72
+ fd = fs.openSync(file, "a");
73
+ let offset = 0;
74
+ while (offset < payload.length) offset += fs.writeSync(fd, payload, offset, payload.length - offset);
75
+ fs.fsyncSync(fd);
76
+ } finally {
77
+ if (fd !== undefined) {
78
+ try { fs.closeSync(fd); } catch { /* already closed */ }
79
+ }
80
+ }
81
+ return fileSize(file, 0, fs);
82
+ }
83
+
84
+ /**
85
+ * Read every parseable journal record. A corrupt line (crash mid-append) is
86
+ * skipped — same semantics as atomic.mjs readJsonl, but with injectable fs.
87
+ * @param {string} root
88
+ * @param {typeof defaultJournalFs} [fs]
89
+ * @returns {Array<{ command: object, result: { status: string, reason: string|null }, materializedRevision: number, at: number }>}
90
+ */
91
+ export function readJournal(root, fs = defaultJournalFs) {
92
+ const data = readAllBytes(journalPath(root), fs);
93
+ const entries = [];
94
+ for (const line of data.toString("utf8").split("\n")) {
95
+ const trimmed = line.trim();
96
+ if (!trimmed) continue;
97
+ try {
98
+ entries.push(JSON.parse(trimmed));
99
+ } catch {
100
+ /* skip corrupt line */
101
+ }
102
+ }
103
+ return entries;
104
+ }
105
+
106
+ /**
107
+ * Idempotency lookup: the recorded result of an already-processed commandId,
108
+ * or null. On the (never-intended) duplicate, the FIRST entry is the original.
109
+ * @param {string} root
110
+ * @param {string} commandId
111
+ * @param {typeof defaultJournalFs} [fs]
112
+ * @returns {{ status: string, reason: string|null } | null}
113
+ */
114
+ export function findProcessedCommand(root, commandId, fs = defaultJournalFs) {
115
+ for (const entry of readJournal(root, fs)) {
116
+ if (entry?.command?.commandId === commandId) return entry.result ?? null;
117
+ }
118
+ return null;
119
+ }
120
+
121
+ /**
122
+ * @param {string} root
123
+ * @param {typeof defaultJournalFs} [fs]
124
+ * @returns {{ materializedRevision: number, journalBytes: number } | null}
125
+ */
126
+ export function readCheckpoint(root, fs = defaultJournalFs) {
127
+ const file = checkpointPath(root);
128
+ if (!fs.existsSync(file)) return null;
129
+ let raw;
130
+ try {
131
+ raw = readAllBytes(file, fs).toString("utf8");
132
+ } catch {
133
+ return null;
134
+ }
135
+ if (!raw.trim()) return null;
136
+ try {
137
+ return JSON.parse(raw);
138
+ } catch {
139
+ return null;
140
+ }
141
+ }
142
+
143
+ /**
144
+ * Atomically replace the checkpoint (temp file + fsync + rename, screen-log
145
+ * pattern). Callers may only run gcJournal after this returns true.
146
+ * @param {string} root
147
+ * @param {{ materializedRevision: number, journalBytes: number }} checkpoint
148
+ * @param {typeof defaultJournalFs} [fs]
149
+ * @returns {boolean} whether the checkpoint is durably on disk
150
+ */
151
+ export function writeCheckpoint(root, checkpoint, fs = defaultJournalFs) {
152
+ const file = checkpointPath(root);
153
+ fs.mkdirSync(dirname(file), { recursive: true });
154
+ return replaceFile(file, Buffer.from(`${JSON.stringify(checkpoint, null, 2)}\n`, "utf8"), fs);
155
+ }
156
+
157
+ /**
158
+ * Drop the journal prefix already covered by a successful checkpoint. Without
159
+ * a checkpoint this is a no-op (nothing proves the prefix is materialized).
160
+ * A checkpoint covering the whole journal truncates the file to empty — every
161
+ * covered record is materialized, and keeping it would grow the journal (and
162
+ * every findProcessedCommand/boot-replay scan) without bound.
163
+ * @param {string} root
164
+ * @param {typeof defaultJournalFs} [fs]
165
+ * @returns {number} journal size in bytes after the call
166
+ */
167
+ export function gcJournal(root, fs = defaultJournalFs) {
168
+ const file = journalPath(root);
169
+ const size = fileSize(file, 0, fs);
170
+ if (size === 0) return 0;
171
+ const checkpoint = readCheckpoint(root, fs);
172
+ if (!checkpoint || typeof checkpoint.journalBytes !== "number") return size;
173
+ const journalBytes = Math.floor(checkpoint.journalBytes);
174
+ if (journalBytes <= 0) return size;
175
+ if (journalBytes >= size) {
176
+ // Full coverage: everything in the file is materialized per the
177
+ // checkpoint. Truncate to empty; dedupe for covered commandIds is the
178
+ // coordinator's in-memory ring's job now.
179
+ if (!replaceFile(file, Buffer.alloc(0), fs)) return fileSize(file, size, fs);
180
+ return 0;
181
+ }
182
+ // journalBytes < size means the file exists with an un-checkpointed tail.
183
+ const tail = readAllBytes(file, fs, journalBytes);
184
+ if (!replaceFile(file, tail, fs)) return fileSize(file, size, fs);
185
+ return fileSize(file, tail.length, fs);
186
+ }
187
+
188
+ /**
189
+ * Truncate a crash-torn journal tail so the next append stays parseable.
190
+ *
191
+ * `appendCommand` writes `JSON+\n` in a partial-write loop; a process killed
192
+ * mid-append leaves bytes without a trailing newline. Without repair, the next
193
+ * append concatenates onto that torn line and the merged line is unparseable — the
194
+ * freshly fsynced+acked record becomes invisible to every readJournal /
195
+ * boot-replay / revision scan. The repair is byte-exact and cheap: scan
196
+ * backwards for the last newline and drop everything after it (a journal with
197
+ * no newline at all is torn from byte 0 and truncates to empty). Complete but
198
+ * corrupt lines are kept — readJournal already skips them.
199
+ * @param {string} root
200
+ * @param {typeof defaultJournalFs} [fs]
201
+ * @returns {number} journal size in bytes after the call
202
+ */
203
+ export function repairJournalTail(root, fs = defaultJournalFs) {
204
+ const file = journalPath(root);
205
+ const size = fileSize(file, 0, fs);
206
+ if (size === 0) return 0;
207
+ const data = readAllBytes(file, fs);
208
+ const lastNewline = data.lastIndexOf("\n");
209
+ if (lastNewline === data.length - 1) return size;
210
+ const keep = lastNewline + 1;
211
+ if (!replaceFile(file, data.subarray(0, keep), fs)) return fileSize(file, size, fs);
212
+ return keep;
213
+ }
214
+
215
+ /**
216
+ * Read the file from `position` (default 0) to EOF via injectable readSync.
217
+ * @param {string} file
218
+ * @param {typeof defaultJournalFs} fs
219
+ * @param {number} [position]
220
+ * @returns {Buffer}
221
+ */
222
+ function readAllBytes(file, fs, position = 0) {
223
+ if (!fs.existsSync(file)) return Buffer.alloc(0);
224
+ let fd;
225
+ try {
226
+ fd = fs.openSync(file, "r");
227
+ const size = Math.max(0, fs.fstatSync(fd).size - position);
228
+ if (size === 0) return Buffer.alloc(0);
229
+ const output = Buffer.allocUnsafe(size);
230
+ let offset = 0;
231
+ while (offset < size) {
232
+ const read = fs.readSync(fd, output, offset, size - offset, position + offset);
233
+ if (read === 0) break;
234
+ offset += read;
235
+ }
236
+ return offset === size ? output : output.subarray(0, offset);
237
+ } finally {
238
+ if (fd !== undefined) {
239
+ try { fs.closeSync(fd); } catch { /* already closed */ }
240
+ }
241
+ }
242
+ }
243
+
244
+ /**
245
+ * Temp file + fsync + rename replacement (screen-log replaceScreenLog pattern).
246
+ * @param {string} file
247
+ * @param {Buffer} data
248
+ * @param {typeof defaultJournalFs} fs
249
+ * @returns {boolean}
250
+ */
251
+ function replaceFile(file, data, fs) {
252
+ const temp = `${file}.${process.pid}.${tempSequence++}.tmp`;
253
+ let fd;
254
+ try {
255
+ fs.mkdirSync(dirname(file), { recursive: true });
256
+ fd = fs.openSync(temp, "w");
257
+ let offset = 0;
258
+ while (offset < data.length) offset += fs.writeSync(fd, data, offset, data.length - offset, offset);
259
+ fs.fsyncSync(fd);
260
+ fs.closeSync(fd);
261
+ fd = undefined;
262
+ fs.renameSync(temp, file);
263
+ return true;
264
+ } catch {
265
+ if (fd !== undefined) {
266
+ try { fs.closeSync(fd); } catch { /* already closed */ }
267
+ }
268
+ try { fs.unlinkSync(temp); } catch { /* best effort */ }
269
+ return false;
270
+ }
271
+ }
272
+
273
+ /** @param {string} file @param {number} fallback @param {typeof defaultJournalFs} fs */
274
+ function fileSize(file, fallback, fs) {
275
+ try {
276
+ return fs.statSync(file).size;
277
+ } catch {
278
+ return fallback;
279
+ }
280
+ }
281
+
282
+ let tempSequence = 0;
@@ -0,0 +1,12 @@
1
+ /**
2
+ * Protocol version of the View State Coordinator control socket (issue #108).
3
+ *
4
+ * Bumped whenever a coordinator release adds command kinds or changes the
5
+ * envelope contract. A client built against version N refuses to talk to a
6
+ * coordinator reporting < N — a pong WITHOUT the field counts as version 1
7
+ * (the pre-#107 baseline) — and replaces the stale instance with a fresh one
8
+ * (SIGTERM via the coordinator lease pid, then respawn) instead of feeding
9
+ * it commands it cannot understand (`sync_foreground rejected (unknown_kind)`
10
+ * used to strand every foreground state write on extension updates).
11
+ */
12
+ export const COORDINATOR_PROTOCOL_VERSION = 2;
@@ -120,3 +120,18 @@ export function launchAutoState(root, config, opts) {
120
120
 
121
121
  return { pid: child.pid ?? null, configPath };
122
122
  }
123
+
124
+ /**
125
+ * Launch the detached view-state coordinator for a board root (issue #91, spec
126
+ * D3). No config file: the coordinator takes the root as its only argument.
127
+ * Idempotent by lease — a second instance loses the coordinator lease and
128
+ * exits silently, so callers may spawn freely on probe failure.
129
+ * @param {string} root
130
+ * @param {{ runnerScript: string, node?: string }} opts
131
+ * @returns {{ pid: number|null }}
132
+ */
133
+ export function launchCoordinator(root, opts) {
134
+ const node = opts.node ?? resolveNode();
135
+ const child = spawnDetached(node, [opts.runnerScript, root], root);
136
+ return { pid: child.pid ?? null };
137
+ }
@@ -89,6 +89,22 @@ export function hostEndpointPathFor(platform, root, viewId, instanceId) {
89
89
  }
90
90
  /** @param {string} root @param {string} viewId */
91
91
  export const screenLogPath = (root, viewId) => path.join(viewDir(root, viewId), "screen.log");
92
+ /**
93
+ * Well-known endpoint for the board-root View State Coordinator (issue #91, spec D3).
94
+ * Exactly one coordinator may own a root (token-fenced lease), so one stable path
95
+ * suffices; a stale POSIX socket left by a crashed coordinator is unlinked by the
96
+ * new lease owner before bind. win32 pipe names embed a 16-hex hash of the root to
97
+ * stay under the 256-char limit and keep per-root isolation.
98
+ * @param {"win32"|"linux"|"darwin"} platform
99
+ * @param {string} root
100
+ */
101
+ export function coordinatorEndpointPathFor(platform, root) {
102
+ if (platform === "win32") {
103
+ const hash = createHash("sha256").update(String(root)).digest("hex").slice(0, 16);
104
+ return `\\\\.\\pipe\\agent-board-coordinator-${hash}`;
105
+ }
106
+ return path.join(root, "coordinator.sock");
107
+ }
92
108
  /** @param {string} root @param {string} viewId */
93
109
  export const hostPidPath = (root, viewId) => path.join(viewDir(root, viewId), "host-pid.json");
94
110
  /** @param {string} root @param {string} viewId */