@zhuxixi/pi-agent-board 0.6.2 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,7 +22,9 @@ import { acquireOwnedViewLock } from "../src/core/locks.mjs";
22
22
  import * as P from "../src/core/paths.mjs";
23
23
  import { appendBoundedScreenLog, reconcileScreenLog } from "../src/core/screen-log.mjs";
24
24
  import { encodePromptForCliArg } from "../src/core/prompt-transport.mjs";
25
- import { readHost, readState, updateOwnedHost, writeHost, writeState } from "../src/core/store.mjs";
25
+ import { readHost, readState, updateOwnedHost, writeHost } from "../src/core/store.mjs";
26
+ import { sendStateCommand } from "../src/core/coordinator-client.mjs";
27
+ import { markRowFailedDirect } from "./pty-runner-legacy.mjs";
26
28
  import { ensureNodePtySpawnHelperExecutable } from "../src/core/pty-support.mjs";
27
29
 
28
30
  const requireForPty = createRequire(import.meta.url);
@@ -240,8 +242,12 @@ function legacyMain(config) {
240
242
  } catch (err) {
241
243
  const message = err instanceof Error ? err.message : String(err);
242
244
  update({ state: "failed", endedAt: Date.now(), exitCode: 1, error: message });
243
- markRowFailed(config.root, config.viewId, `PTY host failed: ${message}`);
244
- process.exit(1);
245
+ // The command settles within the client's own timeout (never throws), so
246
+ // the exit stays bounded while the fenced write gets its chance. Return
247
+ // instead of falling through: child is null here and the rest of this
248
+ // function assumes a spawned child.
249
+ void markRowFailed(config.root, config.viewId, `PTY host failed: ${message}`).finally(() => process.exit(1));
250
+ return;
245
251
  }
246
252
  childPid = child.pid ?? null;
247
253
  update({ childPid });
@@ -754,7 +760,7 @@ async function ownedMain(config) {
754
760
  } catch (err) {
755
761
  const message = err instanceof Error ? err.message : String(err);
756
762
  diag("child_spawn_failed", message);
757
- markRowFailed(config.root, config.viewId, `PTY host failed: ${message}`);
763
+ await markRowFailed(config.root, config.viewId, `PTY host failed: ${message}`);
758
764
  await finishHost("child_spawn_failed", 1);
759
765
  return;
760
766
  }
@@ -1018,34 +1024,66 @@ function captureStartToken(pid) {
1018
1024
  }
1019
1025
 
1020
1026
 
1021
- function markRowFailed(root, viewId, message) {
1022
- const now = Date.now();
1023
- const state = readState(root, viewId) ?? {
1024
- version: 1,
1027
+ /**
1028
+ * Finalize the view row as failed when the PTY host itself fails (child spawn
1029
+ * failure). Routed through the View State Coordinator as a fenced
1030
+ * `host_run_failed` command (issue #91, PR #1 residual risk #2): a manually
1031
+ * completed row rejects with manual_fence and stays completed, and a row
1032
+ * re-pointed to a newer run rejects with stale_run.
1033
+ *
1034
+ * The runId is sourced from the row's current state.json (null = the row had
1035
+ * no run) so the coordinator's stale_run guard stays effective; the
1036
+ * coordinator cannot enforce this from its side because it has no visibility
1037
+ * into what the host knows.
1038
+ *
1039
+ * Never throws: every client outcome is handled so the crash path cannot hang.
1040
+ * - applied → done (journaled + materialized by the coordinator).
1041
+ * - coordinator_disabled → legacy direct write (documented escape hatch).
1042
+ * - decided rejections (manual_fence / stale_run / no_change) → info
1043
+ * diagnostic; the coordinator's verdict governs.
1044
+ * - everything else (timeout / connection reset / unavailable) → warn
1045
+ * diagnostic: the outcome is unknown or the row could not be marked — the
1046
+ * runner still shuts down.
1047
+ * @param {string} root
1048
+ * @param {string} viewId
1049
+ * @param {string} message
1050
+ */
1051
+ async function markRowFailed(root, viewId, message) {
1052
+ const knownRunId = readState(root, viewId)?.currentRunId ?? null;
1053
+ const result = await sendStateCommand(root, {
1054
+ type: "state_command",
1025
1055
  viewId,
1026
- currentRunId: null,
1027
- semanticState: "queued",
1028
- processState: "exited",
1029
- summary: "Queued",
1030
- lastActivityAt: now,
1031
- updatedAt: now,
1032
- needsInput: false,
1033
- hasError: false,
1034
- latestAssistantPreview: "",
1035
- latestTool: null,
1036
- question: null,
1037
- pendingQuestions: [],
1038
- error: null,
1039
- };
1040
- state.semanticState = "failed";
1041
- state.processState = "exited";
1042
- state.summary = message;
1043
- state.hasError = true;
1044
- state.needsInput = false;
1045
- state.error = message;
1046
- state.updatedAt = now;
1047
- state.lastActivityAt = now;
1048
- writeState(root, state);
1056
+ runId: knownRunId,
1057
+ source: "pty-runner",
1058
+ kind: "host_run_failed",
1059
+ payload: { error: message },
1060
+ });
1061
+ if (result.status === "applied") return;
1062
+ if (result.reason === "coordinator_disabled") {
1063
+ markRowFailedDirect(root, viewId, message);
1064
+ return;
1065
+ }
1066
+ if (result.reason === "manual_fence" || result.reason === "stale_run" || result.reason === "no_change") {
1067
+ try {
1068
+ appendDiagnostic(root, viewId, {
1069
+ source: "runner",
1070
+ level: "info",
1071
+ code: "host_run_failed_skipped",
1072
+ message: `Host failure not applied to the row (${result.reason})`,
1073
+ details: { reason: result.reason },
1074
+ });
1075
+ } catch { /* best effort */ }
1076
+ return;
1077
+ }
1078
+ try {
1079
+ appendDiagnostic(root, viewId, {
1080
+ source: "runner",
1081
+ level: "warn",
1082
+ code: "host_run_failed_ambiguous",
1083
+ message: `Host failure outcome unknown (${result.reason}); the row may not reflect the failed host`,
1084
+ details: { reason: result.reason },
1085
+ });
1086
+ } catch { /* best effort */ }
1049
1087
  }
1050
1088
 
1051
1089
  function failEarly(message) {
@@ -0,0 +1,403 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Detached View State Coordinator (issue #91, spec D3) — the single logical
4
+ * writer of state.json/status.json for one board root.
5
+ *
6
+ * Usage: node state-coordinator.mjs <root>
7
+ *
8
+ * Design contract:
9
+ * - Exactly one instance per root: a token-fenced lease (acquireOwnedViewLock on
10
+ * the pseudo-view "_coordinator") is grabbed before anything else; a second
11
+ * instance exits 0 immediately ("busy" is an expected, idempotent outcome).
12
+ * - Every semantic mutation arrives as a `state_command` over the JSONL control
13
+ * socket and is decided by the pure layer (state-commands.mjs). This process
14
+ * owns ALL side effects: durable journal append (fsync BEFORE materialize, so
15
+ * boot replay can repair the crash window), materialization under a per-view
16
+ * file lock, checkpoint + journal GC, and socket lifecycle.
17
+ * - Transient kinds (TRANSIENT_KINDS, run_progress) bypass the durable
18
+ * machinery: validate → decide → materialize → reply, with no journal
19
+ * append, no dedupe, and no checkpoint trigger — a periodic snapshot whose
20
+ * next beat supersedes it. They still bump materializedRevision.
21
+ * - Idempotency: a processed commandId returns its original result forever —
22
+ * from the in-memory ring first (bounded, covers the post-GC window), then
23
+ * from the journal (findProcessedCommand). Duplicates never re-append.
24
+ * - Revisions: one global monotonic materializedRevision counter, initialized at
25
+ * boot to max(checkpoint, max journal revision, max revision across views'
26
+ * state.json), incremented once per applied command, and stamped identically
27
+ * on state.json and the command's run status.json (spec 根治条件 5 scope).
28
+ * - Boot replay re-materializes every applied journal record from its stored
29
+ * patches; per-file guards stamp only files strictly behind the recorded
30
+ * revision, so a half-materialized pair (state.json written, status.json not)
31
+ * repairs just the lagging half without moving the other one backwards. The
32
+ * decision-time status binding (statusRunId) is recorded in the journal, so
33
+ * replay is deterministic — no re-decision, no re-binding.
34
+ * - Legacy adoption: a view without materializedRevision is stamped as part of
35
+ * its first applied command's single materialization write (no extra write).
36
+ * - AGENT_BOARD_COORDINATOR=off makes the process exit 0 immediately (tests and
37
+ * the client's degradation path).
38
+ */
39
+ import { existsSync, readdirSync, readFileSync, statSync, unlinkSync } from "node:fs";
40
+ import { randomBytes } from "node:crypto";
41
+ import { createServer } from "node:net";
42
+ import { readJson } from "../src/core/atomic.mjs";
43
+ import {
44
+ appendCommand,
45
+ findProcessedCommand,
46
+ gcJournal,
47
+ readCheckpoint,
48
+ readJournal,
49
+ repairJournalTail,
50
+ writeCheckpoint,
51
+ } from "../src/core/coordinator-journal.mjs";
52
+ import { ownsEndpoint } from "../src/core/host-coordination.mjs";
53
+ import { acquireOwnedViewLock, withViewLockSync } from "../src/core/locks.mjs";
54
+ import * as P from "../src/core/paths.mjs";
55
+ import { readState, readStatus, writeState, writeStatus } from "../src/core/store.mjs";
56
+ import { TRANSIENT_KINDS, decideStateTransition, validateCommand } from "../src/core/state-commands.mjs";
57
+ import { COORDINATOR_PROTOCOL_VERSION } from "../src/core/coordinator-protocol.mjs";
58
+
59
+ /** In-memory processed-command ring size (FIFO). Beyond the journal, this covers
60
+ * idempotency when the journal prefix has already been GC'd away. */
61
+ const PROCESSED_RING_MAX = 1000;
62
+ /** Journal growth (bytes) that triggers checkpoint + GC. Keeps boot replay bounded. */
63
+ const CHECKPOINT_THRESHOLD_BYTES = 262_144;
64
+ /** Lease heartbeat — matches pty-runner's host-start lease cadence. */
65
+ const HEARTBEAT_MS = 1000;
66
+ /** Lifecycle kinds whose legacy write sites stamped state.lastActivityAt on
67
+ * every persist (service markQueued/archive/adopt, job-runner plan-ready,
68
+ * reconcile terminal sites). The decision layer stays clock-free, so the
69
+ * shell adds the wall-clock stamp — into the patch BEFORE the journal
70
+ * append, so boot replay re-applies the exact same fields deterministically
71
+ * (there is no clock at replay time either). (Task 1 review F5.) */
72
+ const LAST_ACTIVITY_STAMP_KINDS = new Set(["mark_queued", "archive_view", "adopt_session", "reconcile_finalize", "plan_ready", "mark_completed", "host_run_failed"]);
73
+ /** Kinds whose status patch may CREATE the run's status file: they carry the
74
+ * full status content for a run that has no materialized status yet. Every
75
+ * other kind keeps PR #1's "patch presence ≠ file requirement" semantics
76
+ * (e.g. mark_completed's `status: {autoState: null}` on a legacy row without
77
+ * a status file must not fabricate one). */
78
+ const STATUS_BOOTSTRAP_KINDS = new Set(["run_started", "followup_started"]);
79
+
80
+ const root = process.argv[2];
81
+ if (!root) {
82
+ process.stderr.write("state-coordinator: missing root argument\n");
83
+ process.exit(2);
84
+ }
85
+ if (process.env.AGENT_BOARD_COORDINATOR === "off") process.exit(0);
86
+
87
+ main().catch((err) => {
88
+ process.stderr.write(`state-coordinator: fatal ${err?.stack || err}\n`);
89
+ process.exit(1);
90
+ });
91
+
92
+ async function main() {
93
+ const instanceId = randomBytes(8).toString("hex");
94
+ const startedAt = Date.now();
95
+
96
+ // 1. Lease first: a second coordinator is an expected, idempotent no-op.
97
+ // Identity carries a POSIX startToken (pty-runner pattern) so a SIGKILLed
98
+ // coordinator's lease can be reclaimed by the next instance after pid reuse.
99
+ /** @type {import("../src/core/locks.mjs").Lease} */
100
+ let lease;
101
+ try {
102
+ lease = acquireOwnedViewLock(root, "_coordinator", "state-coordinator", {
103
+ waitMs: 500,
104
+ identity: { pid: process.pid, startToken: captureStartToken(process.pid) },
105
+ });
106
+ } catch (err) {
107
+ process.stderr.write(`state-coordinator: lease unavailable (${err?.code ?? err?.message ?? "unknown"}); another instance may own it; exiting\n`);
108
+ process.exit(0);
109
+ }
110
+ const startTouchTimer = setInterval(() => {
111
+ try { lease?.touch(); } catch { /* best effort */ }
112
+ }, HEARTBEAT_MS);
113
+ startTouchTimer.unref?.();
114
+
115
+ // 2. Boot state: journal summary, processed-command ring, replay, counter init.
116
+ // Repair a crash-torn tail FIRST: without it the first post-restart append
117
+ // concatenates onto the torn line and that fsynced record becomes invisible
118
+ // to every readJournal / replay / revision scan below.
119
+ repairJournalTail(root);
120
+ const checkpoint = readCheckpoint(root) ?? { materializedRevision: 0, journalBytes: 0 };
121
+ const journal = readJournal(root);
122
+ /** @type {Map<string, { result: { status: string, reason: string|null }, materializedRevision: number }>} */
123
+ const processedCommands = new Map();
124
+ for (const record of journal) {
125
+ const commandId = record?.command?.commandId;
126
+ if (commandId && !processedCommands.has(commandId)) {
127
+ processedCommands.set(commandId, {
128
+ result: record.result ?? { status: "rejected", reason: "unknown" },
129
+ materializedRevision: record.materializedRevision ?? 0,
130
+ });
131
+ }
132
+ }
133
+
134
+ // Replay in journal order: re-materialize every applied record from its stored
135
+ // patches; materialize's per-file guards write only files strictly behind the
136
+ // recorded revision, so this repairs just the half that missed its write in a
137
+ // crash window and never moves the other half backwards.
138
+ for (const record of journal) {
139
+ if (record?.result?.status !== "applied") continue;
140
+ const viewId = record?.command?.viewId;
141
+ if (!viewId) continue;
142
+ materialize(viewId, record.command ?? {}, record.mutate ?? {}, record.materializedRevision ?? 0, record.statusRunId ?? null);
143
+ }
144
+
145
+ // Global revision counter: never reissue a revision that exists anywhere.
146
+ let revisionCounter = Math.max(checkpoint.materializedRevision ?? 0, 0);
147
+ for (const record of journal) revisionCounter = Math.max(revisionCounter, record.materializedRevision ?? 0);
148
+ if (existsSync(P.viewsDir(root))) {
149
+ for (const entry of readdirSync(P.viewsDir(root))) {
150
+ const state = readJson(P.statePath(root, entry), null);
151
+ revisionCounter = Math.max(revisionCounter, state?.materializedRevision ?? 0);
152
+ }
153
+ }
154
+ let lastCheckpointBytes = 0;
155
+
156
+ // 3. Control socket (JSONL, pty-runner's line-buffered server shape).
157
+ const socketPath = P.coordinatorEndpointPathFor(process.platform, root);
158
+ if (process.platform !== "win32" && existsSync(socketPath)) {
159
+ // Lease is ours: any leftover socket file belongs to a dead coordinator.
160
+ try { unlinkSync(socketPath); } catch { /* best effort */ }
161
+ }
162
+ /** {dev,ino} recorded at bind time; cleanup unlinks only this exact inode. */
163
+ let boundSocketIdentity = null;
164
+
165
+ const server = createServer((socket) => {
166
+ let buffer = "";
167
+ socket.on("data", (chunk) => {
168
+ buffer += chunk.toString("utf8");
169
+ const lines = buffer.split("\n");
170
+ buffer = lines.pop() ?? "";
171
+ for (const line of lines) handleClientLine(line, socket);
172
+ });
173
+ socket.on("error", () => {}); // client vanished mid-request; nothing to answer
174
+ });
175
+
176
+ server.on("error", (err) => {
177
+ process.stderr.write(`state-coordinator: server error ${err instanceof Error ? err.message : String(err)}\n`);
178
+ shutdown(1);
179
+ });
180
+
181
+ server.listen(socketPath, () => {
182
+ if (process.platform !== "win32") {
183
+ try {
184
+ const st = statSync(socketPath);
185
+ boundSocketIdentity = { dev: st.dev, ino: st.ino };
186
+ } catch { /* without the identity, cleanup degrades to no-op */ }
187
+ }
188
+ });
189
+
190
+ let shuttingDown = false;
191
+ process.on("SIGTERM", () => shutdown(0));
192
+ process.on("SIGINT", () => shutdown(0));
193
+
194
+ function shutdown(code) {
195
+ if (shuttingDown) return;
196
+ shuttingDown = true;
197
+ try { clearInterval(startTouchTimer); } catch { /* best effort */ }
198
+ try { server.close(); } catch { /* already closed */ }
199
+ if (process.platform !== "win32" && boundSocketIdentity) {
200
+ try {
201
+ const st = statSync(socketPath);
202
+ if (ownsEndpoint(boundSocketIdentity, { dev: st.dev, ino: st.ino })) {
203
+ try { unlinkSync(socketPath); } catch { /* best effort */ }
204
+ }
205
+ } catch { /* path gone — nothing to clean */ }
206
+ }
207
+ try { lease.release(); } catch { /* best effort */ }
208
+ process.exit(code);
209
+ }
210
+
211
+ function send(socket, msg) {
212
+ socket.write(JSON.stringify(msg) + "\n");
213
+ }
214
+
215
+ function handleClientLine(line, socket) {
216
+ if (!line.trim()) return;
217
+ let msg;
218
+ try { msg = JSON.parse(line); } catch { return send(socket, { type: "error", message: "invalid json" }); }
219
+ switch (msg?.type) {
220
+ case "ping":
221
+ send(socket, { type: "pong", instanceId, startedAt, protocolVersion: COORDINATOR_PROTOCOL_VERSION });
222
+ break;
223
+ case "state_command":
224
+ send(socket, { type: "state_command_result", ...processStateCommand(msg) });
225
+ break;
226
+ default:
227
+ send(socket, { type: "error", message: "unknown type" });
228
+ }
229
+ }
230
+
231
+ /**
232
+ * The command loop: validate → dedupe → decide → journal (fsync) → materialize
233
+ * → remember → reply. Rejections are journaled too, so their exact reason is
234
+ * replayable; duplicates short-circuit before any append.
235
+ * @param {object} msg
236
+ * @returns {{ commandId: string|null, status: string, reason: string|null, materializedRevision: number }}
237
+ */
238
+ function processStateCommand(msg) {
239
+ const checked = validateCommand(msg);
240
+ if (!checked.ok) {
241
+ return {
242
+ commandId: typeof msg?.commandId === "string" ? msg.commandId : null,
243
+ status: "rejected",
244
+ reason: checked.error,
245
+ materializedRevision: 0,
246
+ };
247
+ }
248
+ const command = checked.command;
249
+ const transient = TRANSIENT_KINDS.includes(command.kind);
250
+
251
+ if (!transient) {
252
+ const known = lookupProcessed(command.commandId);
253
+ if (known) {
254
+ return { commandId: command.commandId, status: known.result.status, reason: known.result.reason, materializedRevision: known.materializedRevision };
255
+ }
256
+ }
257
+
258
+ const state = readState(root, command.viewId);
259
+ // Status consistency only binds the view's current run (spec 根治条件 5):
260
+ // no currentRunId → status.json plays no role in this command.
261
+ // Exception (Task 1 review F1): followup_started re-points the row at a
262
+ // NEW run carried in payload.newRunId (command.runId is deliberately null
263
+ // so the stale-run guard cannot fire against the finished parent run).
264
+ // Resolving the status from the parent here would overwrite the parent's
265
+ // status.json with the new run's bootstrap patch; resolving from newRunId
266
+ // finds no file and cleanly bootstraps the new run's status instead.
267
+ const statusRunId = command.kind === "followup_started"
268
+ ? (typeof command.payload?.newRunId === "string" ? command.payload.newRunId : null)
269
+ : (command.runId ?? state?.currentRunId ?? null);
270
+ const status = statusRunId ? readStatus(root, command.viewId, statusRunId) : null;
271
+ const now = Date.now();
272
+ const decision = stampLegacyTimestamps(command, decideStateTransition(command, state, status, now), now);
273
+ const currentRevision = state?.materializedRevision ?? 0;
274
+
275
+ if (decision.action === "reject") {
276
+ if (transient) {
277
+ // Transient rejections are not replayable either — nothing was
278
+ // mutated, and journaling them would grow the journal for noise.
279
+ return { commandId: command.commandId ?? null, status: "rejected", reason: decision.reason, materializedRevision: currentRevision };
280
+ }
281
+ const result = { status: "rejected", reason: decision.reason };
282
+ const record = { command, result, materializedRevision: currentRevision, at: now };
283
+ const journalBytes = appendCommand(root, record);
284
+ rememberProcessed(command.commandId, result, currentRevision);
285
+ maybeCheckpoint(journalBytes);
286
+ return { commandId: command.commandId, status: "rejected", reason: decision.reason, materializedRevision: currentRevision };
287
+ }
288
+
289
+ const newRevision = revisionCounter + 1;
290
+ const result = { status: "applied", reason: decision.reason };
291
+ if (transient) {
292
+ // No journal, no processed ring, no checkpoint: a periodic snapshot —
293
+ // the next beat supersedes it.
294
+ materialize(command.viewId, command, decision.mutate, newRevision, statusRunId);
295
+ revisionCounter = newRevision;
296
+ return { commandId: command.commandId ?? null, status: "applied", reason: decision.reason, materializedRevision: newRevision };
297
+ }
298
+ // Journal first (fsync inside), materialize second: boot replay repairs
299
+ // the window between the two using the stored patches.
300
+ const record = { command, result, materializedRevision: newRevision, at: now, mutate: decision.mutate, statusRunId };
301
+ const journalBytes = appendCommand(root, record);
302
+ materialize(command.viewId, command, decision.mutate, newRevision, statusRunId);
303
+ revisionCounter = newRevision;
304
+ rememberProcessed(command.commandId, result, newRevision);
305
+ maybeCheckpoint(journalBytes);
306
+ return { commandId: command.commandId, status: "applied", reason: decision.reason, materializedRevision: newRevision };
307
+ }
308
+
309
+ /**
310
+ * Merge patches onto state.json (and the run's status.json when it exists)
311
+ * under the view's materialize lock, stamping the shared revision. Per-file
312
+ * guards stamp only files strictly behind the record revision: the live path
313
+ * always qualifies (the global counter is monotonic), while replay repairs
314
+ * just the half that missed its write in a crash window and never moves the
315
+ * other half backwards. A status patch without a status file is skipped —
316
+ * patch presence ≠ file requirement (mark_completed always emits
317
+ * `status: {autoState: null}`). The status run binding is the decision-time
318
+ * one (`statusRunId`, recorded in the journal) so replay cannot re-bind a
319
+ * runId-less patch to whatever run is current on disk at replay time;
320
+ * `state.currentRunId` remains only as a legacy fallback for records written
321
+ * before statusRunId existed.
322
+ */
323
+ function materialize(viewId, command, mutate, revision, statusRunId) {
324
+ withViewLockSync(root, viewId, "state-materialize", () => {
325
+ const state = readState(root, viewId);
326
+ if (mutate?.state && state && (state.materializedRevision ?? 0) < revision) {
327
+ writeState(root, { ...state, ...mutate.state, materializedRevision: revision });
328
+ }
329
+ const runId = command?.runId ?? statusRunId ?? state?.currentRunId ?? null;
330
+ if (mutate?.status && runId) {
331
+ const status = readStatus(root, viewId, runId);
332
+ if (!status && STATUS_BOOTSTRAP_KINDS.has(command?.kind)) {
333
+ // run_started / followup_started carry the full status content for a
334
+ // run that has no status file yet — create it (F1's clean-bootstrap
335
+ // half). Identity fields come from the command context; the patch
336
+ // carries everything meaningful.
337
+ writeStatus(root, { version: 1, runId, viewId, ...mutate.status, materializedRevision: revision });
338
+ } else if (status && (status.materializedRevision ?? 0) < revision) {
339
+ writeStatus(root, { ...status, ...mutate.status, materializedRevision: revision });
340
+ }
341
+ }
342
+ });
343
+ }
344
+
345
+ /** Add the legacy wall-clock stamp to a decided patch (F5): see
346
+ * LAST_ACTIVITY_STAMP_KINDS. Applied before the journal append so the
347
+ * stored patches are exactly what replay re-materializes. */
348
+ function stampLegacyTimestamps(command, decision, now) {
349
+ if (decision.action !== "apply" || !LAST_ACTIVITY_STAMP_KINDS.has(command.kind)) return decision;
350
+ return { ...decision, mutate: { ...decision.mutate, state: { ...(decision.mutate.state ?? {}), lastActivityAt: now } } };
351
+ }
352
+
353
+ /** Memory ring first (covers the post-GC window), journal second. */
354
+ function lookupProcessed(commandId) {
355
+ const inMemory = processedCommands.get(commandId);
356
+ if (inMemory) return inMemory;
357
+ const result = findProcessedCommand(root, commandId);
358
+ if (!result) return null;
359
+ const entry = { result, materializedRevision: revisionOf(commandId) };
360
+ rememberProcessed(commandId, result, entry.materializedRevision);
361
+ return entry;
362
+ }
363
+
364
+ /** Revision recorded for a known commandId (journal scan; 0 if unrecorded). */
365
+ function revisionOf(commandId) {
366
+ for (const record of readJournal(root)) {
367
+ if (record?.command?.commandId === commandId) return record.materializedRevision ?? 0;
368
+ }
369
+ return 0;
370
+ }
371
+
372
+ function rememberProcessed(commandId, result, materializedRevision) {
373
+ processedCommands.set(commandId, { result, materializedRevision });
374
+ while (processedCommands.size > PROCESSED_RING_MAX) {
375
+ processedCommands.delete(processedCommands.keys().next().value);
376
+ }
377
+ }
378
+
379
+ /** Checkpoint + GC once the journal outgrows the threshold. GC only runs when
380
+ * the checkpoint write provably succeeded (journal module enforces this). */
381
+ function maybeCheckpoint(journalBytes) {
382
+ if (journalBytes - lastCheckpointBytes < CHECKPOINT_THRESHOLD_BYTES) return;
383
+ if (!writeCheckpoint(root, { materializedRevision: revisionCounter, journalBytes })) return;
384
+ lastCheckpointBytes = gcJournal(root);
385
+ }
386
+ }
387
+
388
+ /** POSIX process start token — /proc/<pid>/stat field 22 (starttime), stable
389
+ * across exec(2). Same recipe as pty-runner: lets locks.mjs reclaim this
390
+ * coordinator's lease after a SIGKILL even if the pid was reused. null on
391
+ * failure or non-Linux platforms (reclaim degrades to "blocked" there — same
392
+ * platform parity as PTY hosts). */
393
+ function captureStartToken(pid) {
394
+ if (process.platform !== "linux" || !pid) return null;
395
+ try {
396
+ const stat = readFileSync(`/proc/${pid}/stat`, "utf8");
397
+ const afterComm = stat.slice(stat.lastIndexOf(")") + 1).trimStart();
398
+ const fields = afterComm.split(/\s+/);
399
+ return fields[19] ?? null;
400
+ } catch {
401
+ return null;
402
+ }
403
+ }
@@ -9,10 +9,11 @@
9
9
  import { spawn } from "node:child_process";
10
10
  import { readJson } from "../src/core/atomic.mjs";
11
11
  import { appendDiagnostic } from "../src/core/diagnostics.mjs";
12
- import { applyAutoStateToStatus, applyAutoStateToViewState, autoStateEnabled, autoStateFromModelOrHeuristic, autoStateModel, buildAutoStatePrompt, heuristicAutoState } from "../src/core/auto-state.mjs";
12
+ import { applyAutoStateToStatus, applyAutoStateToViewState, autoStateEnabled, autoStateFromModelOrHeuristic, autoStateModel, buildAutoStatePrompt, heuristicAutoState, isManualCompletion } from "../src/core/auto-state.mjs";
13
13
  import { finalizeEvidence, readEvidence, summarizeEvidence, writeEvidence } from "../src/core/evidence.mjs";
14
14
  import { updateCodeRefsFromEvidence } from "../src/core/code-refs-store.mjs";
15
15
  import { readState, readStatus, readMeta, writeState, writeStatus } from "../src/core/store.mjs";
16
+ import { sendStateCommand } from "../src/core/coordinator-client.mjs";
16
17
 
17
18
  async function main() {
18
19
  const configPath = process.argv[2];
@@ -26,6 +27,10 @@ async function main() {
26
27
 
27
28
  const state = readState(config.root, config.viewId);
28
29
  if (!state || state.processState === "alive" || state.semanticState === "failed" || state.semanticState === "stopped") process.exit(0);
30
+ // Cheap pre-check (optimization only): skip a pointless command when the
31
+ // manual verdict is already materialized. The coordinator's manual_fence
32
+ // stays the authoritative guard for races after this read.
33
+ if (isManualCompletion(state)) process.exit(0);
29
34
 
30
35
  const evidence = readEvidence(config.root, config.viewId);
31
36
  const latest = latestEvidenceText(evidence) || state.latestAssistantPreview || state.summary || "";
@@ -45,24 +50,93 @@ async function main() {
45
50
  classification = autoStateFromModelOrHeuristic(out, latest, { lastAgentActivityAt: state.lastAgentActivityAt ?? null });
46
51
  }
47
52
 
48
- let changed = false;
49
- if (config.runId) {
50
- const status = readStatus(config.root, config.viewId, config.runId);
51
- if (status) {
52
- changed = applyAutoStateToStatus(status, classification, Date.now()) || changed;
53
- status.evidenceSummary = summarizeEvidence(finalizeEvidence(evidence, status, Date.now()));
54
- writeStatus(config.root, status);
53
+ // Issue #91 (A8 path 2): the classification lands through the View State
54
+ // Coordinator — the single writer of state.json/status.json. A manual
55
+ // completion is fenced by the coordinator (manual_fence), so this late pass
56
+ // can no longer clobber the user's verdict (#46 class). Ambiguous transport
57
+ // outcomes (timeout / connection_reset) NEVER fall back to a direct write:
58
+ // the command may already be journaled, and the coordinator's boot replay
59
+ // is the recovery path.
60
+ const result = await sendStateCommand(config.root, {
61
+ type: "state_command",
62
+ viewId: config.viewId,
63
+ runId: config.runId ?? null,
64
+ source: "state-runner",
65
+ kind: "auto_state_classified",
66
+ expectedRevision: null,
67
+ payload: { classification },
68
+ });
69
+
70
+ if (result.status === "applied") {
71
+ appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated row state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
72
+ } else if (result.reason === "manual_fence" || result.reason === "no_change" || result.reason === "stale_run") {
73
+ // Designed fences — informational, not errors: the coordinator is the
74
+ // authority and a manual completion wins by design.
75
+ appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "auto_state_classified_skipped", message: `Auto-state classification not applied (${result.reason})`, details: { reason: result.reason } });
76
+ } else if (result.reason !== "coordinator_disabled") {
77
+ appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, level: "warn", code: "auto_state_command_ambiguous", message: `Auto-state classification outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise the next classification pass will converge the row`, details: { reason: result.reason } });
78
+ }
79
+
80
+ if (result.reason === "coordinator_disabled") {
81
+ // Legacy escape hatch (AGENT_BOARD_COORDINATOR=off): pre-coordinator
82
+ // direct-write behavior, unchanged.
83
+ let changed = false;
84
+ if (config.runId) {
85
+ const status = readStatus(config.root, config.viewId, config.runId);
86
+ if (status) {
87
+ changed = applyAutoStateToStatus(status, classification, Date.now()) || changed;
88
+ status.evidenceSummary = summarizeEvidence(finalizeEvidence(evidence, status, Date.now()));
89
+ writeStatus(config.root, status);
90
+ }
55
91
  }
92
+ const latestState = readState(config.root, config.viewId) ?? state;
93
+ changed = applyAutoStateToViewState(latestState, classification, Date.now()) || changed;
94
+ finalizeEvidence(evidence, { semanticState: latestState.semanticState, usage: null }, Date.now());
95
+ latestState.review = summarizeEvidence(evidence);
96
+ writeEvidence(config.root, evidence);
97
+ updateCodeRefsFromEvidence(config.root, config.viewId, evidence, meta);
98
+ writeState(config.root, latestState);
99
+ if (changed) {
100
+ appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated row state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
101
+ }
102
+ process.exit(0);
56
103
  }
57
- const latestState = readState(config.root, config.viewId) ?? state;
58
- changed = applyAutoStateToViewState(latestState, classification, Date.now()) || changed;
59
- finalizeEvidence(evidence, { semanticState: latestState.semanticState, usage: null }, Date.now());
60
- latestState.review = summarizeEvidence(evidence);
104
+
105
+ // Evidence pipeline (coordinator-independent) stays direct: finalize the
106
+ // evidence snapshot from a FRESH state read (reads are not writes — the
107
+ // coordinator owns state.json/status.json writes, not reads), then persist
108
+ // the evidence artifacts and code-refs.
109
+ const postState = readState(config.root, config.viewId);
110
+ finalizeEvidence(evidence, { semanticState: postState?.semanticState ?? state.semanticState, usage: null }, Date.now());
61
111
  writeEvidence(config.root, evidence);
62
112
  updateCodeRefsFromEvidence(config.root, config.viewId, evidence, meta);
63
- writeState(config.root, latestState);
64
- if (changed) {
65
- appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated row state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
113
+
114
+ // Issue #91 PR #2 (Task 4): the evidence mirrors move behind the coordinator
115
+ // too — one patch_fields command materializes review/evidenceSummary on both
116
+ // files under a shared materializedRevision (previously two separately
117
+ // fenced direct writes). The coordinator's generic manual_fence / stale_run
118
+ // guards are authoritative; ambiguous outcomes (timeout / connection_reset)
119
+ // NEVER fall back to a direct write — if the command was journaled, boot
120
+ // replay recovers it, and the next classification pass re-derives mirrors
121
+ // from the evidence files either way.
122
+ const mirrorSummary = summarizeEvidence(evidence);
123
+ const patch = await sendStateCommand(config.root, {
124
+ type: "state_command",
125
+ viewId: config.viewId,
126
+ runId: config.runId ?? null,
127
+ source: "state-runner",
128
+ kind: "patch_fields",
129
+ expectedRevision: null,
130
+ payload: { state: { review: mirrorSummary }, status: { evidenceSummary: mirrorSummary } },
131
+ });
132
+ if (patch.status === "applied") {
133
+ // Quiet success — mirrors materialized by the coordinator.
134
+ } else if (patch.reason === "manual_fence" || patch.reason === "stale_run" || patch.reason === "no_change") {
135
+ // Designed fences — informational: a manual verdict or a newer run owns
136
+ // the row, or the mirrors already match.
137
+ appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "evidence_mirror_patch_skipped", message: `Evidence mirror patch not applied (${patch.reason})`, details: { reason: patch.reason } });
138
+ } else {
139
+ appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, level: "warn", code: "evidence_mirror_patch_ambiguous", message: `Evidence mirror patch outcome unknown (${patch.reason}); if the command was journaled, coordinator replay will recover it; the next classification pass re-derives the mirrors from the evidence files`, details: { reason: patch.reason } });
66
140
  }
67
141
  }
68
142