@zhuxixi/pi-agent-board 0.6.2 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/docs/superpowers/plans/2026-09-08-issue-11-attach-runtime-desync-heal.md +917 -0
- package/docs/superpowers/plans/2026-09-09-harden-runner-architecture.md +603 -0
- package/docs/superpowers/plans/2026-09-09-single-writer-completion.md +252 -0
- package/docs/superpowers/specs/2026-09-07-issue-11-attach-runtime-desync-heal-design.md +130 -0
- package/docs/superpowers/specs/2026-09-09-harden-runner-architecture-design.md +298 -0
- package/package.json +1 -1
- package/runner/job-runner-legacy.mjs +68 -0
- package/runner/job-runner.mjs +370 -67
- package/runner/pty-runner-legacy.mjs +50 -0
- package/runner/pty-runner.mjs +69 -31
- package/runner/state-coordinator.mjs +403 -0
- package/runner/state-runner.mjs +89 -15
- package/src/commands/bg.ts +2 -1
- package/src/core/coordinator-client.mjs +313 -0
- package/src/core/coordinator-journal.mjs +282 -0
- package/src/core/coordinator-protocol.mjs +12 -0
- package/src/core/launch.mjs +15 -0
- package/src/core/paths.mjs +16 -0
- package/src/core/pty-attach-jiggle-controller.mjs +57 -4
- package/src/core/pty-attach-render.mjs +30 -0
- package/src/core/state-commands.mjs +617 -0
- package/src/core/types.mjs +2 -0
- package/src/runtime/service.mjs +416 -111
- package/src/ui/dashboard.ts +77 -110
- package/src/ui/pty-attach.ts +63 -1
package/runner/pty-runner.mjs
CHANGED
|
@@ -22,7 +22,9 @@ import { acquireOwnedViewLock } from "../src/core/locks.mjs";
|
|
|
22
22
|
import * as P from "../src/core/paths.mjs";
|
|
23
23
|
import { appendBoundedScreenLog, reconcileScreenLog } from "../src/core/screen-log.mjs";
|
|
24
24
|
import { encodePromptForCliArg } from "../src/core/prompt-transport.mjs";
|
|
25
|
-
import { readHost, readState, updateOwnedHost, writeHost
|
|
25
|
+
import { readHost, readState, updateOwnedHost, writeHost } from "../src/core/store.mjs";
|
|
26
|
+
import { sendStateCommand } from "../src/core/coordinator-client.mjs";
|
|
27
|
+
import { markRowFailedDirect } from "./pty-runner-legacy.mjs";
|
|
26
28
|
import { ensureNodePtySpawnHelperExecutable } from "../src/core/pty-support.mjs";
|
|
27
29
|
|
|
28
30
|
const requireForPty = createRequire(import.meta.url);
|
|
@@ -240,8 +242,12 @@ function legacyMain(config) {
|
|
|
240
242
|
} catch (err) {
|
|
241
243
|
const message = err instanceof Error ? err.message : String(err);
|
|
242
244
|
update({ state: "failed", endedAt: Date.now(), exitCode: 1, error: message });
|
|
243
|
-
|
|
244
|
-
|
|
245
|
+
// The command settles within the client's own timeout (never throws), so
|
|
246
|
+
// the exit stays bounded while the fenced write gets its chance. Return
|
|
247
|
+
// instead of falling through: child is null here and the rest of this
|
|
248
|
+
// function assumes a spawned child.
|
|
249
|
+
void markRowFailed(config.root, config.viewId, `PTY host failed: ${message}`).finally(() => process.exit(1));
|
|
250
|
+
return;
|
|
245
251
|
}
|
|
246
252
|
childPid = child.pid ?? null;
|
|
247
253
|
update({ childPid });
|
|
@@ -754,7 +760,7 @@ async function ownedMain(config) {
|
|
|
754
760
|
} catch (err) {
|
|
755
761
|
const message = err instanceof Error ? err.message : String(err);
|
|
756
762
|
diag("child_spawn_failed", message);
|
|
757
|
-
markRowFailed(config.root, config.viewId, `PTY host failed: ${message}`);
|
|
763
|
+
await markRowFailed(config.root, config.viewId, `PTY host failed: ${message}`);
|
|
758
764
|
await finishHost("child_spawn_failed", 1);
|
|
759
765
|
return;
|
|
760
766
|
}
|
|
@@ -1018,34 +1024,66 @@ function captureStartToken(pid) {
|
|
|
1018
1024
|
}
|
|
1019
1025
|
|
|
1020
1026
|
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1027
|
+
/**
|
|
1028
|
+
* Finalize the view row as failed when the PTY host itself fails (child spawn
|
|
1029
|
+
* failure). Routed through the View State Coordinator as a fenced
|
|
1030
|
+
* `host_run_failed` command (issue #91, PR #1 residual risk #2): a manually
|
|
1031
|
+
* completed row rejects with manual_fence and stays completed, and a row
|
|
1032
|
+
* re-pointed to a newer run rejects with stale_run.
|
|
1033
|
+
*
|
|
1034
|
+
* The runId is sourced from the row's current state.json (null = the row had
|
|
1035
|
+
* no run) so the coordinator's stale_run guard stays effective; the
|
|
1036
|
+
* coordinator cannot enforce this from its side because it has no visibility
|
|
1037
|
+
* into what the host knows.
|
|
1038
|
+
*
|
|
1039
|
+
* Never throws: every client outcome is handled so the crash path cannot hang.
|
|
1040
|
+
* - applied → done (journaled + materialized by the coordinator).
|
|
1041
|
+
* - coordinator_disabled → legacy direct write (documented escape hatch).
|
|
1042
|
+
* - decided rejections (manual_fence / stale_run / no_change) → info
|
|
1043
|
+
* diagnostic; the coordinator's verdict governs.
|
|
1044
|
+
* - everything else (timeout / connection reset / unavailable) → warn
|
|
1045
|
+
* diagnostic: the outcome is unknown or the row could not be marked — the
|
|
1046
|
+
* runner still shuts down.
|
|
1047
|
+
* @param {string} root
|
|
1048
|
+
* @param {string} viewId
|
|
1049
|
+
* @param {string} message
|
|
1050
|
+
*/
|
|
1051
|
+
async function markRowFailed(root, viewId, message) {
|
|
1052
|
+
const knownRunId = readState(root, viewId)?.currentRunId ?? null;
|
|
1053
|
+
const result = await sendStateCommand(root, {
|
|
1054
|
+
type: "state_command",
|
|
1025
1055
|
viewId,
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1056
|
+
runId: knownRunId,
|
|
1057
|
+
source: "pty-runner",
|
|
1058
|
+
kind: "host_run_failed",
|
|
1059
|
+
payload: { error: message },
|
|
1060
|
+
});
|
|
1061
|
+
if (result.status === "applied") return;
|
|
1062
|
+
if (result.reason === "coordinator_disabled") {
|
|
1063
|
+
markRowFailedDirect(root, viewId, message);
|
|
1064
|
+
return;
|
|
1065
|
+
}
|
|
1066
|
+
if (result.reason === "manual_fence" || result.reason === "stale_run" || result.reason === "no_change") {
|
|
1067
|
+
try {
|
|
1068
|
+
appendDiagnostic(root, viewId, {
|
|
1069
|
+
source: "runner",
|
|
1070
|
+
level: "info",
|
|
1071
|
+
code: "host_run_failed_skipped",
|
|
1072
|
+
message: `Host failure not applied to the row (${result.reason})`,
|
|
1073
|
+
details: { reason: result.reason },
|
|
1074
|
+
});
|
|
1075
|
+
} catch { /* best effort */ }
|
|
1076
|
+
return;
|
|
1077
|
+
}
|
|
1078
|
+
try {
|
|
1079
|
+
appendDiagnostic(root, viewId, {
|
|
1080
|
+
source: "runner",
|
|
1081
|
+
level: "warn",
|
|
1082
|
+
code: "host_run_failed_ambiguous",
|
|
1083
|
+
message: `Host failure outcome unknown (${result.reason}); the row may not reflect the failed host`,
|
|
1084
|
+
details: { reason: result.reason },
|
|
1085
|
+
});
|
|
1086
|
+
} catch { /* best effort */ }
|
|
1049
1087
|
}
|
|
1050
1088
|
|
|
1051
1089
|
function failEarly(message) {
|
|
@@ -0,0 +1,403 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Detached View State Coordinator (issue #91, spec D3) — the single logical
|
|
4
|
+
* writer of state.json/status.json for one board root.
|
|
5
|
+
*
|
|
6
|
+
* Usage: node state-coordinator.mjs <root>
|
|
7
|
+
*
|
|
8
|
+
* Design contract:
|
|
9
|
+
* - Exactly one instance per root: a token-fenced lease (acquireOwnedViewLock on
|
|
10
|
+
* the pseudo-view "_coordinator") is grabbed before anything else; a second
|
|
11
|
+
* instance exits 0 immediately ("busy" is an expected, idempotent outcome).
|
|
12
|
+
* - Every semantic mutation arrives as a `state_command` over the JSONL control
|
|
13
|
+
* socket and is decided by the pure layer (state-commands.mjs). This process
|
|
14
|
+
* owns ALL side effects: durable journal append (fsync BEFORE materialize, so
|
|
15
|
+
* boot replay can repair the crash window), materialization under a per-view
|
|
16
|
+
* file lock, checkpoint + journal GC, and socket lifecycle.
|
|
17
|
+
* - Transient kinds (TRANSIENT_KINDS, run_progress) bypass the durable
|
|
18
|
+
* machinery: validate → decide → materialize → reply, with no journal
|
|
19
|
+
* append, no dedupe, and no checkpoint trigger — a periodic snapshot whose
|
|
20
|
+
* next beat supersedes it. They still bump materializedRevision.
|
|
21
|
+
* - Idempotency: a processed commandId returns its original result forever —
|
|
22
|
+
* from the in-memory ring first (bounded, covers the post-GC window), then
|
|
23
|
+
* from the journal (findProcessedCommand). Duplicates never re-append.
|
|
24
|
+
* - Revisions: one global monotonic materializedRevision counter, initialized at
|
|
25
|
+
* boot to max(checkpoint, max journal revision, max revision across views'
|
|
26
|
+
* state.json), incremented once per applied command, and stamped identically
|
|
27
|
+
* on state.json and the command's run status.json (spec 根治条件 5 scope).
|
|
28
|
+
* - Boot replay re-materializes every applied journal record from its stored
|
|
29
|
+
* patches; per-file guards stamp only files strictly behind the recorded
|
|
30
|
+
* revision, so a half-materialized pair (state.json written, status.json not)
|
|
31
|
+
* repairs just the lagging half without moving the other one backwards. The
|
|
32
|
+
* decision-time status binding (statusRunId) is recorded in the journal, so
|
|
33
|
+
* replay is deterministic — no re-decision, no re-binding.
|
|
34
|
+
* - Legacy adoption: a view without materializedRevision is stamped as part of
|
|
35
|
+
* its first applied command's single materialization write (no extra write).
|
|
36
|
+
* - AGENT_BOARD_COORDINATOR=off makes the process exit 0 immediately (tests and
|
|
37
|
+
* the client's degradation path).
|
|
38
|
+
*/
|
|
39
|
+
import { existsSync, readdirSync, readFileSync, statSync, unlinkSync } from "node:fs";
|
|
40
|
+
import { randomBytes } from "node:crypto";
|
|
41
|
+
import { createServer } from "node:net";
|
|
42
|
+
import { readJson } from "../src/core/atomic.mjs";
|
|
43
|
+
import {
|
|
44
|
+
appendCommand,
|
|
45
|
+
findProcessedCommand,
|
|
46
|
+
gcJournal,
|
|
47
|
+
readCheckpoint,
|
|
48
|
+
readJournal,
|
|
49
|
+
repairJournalTail,
|
|
50
|
+
writeCheckpoint,
|
|
51
|
+
} from "../src/core/coordinator-journal.mjs";
|
|
52
|
+
import { ownsEndpoint } from "../src/core/host-coordination.mjs";
|
|
53
|
+
import { acquireOwnedViewLock, withViewLockSync } from "../src/core/locks.mjs";
|
|
54
|
+
import * as P from "../src/core/paths.mjs";
|
|
55
|
+
import { readState, readStatus, writeState, writeStatus } from "../src/core/store.mjs";
|
|
56
|
+
import { TRANSIENT_KINDS, decideStateTransition, validateCommand } from "../src/core/state-commands.mjs";
|
|
57
|
+
import { COORDINATOR_PROTOCOL_VERSION } from "../src/core/coordinator-protocol.mjs";
|
|
58
|
+
|
|
59
|
+
/** In-memory processed-command ring size (FIFO). Beyond the journal, this covers
|
|
60
|
+
* idempotency when the journal prefix has already been GC'd away. */
|
|
61
|
+
const PROCESSED_RING_MAX = 1000;
|
|
62
|
+
/** Journal growth (bytes) that triggers checkpoint + GC. Keeps boot replay bounded. */
|
|
63
|
+
const CHECKPOINT_THRESHOLD_BYTES = 262_144;
|
|
64
|
+
/** Lease heartbeat — matches pty-runner's host-start lease cadence. */
|
|
65
|
+
const HEARTBEAT_MS = 1000;
|
|
66
|
+
/** Lifecycle kinds whose legacy write sites stamped state.lastActivityAt on
|
|
67
|
+
* every persist (service markQueued/archive/adopt, job-runner plan-ready,
|
|
68
|
+
* reconcile terminal sites). The decision layer stays clock-free, so the
|
|
69
|
+
* shell adds the wall-clock stamp — into the patch BEFORE the journal
|
|
70
|
+
* append, so boot replay re-applies the exact same fields deterministically
|
|
71
|
+
* (there is no clock at replay time either). (Task 1 review F5.) */
|
|
72
|
+
const LAST_ACTIVITY_STAMP_KINDS = new Set(["mark_queued", "archive_view", "adopt_session", "reconcile_finalize", "plan_ready", "mark_completed", "host_run_failed"]);
|
|
73
|
+
/** Kinds whose status patch may CREATE the run's status file: they carry the
|
|
74
|
+
* full status content for a run that has no materialized status yet. Every
|
|
75
|
+
* other kind keeps PR #1's "patch presence ≠ file requirement" semantics
|
|
76
|
+
* (e.g. mark_completed's `status: {autoState: null}` on a legacy row without
|
|
77
|
+
* a status file must not fabricate one). */
|
|
78
|
+
const STATUS_BOOTSTRAP_KINDS = new Set(["run_started", "followup_started"]);
|
|
79
|
+
|
|
80
|
+
const root = process.argv[2];
|
|
81
|
+
if (!root) {
|
|
82
|
+
process.stderr.write("state-coordinator: missing root argument\n");
|
|
83
|
+
process.exit(2);
|
|
84
|
+
}
|
|
85
|
+
if (process.env.AGENT_BOARD_COORDINATOR === "off") process.exit(0);
|
|
86
|
+
|
|
87
|
+
main().catch((err) => {
|
|
88
|
+
process.stderr.write(`state-coordinator: fatal ${err?.stack || err}\n`);
|
|
89
|
+
process.exit(1);
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
async function main() {
|
|
93
|
+
const instanceId = randomBytes(8).toString("hex");
|
|
94
|
+
const startedAt = Date.now();
|
|
95
|
+
|
|
96
|
+
// 1. Lease first: a second coordinator is an expected, idempotent no-op.
|
|
97
|
+
// Identity carries a POSIX startToken (pty-runner pattern) so a SIGKILLed
|
|
98
|
+
// coordinator's lease can be reclaimed by the next instance after pid reuse.
|
|
99
|
+
/** @type {import("../src/core/locks.mjs").Lease} */
|
|
100
|
+
let lease;
|
|
101
|
+
try {
|
|
102
|
+
lease = acquireOwnedViewLock(root, "_coordinator", "state-coordinator", {
|
|
103
|
+
waitMs: 500,
|
|
104
|
+
identity: { pid: process.pid, startToken: captureStartToken(process.pid) },
|
|
105
|
+
});
|
|
106
|
+
} catch (err) {
|
|
107
|
+
process.stderr.write(`state-coordinator: lease unavailable (${err?.code ?? err?.message ?? "unknown"}); another instance may own it; exiting\n`);
|
|
108
|
+
process.exit(0);
|
|
109
|
+
}
|
|
110
|
+
const startTouchTimer = setInterval(() => {
|
|
111
|
+
try { lease?.touch(); } catch { /* best effort */ }
|
|
112
|
+
}, HEARTBEAT_MS);
|
|
113
|
+
startTouchTimer.unref?.();
|
|
114
|
+
|
|
115
|
+
// 2. Boot state: journal summary, processed-command ring, replay, counter init.
|
|
116
|
+
// Repair a crash-torn tail FIRST: without it the first post-restart append
|
|
117
|
+
// concatenates onto the torn line and that fsynced record becomes invisible
|
|
118
|
+
// to every readJournal / replay / revision scan below.
|
|
119
|
+
repairJournalTail(root);
|
|
120
|
+
const checkpoint = readCheckpoint(root) ?? { materializedRevision: 0, journalBytes: 0 };
|
|
121
|
+
const journal = readJournal(root);
|
|
122
|
+
/** @type {Map<string, { result: { status: string, reason: string|null }, materializedRevision: number }>} */
|
|
123
|
+
const processedCommands = new Map();
|
|
124
|
+
for (const record of journal) {
|
|
125
|
+
const commandId = record?.command?.commandId;
|
|
126
|
+
if (commandId && !processedCommands.has(commandId)) {
|
|
127
|
+
processedCommands.set(commandId, {
|
|
128
|
+
result: record.result ?? { status: "rejected", reason: "unknown" },
|
|
129
|
+
materializedRevision: record.materializedRevision ?? 0,
|
|
130
|
+
});
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// Replay in journal order: re-materialize every applied record from its stored
|
|
135
|
+
// patches; materialize's per-file guards write only files strictly behind the
|
|
136
|
+
// recorded revision, so this repairs just the half that missed its write in a
|
|
137
|
+
// crash window and never moves the other half backwards.
|
|
138
|
+
for (const record of journal) {
|
|
139
|
+
if (record?.result?.status !== "applied") continue;
|
|
140
|
+
const viewId = record?.command?.viewId;
|
|
141
|
+
if (!viewId) continue;
|
|
142
|
+
materialize(viewId, record.command ?? {}, record.mutate ?? {}, record.materializedRevision ?? 0, record.statusRunId ?? null);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
// Global revision counter: never reissue a revision that exists anywhere.
|
|
146
|
+
let revisionCounter = Math.max(checkpoint.materializedRevision ?? 0, 0);
|
|
147
|
+
for (const record of journal) revisionCounter = Math.max(revisionCounter, record.materializedRevision ?? 0);
|
|
148
|
+
if (existsSync(P.viewsDir(root))) {
|
|
149
|
+
for (const entry of readdirSync(P.viewsDir(root))) {
|
|
150
|
+
const state = readJson(P.statePath(root, entry), null);
|
|
151
|
+
revisionCounter = Math.max(revisionCounter, state?.materializedRevision ?? 0);
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
let lastCheckpointBytes = 0;
|
|
155
|
+
|
|
156
|
+
// 3. Control socket (JSONL, pty-runner's line-buffered server shape).
|
|
157
|
+
const socketPath = P.coordinatorEndpointPathFor(process.platform, root);
|
|
158
|
+
if (process.platform !== "win32" && existsSync(socketPath)) {
|
|
159
|
+
// Lease is ours: any leftover socket file belongs to a dead coordinator.
|
|
160
|
+
try { unlinkSync(socketPath); } catch { /* best effort */ }
|
|
161
|
+
}
|
|
162
|
+
/** {dev,ino} recorded at bind time; cleanup unlinks only this exact inode. */
|
|
163
|
+
let boundSocketIdentity = null;
|
|
164
|
+
|
|
165
|
+
const server = createServer((socket) => {
|
|
166
|
+
let buffer = "";
|
|
167
|
+
socket.on("data", (chunk) => {
|
|
168
|
+
buffer += chunk.toString("utf8");
|
|
169
|
+
const lines = buffer.split("\n");
|
|
170
|
+
buffer = lines.pop() ?? "";
|
|
171
|
+
for (const line of lines) handleClientLine(line, socket);
|
|
172
|
+
});
|
|
173
|
+
socket.on("error", () => {}); // client vanished mid-request; nothing to answer
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
server.on("error", (err) => {
|
|
177
|
+
process.stderr.write(`state-coordinator: server error ${err instanceof Error ? err.message : String(err)}\n`);
|
|
178
|
+
shutdown(1);
|
|
179
|
+
});
|
|
180
|
+
|
|
181
|
+
server.listen(socketPath, () => {
|
|
182
|
+
if (process.platform !== "win32") {
|
|
183
|
+
try {
|
|
184
|
+
const st = statSync(socketPath);
|
|
185
|
+
boundSocketIdentity = { dev: st.dev, ino: st.ino };
|
|
186
|
+
} catch { /* without the identity, cleanup degrades to no-op */ }
|
|
187
|
+
}
|
|
188
|
+
});
|
|
189
|
+
|
|
190
|
+
let shuttingDown = false;
|
|
191
|
+
process.on("SIGTERM", () => shutdown(0));
|
|
192
|
+
process.on("SIGINT", () => shutdown(0));
|
|
193
|
+
|
|
194
|
+
function shutdown(code) {
|
|
195
|
+
if (shuttingDown) return;
|
|
196
|
+
shuttingDown = true;
|
|
197
|
+
try { clearInterval(startTouchTimer); } catch { /* best effort */ }
|
|
198
|
+
try { server.close(); } catch { /* already closed */ }
|
|
199
|
+
if (process.platform !== "win32" && boundSocketIdentity) {
|
|
200
|
+
try {
|
|
201
|
+
const st = statSync(socketPath);
|
|
202
|
+
if (ownsEndpoint(boundSocketIdentity, { dev: st.dev, ino: st.ino })) {
|
|
203
|
+
try { unlinkSync(socketPath); } catch { /* best effort */ }
|
|
204
|
+
}
|
|
205
|
+
} catch { /* path gone — nothing to clean */ }
|
|
206
|
+
}
|
|
207
|
+
try { lease.release(); } catch { /* best effort */ }
|
|
208
|
+
process.exit(code);
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
function send(socket, msg) {
|
|
212
|
+
socket.write(JSON.stringify(msg) + "\n");
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
function handleClientLine(line, socket) {
|
|
216
|
+
if (!line.trim()) return;
|
|
217
|
+
let msg;
|
|
218
|
+
try { msg = JSON.parse(line); } catch { return send(socket, { type: "error", message: "invalid json" }); }
|
|
219
|
+
switch (msg?.type) {
|
|
220
|
+
case "ping":
|
|
221
|
+
send(socket, { type: "pong", instanceId, startedAt, protocolVersion: COORDINATOR_PROTOCOL_VERSION });
|
|
222
|
+
break;
|
|
223
|
+
case "state_command":
|
|
224
|
+
send(socket, { type: "state_command_result", ...processStateCommand(msg) });
|
|
225
|
+
break;
|
|
226
|
+
default:
|
|
227
|
+
send(socket, { type: "error", message: "unknown type" });
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* The command loop: validate → dedupe → decide → journal (fsync) → materialize
|
|
233
|
+
* → remember → reply. Rejections are journaled too, so their exact reason is
|
|
234
|
+
* replayable; duplicates short-circuit before any append.
|
|
235
|
+
* @param {object} msg
|
|
236
|
+
* @returns {{ commandId: string|null, status: string, reason: string|null, materializedRevision: number }}
|
|
237
|
+
*/
|
|
238
|
+
function processStateCommand(msg) {
|
|
239
|
+
const checked = validateCommand(msg);
|
|
240
|
+
if (!checked.ok) {
|
|
241
|
+
return {
|
|
242
|
+
commandId: typeof msg?.commandId === "string" ? msg.commandId : null,
|
|
243
|
+
status: "rejected",
|
|
244
|
+
reason: checked.error,
|
|
245
|
+
materializedRevision: 0,
|
|
246
|
+
};
|
|
247
|
+
}
|
|
248
|
+
const command = checked.command;
|
|
249
|
+
const transient = TRANSIENT_KINDS.includes(command.kind);
|
|
250
|
+
|
|
251
|
+
if (!transient) {
|
|
252
|
+
const known = lookupProcessed(command.commandId);
|
|
253
|
+
if (known) {
|
|
254
|
+
return { commandId: command.commandId, status: known.result.status, reason: known.result.reason, materializedRevision: known.materializedRevision };
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
const state = readState(root, command.viewId);
|
|
259
|
+
// Status consistency only binds the view's current run (spec 根治条件 5):
|
|
260
|
+
// no currentRunId → status.json plays no role in this command.
|
|
261
|
+
// Exception (Task 1 review F1): followup_started re-points the row at a
|
|
262
|
+
// NEW run carried in payload.newRunId (command.runId is deliberately null
|
|
263
|
+
// so the stale-run guard cannot fire against the finished parent run).
|
|
264
|
+
// Resolving the status from the parent here would overwrite the parent's
|
|
265
|
+
// status.json with the new run's bootstrap patch; resolving from newRunId
|
|
266
|
+
// finds no file and cleanly bootstraps the new run's status instead.
|
|
267
|
+
const statusRunId = command.kind === "followup_started"
|
|
268
|
+
? (typeof command.payload?.newRunId === "string" ? command.payload.newRunId : null)
|
|
269
|
+
: (command.runId ?? state?.currentRunId ?? null);
|
|
270
|
+
const status = statusRunId ? readStatus(root, command.viewId, statusRunId) : null;
|
|
271
|
+
const now = Date.now();
|
|
272
|
+
const decision = stampLegacyTimestamps(command, decideStateTransition(command, state, status, now), now);
|
|
273
|
+
const currentRevision = state?.materializedRevision ?? 0;
|
|
274
|
+
|
|
275
|
+
if (decision.action === "reject") {
|
|
276
|
+
if (transient) {
|
|
277
|
+
// Transient rejections are not replayable either — nothing was
|
|
278
|
+
// mutated, and journaling them would grow the journal for noise.
|
|
279
|
+
return { commandId: command.commandId ?? null, status: "rejected", reason: decision.reason, materializedRevision: currentRevision };
|
|
280
|
+
}
|
|
281
|
+
const result = { status: "rejected", reason: decision.reason };
|
|
282
|
+
const record = { command, result, materializedRevision: currentRevision, at: now };
|
|
283
|
+
const journalBytes = appendCommand(root, record);
|
|
284
|
+
rememberProcessed(command.commandId, result, currentRevision);
|
|
285
|
+
maybeCheckpoint(journalBytes);
|
|
286
|
+
return { commandId: command.commandId, status: "rejected", reason: decision.reason, materializedRevision: currentRevision };
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
const newRevision = revisionCounter + 1;
|
|
290
|
+
const result = { status: "applied", reason: decision.reason };
|
|
291
|
+
if (transient) {
|
|
292
|
+
// No journal, no processed ring, no checkpoint: a periodic snapshot —
|
|
293
|
+
// the next beat supersedes it.
|
|
294
|
+
materialize(command.viewId, command, decision.mutate, newRevision, statusRunId);
|
|
295
|
+
revisionCounter = newRevision;
|
|
296
|
+
return { commandId: command.commandId ?? null, status: "applied", reason: decision.reason, materializedRevision: newRevision };
|
|
297
|
+
}
|
|
298
|
+
// Journal first (fsync inside), materialize second: boot replay repairs
|
|
299
|
+
// the window between the two using the stored patches.
|
|
300
|
+
const record = { command, result, materializedRevision: newRevision, at: now, mutate: decision.mutate, statusRunId };
|
|
301
|
+
const journalBytes = appendCommand(root, record);
|
|
302
|
+
materialize(command.viewId, command, decision.mutate, newRevision, statusRunId);
|
|
303
|
+
revisionCounter = newRevision;
|
|
304
|
+
rememberProcessed(command.commandId, result, newRevision);
|
|
305
|
+
maybeCheckpoint(journalBytes);
|
|
306
|
+
return { commandId: command.commandId, status: "applied", reason: decision.reason, materializedRevision: newRevision };
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
/**
|
|
310
|
+
* Merge patches onto state.json (and the run's status.json when it exists)
|
|
311
|
+
* under the view's materialize lock, stamping the shared revision. Per-file
|
|
312
|
+
* guards stamp only files strictly behind the record revision: the live path
|
|
313
|
+
* always qualifies (the global counter is monotonic), while replay repairs
|
|
314
|
+
* just the half that missed its write in a crash window and never moves the
|
|
315
|
+
* other half backwards. A status patch without a status file is skipped —
|
|
316
|
+
* patch presence ≠ file requirement (mark_completed always emits
|
|
317
|
+
* `status: {autoState: null}`). The status run binding is the decision-time
|
|
318
|
+
* one (`statusRunId`, recorded in the journal) so replay cannot re-bind a
|
|
319
|
+
* runId-less patch to whatever run is current on disk at replay time;
|
|
320
|
+
* `state.currentRunId` remains only as a legacy fallback for records written
|
|
321
|
+
* before statusRunId existed.
|
|
322
|
+
*/
|
|
323
|
+
function materialize(viewId, command, mutate, revision, statusRunId) {
|
|
324
|
+
withViewLockSync(root, viewId, "state-materialize", () => {
|
|
325
|
+
const state = readState(root, viewId);
|
|
326
|
+
if (mutate?.state && state && (state.materializedRevision ?? 0) < revision) {
|
|
327
|
+
writeState(root, { ...state, ...mutate.state, materializedRevision: revision });
|
|
328
|
+
}
|
|
329
|
+
const runId = command?.runId ?? statusRunId ?? state?.currentRunId ?? null;
|
|
330
|
+
if (mutate?.status && runId) {
|
|
331
|
+
const status = readStatus(root, viewId, runId);
|
|
332
|
+
if (!status && STATUS_BOOTSTRAP_KINDS.has(command?.kind)) {
|
|
333
|
+
// run_started / followup_started carry the full status content for a
|
|
334
|
+
// run that has no status file yet — create it (F1's clean-bootstrap
|
|
335
|
+
// half). Identity fields come from the command context; the patch
|
|
336
|
+
// carries everything meaningful.
|
|
337
|
+
writeStatus(root, { version: 1, runId, viewId, ...mutate.status, materializedRevision: revision });
|
|
338
|
+
} else if (status && (status.materializedRevision ?? 0) < revision) {
|
|
339
|
+
writeStatus(root, { ...status, ...mutate.status, materializedRevision: revision });
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
});
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/** Add the legacy wall-clock stamp to a decided patch (F5): see
|
|
346
|
+
* LAST_ACTIVITY_STAMP_KINDS. Applied before the journal append so the
|
|
347
|
+
* stored patches are exactly what replay re-materializes. */
|
|
348
|
+
function stampLegacyTimestamps(command, decision, now) {
|
|
349
|
+
if (decision.action !== "apply" || !LAST_ACTIVITY_STAMP_KINDS.has(command.kind)) return decision;
|
|
350
|
+
return { ...decision, mutate: { ...decision.mutate, state: { ...(decision.mutate.state ?? {}), lastActivityAt: now } } };
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
/** Memory ring first (covers the post-GC window), journal second. */
|
|
354
|
+
function lookupProcessed(commandId) {
|
|
355
|
+
const inMemory = processedCommands.get(commandId);
|
|
356
|
+
if (inMemory) return inMemory;
|
|
357
|
+
const result = findProcessedCommand(root, commandId);
|
|
358
|
+
if (!result) return null;
|
|
359
|
+
const entry = { result, materializedRevision: revisionOf(commandId) };
|
|
360
|
+
rememberProcessed(commandId, result, entry.materializedRevision);
|
|
361
|
+
return entry;
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
/** Revision recorded for a known commandId (journal scan; 0 if unrecorded). */
|
|
365
|
+
function revisionOf(commandId) {
|
|
366
|
+
for (const record of readJournal(root)) {
|
|
367
|
+
if (record?.command?.commandId === commandId) return record.materializedRevision ?? 0;
|
|
368
|
+
}
|
|
369
|
+
return 0;
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
function rememberProcessed(commandId, result, materializedRevision) {
|
|
373
|
+
processedCommands.set(commandId, { result, materializedRevision });
|
|
374
|
+
while (processedCommands.size > PROCESSED_RING_MAX) {
|
|
375
|
+
processedCommands.delete(processedCommands.keys().next().value);
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
/** Checkpoint + GC once the journal outgrows the threshold. GC only runs when
|
|
380
|
+
* the checkpoint write provably succeeded (journal module enforces this). */
|
|
381
|
+
function maybeCheckpoint(journalBytes) {
|
|
382
|
+
if (journalBytes - lastCheckpointBytes < CHECKPOINT_THRESHOLD_BYTES) return;
|
|
383
|
+
if (!writeCheckpoint(root, { materializedRevision: revisionCounter, journalBytes })) return;
|
|
384
|
+
lastCheckpointBytes = gcJournal(root);
|
|
385
|
+
}
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/** POSIX process start token — /proc/<pid>/stat field 22 (starttime), stable
|
|
389
|
+
* across exec(2). Same recipe as pty-runner: lets locks.mjs reclaim this
|
|
390
|
+
* coordinator's lease after a SIGKILL even if the pid was reused. null on
|
|
391
|
+
* failure or non-Linux platforms (reclaim degrades to "blocked" there — same
|
|
392
|
+
* platform parity as PTY hosts). */
|
|
393
|
+
function captureStartToken(pid) {
|
|
394
|
+
if (process.platform !== "linux" || !pid) return null;
|
|
395
|
+
try {
|
|
396
|
+
const stat = readFileSync(`/proc/${pid}/stat`, "utf8");
|
|
397
|
+
const afterComm = stat.slice(stat.lastIndexOf(")") + 1).trimStart();
|
|
398
|
+
const fields = afterComm.split(/\s+/);
|
|
399
|
+
return fields[19] ?? null;
|
|
400
|
+
} catch {
|
|
401
|
+
return null;
|
|
402
|
+
}
|
|
403
|
+
}
|
package/runner/state-runner.mjs
CHANGED
|
@@ -9,10 +9,11 @@
|
|
|
9
9
|
import { spawn } from "node:child_process";
|
|
10
10
|
import { readJson } from "../src/core/atomic.mjs";
|
|
11
11
|
import { appendDiagnostic } from "../src/core/diagnostics.mjs";
|
|
12
|
-
import { applyAutoStateToStatus, applyAutoStateToViewState, autoStateEnabled, autoStateFromModelOrHeuristic, autoStateModel, buildAutoStatePrompt, heuristicAutoState } from "../src/core/auto-state.mjs";
|
|
12
|
+
import { applyAutoStateToStatus, applyAutoStateToViewState, autoStateEnabled, autoStateFromModelOrHeuristic, autoStateModel, buildAutoStatePrompt, heuristicAutoState, isManualCompletion } from "../src/core/auto-state.mjs";
|
|
13
13
|
import { finalizeEvidence, readEvidence, summarizeEvidence, writeEvidence } from "../src/core/evidence.mjs";
|
|
14
14
|
import { updateCodeRefsFromEvidence } from "../src/core/code-refs-store.mjs";
|
|
15
15
|
import { readState, readStatus, readMeta, writeState, writeStatus } from "../src/core/store.mjs";
|
|
16
|
+
import { sendStateCommand } from "../src/core/coordinator-client.mjs";
|
|
16
17
|
|
|
17
18
|
async function main() {
|
|
18
19
|
const configPath = process.argv[2];
|
|
@@ -26,6 +27,10 @@ async function main() {
|
|
|
26
27
|
|
|
27
28
|
const state = readState(config.root, config.viewId);
|
|
28
29
|
if (!state || state.processState === "alive" || state.semanticState === "failed" || state.semanticState === "stopped") process.exit(0);
|
|
30
|
+
// Cheap pre-check (optimization only): skip a pointless command when the
|
|
31
|
+
// manual verdict is already materialized. The coordinator's manual_fence
|
|
32
|
+
// stays the authoritative guard for races after this read.
|
|
33
|
+
if (isManualCompletion(state)) process.exit(0);
|
|
29
34
|
|
|
30
35
|
const evidence = readEvidence(config.root, config.viewId);
|
|
31
36
|
const latest = latestEvidenceText(evidence) || state.latestAssistantPreview || state.summary || "";
|
|
@@ -45,24 +50,93 @@ async function main() {
|
|
|
45
50
|
classification = autoStateFromModelOrHeuristic(out, latest, { lastAgentActivityAt: state.lastAgentActivityAt ?? null });
|
|
46
51
|
}
|
|
47
52
|
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
53
|
+
// Issue #91 (A8 path 2): the classification lands through the View State
|
|
54
|
+
// Coordinator — the single writer of state.json/status.json. A manual
|
|
55
|
+
// completion is fenced by the coordinator (manual_fence), so this late pass
|
|
56
|
+
// can no longer clobber the user's verdict (#46 class). Ambiguous transport
|
|
57
|
+
// outcomes (timeout / connection_reset) NEVER fall back to a direct write:
|
|
58
|
+
// the command may already be journaled, and the coordinator's boot replay
|
|
59
|
+
// is the recovery path.
|
|
60
|
+
const result = await sendStateCommand(config.root, {
|
|
61
|
+
type: "state_command",
|
|
62
|
+
viewId: config.viewId,
|
|
63
|
+
runId: config.runId ?? null,
|
|
64
|
+
source: "state-runner",
|
|
65
|
+
kind: "auto_state_classified",
|
|
66
|
+
expectedRevision: null,
|
|
67
|
+
payload: { classification },
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
if (result.status === "applied") {
|
|
71
|
+
appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated row state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
|
|
72
|
+
} else if (result.reason === "manual_fence" || result.reason === "no_change" || result.reason === "stale_run") {
|
|
73
|
+
// Designed fences — informational, not errors: the coordinator is the
|
|
74
|
+
// authority and a manual completion wins by design.
|
|
75
|
+
appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "auto_state_classified_skipped", message: `Auto-state classification not applied (${result.reason})`, details: { reason: result.reason } });
|
|
76
|
+
} else if (result.reason !== "coordinator_disabled") {
|
|
77
|
+
appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, level: "warn", code: "auto_state_command_ambiguous", message: `Auto-state classification outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise the next classification pass will converge the row`, details: { reason: result.reason } });
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
if (result.reason === "coordinator_disabled") {
|
|
81
|
+
// Legacy escape hatch (AGENT_BOARD_COORDINATOR=off): pre-coordinator
|
|
82
|
+
// direct-write behavior, unchanged.
|
|
83
|
+
let changed = false;
|
|
84
|
+
if (config.runId) {
|
|
85
|
+
const status = readStatus(config.root, config.viewId, config.runId);
|
|
86
|
+
if (status) {
|
|
87
|
+
changed = applyAutoStateToStatus(status, classification, Date.now()) || changed;
|
|
88
|
+
status.evidenceSummary = summarizeEvidence(finalizeEvidence(evidence, status, Date.now()));
|
|
89
|
+
writeStatus(config.root, status);
|
|
90
|
+
}
|
|
55
91
|
}
|
|
92
|
+
const latestState = readState(config.root, config.viewId) ?? state;
|
|
93
|
+
changed = applyAutoStateToViewState(latestState, classification, Date.now()) || changed;
|
|
94
|
+
finalizeEvidence(evidence, { semanticState: latestState.semanticState, usage: null }, Date.now());
|
|
95
|
+
latestState.review = summarizeEvidence(evidence);
|
|
96
|
+
writeEvidence(config.root, evidence);
|
|
97
|
+
updateCodeRefsFromEvidence(config.root, config.viewId, evidence, meta);
|
|
98
|
+
writeState(config.root, latestState);
|
|
99
|
+
if (changed) {
|
|
100
|
+
appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated row state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
|
|
101
|
+
}
|
|
102
|
+
process.exit(0);
|
|
56
103
|
}
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
104
|
+
|
|
105
|
+
// Evidence pipeline (coordinator-independent) stays direct: finalize the
|
|
106
|
+
// evidence snapshot from a FRESH state read (reads are not writes — the
|
|
107
|
+
// coordinator owns state.json/status.json writes, not reads), then persist
|
|
108
|
+
// the evidence artifacts and code-refs.
|
|
109
|
+
const postState = readState(config.root, config.viewId);
|
|
110
|
+
finalizeEvidence(evidence, { semanticState: postState?.semanticState ?? state.semanticState, usage: null }, Date.now());
|
|
61
111
|
writeEvidence(config.root, evidence);
|
|
62
112
|
updateCodeRefsFromEvidence(config.root, config.viewId, evidence, meta);
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
113
|
+
|
|
114
|
+
// Issue #91 PR #2 (Task 4): the evidence mirrors move behind the coordinator
|
|
115
|
+
// too — one patch_fields command materializes review/evidenceSummary on both
|
|
116
|
+
// files under a shared materializedRevision (previously two separately
|
|
117
|
+
// fenced direct writes). The coordinator's generic manual_fence / stale_run
|
|
118
|
+
// guards are authoritative; ambiguous outcomes (timeout / connection_reset)
|
|
119
|
+
// NEVER fall back to a direct write — if the command was journaled, boot
|
|
120
|
+
// replay recovers it, and the next classification pass re-derives mirrors
|
|
121
|
+
// from the evidence files either way.
|
|
122
|
+
const mirrorSummary = summarizeEvidence(evidence);
|
|
123
|
+
const patch = await sendStateCommand(config.root, {
|
|
124
|
+
type: "state_command",
|
|
125
|
+
viewId: config.viewId,
|
|
126
|
+
runId: config.runId ?? null,
|
|
127
|
+
source: "state-runner",
|
|
128
|
+
kind: "patch_fields",
|
|
129
|
+
expectedRevision: null,
|
|
130
|
+
payload: { state: { review: mirrorSummary }, status: { evidenceSummary: mirrorSummary } },
|
|
131
|
+
});
|
|
132
|
+
if (patch.status === "applied") {
|
|
133
|
+
// Quiet success — mirrors materialized by the coordinator.
|
|
134
|
+
} else if (patch.reason === "manual_fence" || patch.reason === "stale_run" || patch.reason === "no_change") {
|
|
135
|
+
// Designed fences — informational: a manual verdict or a newer run owns
|
|
136
|
+
// the row, or the mirrors already match.
|
|
137
|
+
appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, code: "evidence_mirror_patch_skipped", message: `Evidence mirror patch not applied (${patch.reason})`, details: { reason: patch.reason } });
|
|
138
|
+
} else {
|
|
139
|
+
appendDiagnostic(config.root, config.viewId, { source: "service", runId: config.runId, level: "warn", code: "evidence_mirror_patch_ambiguous", message: `Evidence mirror patch outcome unknown (${patch.reason}); if the command was journaled, coordinator replay will recover it; the next classification pass re-derives the mirrors from the evidence files`, details: { reason: patch.reason } });
|
|
66
140
|
}
|
|
67
141
|
}
|
|
68
142
|
|