@zhuxixi/pi-agent-board 0.6.2 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/docs/superpowers/plans/2026-09-08-issue-11-attach-runtime-desync-heal.md +917 -0
- package/docs/superpowers/plans/2026-09-09-harden-runner-architecture.md +603 -0
- package/docs/superpowers/plans/2026-09-09-single-writer-completion.md +252 -0
- package/docs/superpowers/specs/2026-09-07-issue-11-attach-runtime-desync-heal-design.md +130 -0
- package/docs/superpowers/specs/2026-09-09-harden-runner-architecture-design.md +298 -0
- package/package.json +1 -1
- package/runner/job-runner-legacy.mjs +68 -0
- package/runner/job-runner.mjs +370 -67
- package/runner/pty-runner-legacy.mjs +50 -0
- package/runner/pty-runner.mjs +69 -31
- package/runner/state-coordinator.mjs +403 -0
- package/runner/state-runner.mjs +89 -15
- package/src/commands/bg.ts +2 -1
- package/src/core/coordinator-client.mjs +313 -0
- package/src/core/coordinator-journal.mjs +282 -0
- package/src/core/coordinator-protocol.mjs +12 -0
- package/src/core/launch.mjs +15 -0
- package/src/core/paths.mjs +16 -0
- package/src/core/pty-attach-jiggle-controller.mjs +57 -4
- package/src/core/pty-attach-render.mjs +30 -0
- package/src/core/state-commands.mjs +617 -0
- package/src/core/types.mjs +2 -0
- package/src/runtime/service.mjs +416 -111
- package/src/ui/dashboard.ts +77 -110
- package/src/ui/pty-attach.ts +63 -1
package/runner/job-runner.mjs
CHANGED
|
@@ -5,13 +5,18 @@
|
|
|
5
5
|
* Usage: node job-runner.mjs <configPath>
|
|
6
6
|
*
|
|
7
7
|
* Owns one run: spawns a headless Pi worker (`pi --mode json -p --session <file> <prompt>`),
|
|
8
|
-
* streams its JSON events into events.jsonl,
|
|
9
|
-
*
|
|
8
|
+
* streams its JSON events into events.jsonl, and routes every semantic-state
|
|
9
|
+
* mutation through the View State Coordinator as commands (issue #91): boot
|
|
10
|
+
* run_started, transient run_progress beats on the throttled hot path,
|
|
11
|
+
* auto-state classifications, run_finalized on exit, and the post-exit
|
|
12
|
+
* patch_fields (evidence mirrors / model summary). Evidence artifacts
|
|
13
|
+
* (events/stdout/stderr logs, evidence files, code-refs) stay runner-owned and
|
|
14
|
+
* are written directly. Survives the parent Pi process exiting/reloading.
|
|
10
15
|
*/
|
|
11
16
|
import { spawn } from "node:child_process";
|
|
12
17
|
import { fileURLToPath } from "node:url";
|
|
13
18
|
import { appendLine, readJson } from "../src/core/atomic.mjs";
|
|
14
|
-
import { createRunStatus, finalizeRun,
|
|
19
|
+
import { createRunStatus, finalizeRun, reduceEvent } from "../src/core/events.mjs";
|
|
15
20
|
import { encodePromptForCliArg } from "../src/core/prompt-transport.mjs";
|
|
16
21
|
import { applyAutoStateToStatus, autoStateEnabled, autoStateFromModelOrHeuristic, autoStateModel, buildAutoStatePrompt, heuristicAutoState, isManualCompletion } from "../src/core/auto-state.mjs";
|
|
17
22
|
import { appendDiagnostic } from "../src/core/diagnostics.mjs";
|
|
@@ -21,8 +26,10 @@ import { claimNextFollowUp, completeFollowUp, releaseFollowUp } from "../src/cor
|
|
|
21
26
|
import { newRunId } from "../src/core/ids.mjs";
|
|
22
27
|
import { launchRun } from "../src/core/launch.mjs";
|
|
23
28
|
import * as P from "../src/core/paths.mjs";
|
|
24
|
-
import { readState, readStatus, readMeta
|
|
29
|
+
import { readState, readStatus, readMeta } from "../src/core/store.mjs";
|
|
25
30
|
import { readSteering, recordPlanReady } from "../src/core/steering.mjs";
|
|
31
|
+
import { sendStateCommand, coordinatorDisabled } from "../src/core/coordinator-client.mjs";
|
|
32
|
+
import { legacyFollowupBootstrap, legacyPersistState, legacyPlanReadyStateWrite, legacyWriteState, legacyWriteStatus } from "./job-runner-legacy.mjs";
|
|
26
33
|
import { buildApprovePlanPrompt, buildPlanChangesPrompt, buildPlanRequestPrompt } from "../src/core/steering-prompts.mjs";
|
|
27
34
|
|
|
28
35
|
const WRITE_THROTTLE_MS = 250;
|
|
@@ -59,13 +66,70 @@ function main() {
|
|
|
59
66
|
const meta = readMeta(root, viewId);
|
|
60
67
|
|
|
61
68
|
let status = createRunStatus(config, null, Date.now());
|
|
69
|
+
// The run starts working the moment the runner is up; seeding the boot status
|
|
70
|
+
// with "working" (instead of createRunStatus's "queued") avoids the stale
|
|
71
|
+
// queued frame flickering through run_started's first materialization.
|
|
72
|
+
status.semanticState = "working";
|
|
62
73
|
let evidence = emptyEvidenceSnapshot({ viewId, runId, source: "json-runner" });
|
|
63
74
|
appendDiagnostic(root, viewId, { source: "runner", runId, code: "runner_start", message: "Runner started", details: { kind: config.kind, cwd: config.cwd, model: config.model } });
|
|
64
|
-
writeStatus(root, status);
|
|
65
75
|
writeRunEvidence(root, evidence);
|
|
66
76
|
writeEvidence(root, evidence);
|
|
67
77
|
updateCodeRefsFromEvidence(root, viewId, evidence, meta);
|
|
68
|
-
|
|
78
|
+
if (coordinatorDisabled()) {
|
|
79
|
+
legacyPersistState({ root, viewId, runId, status });
|
|
80
|
+
}
|
|
81
|
+
return bootstrapRun({ root, viewId, runId, config, status, meta, evidence, stdoutLog, stderrLog, eventsLog });
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Async continuation of main: route the run_started bootstrap through the
|
|
86
|
+
* coordinator (the status file it creates is what every later run_progress
|
|
87
|
+
* beat patches onto), then spawn the worker and wire the event handlers.
|
|
88
|
+
* @param {{ root: string, viewId: string, runId: string, config: object, status: object, meta: object, evidence: object, stdoutLog: string, stderrLog: string, eventsLog: string }} ctx
|
|
89
|
+
*/
|
|
90
|
+
async function bootstrapRun({ root, viewId, runId, config, status, meta, evidence, stdoutLog, stderrLog, eventsLog }) {
|
|
91
|
+
if (!coordinatorDisabled()) {
|
|
92
|
+
// Journaled bootstrap: creates the run's status.json (STATUS_BOOTSTRAP_KINDS)
|
|
93
|
+
// and pins the row to working/alive/currentRunId. Ambiguous outcomes never
|
|
94
|
+
// block the run: if the command was journaled, coordinator replay recovers
|
|
95
|
+
// it; otherwise dashboard reconcile converges the row.
|
|
96
|
+
const started = await sendStateCommand(root, {
|
|
97
|
+
type: "state_command",
|
|
98
|
+
viewId,
|
|
99
|
+
runId,
|
|
100
|
+
source: "job-runner",
|
|
101
|
+
kind: "run_started",
|
|
102
|
+
expectedRevision: null,
|
|
103
|
+
payload: { status: { ...status } },
|
|
104
|
+
});
|
|
105
|
+
if (started.status !== "applied" && started.reason !== "coordinator_disabled") {
|
|
106
|
+
appendDiagnostic(root, viewId, { source: "runner", runId, level: "warn", code: "run_started_command_ambiguous", message: `Run bootstrap outcome unknown (${started.reason}); if the command was journaled, coordinator replay will recover it; otherwise dashboard reconcile will converge the row`, details: { reason: started.reason } });
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* One transient progress beat: ship the full in-memory status as a sparse
|
|
112
|
+
* patch — the coordinator merges it onto the materialized status and lets
|
|
113
|
+
* projectViewState recompute the row state (delegation, not copied rules).
|
|
114
|
+
*
|
|
115
|
+
* Fire-and-forget by design (the ONLY such command in the runner): the hot
|
|
116
|
+
* path is a periodic self-healing snapshot (~4/sec), a lost beat is
|
|
117
|
+
* superseded by the next one, and failures are silently ignored — logging
|
|
118
|
+
* them would spam diagnostics at beat frequency. Transient commands are
|
|
119
|
+
* never journaled, so an ambiguous outcome has no replay to await.
|
|
120
|
+
*/
|
|
121
|
+
const sendProgressBeat = () => {
|
|
122
|
+
const { materializedRevision: _fileStamp, ...patch } = status;
|
|
123
|
+
void sendStateCommand(root, {
|
|
124
|
+
type: "state_command",
|
|
125
|
+
viewId,
|
|
126
|
+
runId,
|
|
127
|
+
source: "job-runner",
|
|
128
|
+
kind: "run_progress",
|
|
129
|
+
expectedRevision: null,
|
|
130
|
+
payload: { statusPatch: patch },
|
|
131
|
+
}).catch(() => {});
|
|
132
|
+
};
|
|
69
133
|
|
|
70
134
|
// Build worker args: pi --mode json -p --session <file> [--model m] [--thinking l] [--tools t] <prompt>
|
|
71
135
|
const args = [
|
|
@@ -91,7 +155,9 @@ function main() {
|
|
|
91
155
|
|
|
92
156
|
status.pid = worker.pid ?? null;
|
|
93
157
|
appendDiagnostic(root, viewId, { source: "runner", runId, code: "worker_pid", message: "Worker pid recorded", details: { pid: status.pid } });
|
|
94
|
-
|
|
158
|
+
// The pid lands on disk via a transient progress beat (the worker's first
|
|
159
|
+
// events would carry it too, but a silent worker must still be observable).
|
|
160
|
+
sendProgressBeat();
|
|
95
161
|
|
|
96
162
|
let stoppedByUser = false;
|
|
97
163
|
let dirty = false;
|
|
@@ -99,15 +165,12 @@ function main() {
|
|
|
99
165
|
|
|
100
166
|
const persist = (force = false) => {
|
|
101
167
|
void force;
|
|
102
|
-
const now = Date.now();
|
|
103
168
|
status.evidenceSummary = summarizeEvidence(evidence);
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
writeState(root, projectViewState(status, now, readState(root, viewId)));
|
|
169
|
+
persistEvidenceArtifacts();
|
|
170
|
+
if (coordinatorDisabled()) legacyPersistState({ root, viewId, runId, status });
|
|
171
|
+
else sendProgressBeat();
|
|
108
172
|
// Best-effort code-refs extraction shells out to git and can take hundreds of
|
|
109
|
-
// ms;
|
|
110
|
-
// The extraction only depends on evidence + git, never on state.json.
|
|
173
|
+
// ms; it only depends on evidence + git, never on state.json.
|
|
111
174
|
updateCodeRefsFromEvidence(root, viewId, evidence, meta);
|
|
112
175
|
dirty = false;
|
|
113
176
|
};
|
|
@@ -125,6 +188,116 @@ function main() {
|
|
|
125
188
|
return true;
|
|
126
189
|
};
|
|
127
190
|
|
|
191
|
+
/**
|
|
192
|
+
* Runner-owned evidence artifacts (evidence files + code-refs). Written
|
|
193
|
+
* directly — the coordinator owns state.json/status.json, not these.
|
|
194
|
+
*/
|
|
195
|
+
const persistEvidenceArtifacts = () => {
|
|
196
|
+
status.evidenceSummary = summarizeEvidence(evidence);
|
|
197
|
+
writeRunEvidence(root, evidence);
|
|
198
|
+
writeEvidence(root, evidence);
|
|
199
|
+
updateCodeRefsFromEvidence(root, viewId, evidence, meta);
|
|
200
|
+
};
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Refresh the evidence mirrors (status.evidenceSummary / state.review)
|
|
204
|
+
* through the coordinator: `patch_fields` carries only whitelisted mirror
|
|
205
|
+
* fields, and the coordinator's generic manual fence (source != user on a
|
|
206
|
+
* manually-completed row) is the authoritative guard — the old fresh-read
|
|
207
|
+
* + isManualCompletion pre-checks are no longer needed. Designed fences
|
|
208
|
+
* (manual_fence / no_change) are informational; ambiguous outcomes never
|
|
209
|
+
* fall back to a direct write.
|
|
210
|
+
*/
|
|
211
|
+
const refreshEvidenceMirrors = async () => {
|
|
212
|
+
status.evidenceSummary = summarizeEvidence(evidence);
|
|
213
|
+
const result = await sendStateCommand(root, {
|
|
214
|
+
type: "state_command",
|
|
215
|
+
viewId,
|
|
216
|
+
runId: null,
|
|
217
|
+
source: "job-runner",
|
|
218
|
+
kind: "patch_fields",
|
|
219
|
+
expectedRevision: null,
|
|
220
|
+
payload: {
|
|
221
|
+
state: { review: status.evidenceSummary },
|
|
222
|
+
status: { evidenceSummary: status.evidenceSummary },
|
|
223
|
+
},
|
|
224
|
+
});
|
|
225
|
+
if (result.status === "applied") {
|
|
226
|
+
const fresh = readStatus(root, viewId, runId);
|
|
227
|
+
if (fresh) Object.assign(status, fresh);
|
|
228
|
+
return true;
|
|
229
|
+
}
|
|
230
|
+
if (result.reason === "manual_fence" || result.reason === "no_change") return false;
|
|
231
|
+
if (result.reason === "coordinator_disabled") {
|
|
232
|
+
// Legacy escape hatch: fresh-read + fence, pre-coordinator semantics.
|
|
233
|
+
const freshStatus = readStatus(root, viewId, runId);
|
|
234
|
+
if (freshStatus && !isManualCompletion(freshStatus)) {
|
|
235
|
+
freshStatus.evidenceSummary = status.evidenceSummary;
|
|
236
|
+
legacyWriteStatus(root, viewId, runId, freshStatus);
|
|
237
|
+
}
|
|
238
|
+
const freshState = readState(root, viewId);
|
|
239
|
+
if (freshState && !isManualCompletion(freshState)) {
|
|
240
|
+
freshState.review = status.evidenceSummary;
|
|
241
|
+
legacyWriteState(root, viewId, freshState);
|
|
242
|
+
}
|
|
243
|
+
return true;
|
|
244
|
+
}
|
|
245
|
+
appendDiagnostic(root, viewId, { source: "runner", runId, level: "warn", code: "evidence_mirror_command_ambiguous", message: `Evidence mirror outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise the next mirror refresh will converge`, details: { reason: result.reason } });
|
|
246
|
+
return false;
|
|
247
|
+
};
|
|
248
|
+
|
|
249
|
+
/**
|
|
250
|
+
* Materialize the run's terminal state through the View State Coordinator
|
|
251
|
+
* (issue #91, A8 path 3): the runner submits minimal facts and the
|
|
252
|
+
* coordinator computes terminal semantics via finalizeRun, so a manual
|
|
253
|
+
* completion landing before the command is fenced by the coordinator
|
|
254
|
+
* (manual_fence), not by a file re-read (#46 class). Evidence artifacts stay
|
|
255
|
+
* direct (runner-owned). Ambiguous outcomes (timeout / connection_reset)
|
|
256
|
+
* NEVER fall back to a direct write: the command may already be journaled,
|
|
257
|
+
* and the coordinator's boot replay is the recovery path.
|
|
258
|
+
* coordinator_disabled keeps the pre-coordinator direct persist.
|
|
259
|
+
* @param {{ exitCode: number|null, stoppedByUser: boolean }} facts
|
|
260
|
+
* @returns {Promise<boolean>} whether the final state is known materialized
|
|
261
|
+
*/
|
|
262
|
+
const finalizeThroughCoordinator = async ({ exitCode, stoppedByUser: stopped }) => {
|
|
263
|
+
const payload = { exitCode, stoppedByUser: stopped };
|
|
264
|
+
if (status.endedAt != null) payload.endedAt = status.endedAt;
|
|
265
|
+
// The close path cancels the pending throttled flush after the final
|
|
266
|
+
// buffer flush, so a stopReason observed in the last burst (reduceEvent
|
|
267
|
+
// sets it in memory only) never reached disk. Overlay it onto the payload:
|
|
268
|
+
// finalizeSemanticState keys on stopReason alone for exit-0 exits, and the
|
|
269
|
+
// coordinator already supports the payload overlay (issue #91).
|
|
270
|
+
if (status.stopReason != null) payload.stopReason = status.stopReason;
|
|
271
|
+
if (status.latestAssistantPreview) payload.latestAssistantPreview = status.latestAssistantPreview;
|
|
272
|
+
if (status.lastAgentActivityAt != null) payload.lastAgentActivityAt = status.lastAgentActivityAt;
|
|
273
|
+
const result = await sendStateCommand(root, {
|
|
274
|
+
type: "state_command",
|
|
275
|
+
viewId,
|
|
276
|
+
runId,
|
|
277
|
+
source: "job-runner",
|
|
278
|
+
kind: "run_finalized",
|
|
279
|
+
expectedRevision: null,
|
|
280
|
+
payload,
|
|
281
|
+
});
|
|
282
|
+
if (result.status === "applied") {
|
|
283
|
+
const fresh = readStatus(root, viewId, runId);
|
|
284
|
+
if (fresh) Object.assign(status, fresh);
|
|
285
|
+
await refreshEvidenceMirrors();
|
|
286
|
+
return true;
|
|
287
|
+
}
|
|
288
|
+
if (result.reason === "coordinator_disabled") {
|
|
289
|
+
persistEvidenceArtifacts();
|
|
290
|
+
legacyPersistState({ root, viewId, runId, status });
|
|
291
|
+
return true;
|
|
292
|
+
}
|
|
293
|
+
if (result.reason === "stale_run") {
|
|
294
|
+
// Duplicate finalize or the run was already superseded — nothing to do.
|
|
295
|
+
return false;
|
|
296
|
+
}
|
|
297
|
+
appendDiagnostic(root, viewId, { source: "runner", runId, level: "warn", code: "run_finalize_command_ambiguous", message: `Run finalization outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise dashboard reconcile will converge the row`, details: { reason: result.reason } });
|
|
298
|
+
return false;
|
|
299
|
+
};
|
|
300
|
+
|
|
128
301
|
const scheduleFlush = () => {
|
|
129
302
|
if (flushTimer) {
|
|
130
303
|
dirty = true;
|
|
@@ -196,16 +369,26 @@ function main() {
|
|
|
196
369
|
process.on("SIGTERM", stop);
|
|
197
370
|
process.on("SIGINT", stop);
|
|
198
371
|
|
|
199
|
-
worker.on("error", (err) => {
|
|
372
|
+
worker.on("error", async (err) => {
|
|
373
|
+
// Cancel any in-flight throttled flush before finalizing: if the timer
|
|
374
|
+
// callback lands during the sendStateCommand await below, the stale
|
|
375
|
+
// persist() would overwrite the coordinator-materialized terminal state
|
|
376
|
+
// and the close-path run_finalized would then bounce as stale_run
|
|
377
|
+
// (symmetric with the close path's cancel; issue #91 fix round 1).
|
|
378
|
+
if (flushTimer) {
|
|
379
|
+
clearTimeout(flushTimer);
|
|
380
|
+
flushTimer = null;
|
|
381
|
+
}
|
|
200
382
|
status.error = `Failed to launch worker: ${err instanceof Error ? err.message : String(err)}`;
|
|
201
383
|
appendDiagnostic(root, viewId, { source: "runner", runId, level: "error", code: "worker_error", message: status.error, details: {} });
|
|
202
384
|
finalizeRun(status, { exitCode: 1, stoppedByUser }, Date.now());
|
|
203
385
|
finalizeEvidence(evidence, status, Date.now());
|
|
204
|
-
|
|
386
|
+
persistEvidenceArtifacts();
|
|
387
|
+
await finalizeThroughCoordinator({ exitCode: 1, stoppedByUser });
|
|
205
388
|
process.exit(1);
|
|
206
389
|
});
|
|
207
390
|
|
|
208
|
-
worker.on("close", (code) => {
|
|
391
|
+
worker.on("close", async (code) => {
|
|
209
392
|
if (buffer.trim()) onLine(buffer);
|
|
210
393
|
if (flushTimer) {
|
|
211
394
|
clearTimeout(flushTimer);
|
|
@@ -218,45 +401,67 @@ function main() {
|
|
|
218
401
|
// dashboard flips to its final state at once. Then try to classify the final
|
|
219
402
|
// bucket and upgrade the summary with cheap model passes. Slow/unreachable
|
|
220
403
|
// model calls must never stall the row indefinitely.
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
404
|
+
//
|
|
405
|
+
// Issue #91 (A8 path 3): the terminal status/state materialize through the
|
|
406
|
+
// View State Coordinator (finalizeThroughCoordinator) — only evidence
|
|
407
|
+
// artifacts are written directly here. The in-flight hot-path flush was
|
|
408
|
+
// cancelled above, so no throttled write can race the coordinator's patch.
|
|
409
|
+
persistEvidenceArtifacts();
|
|
410
|
+
await finalizeThroughCoordinator({ exitCode: code ?? 0, stoppedByUser });
|
|
411
|
+
applyHeuristicAutoState(config, status, evidence)
|
|
228
412
|
.then((changed) => {
|
|
229
413
|
if (changed) {
|
|
230
414
|
finalizeEvidence(evidence, status, Date.now());
|
|
231
|
-
|
|
232
|
-
|
|
415
|
+
if (coordinatorDisabled()) persistUnlessManual(true);
|
|
416
|
+
else refreshEvidenceMirrors();
|
|
233
417
|
}
|
|
234
|
-
return
|
|
418
|
+
return maybeModelAutoState(config, status, evidence);
|
|
235
419
|
})
|
|
236
420
|
.then((changed) => {
|
|
237
|
-
if (changed)
|
|
421
|
+
if (changed) {
|
|
422
|
+
finalizeEvidence(evidence, status, Date.now());
|
|
423
|
+
if (coordinatorDisabled()) persistUnlessManual(true);
|
|
424
|
+
else refreshEvidenceMirrors();
|
|
425
|
+
}
|
|
426
|
+
return maybeModelSummary(config, status);
|
|
427
|
+
})
|
|
428
|
+
.then(async (changed) => {
|
|
429
|
+
if (!changed) return;
|
|
430
|
+
if (coordinatorDisabled()) persistUnlessManual(true);
|
|
431
|
+
else await patchSummaryThroughCoordinator(config, status);
|
|
238
432
|
})
|
|
239
433
|
.catch(() => {})
|
|
240
|
-
.
|
|
434
|
+
.then(async () => {
|
|
241
435
|
// The finalize chain must never prevent process.exit: a lock/fs failure
|
|
242
|
-
// here used to pin the runner as a 100% CPU zombie (issue #33).
|
|
436
|
+
// here used to pin the runner as a 100% CPU zombie (issue #33). Both
|
|
437
|
+
// steps now await coordinator commands, so they run before the exit.
|
|
243
438
|
try {
|
|
244
|
-
finalizeSteeringIfNeeded(config, status, evidence);
|
|
439
|
+
await finalizeSteeringIfNeeded(config, status, evidence);
|
|
245
440
|
} catch (err) {
|
|
246
441
|
tryAppendDiagnostic(config, "finalize_steering_failed", err);
|
|
247
442
|
}
|
|
248
443
|
try {
|
|
249
|
-
drainQueuedFollowUp(config, status);
|
|
444
|
+
await drainQueuedFollowUp(config, status);
|
|
250
445
|
} catch (err) {
|
|
251
446
|
tryAppendDiagnostic(config, "follow_up_drain_failed", err);
|
|
252
447
|
}
|
|
448
|
+
})
|
|
449
|
+
.finally(() => {
|
|
253
450
|
process.exit(stoppedByUser ? 0 : (code ?? 0));
|
|
254
451
|
});
|
|
255
452
|
});
|
|
256
453
|
}
|
|
257
454
|
|
|
258
|
-
/**
|
|
259
|
-
|
|
455
|
+
/**
|
|
456
|
+
* Route the plan-ready row flip through the coordinator (`plan_ready`): the
|
|
457
|
+
* decision layer carries the exact legacy patch (needs_input/exited/"Approve
|
|
458
|
+
* this plan?") and its manual fence replaces the old file re-read guard.
|
|
459
|
+
* recordPlanReady's steering.json write STAYS direct — steering is not a
|
|
460
|
+
* coordinator artifact. The cheap pre-check remains as an optimization; the
|
|
461
|
+
* coordinator's manual_fence is authoritative. Async because the command must
|
|
462
|
+
* land before the exit-chain process.exit.
|
|
463
|
+
* @param {import("../src/core/types.mjs").RunConfig} config @param {import("../src/core/types.mjs").RunStatus} status @param {import("../src/core/types.mjs").EvidenceSnapshot} evidence */
|
|
464
|
+
async function finalizeSteeringIfNeeded(config, status, evidence) {
|
|
260
465
|
if (config.kind !== "plan" && config.kind !== "plan_change") return;
|
|
261
466
|
if (status.semanticState === "failed" || status.semanticState === "stopped") return;
|
|
262
467
|
// A manual completion racing the exit chain must not be resurrected for
|
|
@@ -267,21 +472,26 @@ function finalizeSteeringIfNeeded(config, status, evidence) {
|
|
|
267
472
|
runId: config.runId,
|
|
268
473
|
planText: latestEvidenceText(evidence) || status.latestAssistantPreview || status.summary || "Plan ready",
|
|
269
474
|
});
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
prev.processState = "exited";
|
|
274
|
-
prev.needsInput = true;
|
|
275
|
-
prev.question = "Approve this plan?";
|
|
276
|
-
prev.summary = "Plan ready for approval";
|
|
277
|
-
prev.currentRunId = config.runId;
|
|
278
|
-
prev.updatedAt = Date.now();
|
|
279
|
-
writeState(config.root, prev);
|
|
475
|
+
if (coordinatorDisabled()) {
|
|
476
|
+
legacyPlanReadyStateWrite(config.root, config.viewId, config.runId);
|
|
477
|
+
return;
|
|
280
478
|
}
|
|
479
|
+
const result = await sendStateCommand(config.root, {
|
|
480
|
+
type: "state_command",
|
|
481
|
+
viewId: config.viewId,
|
|
482
|
+
runId: config.runId,
|
|
483
|
+
source: "job-runner",
|
|
484
|
+
kind: "plan_ready",
|
|
485
|
+
expectedRevision: null,
|
|
486
|
+
payload: { runId: config.runId },
|
|
487
|
+
});
|
|
488
|
+
if (result.status === "applied") return;
|
|
489
|
+
if (result.reason === "manual_fence" || result.reason === "no_change") return;
|
|
490
|
+
appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, level: "warn", code: "plan_ready_command_ambiguous", message: `Plan-ready outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise dashboard reconcile will converge the row`, details: { reason: result.reason } });
|
|
281
491
|
}
|
|
282
492
|
|
|
283
493
|
/** @param {import("../src/core/types.mjs").RunConfig} config @param {import("../src/core/types.mjs").RunStatus} status */
|
|
284
|
-
function drainQueuedFollowUp(config, status) {
|
|
494
|
+
async function drainQueuedFollowUp(config, status) {
|
|
285
495
|
if (status.semanticState !== "idle" && status.semanticState !== "completed") return;
|
|
286
496
|
// A manual completion racing the exit chain must never be followed up: the
|
|
287
497
|
// user just finished this row, so don't launch a new run over it. The
|
|
@@ -302,8 +512,35 @@ function drainQueuedFollowUp(config, status) {
|
|
|
302
512
|
try {
|
|
303
513
|
const { pid } = launchRun(config.root, nextConfig, { runnerScript: fileURLToPath(import.meta.url) });
|
|
304
514
|
const nextStatus = createRunStatus(nextConfig, pid ?? null, Date.now());
|
|
305
|
-
|
|
306
|
-
|
|
515
|
+
// Bootstrap the follow-up run through the coordinator: command.runId is
|
|
516
|
+
// deliberately omitted (a parent-run runId would trip the generic stale-run
|
|
517
|
+
// guard against the just-finalized parent); payload.newRunId governs the
|
|
518
|
+
// state-side currentRunId. The new runner's own run_started then lands on
|
|
519
|
+
// top of this bootstrap with the real pid.
|
|
520
|
+
if (coordinatorDisabled()) {
|
|
521
|
+
legacyFollowupBootstrap(config.root, config.viewId, nextStatus);
|
|
522
|
+
} else {
|
|
523
|
+
const result = await sendStateCommand(config.root, {
|
|
524
|
+
type: "state_command",
|
|
525
|
+
viewId: config.viewId,
|
|
526
|
+
source: "job-runner",
|
|
527
|
+
kind: "followup_started",
|
|
528
|
+
expectedRevision: null,
|
|
529
|
+
payload: { newRunId: nextRunId, statusPatch: { ...nextStatus } },
|
|
530
|
+
});
|
|
531
|
+
if (result.reason === "manual_fence") {
|
|
532
|
+
// The user completed the row between the pre-check and this command.
|
|
533
|
+
// The fence preserved their verdict (legacy clobbered it); the child
|
|
534
|
+
// is already launched, so complete the item to avoid a double fire
|
|
535
|
+
// and surface the lost follow-up.
|
|
536
|
+
appendDiagnostic(config.root, config.viewId, { source: "queue", runId: nextRunId, level: "warn", code: "follow_up_fenced", message: "Manual completion fenced the follow-up bootstrap; the launched run continues but the row keeps its manual verdict", details: { kind: item.kind } });
|
|
537
|
+
completeFollowUp(config.root, config.viewId, item.id, { runId: nextRunId });
|
|
538
|
+
return;
|
|
539
|
+
}
|
|
540
|
+
if (result.status !== "applied" && result.reason !== "no_change" && result.reason !== "stale_run") {
|
|
541
|
+
appendDiagnostic(config.root, config.viewId, { source: "queue", runId: nextRunId, level: "warn", code: "follow_up_bootstrap_ambiguous", message: `Follow-up bootstrap outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise the launched runner's own run_started converges the row`, details: { reason: result.reason } });
|
|
542
|
+
}
|
|
543
|
+
}
|
|
307
544
|
completeFollowUp(config.root, config.viewId, item.id, { runId: nextRunId });
|
|
308
545
|
appendDiagnostic(config.root, config.viewId, { source: "queue", runId: nextRunId, code: "follow_up_started", message: "Queued follow-up started by JSON runner", details: { kind: item.kind } });
|
|
309
546
|
} catch (err) {
|
|
@@ -369,22 +606,59 @@ function canAutoState(config, status, evidence) {
|
|
|
369
606
|
return Boolean((latestEvidenceText(evidence) || status.latestAssistantPreview || status.summary || "").trim());
|
|
370
607
|
}
|
|
371
608
|
|
|
372
|
-
|
|
609
|
+
/**
|
|
610
|
+
* Submit one classification to the View State Coordinator (issue #91, A8 path 2).
|
|
611
|
+
* The coordinator owns semantic state: applied patches are materialized by it and
|
|
612
|
+
* this runner only refreshes its in-memory status from disk so any remaining
|
|
613
|
+
* direct persist (PR #1 hot path) starts from authoritative fields. Designed
|
|
614
|
+
* fences (manual_fence / no_change / stale_run) are informational, not errors.
|
|
615
|
+
* Ambiguous transport outcomes (timeout / connection_reset) never fall back to a
|
|
616
|
+
* direct write — the command may already be journaled, and the coordinator's
|
|
617
|
+
* boot replay is the recovery path.
|
|
618
|
+
* @param {import("../src/core/types.mjs").RunConfig} config
|
|
619
|
+
* @param {import("../src/core/types.mjs").RunStatus} status mutated in place on apply (fresh coordinator fields)
|
|
620
|
+
* @param {import("../src/core/types.mjs").AutoStateClassification} classification
|
|
621
|
+
* @returns {Promise<boolean>} whether the classification was applied
|
|
622
|
+
*/
|
|
623
|
+
async function classifyThroughCoordinator(config, status, classification) {
|
|
624
|
+
const result = await sendStateCommand(config.root, {
|
|
625
|
+
type: "state_command",
|
|
626
|
+
viewId: config.viewId,
|
|
627
|
+
runId: config.runId,
|
|
628
|
+
source: "job-runner",
|
|
629
|
+
kind: "auto_state_classified",
|
|
630
|
+
expectedRevision: null,
|
|
631
|
+
payload: { classification },
|
|
632
|
+
});
|
|
633
|
+
if (result.status === "applied") {
|
|
634
|
+
appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated terminal state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
|
|
635
|
+
const fresh = readStatus(config.root, config.viewId, config.runId);
|
|
636
|
+
if (fresh) Object.assign(status, fresh);
|
|
637
|
+
return true;
|
|
638
|
+
}
|
|
639
|
+
if (result.reason === "coordinator_disabled") {
|
|
640
|
+
// Legacy escape hatch (AGENT_BOARD_COORDINATOR=off): apply locally; the
|
|
641
|
+
// caller's persistUnlessManual keeps the pre-coordinator fence for this path.
|
|
642
|
+
return applyAutoStateToStatus(status, classification, Date.now());
|
|
643
|
+
}
|
|
644
|
+
if (result.reason === "manual_fence" || result.reason === "no_change" || result.reason === "stale_run") {
|
|
645
|
+
// Designed fences — informational, not errors.
|
|
646
|
+
return false;
|
|
647
|
+
}
|
|
648
|
+
appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, level: "warn", code: "auto_state_command_ambiguous", message: `Auto-state classification outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise the next classification pass will converge the row`, details: { reason: result.reason } });
|
|
649
|
+
return false;
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
async function applyHeuristicAutoState(config, status, evidence) {
|
|
373
653
|
if (!canAutoState(config, status, evidence)) return false;
|
|
374
|
-
//
|
|
375
|
-
//
|
|
376
|
-
//
|
|
377
|
-
// the completed+null pair. If the user marked the row done while the worker
|
|
378
|
-
// was exiting, skip classification so the persist path can't clobber it.
|
|
654
|
+
// Cheap pre-check kept as an optimization (avoids a pointless command);
|
|
655
|
+
// correctness no longer depends on it — the coordinator fences manual
|
|
656
|
+
// completions authoritatively (manual_fence).
|
|
379
657
|
const latestState = readState(config.root, config.viewId);
|
|
380
658
|
if (isManualCompletion(latestState)) return false;
|
|
381
659
|
const latest = latestEvidenceText(evidence) || status.latestAssistantPreview || status.summary || "";
|
|
382
660
|
const classification = heuristicAutoState(latest, { lastAgentActivityAt: status.lastAgentActivityAt ?? null });
|
|
383
|
-
|
|
384
|
-
if (changed) {
|
|
385
|
-
appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated terminal state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
|
|
386
|
-
}
|
|
387
|
-
return changed;
|
|
661
|
+
return classifyThroughCoordinator(config, status, classification);
|
|
388
662
|
}
|
|
389
663
|
|
|
390
664
|
async function maybeModelAutoState(config, status, evidence) {
|
|
@@ -398,19 +672,14 @@ async function maybeModelAutoState(config, status, evidence) {
|
|
|
398
672
|
[...config.piArgsPrefix, "--mode", "json", "-p", "--no-session", "--model", model, prompt],
|
|
399
673
|
15000,
|
|
400
674
|
);
|
|
401
|
-
//
|
|
402
|
-
//
|
|
403
|
-
//
|
|
404
|
-
// in-memory status would clobber the user's verdict.
|
|
675
|
+
// The user may have marked the row done manually during the model call. The
|
|
676
|
+
// cheap pre-check avoids a pointless command; the coordinator's manual_fence
|
|
677
|
+
// is the authoritative guard for races after this read.
|
|
405
678
|
const fresh = readStatus(config.root, config.viewId, config.runId);
|
|
406
679
|
if (!fresh || isManualCompletion(fresh)) return false;
|
|
407
680
|
Object.assign(status, fresh);
|
|
408
681
|
const classification = autoStateFromModelOrHeuristic(out, latest, { lastAgentActivityAt: status.lastAgentActivityAt ?? null });
|
|
409
|
-
|
|
410
|
-
if (changed) {
|
|
411
|
-
appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier refined terminal state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
|
|
412
|
-
}
|
|
413
|
-
return changed;
|
|
682
|
+
return classifyThroughCoordinator(config, status, classification);
|
|
414
683
|
}
|
|
415
684
|
|
|
416
685
|
/** Default cheap model for terminal summaries. Override/disable via $AGENT_BOARD_SUMMARY_MODEL. */
|
|
@@ -450,6 +719,40 @@ async function maybeModelSummary(config, status) {
|
|
|
450
719
|
return false;
|
|
451
720
|
}
|
|
452
721
|
|
|
722
|
+
/**
|
|
723
|
+
* Route the post-exit model-summary upgrade through the coordinator as a
|
|
724
|
+
* `patch_fields` command (summary + latestAssistantPreview are whitelisted for
|
|
725
|
+
* the job-runner source). The generic manual fence replaces the old
|
|
726
|
+
* persistUnlessManual file re-read. runId stays null: the finished run's
|
|
727
|
+
* currentRunId still points at it, so the coordinator binds the status patch
|
|
728
|
+
* to the right file without tripping the stale-run guard.
|
|
729
|
+
* @param {import("../src/core/types.mjs").RunConfig} config
|
|
730
|
+
* @param {import("../src/core/types.mjs").RunStatus} status mutated in place on apply
|
|
731
|
+
* @returns {Promise<boolean>} whether the summary patch was applied
|
|
732
|
+
*/
|
|
733
|
+
async function patchSummaryThroughCoordinator(config, status) {
|
|
734
|
+
const result = await sendStateCommand(config.root, {
|
|
735
|
+
type: "state_command",
|
|
736
|
+
viewId: config.viewId,
|
|
737
|
+
runId: null,
|
|
738
|
+
source: "job-runner",
|
|
739
|
+
kind: "patch_fields",
|
|
740
|
+
expectedRevision: null,
|
|
741
|
+
payload: {
|
|
742
|
+
state: { summary: status.summary, latestAssistantPreview: status.latestAssistantPreview },
|
|
743
|
+
status: { summary: status.summary },
|
|
744
|
+
},
|
|
745
|
+
});
|
|
746
|
+
if (result.status === "applied") {
|
|
747
|
+
const fresh = readStatus(config.root, config.viewId, config.runId);
|
|
748
|
+
if (fresh) Object.assign(status, fresh);
|
|
749
|
+
return true;
|
|
750
|
+
}
|
|
751
|
+
if (result.reason === "manual_fence" || result.reason === "no_change") return false;
|
|
752
|
+
appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, level: "warn", code: "summary_patch_command_ambiguous", message: `Summary patch outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise dashboard reconcile will converge the row`, details: { reason: result.reason } });
|
|
753
|
+
return false;
|
|
754
|
+
}
|
|
755
|
+
|
|
453
756
|
/**
|
|
454
757
|
* Run a pi one-shot and return the concatenated assistant text from message_end events.
|
|
455
758
|
* @param {string} command
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pre-coordinator direct write for host-failure row finalization (issue #91).
|
|
3
|
+
*
|
|
4
|
+
* Only reachable via `AGENT_BOARD_COORDINATOR=off` — the documented escape
|
|
5
|
+
* hatch. The normal path routes `host_run_failed` through the View State
|
|
6
|
+
* Coordinator (`runner/state-coordinator.mjs`), whose manual_fence / stale_run
|
|
7
|
+
* guards own the decision; this direct write has NO manual-completion fence,
|
|
8
|
+
* which is exactly why it must stay unreachable in the default configuration.
|
|
9
|
+
*
|
|
10
|
+
* Lives in its own module so `runner/pty-runner.mjs` itself never imports the
|
|
11
|
+
* state materializers (writer-boundary test, spec D3). `writeHost` is a
|
|
12
|
+
* different artifact: host.json is owned by the pty-runner per spec D3.
|
|
13
|
+
*/
|
|
14
|
+
import { readState, writeState } from "../src/core/store.mjs";
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Legacy direct write of the view-failed row (pre-coordinator markRowFailed).
|
|
18
|
+
* @param {string} root
|
|
19
|
+
* @param {string} viewId
|
|
20
|
+
* @param {string} message
|
|
21
|
+
*/
|
|
22
|
+
export function markRowFailedDirect(root, viewId, message) {
|
|
23
|
+
const now = Date.now();
|
|
24
|
+
const state = readState(root, viewId) ?? {
|
|
25
|
+
version: 1,
|
|
26
|
+
viewId,
|
|
27
|
+
currentRunId: null,
|
|
28
|
+
semanticState: "queued",
|
|
29
|
+
processState: "exited",
|
|
30
|
+
summary: "Queued",
|
|
31
|
+
lastActivityAt: now,
|
|
32
|
+
updatedAt: now,
|
|
33
|
+
needsInput: false,
|
|
34
|
+
hasError: false,
|
|
35
|
+
latestAssistantPreview: "",
|
|
36
|
+
latestTool: null,
|
|
37
|
+
question: null,
|
|
38
|
+
pendingQuestions: [],
|
|
39
|
+
error: null,
|
|
40
|
+
};
|
|
41
|
+
state.semanticState = "failed";
|
|
42
|
+
state.processState = "exited";
|
|
43
|
+
state.summary = message;
|
|
44
|
+
state.hasError = true;
|
|
45
|
+
state.needsInput = false;
|
|
46
|
+
state.error = message;
|
|
47
|
+
state.updatedAt = now;
|
|
48
|
+
state.lastActivityAt = now;
|
|
49
|
+
writeState(root, state);
|
|
50
|
+
}
|