@zhuxixi/pi-agent-board 0.6.2 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,13 +5,18 @@
5
5
  * Usage: node job-runner.mjs <configPath>
6
6
  *
7
7
  * Owns one run: spawns a headless Pi worker (`pi --mode json -p --session <file> <prompt>`),
8
- * streams its JSON events into events.jsonl, reduces them into status.json + the row's
9
- * state.json, and finalizes on exit. Survives the parent Pi process exiting/reloading.
8
+ * streams its JSON events into events.jsonl, and routes every semantic-state
9
+ * mutation through the View State Coordinator as commands (issue #91): boot
10
+ * run_started, transient run_progress beats on the throttled hot path,
11
+ * auto-state classifications, run_finalized on exit, and the post-exit
12
+ * patch_fields (evidence mirrors / model summary). Evidence artifacts
13
+ * (events/stdout/stderr logs, evidence files, code-refs) stay runner-owned and
14
+ * are written directly. Survives the parent Pi process exiting/reloading.
10
15
  */
11
16
  import { spawn } from "node:child_process";
12
17
  import { fileURLToPath } from "node:url";
13
18
  import { appendLine, readJson } from "../src/core/atomic.mjs";
14
- import { createRunStatus, finalizeRun, projectViewState, reduceEvent } from "../src/core/events.mjs";
19
+ import { createRunStatus, finalizeRun, reduceEvent } from "../src/core/events.mjs";
15
20
  import { encodePromptForCliArg } from "../src/core/prompt-transport.mjs";
16
21
  import { applyAutoStateToStatus, autoStateEnabled, autoStateFromModelOrHeuristic, autoStateModel, buildAutoStatePrompt, heuristicAutoState, isManualCompletion } from "../src/core/auto-state.mjs";
17
22
  import { appendDiagnostic } from "../src/core/diagnostics.mjs";
@@ -21,8 +26,10 @@ import { claimNextFollowUp, completeFollowUp, releaseFollowUp } from "../src/cor
21
26
  import { newRunId } from "../src/core/ids.mjs";
22
27
  import { launchRun } from "../src/core/launch.mjs";
23
28
  import * as P from "../src/core/paths.mjs";
24
- import { readState, readStatus, readMeta, writeState, writeStatus } from "../src/core/store.mjs";
29
+ import { readState, readStatus, readMeta } from "../src/core/store.mjs";
25
30
  import { readSteering, recordPlanReady } from "../src/core/steering.mjs";
31
+ import { sendStateCommand, coordinatorDisabled } from "../src/core/coordinator-client.mjs";
32
+ import { legacyFollowupBootstrap, legacyPersistState, legacyPlanReadyStateWrite, legacyWriteState, legacyWriteStatus } from "./job-runner-legacy.mjs";
26
33
  import { buildApprovePlanPrompt, buildPlanChangesPrompt, buildPlanRequestPrompt } from "../src/core/steering-prompts.mjs";
27
34
 
28
35
  const WRITE_THROTTLE_MS = 250;
@@ -59,13 +66,70 @@ function main() {
59
66
  const meta = readMeta(root, viewId);
60
67
 
61
68
  let status = createRunStatus(config, null, Date.now());
69
+ // The run starts working the moment the runner is up; seeding the boot status
70
+ // with "working" (instead of createRunStatus's "queued") avoids the stale
71
+ // queued frame flickering through run_started's first materialization.
72
+ status.semanticState = "working";
62
73
  let evidence = emptyEvidenceSnapshot({ viewId, runId, source: "json-runner" });
63
74
  appendDiagnostic(root, viewId, { source: "runner", runId, code: "runner_start", message: "Runner started", details: { kind: config.kind, cwd: config.cwd, model: config.model } });
64
- writeStatus(root, status);
65
75
  writeRunEvidence(root, evidence);
66
76
  writeEvidence(root, evidence);
67
77
  updateCodeRefsFromEvidence(root, viewId, evidence, meta);
68
- writeState(root, projectViewState(status, Date.now(), readState(root, viewId)));
78
+ if (coordinatorDisabled()) {
79
+ legacyPersistState({ root, viewId, runId, status });
80
+ }
81
+ return bootstrapRun({ root, viewId, runId, config, status, meta, evidence, stdoutLog, stderrLog, eventsLog });
82
+ }
83
+
84
+ /**
85
+ * Async continuation of main: route the run_started bootstrap through the
86
+ * coordinator (the status file it creates is what every later run_progress
87
+ * beat patches onto), then spawn the worker and wire the event handlers.
88
+ * @param {{ root: string, viewId: string, runId: string, config: object, status: object, meta: object, evidence: object, stdoutLog: string, stderrLog: string, eventsLog: string }} ctx
89
+ */
90
+ async function bootstrapRun({ root, viewId, runId, config, status, meta, evidence, stdoutLog, stderrLog, eventsLog }) {
91
+ if (!coordinatorDisabled()) {
92
+ // Journaled bootstrap: creates the run's status.json (STATUS_BOOTSTRAP_KINDS)
93
+ // and pins the row to working/alive/currentRunId. Ambiguous outcomes never
94
+ // block the run: if the command was journaled, coordinator replay recovers
95
+ // it; otherwise dashboard reconcile converges the row.
96
+ const started = await sendStateCommand(root, {
97
+ type: "state_command",
98
+ viewId,
99
+ runId,
100
+ source: "job-runner",
101
+ kind: "run_started",
102
+ expectedRevision: null,
103
+ payload: { status: { ...status } },
104
+ });
105
+ if (started.status !== "applied" && started.reason !== "coordinator_disabled") {
106
+ appendDiagnostic(root, viewId, { source: "runner", runId, level: "warn", code: "run_started_command_ambiguous", message: `Run bootstrap outcome unknown (${started.reason}); if the command was journaled, coordinator replay will recover it; otherwise dashboard reconcile will converge the row`, details: { reason: started.reason } });
107
+ }
108
+ }
109
+
110
+ /**
111
+ * One transient progress beat: ship the full in-memory status as a sparse
112
+ * patch — the coordinator merges it onto the materialized status and lets
113
+ * projectViewState recompute the row state (delegation, not copied rules).
114
+ *
115
+ * Fire-and-forget by design (the ONLY such command in the runner): the hot
116
+ * path is a periodic self-healing snapshot (~4/sec), a lost beat is
117
+ * superseded by the next one, and failures are silently ignored — logging
118
+ * them would spam diagnostics at beat frequency. Transient commands are
119
+ * never journaled, so an ambiguous outcome has no replay to await.
120
+ */
121
+ const sendProgressBeat = () => {
122
+ const { materializedRevision: _fileStamp, ...patch } = status;
123
+ void sendStateCommand(root, {
124
+ type: "state_command",
125
+ viewId,
126
+ runId,
127
+ source: "job-runner",
128
+ kind: "run_progress",
129
+ expectedRevision: null,
130
+ payload: { statusPatch: patch },
131
+ }).catch(() => {});
132
+ };
69
133
 
70
134
  // Build worker args: pi --mode json -p --session <file> [--model m] [--thinking l] [--tools t] <prompt>
71
135
  const args = [
@@ -91,7 +155,9 @@ function main() {
91
155
 
92
156
  status.pid = worker.pid ?? null;
93
157
  appendDiagnostic(root, viewId, { source: "runner", runId, code: "worker_pid", message: "Worker pid recorded", details: { pid: status.pid } });
94
- writeStatus(root, status);
158
+ // The pid lands on disk via a transient progress beat (the worker's first
159
+ // events would carry it too, but a silent worker must still be observable).
160
+ sendProgressBeat();
95
161
 
96
162
  let stoppedByUser = false;
97
163
  let dirty = false;
@@ -99,15 +165,12 @@ function main() {
99
165
 
100
166
  const persist = (force = false) => {
101
167
  void force;
102
- const now = Date.now();
103
168
  status.evidenceSummary = summarizeEvidence(evidence);
104
- writeStatus(root, status);
105
- writeRunEvidence(root, evidence);
106
- writeEvidence(root, evidence);
107
- writeState(root, projectViewState(status, now, readState(root, viewId)));
169
+ persistEvidenceArtifacts();
170
+ if (coordinatorDisabled()) legacyPersistState({ root, viewId, runId, status });
171
+ else sendProgressBeat();
108
172
  // Best-effort code-refs extraction shells out to git and can take hundreds of
109
- // ms; run it after the state write so endedAt-visible state converges first.
110
- // The extraction only depends on evidence + git, never on state.json.
173
+ // ms; it only depends on evidence + git, never on state.json.
111
174
  updateCodeRefsFromEvidence(root, viewId, evidence, meta);
112
175
  dirty = false;
113
176
  };
@@ -125,6 +188,116 @@ function main() {
125
188
  return true;
126
189
  };
127
190
 
191
+ /**
192
+ * Runner-owned evidence artifacts (evidence files + code-refs). Written
193
+ * directly — the coordinator owns state.json/status.json, not these.
194
+ */
195
+ const persistEvidenceArtifacts = () => {
196
+ status.evidenceSummary = summarizeEvidence(evidence);
197
+ writeRunEvidence(root, evidence);
198
+ writeEvidence(root, evidence);
199
+ updateCodeRefsFromEvidence(root, viewId, evidence, meta);
200
+ };
201
+
202
+ /**
203
+ * Refresh the evidence mirrors (status.evidenceSummary / state.review)
204
+ * through the coordinator: `patch_fields` carries only whitelisted mirror
205
+ * fields, and the coordinator's generic manual fence (source != user on a
206
+ * manually-completed row) is the authoritative guard — the old fresh-read
207
+ * + isManualCompletion pre-checks are no longer needed. Designed fences
208
+ * (manual_fence / no_change) are informational; ambiguous outcomes never
209
+ * fall back to a direct write.
210
+ */
211
+ const refreshEvidenceMirrors = async () => {
212
+ status.evidenceSummary = summarizeEvidence(evidence);
213
+ const result = await sendStateCommand(root, {
214
+ type: "state_command",
215
+ viewId,
216
+ runId: null,
217
+ source: "job-runner",
218
+ kind: "patch_fields",
219
+ expectedRevision: null,
220
+ payload: {
221
+ state: { review: status.evidenceSummary },
222
+ status: { evidenceSummary: status.evidenceSummary },
223
+ },
224
+ });
225
+ if (result.status === "applied") {
226
+ const fresh = readStatus(root, viewId, runId);
227
+ if (fresh) Object.assign(status, fresh);
228
+ return true;
229
+ }
230
+ if (result.reason === "manual_fence" || result.reason === "no_change") return false;
231
+ if (result.reason === "coordinator_disabled") {
232
+ // Legacy escape hatch: fresh-read + fence, pre-coordinator semantics.
233
+ const freshStatus = readStatus(root, viewId, runId);
234
+ if (freshStatus && !isManualCompletion(freshStatus)) {
235
+ freshStatus.evidenceSummary = status.evidenceSummary;
236
+ legacyWriteStatus(root, viewId, runId, freshStatus);
237
+ }
238
+ const freshState = readState(root, viewId);
239
+ if (freshState && !isManualCompletion(freshState)) {
240
+ freshState.review = status.evidenceSummary;
241
+ legacyWriteState(root, viewId, freshState);
242
+ }
243
+ return true;
244
+ }
245
+ appendDiagnostic(root, viewId, { source: "runner", runId, level: "warn", code: "evidence_mirror_command_ambiguous", message: `Evidence mirror outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise the next mirror refresh will converge`, details: { reason: result.reason } });
246
+ return false;
247
+ };
248
+
249
+ /**
250
+ * Materialize the run's terminal state through the View State Coordinator
251
+ * (issue #91, A8 path 3): the runner submits minimal facts and the
252
+ * coordinator computes terminal semantics via finalizeRun, so a manual
253
+ * completion landing before the command is fenced by the coordinator
254
+ * (manual_fence), not by a file re-read (#46 class). Evidence artifacts stay
255
+ * direct (runner-owned). Ambiguous outcomes (timeout / connection_reset)
256
+ * NEVER fall back to a direct write: the command may already be journaled,
257
+ * and the coordinator's boot replay is the recovery path.
258
+ * coordinator_disabled keeps the pre-coordinator direct persist.
259
+ * @param {{ exitCode: number|null, stoppedByUser: boolean }} facts
260
+ * @returns {Promise<boolean>} whether the final state is known materialized
261
+ */
262
+ const finalizeThroughCoordinator = async ({ exitCode, stoppedByUser: stopped }) => {
263
+ const payload = { exitCode, stoppedByUser: stopped };
264
+ if (status.endedAt != null) payload.endedAt = status.endedAt;
265
+ // The close path cancels the pending throttled flush after the final
266
+ // buffer flush, so a stopReason observed in the last burst (reduceEvent
267
+ // sets it in memory only) never reached disk. Overlay it onto the payload:
268
+ // finalizeSemanticState keys on stopReason alone for exit-0 exits, and the
269
+ // coordinator already supports the payload overlay (issue #91).
270
+ if (status.stopReason != null) payload.stopReason = status.stopReason;
271
+ if (status.latestAssistantPreview) payload.latestAssistantPreview = status.latestAssistantPreview;
272
+ if (status.lastAgentActivityAt != null) payload.lastAgentActivityAt = status.lastAgentActivityAt;
273
+ const result = await sendStateCommand(root, {
274
+ type: "state_command",
275
+ viewId,
276
+ runId,
277
+ source: "job-runner",
278
+ kind: "run_finalized",
279
+ expectedRevision: null,
280
+ payload,
281
+ });
282
+ if (result.status === "applied") {
283
+ const fresh = readStatus(root, viewId, runId);
284
+ if (fresh) Object.assign(status, fresh);
285
+ await refreshEvidenceMirrors();
286
+ return true;
287
+ }
288
+ if (result.reason === "coordinator_disabled") {
289
+ persistEvidenceArtifacts();
290
+ legacyPersistState({ root, viewId, runId, status });
291
+ return true;
292
+ }
293
+ if (result.reason === "stale_run") {
294
+ // Duplicate finalize or the run was already superseded — nothing to do.
295
+ return false;
296
+ }
297
+ appendDiagnostic(root, viewId, { source: "runner", runId, level: "warn", code: "run_finalize_command_ambiguous", message: `Run finalization outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise dashboard reconcile will converge the row`, details: { reason: result.reason } });
298
+ return false;
299
+ };
300
+
128
301
  const scheduleFlush = () => {
129
302
  if (flushTimer) {
130
303
  dirty = true;
@@ -196,16 +369,26 @@ function main() {
196
369
  process.on("SIGTERM", stop);
197
370
  process.on("SIGINT", stop);
198
371
 
199
- worker.on("error", (err) => {
372
+ worker.on("error", async (err) => {
373
+ // Cancel any in-flight throttled flush before finalizing: if the timer
374
+ // callback lands during the sendStateCommand await below, the stale
375
+ // persist() would overwrite the coordinator-materialized terminal state
376
+ // and the close-path run_finalized would then bounce as stale_run
377
+ // (symmetric with the close path's cancel; issue #91 fix round 1).
378
+ if (flushTimer) {
379
+ clearTimeout(flushTimer);
380
+ flushTimer = null;
381
+ }
200
382
  status.error = `Failed to launch worker: ${err instanceof Error ? err.message : String(err)}`;
201
383
  appendDiagnostic(root, viewId, { source: "runner", runId, level: "error", code: "worker_error", message: status.error, details: {} });
202
384
  finalizeRun(status, { exitCode: 1, stoppedByUser }, Date.now());
203
385
  finalizeEvidence(evidence, status, Date.now());
204
- persist(true);
386
+ persistEvidenceArtifacts();
387
+ await finalizeThroughCoordinator({ exitCode: 1, stoppedByUser });
205
388
  process.exit(1);
206
389
  });
207
390
 
208
- worker.on("close", (code) => {
391
+ worker.on("close", async (code) => {
209
392
  if (buffer.trim()) onLine(buffer);
210
393
  if (flushTimer) {
211
394
  clearTimeout(flushTimer);
@@ -218,45 +401,67 @@ function main() {
218
401
  // dashboard flips to its final state at once. Then try to classify the final
219
402
  // bucket and upgrade the summary with cheap model passes. Slow/unreachable
220
403
  // model calls must never stall the row indefinitely.
221
- persist(true);
222
- if (applyHeuristicAutoState(config, status, evidence)) {
223
- finalizeEvidence(evidence, status, Date.now());
224
- status.evidenceSummary = summarizeEvidence(evidence);
225
- persistUnlessManual(true);
226
- }
227
- maybeModelAutoState(config, status, evidence)
404
+ //
405
+ // Issue #91 (A8 path 3): the terminal status/state materialize through the
406
+ // View State Coordinator (finalizeThroughCoordinator) — only evidence
407
+ // artifacts are written directly here. The in-flight hot-path flush was
408
+ // cancelled above, so no throttled write can race the coordinator's patch.
409
+ persistEvidenceArtifacts();
410
+ await finalizeThroughCoordinator({ exitCode: code ?? 0, stoppedByUser });
411
+ applyHeuristicAutoState(config, status, evidence)
228
412
  .then((changed) => {
229
413
  if (changed) {
230
414
  finalizeEvidence(evidence, status, Date.now());
231
- status.evidenceSummary = summarizeEvidence(evidence);
232
- persistUnlessManual(true);
415
+ if (coordinatorDisabled()) persistUnlessManual(true);
416
+ else refreshEvidenceMirrors();
233
417
  }
234
- return maybeModelSummary(config, status);
418
+ return maybeModelAutoState(config, status, evidence);
235
419
  })
236
420
  .then((changed) => {
237
- if (changed) persistUnlessManual(true);
421
+ if (changed) {
422
+ finalizeEvidence(evidence, status, Date.now());
423
+ if (coordinatorDisabled()) persistUnlessManual(true);
424
+ else refreshEvidenceMirrors();
425
+ }
426
+ return maybeModelSummary(config, status);
427
+ })
428
+ .then(async (changed) => {
429
+ if (!changed) return;
430
+ if (coordinatorDisabled()) persistUnlessManual(true);
431
+ else await patchSummaryThroughCoordinator(config, status);
238
432
  })
239
433
  .catch(() => {})
240
- .finally(() => {
434
+ .then(async () => {
241
435
  // The finalize chain must never prevent process.exit: a lock/fs failure
242
- // here used to pin the runner as a 100% CPU zombie (issue #33).
436
+ // here used to pin the runner as a 100% CPU zombie (issue #33). Both
437
+ // steps now await coordinator commands, so they run before the exit.
243
438
  try {
244
- finalizeSteeringIfNeeded(config, status, evidence);
439
+ await finalizeSteeringIfNeeded(config, status, evidence);
245
440
  } catch (err) {
246
441
  tryAppendDiagnostic(config, "finalize_steering_failed", err);
247
442
  }
248
443
  try {
249
- drainQueuedFollowUp(config, status);
444
+ await drainQueuedFollowUp(config, status);
250
445
  } catch (err) {
251
446
  tryAppendDiagnostic(config, "follow_up_drain_failed", err);
252
447
  }
448
+ })
449
+ .finally(() => {
253
450
  process.exit(stoppedByUser ? 0 : (code ?? 0));
254
451
  });
255
452
  });
256
453
  }
257
454
 
258
- /** @param {import("../src/core/types.mjs").RunConfig} config @param {import("../src/core/types.mjs").RunStatus} status @param {import("../src/core/types.mjs").EvidenceSnapshot} evidence */
259
- function finalizeSteeringIfNeeded(config, status, evidence) {
455
+ /**
456
+ * Route the plan-ready row flip through the coordinator (`plan_ready`): the
457
+ * decision layer carries the exact legacy patch (needs_input/exited/"Approve
458
+ * this plan?") and its manual fence replaces the old file re-read guard.
459
+ * recordPlanReady's steering.json write STAYS direct — steering is not a
460
+ * coordinator artifact. The cheap pre-check remains as an optimization; the
461
+ * coordinator's manual_fence is authoritative. Async because the command must
462
+ * land before the exit-chain process.exit.
463
+ * @param {import("../src/core/types.mjs").RunConfig} config @param {import("../src/core/types.mjs").RunStatus} status @param {import("../src/core/types.mjs").EvidenceSnapshot} evidence */
464
+ async function finalizeSteeringIfNeeded(config, status, evidence) {
260
465
  if (config.kind !== "plan" && config.kind !== "plan_change") return;
261
466
  if (status.semanticState === "failed" || status.semanticState === "stopped") return;
262
467
  // A manual completion racing the exit chain must not be resurrected for
@@ -267,21 +472,26 @@ function finalizeSteeringIfNeeded(config, status, evidence) {
267
472
  runId: config.runId,
268
473
  planText: latestEvidenceText(evidence) || status.latestAssistantPreview || status.summary || "Plan ready",
269
474
  });
270
- const prev = readState(config.root, config.viewId);
271
- if (prev) {
272
- prev.semanticState = "needs_input";
273
- prev.processState = "exited";
274
- prev.needsInput = true;
275
- prev.question = "Approve this plan?";
276
- prev.summary = "Plan ready for approval";
277
- prev.currentRunId = config.runId;
278
- prev.updatedAt = Date.now();
279
- writeState(config.root, prev);
475
+ if (coordinatorDisabled()) {
476
+ legacyPlanReadyStateWrite(config.root, config.viewId, config.runId);
477
+ return;
280
478
  }
479
+ const result = await sendStateCommand(config.root, {
480
+ type: "state_command",
481
+ viewId: config.viewId,
482
+ runId: config.runId,
483
+ source: "job-runner",
484
+ kind: "plan_ready",
485
+ expectedRevision: null,
486
+ payload: { runId: config.runId },
487
+ });
488
+ if (result.status === "applied") return;
489
+ if (result.reason === "manual_fence" || result.reason === "no_change") return;
490
+ appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, level: "warn", code: "plan_ready_command_ambiguous", message: `Plan-ready outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise dashboard reconcile will converge the row`, details: { reason: result.reason } });
281
491
  }
282
492
 
283
493
  /** @param {import("../src/core/types.mjs").RunConfig} config @param {import("../src/core/types.mjs").RunStatus} status */
284
- function drainQueuedFollowUp(config, status) {
494
+ async function drainQueuedFollowUp(config, status) {
285
495
  if (status.semanticState !== "idle" && status.semanticState !== "completed") return;
286
496
  // A manual completion racing the exit chain must never be followed up: the
287
497
  // user just finished this row, so don't launch a new run over it. The
@@ -302,8 +512,35 @@ function drainQueuedFollowUp(config, status) {
302
512
  try {
303
513
  const { pid } = launchRun(config.root, nextConfig, { runnerScript: fileURLToPath(import.meta.url) });
304
514
  const nextStatus = createRunStatus(nextConfig, pid ?? null, Date.now());
305
- writeStatus(config.root, nextStatus);
306
- writeState(config.root, projectViewState(nextStatus, Date.now(), readState(config.root, config.viewId)));
515
+ // Bootstrap the follow-up run through the coordinator: command.runId is
516
+ // deliberately omitted (a parent-run runId would trip the generic stale-run
517
+ // guard against the just-finalized parent); payload.newRunId governs the
518
+ // state-side currentRunId. The new runner's own run_started then lands on
519
+ // top of this bootstrap with the real pid.
520
+ if (coordinatorDisabled()) {
521
+ legacyFollowupBootstrap(config.root, config.viewId, nextStatus);
522
+ } else {
523
+ const result = await sendStateCommand(config.root, {
524
+ type: "state_command",
525
+ viewId: config.viewId,
526
+ source: "job-runner",
527
+ kind: "followup_started",
528
+ expectedRevision: null,
529
+ payload: { newRunId: nextRunId, statusPatch: { ...nextStatus } },
530
+ });
531
+ if (result.reason === "manual_fence") {
532
+ // The user completed the row between the pre-check and this command.
533
+ // The fence preserved their verdict (legacy clobbered it); the child
534
+ // is already launched, so complete the item to avoid a double fire
535
+ // and surface the lost follow-up.
536
+ appendDiagnostic(config.root, config.viewId, { source: "queue", runId: nextRunId, level: "warn", code: "follow_up_fenced", message: "Manual completion fenced the follow-up bootstrap; the launched run continues but the row keeps its manual verdict", details: { kind: item.kind } });
537
+ completeFollowUp(config.root, config.viewId, item.id, { runId: nextRunId });
538
+ return;
539
+ }
540
+ if (result.status !== "applied" && result.reason !== "no_change" && result.reason !== "stale_run") {
541
+ appendDiagnostic(config.root, config.viewId, { source: "queue", runId: nextRunId, level: "warn", code: "follow_up_bootstrap_ambiguous", message: `Follow-up bootstrap outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise the launched runner's own run_started converges the row`, details: { reason: result.reason } });
542
+ }
543
+ }
307
544
  completeFollowUp(config.root, config.viewId, item.id, { runId: nextRunId });
308
545
  appendDiagnostic(config.root, config.viewId, { source: "queue", runId: nextRunId, code: "follow_up_started", message: "Queued follow-up started by JSON runner", details: { kind: item.kind } });
309
546
  } catch (err) {
@@ -369,22 +606,59 @@ function canAutoState(config, status, evidence) {
369
606
  return Boolean((latestEvidenceText(evidence) || status.latestAssistantPreview || status.summary || "").trim());
370
607
  }
371
608
 
372
- function applyHeuristicAutoState(config, status, evidence) {
609
+ /**
610
+ * Submit one classification to the View State Coordinator (issue #91, A8 path 2).
611
+ * The coordinator owns semantic state: applied patches are materialized by it and
612
+ * this runner only refreshes its in-memory status from disk so any remaining
613
+ * direct persist (PR #1 hot path) starts from authoritative fields. Designed
614
+ * fences (manual_fence / no_change / stale_run) are informational, not errors.
615
+ * Ambiguous transport outcomes (timeout / connection_reset) never fall back to a
616
+ * direct write — the command may already be journaled, and the coordinator's
617
+ * boot replay is the recovery path.
618
+ * @param {import("../src/core/types.mjs").RunConfig} config
619
+ * @param {import("../src/core/types.mjs").RunStatus} status mutated in place on apply (fresh coordinator fields)
620
+ * @param {import("../src/core/types.mjs").AutoStateClassification} classification
621
+ * @returns {Promise<boolean>} whether the classification was applied
622
+ */
623
+ async function classifyThroughCoordinator(config, status, classification) {
624
+ const result = await sendStateCommand(config.root, {
625
+ type: "state_command",
626
+ viewId: config.viewId,
627
+ runId: config.runId,
628
+ source: "job-runner",
629
+ kind: "auto_state_classified",
630
+ expectedRevision: null,
631
+ payload: { classification },
632
+ });
633
+ if (result.status === "applied") {
634
+ appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated terminal state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
635
+ const fresh = readStatus(config.root, config.viewId, config.runId);
636
+ if (fresh) Object.assign(status, fresh);
637
+ return true;
638
+ }
639
+ if (result.reason === "coordinator_disabled") {
640
+ // Legacy escape hatch (AGENT_BOARD_COORDINATOR=off): apply locally; the
641
+ // caller's persistUnlessManual keeps the pre-coordinator fence for this path.
642
+ return applyAutoStateToStatus(status, classification, Date.now());
643
+ }
644
+ if (result.reason === "manual_fence" || result.reason === "no_change" || result.reason === "stale_run") {
645
+ // Designed fences — informational, not errors.
646
+ return false;
647
+ }
648
+ appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, level: "warn", code: "auto_state_command_ambiguous", message: `Auto-state classification outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise the next classification pass will converge the row`, details: { reason: result.reason } });
649
+ return false;
650
+ }
651
+
652
+ async function applyHeuristicAutoState(config, status, evidence) {
373
653
  if (!canAutoState(config, status, evidence)) return false;
374
- // Fresh read of state.json (not status.json): completeView writes the manual
375
- // completion signal (semanticState "completed" + autoState null) to state.json
376
- // and only clears autoState in status.json, so status.json can never carry
377
- // the completed+null pair. If the user marked the row done while the worker
378
- // was exiting, skip classification so the persist path can't clobber it.
654
+ // Cheap pre-check kept as an optimization (avoids a pointless command);
655
+ // correctness no longer depends on it — the coordinator fences manual
656
+ // completions authoritatively (manual_fence).
379
657
  const latestState = readState(config.root, config.viewId);
380
658
  if (isManualCompletion(latestState)) return false;
381
659
  const latest = latestEvidenceText(evidence) || status.latestAssistantPreview || status.summary || "";
382
660
  const classification = heuristicAutoState(latest, { lastAgentActivityAt: status.lastAgentActivityAt ?? null });
383
- const changed = applyAutoStateToStatus(status, classification, Date.now());
384
- if (changed) {
385
- appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier updated terminal state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
386
- }
387
- return changed;
661
+ return classifyThroughCoordinator(config, status, classification);
388
662
  }
389
663
 
390
664
  async function maybeModelAutoState(config, status, evidence) {
@@ -398,19 +672,14 @@ async function maybeModelAutoState(config, status, evidence) {
398
672
  [...config.piArgsPrefix, "--mode", "json", "-p", "--no-session", "--model", model, prompt],
399
673
  15000,
400
674
  );
401
- // Fresh read: the user may have marked the row done manually during the model
402
- // call. completeView clears autoState in both state.json and status.json, so a
403
- // manual completion is detectable here; applying the classification to the stale
404
- // in-memory status would clobber the user's verdict.
675
+ // The user may have marked the row done manually during the model call. The
676
+ // cheap pre-check avoids a pointless command; the coordinator's manual_fence
677
+ // is the authoritative guard for races after this read.
405
678
  const fresh = readStatus(config.root, config.viewId, config.runId);
406
679
  if (!fresh || isManualCompletion(fresh)) return false;
407
680
  Object.assign(status, fresh);
408
681
  const classification = autoStateFromModelOrHeuristic(out, latest, { lastAgentActivityAt: status.lastAgentActivityAt ?? null });
409
- const changed = applyAutoStateToStatus(status, classification, Date.now());
410
- if (changed) {
411
- appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, code: "auto_state_classified", message: "Auto-state classifier refined terminal state", details: { kind: classification.kind, confidence: classification.confidence, source: classification.source, reason: classification.reason } });
412
- }
413
- return changed;
682
+ return classifyThroughCoordinator(config, status, classification);
414
683
  }
415
684
 
416
685
  /** Default cheap model for terminal summaries. Override/disable via $AGENT_BOARD_SUMMARY_MODEL. */
@@ -450,6 +719,40 @@ async function maybeModelSummary(config, status) {
450
719
  return false;
451
720
  }
452
721
 
722
+ /**
723
+ * Route the post-exit model-summary upgrade through the coordinator as a
724
+ * `patch_fields` command (summary + latestAssistantPreview are whitelisted for
725
+ * the job-runner source). The generic manual fence replaces the old
726
+ * persistUnlessManual file re-read. runId stays null: the finished run's
727
+ * currentRunId still points at it, so the coordinator binds the status patch
728
+ * to the right file without tripping the stale-run guard.
729
+ * @param {import("../src/core/types.mjs").RunConfig} config
730
+ * @param {import("../src/core/types.mjs").RunStatus} status mutated in place on apply
731
+ * @returns {Promise<boolean>} whether the summary patch was applied
732
+ */
733
+ async function patchSummaryThroughCoordinator(config, status) {
734
+ const result = await sendStateCommand(config.root, {
735
+ type: "state_command",
736
+ viewId: config.viewId,
737
+ runId: null,
738
+ source: "job-runner",
739
+ kind: "patch_fields",
740
+ expectedRevision: null,
741
+ payload: {
742
+ state: { summary: status.summary, latestAssistantPreview: status.latestAssistantPreview },
743
+ status: { summary: status.summary },
744
+ },
745
+ });
746
+ if (result.status === "applied") {
747
+ const fresh = readStatus(config.root, config.viewId, config.runId);
748
+ if (fresh) Object.assign(status, fresh);
749
+ return true;
750
+ }
751
+ if (result.reason === "manual_fence" || result.reason === "no_change") return false;
752
+ appendDiagnostic(config.root, config.viewId, { source: "runner", runId: config.runId, level: "warn", code: "summary_patch_command_ambiguous", message: `Summary patch outcome unknown (${result.reason}); if the command was journaled, coordinator replay will recover it; otherwise dashboard reconcile will converge the row`, details: { reason: result.reason } });
753
+ return false;
754
+ }
755
+
453
756
  /**
454
757
  * Run a pi one-shot and return the concatenated assistant text from message_end events.
455
758
  * @param {string} command
@@ -0,0 +1,50 @@
1
+ /**
2
+ * Pre-coordinator direct write for host-failure row finalization (issue #91).
3
+ *
4
+ * Only reachable via `AGENT_BOARD_COORDINATOR=off` — the documented escape
5
+ * hatch. The normal path routes `host_run_failed` through the View State
6
+ * Coordinator (`runner/state-coordinator.mjs`), whose manual_fence / stale_run
7
+ * guards own the decision; this direct write has NO manual-completion fence,
8
+ * which is exactly why it must stay unreachable in the default configuration.
9
+ *
10
+ * Lives in its own module so `runner/pty-runner.mjs` itself never imports the
11
+ * state materializers (writer-boundary test, spec D3). `writeHost` is a
12
+ * different artifact: host.json is owned by the pty-runner per spec D3.
13
+ */
14
+ import { readState, writeState } from "../src/core/store.mjs";
15
+
16
+ /**
17
+ * Legacy direct write of the view-failed row (pre-coordinator markRowFailed).
18
+ * @param {string} root
19
+ * @param {string} viewId
20
+ * @param {string} message
21
+ */
22
+ export function markRowFailedDirect(root, viewId, message) {
23
+ const now = Date.now();
24
+ const state = readState(root, viewId) ?? {
25
+ version: 1,
26
+ viewId,
27
+ currentRunId: null,
28
+ semanticState: "queued",
29
+ processState: "exited",
30
+ summary: "Queued",
31
+ lastActivityAt: now,
32
+ updatedAt: now,
33
+ needsInput: false,
34
+ hasError: false,
35
+ latestAssistantPreview: "",
36
+ latestTool: null,
37
+ question: null,
38
+ pendingQuestions: [],
39
+ error: null,
40
+ };
41
+ state.semanticState = "failed";
42
+ state.processState = "exited";
43
+ state.summary = message;
44
+ state.hasError = true;
45
+ state.needsInput = false;
46
+ state.error = message;
47
+ state.updatedAt = now;
48
+ state.lastActivityAt = now;
49
+ writeState(root, state);
50
+ }