@zhuxixi/pi-agent-board 0.6.2 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/docs/superpowers/plans/2026-09-08-issue-11-attach-runtime-desync-heal.md +917 -0
- package/docs/superpowers/plans/2026-09-09-harden-runner-architecture.md +603 -0
- package/docs/superpowers/plans/2026-09-09-single-writer-completion.md +252 -0
- package/docs/superpowers/specs/2026-09-07-issue-11-attach-runtime-desync-heal-design.md +130 -0
- package/docs/superpowers/specs/2026-09-09-harden-runner-architecture-design.md +298 -0
- package/package.json +1 -1
- package/runner/job-runner-legacy.mjs +68 -0
- package/runner/job-runner.mjs +370 -67
- package/runner/pty-runner-legacy.mjs +50 -0
- package/runner/pty-runner.mjs +69 -31
- package/runner/state-coordinator.mjs +403 -0
- package/runner/state-runner.mjs +89 -15
- package/src/commands/bg.ts +2 -1
- package/src/core/coordinator-client.mjs +313 -0
- package/src/core/coordinator-journal.mjs +282 -0
- package/src/core/coordinator-protocol.mjs +12 -0
- package/src/core/launch.mjs +15 -0
- package/src/core/paths.mjs +16 -0
- package/src/core/pty-attach-jiggle-controller.mjs +57 -4
- package/src/core/pty-attach-render.mjs +30 -0
- package/src/core/state-commands.mjs +617 -0
- package/src/core/types.mjs +2 -0
- package/src/runtime/service.mjs +416 -111
- package/src/ui/dashboard.ts +77 -110
- package/src/ui/pty-attach.ts +63 -1
|
@@ -0,0 +1,617 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure decision layer for View State Coordinator commands (issue #91, spec D3).
|
|
3
|
+
*
|
|
4
|
+
* This module is the single source of truth for "which semantic-state mutations
|
|
5
|
+
* are allowed". The coordinator process shell (runner/state-coordinator.mjs)
|
|
6
|
+
* owns all side effects — journal, socket, file writes — and calls into this
|
|
7
|
+
* module for every decision. No fs, no net, no Date.now(): every timestamp is
|
|
8
|
+
* taken from an explicit `now` argument or from command payload fields, so
|
|
9
|
+
* decisions are deterministic and exhaustively unit-testable (same pattern as
|
|
10
|
+
* host-coordination.mjs from issue #70).
|
|
11
|
+
*
|
|
12
|
+
* Delegation contract (rules are never copied here):
|
|
13
|
+
* - `auto_state_classified` → auto-state.mjs applyAutoStateToViewState/Status
|
|
14
|
+
* run on deep clones; changed fields are diffed into field patches.
|
|
15
|
+
* - `run_finalized` → events.mjs finalizeRun + projectViewState run on a deep
|
|
16
|
+
* clone of the on-disk status; diffs produce the status/state patches.
|
|
17
|
+
* - `run_progress` / `followup_started` → payload.statusPatch merges onto the
|
|
18
|
+
* status clone, then projectViewState recomputes the state side (shared
|
|
19
|
+
* applyStatusProjection helper).
|
|
20
|
+
* - `reconcile_finalize` → stamps finalizeRun's exact field names on the
|
|
21
|
+
* status (endedAt/exitCode/processState/pid) but does NOT run finalizeRun:
|
|
22
|
+
* the reconciler's explicitly observed semanticState governs, and a full
|
|
23
|
+
* finalize would recompute it from a possibly stale preview.
|
|
24
|
+
*
|
|
25
|
+
* Reject reasons: "unknown_view" | "revision_conflict" | "stale_run" |
|
|
26
|
+
* "manual_fence" | "busy" | "no_change" | "field_not_allowed".
|
|
27
|
+
*/
|
|
28
|
+
import {
|
|
29
|
+
applyAutoStateToStatus,
|
|
30
|
+
applyAutoStateToViewState,
|
|
31
|
+
isManualCompletion,
|
|
32
|
+
} from "./auto-state.mjs";
|
|
33
|
+
import { finalizeRun, projectViewState } from "./events.mjs";
|
|
34
|
+
|
|
35
|
+
/** Command kinds accepted by the View State Coordinator (issue #91 scope). */
|
|
36
|
+
export const STATE_COMMAND_KINDS = Object.freeze([
|
|
37
|
+
"mark_completed",
|
|
38
|
+
"auto_state_classified",
|
|
39
|
+
"run_finalized",
|
|
40
|
+
"mark_queued",
|
|
41
|
+
"run_started",
|
|
42
|
+
"run_progress",
|
|
43
|
+
"reconcile_finalize",
|
|
44
|
+
"host_run_failed",
|
|
45
|
+
"archive_view",
|
|
46
|
+
"adopt_session",
|
|
47
|
+
"sync_foreground",
|
|
48
|
+
"plan_ready",
|
|
49
|
+
"followup_started",
|
|
50
|
+
"patch_fields",
|
|
51
|
+
]);
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Kinds the coordinator applies WITHOUT journaling (plan D3/Task-3 decision:
|
|
55
|
+
* run_progress is a periodic self-healing snapshot — ~4 writes/sec/run would
|
|
56
|
+
* bloat the journal unboundedly; a lost beat is overwritten by the next one).
|
|
57
|
+
* Transient commands still materialize and bump materializedRevision, but they
|
|
58
|
+
* are not replayed and not deduped.
|
|
59
|
+
*/
|
|
60
|
+
export const TRANSIENT_KINDS = Object.freeze(["run_progress"]);
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Per-source whitelist for `patch_fields` (metadata/evidence-mirror merges).
|
|
64
|
+
* Anything not listed here is rejected with "field_not_allowed" — semantic
|
|
65
|
+
* fields must travel through their dedicated kinds so the guard table in the
|
|
66
|
+
* plan stays exhaustive.
|
|
67
|
+
* @typedef {{ state: string[], status: string[] }} PatchableFields
|
|
68
|
+
* @type {Record<string, PatchableFields>}
|
|
69
|
+
*/
|
|
70
|
+
export const PATCHABLE_FIELDS = Object.freeze({
|
|
71
|
+
// summary/latestAssistantPreview: the post-exit model-summary persist routes
|
|
72
|
+
// through patch_fields (controller ruling R1 — plan oversight, Task 3).
|
|
73
|
+
"job-runner": Object.freeze({ state: Object.freeze(["review", "evidenceSummary", "summary", "latestAssistantPreview"]), status: Object.freeze(["evidenceSummary", "summary", "latestAssistantPreview"]) }),
|
|
74
|
+
"state-runner": Object.freeze({ state: Object.freeze(["review", "evidenceSummary"]), status: Object.freeze(["evidenceSummary"]) }),
|
|
75
|
+
"service": Object.freeze({ state: Object.freeze(["lastVisitedAt"]), status: Object.freeze([]) }),
|
|
76
|
+
// lastVisitedAt (markVisited): visiting is a user action, so it routes as
|
|
77
|
+
// dashboard-user — the manual fence only fences non-human sources, and
|
|
78
|
+
// legacy stamped lastVisitedAt unconditionally (visit-recency tracking
|
|
79
|
+
// must keep working on manually-completed rows).
|
|
80
|
+
"dashboard-user": Object.freeze({ state: Object.freeze(["lastVisitedAt"]), status: Object.freeze([]) }),
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
/** Who may originate a state command. Non-human sources are fenced by manual completions. */
|
|
84
|
+
export const COMMAND_SOURCES = Object.freeze([
|
|
85
|
+
"dashboard-user",
|
|
86
|
+
"service",
|
|
87
|
+
"job-runner",
|
|
88
|
+
"state-runner",
|
|
89
|
+
"pty-runner",
|
|
90
|
+
]);
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* A semantic-state mutation request routed through the coordinator.
|
|
94
|
+
* @typedef {Object} StateCommand
|
|
95
|
+
* @property {"state_command"} type
|
|
96
|
+
* @property {string} commandId Stable id — journal replays return the original result.
|
|
97
|
+
* @property {string} viewId
|
|
98
|
+
* @property {string} [runId] Active run this command belongs to (stale-run fenced).
|
|
99
|
+
* @property {typeof COMMAND_SOURCES[number]} source
|
|
100
|
+
* @property {number|null} [expectedRevision] Optimistic concurrency on materializedRevision.
|
|
101
|
+
* @property {typeof STATE_COMMAND_KINDS[number]} kind
|
|
102
|
+
* @property {Record<string, unknown>} payload Kind-specific; validated per kind.
|
|
103
|
+
*/
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Validate the command envelope plus kind-specific payload presence. Pure.
|
|
107
|
+
* @param {any} raw
|
|
108
|
+
* @returns {{ ok: true, command: object } | { ok: false, error: string }}
|
|
109
|
+
*/
|
|
110
|
+
export function validateCommand(raw) {
|
|
111
|
+
if (!raw || raw.type !== "state_command") return { ok: false, error: "bad_type" };
|
|
112
|
+
// Transient kinds have no idempotency semantics — the shell neither dedupes
|
|
113
|
+
// nor replays them, so commandId is optional there (echoed in the reply only).
|
|
114
|
+
if ((typeof raw.commandId !== "string" || !raw.commandId) && !TRANSIENT_KINDS.includes(raw.kind)) {
|
|
115
|
+
return { ok: false, error: "missing_commandId" };
|
|
116
|
+
}
|
|
117
|
+
if (typeof raw.viewId !== "string" || !raw.viewId) return { ok: false, error: "missing_viewId" };
|
|
118
|
+
if (!STATE_COMMAND_KINDS.includes(raw.kind)) return { ok: false, error: "unknown_kind" };
|
|
119
|
+
if (!COMMAND_SOURCES.includes(raw.source)) return { ok: false, error: "unknown_source" };
|
|
120
|
+
if (raw.expectedRevision != null && typeof raw.expectedRevision !== "number") return { ok: false, error: "bad_expectedRevision" };
|
|
121
|
+
if (raw.runId != null && typeof raw.runId !== "string") return { ok: false, error: "bad_runId" };
|
|
122
|
+
if (raw.kind === "auto_state_classified") {
|
|
123
|
+
const classification = raw.payload?.classification;
|
|
124
|
+
if (!classification || typeof classification !== "object") return { ok: false, error: "missing_classification" };
|
|
125
|
+
if (typeof classification.classifiedAt !== "number") return { ok: false, error: "bad_classification" };
|
|
126
|
+
}
|
|
127
|
+
if (raw.kind === "run_finalized") {
|
|
128
|
+
if (!raw.payload || typeof raw.payload !== "object") return { ok: false, error: "missing_payload" };
|
|
129
|
+
if (typeof raw.payload.exitCode !== "number" && raw.payload.exitCode !== null) return { ok: false, error: "missing_exitCode" };
|
|
130
|
+
if (raw.payload.endedAt != null && typeof raw.payload.endedAt !== "number") return { ok: false, error: "bad_endedAt" };
|
|
131
|
+
if (raw.payload.lastAgentActivityAt != null && typeof raw.payload.lastAgentActivityAt !== "number") return { ok: false, error: "bad_lastAgentActivityAt" };
|
|
132
|
+
if (raw.payload.stoppedByUser != null && typeof raw.payload.stoppedByUser !== "boolean") return { ok: false, error: "bad_stoppedByUser" };
|
|
133
|
+
if (raw.payload.stopReason != null && typeof raw.payload.stopReason !== "string") return { ok: false, error: "bad_stopReason" };
|
|
134
|
+
}
|
|
135
|
+
switch (raw.kind) {
|
|
136
|
+
case "mark_queued":
|
|
137
|
+
// runId may be null (PTY host launch pins no run — legacy
|
|
138
|
+
// markQueued(id, null)); the key must be present, and when non-null
|
|
139
|
+
// it must be a non-empty string.
|
|
140
|
+
if (!raw.payload || !("runId" in raw.payload)) return { ok: false, error: "missing_runId" };
|
|
141
|
+
if (raw.payload.runId != null && (typeof raw.payload.runId !== "string" || !raw.payload.runId)) return { ok: false, error: "missing_runId" };
|
|
142
|
+
break;
|
|
143
|
+
case "run_started": {
|
|
144
|
+
if (raw.runId == null) return { ok: false, error: "missing_runId" };
|
|
145
|
+
const status = raw.payload?.status;
|
|
146
|
+
if (!status || typeof status !== "object") return { ok: false, error: "missing_status" };
|
|
147
|
+
if (typeof status.runId !== "string" || status.runId !== raw.runId) return { ok: false, error: "bad_status" };
|
|
148
|
+
break;
|
|
149
|
+
}
|
|
150
|
+
case "run_progress":
|
|
151
|
+
if (!raw.payload || typeof raw.payload.statusPatch !== "object" || raw.payload.statusPatch == null) return { ok: false, error: "missing_statusPatch" };
|
|
152
|
+
break;
|
|
153
|
+
case "followup_started":
|
|
154
|
+
// The command carries NO runId (a null runId skips the generic stale-run
|
|
155
|
+
// guard against the finished parent run); the NEW run's id travels here
|
|
156
|
+
// and governs the state-side currentRunId (ruling R2).
|
|
157
|
+
if (!raw.payload || typeof raw.payload.statusPatch !== "object" || raw.payload.statusPatch == null) return { ok: false, error: "missing_statusPatch" };
|
|
158
|
+
if (typeof raw.payload.newRunId !== "string" || !raw.payload.newRunId) return { ok: false, error: "missing_newRunId" };
|
|
159
|
+
break;
|
|
160
|
+
case "reconcile_finalize": {
|
|
161
|
+
// Project mode (service.mjs dead-runner path): the run's terminal status
|
|
162
|
+
// exists but the row was never materialized from it — the status itself
|
|
163
|
+
// is the verdict, so semanticState/summary are derived, not passed.
|
|
164
|
+
if (raw.payload?.project === true) {
|
|
165
|
+
if (raw.payload.reason != null && typeof raw.payload.reason !== "string") return { ok: false, error: "bad_reason" };
|
|
166
|
+
break;
|
|
167
|
+
}
|
|
168
|
+
const semanticState = raw.payload?.semanticState;
|
|
169
|
+
if (semanticState !== "failed" && semanticState !== "idle") return { ok: false, error: "bad_semanticState" };
|
|
170
|
+
// The reconciler's summary is caller-provided (legacy parity: both
|
|
171
|
+
// service reconcile sites always stamp a summary, ruling R4).
|
|
172
|
+
if (typeof raw.payload?.summary !== "string" || !raw.payload.summary) return { ok: false, error: "missing_summary" };
|
|
173
|
+
if (raw.payload.reason != null && typeof raw.payload.reason !== "string") return { ok: false, error: "bad_reason" };
|
|
174
|
+
if (raw.payload.exitCode != null && typeof raw.payload.exitCode !== "number") return { ok: false, error: "bad_exitCode" };
|
|
175
|
+
break;
|
|
176
|
+
}
|
|
177
|
+
case "host_run_failed":
|
|
178
|
+
if (raw.payload?.error != null && typeof raw.payload.error !== "string") return { ok: false, error: "bad_error" };
|
|
179
|
+
if (raw.payload?.exitCode != null && typeof raw.payload.exitCode !== "number") return { ok: false, error: "bad_exitCode" };
|
|
180
|
+
break;
|
|
181
|
+
case "sync_foreground":
|
|
182
|
+
if (!raw.payload?.projection || typeof raw.payload.projection !== "object") return { ok: false, error: "missing_projection" };
|
|
183
|
+
break;
|
|
184
|
+
case "plan_ready":
|
|
185
|
+
// The producer (job-runner's plan-ready pass) fires post-finalization and
|
|
186
|
+
// re-points the row at the plan-producing run (ruling R3).
|
|
187
|
+
if (raw.payload?.question != null && typeof raw.payload.question !== "string") return { ok: false, error: "bad_question" };
|
|
188
|
+
if (typeof raw.payload?.runId !== "string" || !raw.payload.runId) return { ok: false, error: "missing_runId" };
|
|
189
|
+
break;
|
|
190
|
+
case "patch_fields": {
|
|
191
|
+
const hasState = raw.payload?.state != null && typeof raw.payload.state === "object";
|
|
192
|
+
const hasStatus = raw.payload?.status != null && typeof raw.payload.status === "object";
|
|
193
|
+
if (!hasState && !hasStatus) return { ok: false, error: "missing_payload" };
|
|
194
|
+
break;
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
return { ok: true, command: raw };
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Decide whether a command may mutate the view state, and which field patches
|
|
202
|
+
* to apply. Pure: deep-clones inputs before delegating, never mutates arguments,
|
|
203
|
+
* never touches the clock (pass `now` for wall-clock timestamps; falls back to
|
|
204
|
+
* payload-provided timestamps for determinism).
|
|
205
|
+
*
|
|
206
|
+
* @param {object} command
|
|
207
|
+
* @param {object|null} currentState ViewState as currently materialized (state.json).
|
|
208
|
+
* @param {object|null} currentStatus RunStatus for command.runId as currently
|
|
209
|
+
* materialized (status.json), when a status file exists; null otherwise.
|
|
210
|
+
* @param {number} [now] Wall-clock epoch ms supplied by the coordinator shell.
|
|
211
|
+
* @returns {{ action: "apply", mutate: { state?: object, status?: object }, reason: string }
|
|
212
|
+
* | { action: "reject", reason: string }}
|
|
213
|
+
* `mutate.state`/`mutate.status` are sparse field patches — only fields whose
|
|
214
|
+
* value actually changed. The coordinator merges them onto the materialized
|
|
215
|
+
* files and stamps the shared materializedRevision.
|
|
216
|
+
*/
|
|
217
|
+
export function decideStateTransition(command, currentState, currentStatus, now = undefined) {
|
|
218
|
+
if (!currentState) return reject("unknown_view");
|
|
219
|
+
if (command.expectedRevision != null && command.expectedRevision !== (currentState.materializedRevision ?? 0)) {
|
|
220
|
+
return reject("revision_conflict");
|
|
221
|
+
}
|
|
222
|
+
// Generic stale-run guard: a command for a run other than the row's current
|
|
223
|
+
// run is stale. run_started is EXEMPT — it carries its own liveness-scoped
|
|
224
|
+
// guard below (launch-order inversion, cold-start race): the runner is
|
|
225
|
+
// authoritative that its run just started, so a bootstrap landing on a
|
|
226
|
+
// not-alive row applies and re-pins currentRunId. Only two SIMULTANEOUS live
|
|
227
|
+
// runs on one view are the hazard this guard exists for.
|
|
228
|
+
if (
|
|
229
|
+
command.kind !== "run_started" &&
|
|
230
|
+
command.runId &&
|
|
231
|
+
currentState.currentRunId &&
|
|
232
|
+
command.runId !== currentState.currentRunId
|
|
233
|
+
) {
|
|
234
|
+
return reject("stale_run");
|
|
235
|
+
}
|
|
236
|
+
// Manual completions are user verdicts: only a human source may act on a
|
|
237
|
+
// fenced row (this is the #46 invariant — late classifications lose).
|
|
238
|
+
if (command.source !== "dashboard-user" && isManualCompletion(currentState)) {
|
|
239
|
+
return reject("manual_fence");
|
|
240
|
+
}
|
|
241
|
+
switch (command.kind) {
|
|
242
|
+
case "mark_completed": {
|
|
243
|
+
if (currentState.processState === "alive") return reject("busy");
|
|
244
|
+
// autoState: null on both artifacts is the manual-completion fence
|
|
245
|
+
// signal that later auto-state commands (and the auto-state rules
|
|
246
|
+
// themselves) key off — see isManualCompletion().
|
|
247
|
+
return {
|
|
248
|
+
action: "apply",
|
|
249
|
+
reason: "manual_completion",
|
|
250
|
+
mutate: {
|
|
251
|
+
state: {
|
|
252
|
+
semanticState: "completed",
|
|
253
|
+
processState: "exited",
|
|
254
|
+
needsInput: false,
|
|
255
|
+
hasError: false,
|
|
256
|
+
question: null,
|
|
257
|
+
pendingQuestions: [],
|
|
258
|
+
error: null,
|
|
259
|
+
autoState: null,
|
|
260
|
+
},
|
|
261
|
+
status: { autoState: null },
|
|
262
|
+
},
|
|
263
|
+
};
|
|
264
|
+
}
|
|
265
|
+
case "auto_state_classified": {
|
|
266
|
+
const classification = command.payload?.classification;
|
|
267
|
+
const at = now ?? classification?.classifiedAt;
|
|
268
|
+
const stateClone = cloneJson(currentState);
|
|
269
|
+
const statusClone = cloneJson(currentStatus);
|
|
270
|
+
const stateChanged = applyAutoStateToViewState(stateClone, classification, at);
|
|
271
|
+
const statusChanged = statusClone ? applyAutoStateToStatus(statusClone, classification, at) : false;
|
|
272
|
+
if (!stateChanged && !statusChanged) return reject("no_change");
|
|
273
|
+
return { action: "apply", reason: command.kind, mutate: buildPatches(currentState, stateClone, stateChanged, currentStatus, statusClone, statusChanged) };
|
|
274
|
+
}
|
|
275
|
+
case "run_finalized": {
|
|
276
|
+
const payload = command.payload ?? {};
|
|
277
|
+
// Only a live run record for this view's current run may be finalized;
|
|
278
|
+
// anything else is stale (duplicate finalize commands included).
|
|
279
|
+
if (!currentStatus) return reject("stale_run");
|
|
280
|
+
if (currentState.currentRunId !== command.runId) return reject("stale_run");
|
|
281
|
+
if (currentState.processState !== "alive") return reject("stale_run");
|
|
282
|
+
const at = now ?? payload.endedAt ?? 0;
|
|
283
|
+
const statusClone = cloneJson(currentStatus);
|
|
284
|
+
// Overlay fresher fields the throttled on-disk write may lag behind.
|
|
285
|
+
// stopReason must be overlaid onto the status itself: finalizeSemanticState
|
|
286
|
+
// reads status.stopReason (finalizeRun's opts have no stopReason slot), so
|
|
287
|
+
// a fresh "aborted" reported by the runner would otherwise be lost and the
|
|
288
|
+
// run would land on "idle" instead of "failed".
|
|
289
|
+
if (typeof payload.latestAssistantPreview === "string") statusClone.latestAssistantPreview = payload.latestAssistantPreview;
|
|
290
|
+
if (payload.lastAgentActivityAt != null) statusClone.lastAgentActivityAt = payload.lastAgentActivityAt;
|
|
291
|
+
if (payload.stopReason != null) statusClone.stopReason = payload.stopReason;
|
|
292
|
+
finalizeRun(statusClone, {
|
|
293
|
+
exitCode: payload.exitCode,
|
|
294
|
+
stoppedByUser: payload.stoppedByUser,
|
|
295
|
+
stopReason: payload.stopReason,
|
|
296
|
+
openEnded: payload.openEnded,
|
|
297
|
+
}, at);
|
|
298
|
+
const projectedState = projectViewState(statusClone, at, currentState);
|
|
299
|
+
return {
|
|
300
|
+
action: "apply",
|
|
301
|
+
reason: command.kind,
|
|
302
|
+
mutate: {
|
|
303
|
+
state: diffFields(currentState, projectedState),
|
|
304
|
+
status: diffFields(currentStatus, statusClone),
|
|
305
|
+
},
|
|
306
|
+
};
|
|
307
|
+
}
|
|
308
|
+
case "mark_queued": {
|
|
309
|
+
return {
|
|
310
|
+
action: "apply",
|
|
311
|
+
reason: command.kind,
|
|
312
|
+
mutate: {
|
|
313
|
+
state: {
|
|
314
|
+
currentRunId: command.payload.runId,
|
|
315
|
+
semanticState: "queued",
|
|
316
|
+
processState: "alive",
|
|
317
|
+
summary: "Queued",
|
|
318
|
+
needsInput: false,
|
|
319
|
+
hasError: false,
|
|
320
|
+
question: null,
|
|
321
|
+
pendingQuestions: [],
|
|
322
|
+
error: null,
|
|
323
|
+
autoState: null,
|
|
324
|
+
},
|
|
325
|
+
},
|
|
326
|
+
};
|
|
327
|
+
}
|
|
328
|
+
case "run_started": {
|
|
329
|
+
// Liveness-scoped stale guard (the generic guard above exempts this
|
|
330
|
+
// kind): only TWO LIVE RUNS on one view are the hazard — a bootstrap for
|
|
331
|
+
// a different runId while this row already runs something else is
|
|
332
|
+
// out-of-order and rejected. When the row is NOT alive, the runner is
|
|
333
|
+
// authoritative that its run just started: apply and re-pin
|
|
334
|
+
// currentRunId. This covers the cold-start race where run_started lands
|
|
335
|
+
// before mark_queued (the runner was spawned first), and re-launches of
|
|
336
|
+
// rows still pointing at the previous run.
|
|
337
|
+
if (
|
|
338
|
+
currentState.processState === "alive" &&
|
|
339
|
+
currentState.currentRunId &&
|
|
340
|
+
currentState.currentRunId !== command.runId
|
|
341
|
+
) {
|
|
342
|
+
return reject("stale_run");
|
|
343
|
+
}
|
|
344
|
+
const statusAfter = cloneJson(command.payload.status);
|
|
345
|
+
return {
|
|
346
|
+
action: "apply",
|
|
347
|
+
reason: command.kind,
|
|
348
|
+
mutate: {
|
|
349
|
+
state: {
|
|
350
|
+
processState: "alive",
|
|
351
|
+
semanticState: "working",
|
|
352
|
+
currentRunId: command.runId,
|
|
353
|
+
},
|
|
354
|
+
status: diffFields(currentStatus, statusAfter),
|
|
355
|
+
},
|
|
356
|
+
};
|
|
357
|
+
}
|
|
358
|
+
case "run_progress": {
|
|
359
|
+
// Liveness semantics: only the row's current live run may move forward.
|
|
360
|
+
// Late/duplicate progress from a finished run is dropped (next run's
|
|
361
|
+
// progress supersedes it anyway — this kind is transient by design).
|
|
362
|
+
// A live run with no materialized status cannot legitimately progress
|
|
363
|
+
// (F2): the first beat always follows run_started's bootstrap write, so
|
|
364
|
+
// a missing status here means the beat is stale or out of order — and
|
|
365
|
+
// projecting onto `null` would materialize undefined processState.
|
|
366
|
+
if (!currentStatus) return reject("stale_run");
|
|
367
|
+
if (currentState.currentRunId !== command.runId || currentState.processState !== "alive") return reject("stale_run");
|
|
368
|
+
return applyStatusProjection(command, currentState, currentStatus, now);
|
|
369
|
+
}
|
|
370
|
+
case "followup_started": {
|
|
371
|
+
// No liveness guard and no command.runId: the follow-up starts from a
|
|
372
|
+
// just-finalized parent run, so the generic stale-run guard must not fire
|
|
373
|
+
// against it. The NEW run's identity (payload.newRunId) governs the
|
|
374
|
+
// state-side currentRunId regardless of what the status patch carries,
|
|
375
|
+
// and a follow-up starting is definitionally the row running again —
|
|
376
|
+
// pin processState so a sparse bootstrap patch can never materialize
|
|
377
|
+
// an undefined (key-dropping) processState on a re-pointed row.
|
|
378
|
+
const result = applyStatusProjection(command, currentState, currentStatus, now);
|
|
379
|
+
return {
|
|
380
|
+
...result,
|
|
381
|
+
mutate: { ...result.mutate, state: { ...result.mutate.state, currentRunId: command.payload.newRunId, processState: "alive" } },
|
|
382
|
+
};
|
|
383
|
+
}
|
|
384
|
+
case "reconcile_finalize": {
|
|
385
|
+
if (currentState.processState !== "alive") return reject("no_change");
|
|
386
|
+
const at = now ?? 0;
|
|
387
|
+
if (command.payload.project === true) {
|
|
388
|
+
// Project mode: faithfully re-materialize the row from the run's
|
|
389
|
+
// terminal status (projectViewState delegation — never copied rules).
|
|
390
|
+
// A missing status means there is nothing to project — stale.
|
|
391
|
+
if (!currentStatus) return reject("stale_run");
|
|
392
|
+
const projected = projectViewState(cloneJson(currentStatus), at, currentState);
|
|
393
|
+
return { action: "apply", reason: command.kind, mutate: { state: diffFields(currentState, projected) } };
|
|
394
|
+
}
|
|
395
|
+
const failed = command.payload.semanticState === "failed";
|
|
396
|
+
const stateClone = cloneJson(currentState);
|
|
397
|
+
stateClone.semanticState = command.payload.semanticState;
|
|
398
|
+
stateClone.processState = "exited";
|
|
399
|
+
// Derived-field clearing at legacy parity (service.mjs reconcile sites,
|
|
400
|
+
// ruling R4): both legacy branches always stamp these fields.
|
|
401
|
+
stateClone.needsInput = false;
|
|
402
|
+
stateClone.hasError = failed;
|
|
403
|
+
stateClone.question = null;
|
|
404
|
+
stateClone.pendingQuestions = [];
|
|
405
|
+
stateClone.error = command.payload.reason ?? null;
|
|
406
|
+
stateClone.summary = command.payload.summary;
|
|
407
|
+
const mutate = { state: diffFields(currentState, stateClone) };
|
|
408
|
+
if (currentStatus) {
|
|
409
|
+
const statusClone = cloneJson(currentStatus);
|
|
410
|
+
// Same field names finalizeRun stamps, minus the semantic recomputation:
|
|
411
|
+
// the reconciler observed the outcome explicitly and its verdict governs.
|
|
412
|
+
statusClone.endedAt = at;
|
|
413
|
+
statusClone.exitCode = command.payload.exitCode ?? null;
|
|
414
|
+
statusClone.processState = "exited";
|
|
415
|
+
statusClone.pid = null;
|
|
416
|
+
statusClone.semanticState = command.payload.semanticState;
|
|
417
|
+
mutate.status = diffFields(currentStatus, statusClone);
|
|
418
|
+
}
|
|
419
|
+
return { action: "apply", reason: command.kind, mutate };
|
|
420
|
+
}
|
|
421
|
+
case "host_run_failed": {
|
|
422
|
+
// No kind-specific guard: the generic manual_fence above is exactly the
|
|
423
|
+
// PR #1 residual-risk closure — a late host crash must not flip a row
|
|
424
|
+
// the user already completed by hand.
|
|
425
|
+
const message = command.payload?.error ?? "PTY host failed";
|
|
426
|
+
const mutate = {
|
|
427
|
+
state: {
|
|
428
|
+
semanticState: "failed",
|
|
429
|
+
processState: "exited",
|
|
430
|
+
// Legacy parity with markRowFailedDirect (runner/pty-runner-legacy.mjs):
|
|
431
|
+
// both summary and error carry the message, hasError/needsInput are
|
|
432
|
+
// stamped so row rendering and warm-host eviction match the direct era.
|
|
433
|
+
summary: message,
|
|
434
|
+
hasError: true,
|
|
435
|
+
needsInput: false,
|
|
436
|
+
error: command.payload?.error ?? null,
|
|
437
|
+
},
|
|
438
|
+
};
|
|
439
|
+
if (currentStatus) {
|
|
440
|
+
mutate.status = {
|
|
441
|
+
semanticState: "failed",
|
|
442
|
+
processState: "exited",
|
|
443
|
+
error: command.payload?.error ?? null,
|
|
444
|
+
};
|
|
445
|
+
}
|
|
446
|
+
return { action: "apply", reason: command.kind, mutate };
|
|
447
|
+
}
|
|
448
|
+
case "archive_view": {
|
|
449
|
+
// Busy rows are allowed: archiving a working row stops it (matches the
|
|
450
|
+
// legacy archiveView behavior this kind replaces).
|
|
451
|
+
return {
|
|
452
|
+
action: "apply",
|
|
453
|
+
reason: command.kind,
|
|
454
|
+
mutate: {
|
|
455
|
+
state: {
|
|
456
|
+
semanticState: "stopped",
|
|
457
|
+
processState: "exited",
|
|
458
|
+
needsInput: false,
|
|
459
|
+
hasError: false,
|
|
460
|
+
question: null,
|
|
461
|
+
pendingQuestions: [],
|
|
462
|
+
error: null,
|
|
463
|
+
autoState: null,
|
|
464
|
+
summary: "Stopped",
|
|
465
|
+
},
|
|
466
|
+
},
|
|
467
|
+
};
|
|
468
|
+
}
|
|
469
|
+
case "adopt_session": {
|
|
470
|
+
if (currentState.processState === "alive") return reject("busy");
|
|
471
|
+
// Derived-field clearing at legacy parity (service.mjs adoptSession reuse
|
|
472
|
+
// path, ruling R4). autoState is intentionally untouched: the legacy
|
|
473
|
+
// adopt path leaves it alone.
|
|
474
|
+
return {
|
|
475
|
+
action: "apply",
|
|
476
|
+
reason: command.kind,
|
|
477
|
+
mutate: {
|
|
478
|
+
state: {
|
|
479
|
+
semanticState: "idle",
|
|
480
|
+
processState: "exited",
|
|
481
|
+
needsInput: false,
|
|
482
|
+
hasError: false,
|
|
483
|
+
question: null,
|
|
484
|
+
pendingQuestions: [],
|
|
485
|
+
error: null,
|
|
486
|
+
summary: "Backgrounded session",
|
|
487
|
+
},
|
|
488
|
+
},
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
case "sync_foreground": {
|
|
492
|
+
// Foreground mirrors never own a background run: the projection is the
|
|
493
|
+
// caller's, but currentRunId stays null regardless of what it carries.
|
|
494
|
+
const stateClone = cloneJson(currentState);
|
|
495
|
+
Object.assign(stateClone, command.payload.projection);
|
|
496
|
+
stateClone.currentRunId = null;
|
|
497
|
+
return { action: "apply", reason: command.kind, mutate: { state: diffFields(currentState, stateClone) } };
|
|
498
|
+
}
|
|
499
|
+
case "plan_ready": {
|
|
500
|
+
// The producer (job-runner's plan-ready pass) fires POST-finalization:
|
|
501
|
+
// the run has exited by the time a plan is ready for approval, so a live
|
|
502
|
+
// row means out-of-order delivery — drop it (ruling R3 inverts the guard).
|
|
503
|
+
if (currentState.processState === "alive") return reject("no_change");
|
|
504
|
+
// Exact legacy parity with runner/job-runner.mjs's plan-ready write.
|
|
505
|
+
return {
|
|
506
|
+
action: "apply",
|
|
507
|
+
reason: command.kind,
|
|
508
|
+
mutate: {
|
|
509
|
+
state: {
|
|
510
|
+
semanticState: "needs_input",
|
|
511
|
+
processState: "exited",
|
|
512
|
+
needsInput: true,
|
|
513
|
+
question: command.payload?.question ?? "Approve this plan?",
|
|
514
|
+
summary: "Plan ready for approval",
|
|
515
|
+
currentRunId: command.payload.runId,
|
|
516
|
+
},
|
|
517
|
+
},
|
|
518
|
+
};
|
|
519
|
+
}
|
|
520
|
+
case "patch_fields": {
|
|
521
|
+
const allowed = PATCHABLE_FIELDS[command.source];
|
|
522
|
+
if (!allowed) return reject("field_not_allowed");
|
|
523
|
+
const requestedState = command.payload.state ?? {};
|
|
524
|
+
const requestedStatus = command.payload.status ?? {};
|
|
525
|
+
for (const key of Object.keys(requestedState)) {
|
|
526
|
+
if (!allowed.state.includes(key)) return reject("field_not_allowed");
|
|
527
|
+
}
|
|
528
|
+
for (const key of Object.keys(requestedStatus)) {
|
|
529
|
+
if (!allowed.status.includes(key)) return reject("field_not_allowed");
|
|
530
|
+
}
|
|
531
|
+
const stateClone = cloneJson(currentState);
|
|
532
|
+
Object.assign(stateClone, requestedState);
|
|
533
|
+
const statePatch = diffFields(currentState, stateClone);
|
|
534
|
+
let statusPatch;
|
|
535
|
+
if (currentStatus && Object.keys(requestedStatus).length > 0) {
|
|
536
|
+
const statusClone = cloneJson(currentStatus);
|
|
537
|
+
Object.assign(statusClone, requestedStatus);
|
|
538
|
+
statusPatch = diffFields(currentStatus, statusClone);
|
|
539
|
+
}
|
|
540
|
+
if (Object.keys(statePatch).length === 0 && (!statusPatch || Object.keys(statusPatch).length === 0)) {
|
|
541
|
+
return reject("no_change");
|
|
542
|
+
}
|
|
543
|
+
return { action: "apply", reason: command.kind, mutate: statusPatch ? { state: statePatch, status: statusPatch } : { state: statePatch } };
|
|
544
|
+
}
|
|
545
|
+
default:
|
|
546
|
+
// validateCommand already rejects unknown kinds; defensive only.
|
|
547
|
+
return reject("unknown_kind");
|
|
548
|
+
}
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
/** @param {string} reason @returns {{ action: "reject", reason: string }} */
|
|
552
|
+
function reject(reason) {
|
|
553
|
+
return { action: "reject", reason };
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
/** JSON round-trip clone: patches stay JSON-serializable (they are written to disk). */
|
|
557
|
+
function cloneJson(value) {
|
|
558
|
+
return value == null ? value : JSON.parse(JSON.stringify(value));
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
/**
|
|
562
|
+
* Shallow top-level field diff (deep-equality per field via JSON form), so a
|
|
563
|
+
* cloned-but-unchanged nested object never lands in a patch.
|
|
564
|
+
* @param {object|null} before
|
|
565
|
+
* @param {object} after
|
|
566
|
+
*/
|
|
567
|
+
function diffFields(before, after) {
|
|
568
|
+
const patch = {};
|
|
569
|
+
for (const key of Object.keys(after)) {
|
|
570
|
+
if (JSON.stringify(before?.[key]) !== JSON.stringify(after[key])) patch[key] = after[key];
|
|
571
|
+
}
|
|
572
|
+
return patch;
|
|
573
|
+
}
|
|
574
|
+
|
|
575
|
+
/** @param {boolean} stateChanged @param {boolean} statusChanged */
|
|
576
|
+
function buildPatches(currentState, stateClone, stateChanged, currentStatus, statusClone, statusChanged) {
|
|
577
|
+
const mutate = {};
|
|
578
|
+
if (stateChanged) mutate.state = diffFields(currentState, stateClone);
|
|
579
|
+
if (statusChanged) mutate.status = diffFields(currentStatus, statusClone);
|
|
580
|
+
return mutate;
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
/**
|
|
584
|
+
* Shared body for the status-patch kinds (`run_progress`, `followup_started`):
|
|
585
|
+
* merge payload.statusPatch onto the materialized status, then let
|
|
586
|
+
* projectViewState recompute the state side — the projection rules live in
|
|
587
|
+
* events.mjs and are never copied here (same delegation contract as
|
|
588
|
+
* run_finalized).
|
|
589
|
+
*
|
|
590
|
+
* A missing status file is tolerated: the patch merges onto an empty object so
|
|
591
|
+
* the coordinator materializes a fresh status from the patch fields (this is
|
|
592
|
+
* the followup_started bootstrap path — a new run's status file does not exist
|
|
593
|
+
* yet). The caller should send enough fields to make that object meaningful.
|
|
594
|
+
*
|
|
595
|
+
* Pure; timestamp comes from `now`, falling back to the patch's
|
|
596
|
+
* lastActivityAt (payload-carried determinism, same rule as run_finalized).
|
|
597
|
+
*
|
|
598
|
+
* @param {object} command
|
|
599
|
+
* @param {object} currentState
|
|
600
|
+
* @param {object|null} currentStatus
|
|
601
|
+
* @param {number|undefined} now
|
|
602
|
+
*/
|
|
603
|
+
function applyStatusProjection(command, currentState, currentStatus, now) {
|
|
604
|
+
const patch = command.payload.statusPatch;
|
|
605
|
+
const at = now ?? patch.lastActivityAt ?? 0;
|
|
606
|
+
const statusClone = cloneJson(currentStatus) ?? {};
|
|
607
|
+
Object.assign(statusClone, patch);
|
|
608
|
+
const projectedState = projectViewState(statusClone, at, currentState);
|
|
609
|
+
return {
|
|
610
|
+
action: "apply",
|
|
611
|
+
reason: command.kind,
|
|
612
|
+
mutate: {
|
|
613
|
+
state: diffFields(currentState, projectedState),
|
|
614
|
+
status: diffFields(currentStatus ?? {}, statusClone),
|
|
615
|
+
},
|
|
616
|
+
};
|
|
617
|
+
}
|
package/src/core/types.mjs
CHANGED
|
@@ -120,6 +120,7 @@ export const GROUP_LABELS = {
|
|
|
120
120
|
* @property {SteeringSummary} [steering] Compact plan/approval steering summary.
|
|
121
121
|
* @property {CodeRefsSummary} [codeRefs] Compact issue/PR reference summary.
|
|
122
122
|
* @property {AutoStateClassification|null} [autoState] Latest automatic terminal-state classification.
|
|
123
|
+
* @property {number} [materializedRevision] Coordinator-stamped monotonic revision (issue #91). Absent on legacy rows until the View State Coordinator first materializes the view.
|
|
123
124
|
*/
|
|
124
125
|
|
|
125
126
|
/**
|
|
@@ -188,6 +189,7 @@ export const GROUP_LABELS = {
|
|
|
188
189
|
* @property {EvidenceUsage|null} [usage]
|
|
189
190
|
* @property {string|null} [stallReason]
|
|
190
191
|
* @property {ReviewSummary|null} [evidenceSummary]
|
|
192
|
+
* @property {number} [materializedRevision] Coordinator-stamped monotonic revision shared with the view's state.json (issue #91). Absent on legacy rows until first coordinator materialization.
|
|
191
193
|
* @property {AutoStateClassification|null} [autoState]
|
|
192
194
|
*/
|
|
193
195
|
|