muse-crew 0.7.10 → 0.7.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/crew-api.js CHANGED
@@ -12,17 +12,18 @@
12
12
  //
13
13
  // Commands (kebab-case) map to API.md actions:
14
14
  // get-dispatch-state, get-state, create-task, update-task, claim-task,
15
- // park-task, recover-task, upsert-session, log-event, get-events,
15
+ // park-task, recover-task, consume-next-phase, upsert-session, log-event, get-events,
16
16
  // create-project, update-project, delete-project, list-projects,
17
17
  // get-project, kill-switch, get-config, update-config,
18
18
  // set-provenance, get-provenance, acknowledge-poll,
19
+ // scan-verification-pending, resolve-publish-unknown,
19
20
  // record-phase (composite), migrate (one-time import)
20
21
  //
21
22
  // Output is JSON on stdout. Errors are JSON on stderr.
22
23
  // Exit codes: 0 ok · 2 usage/validation · 3 not found · 4 conflict/guard.
23
24
 
24
- import { readFileSync, existsSync, statSync } from "node:fs";
25
- import { join, resolve, sep } from "node:path";
25
+ import { readFileSync, existsSync, statSync, readlinkSync, readdirSync, appendFileSync } from "node:fs";
26
+ import { join, resolve, sep, basename } from "node:path";
26
27
  import { randomUUID } from "node:crypto";
27
28
  import { DatabaseSync } from "node:sqlite";
28
29
  import { homedir } from "node:os";
@@ -259,6 +260,53 @@ function workflowExists(crewHome, slug) {
259
260
  );
260
261
  }
261
262
 
263
+ // Workflow step names for a slug, from the release registry when present,
264
+ // falling back to the workflow file's own `export const meta` block (the
265
+ // single source of truth the registry is built from). Returns null when the
266
+ // workflow file cannot be found or parsed. Used by recover-task to validate
267
+ // target_phase, and by the dispatcher-side contract tests.
268
+ function workflowStepNames(crewHome, slug) {
269
+ if (!slug) return null;
270
+ const candidates = [
271
+ join(crewHome, "current", "workflows", "registry.json"),
272
+ join(crewHome, "workflows", "registry.json"),
273
+ ];
274
+ for (const regPath of candidates) {
275
+ try {
276
+ const reg = JSON.parse(readFileSync(regPath, "utf8"));
277
+ const entry = reg && reg[slug];
278
+ if (entry && Array.isArray(entry.steps)) {
279
+ const names = entry.steps.map((s) => s && s.name).filter((n) => typeof n === "string");
280
+ if (names.length > 0) return names;
281
+ }
282
+ } catch (e) { /* fall through to the meta-block parse */ }
283
+ }
284
+ const wfCandidates = [
285
+ join(crewHome, "current", "workflows", `${slug}.js`),
286
+ join(crewHome, "workflows", `${slug}.js`),
287
+ ];
288
+ for (const wfPath of wfCandidates) {
289
+ let source;
290
+ try { source = readFileSync(wfPath, "utf8"); } catch (e) { continue; }
291
+ const marker = "export const meta = ";
292
+ const start = source.indexOf(marker);
293
+ if (start < 0) continue;
294
+ const bodyStart = start + marker.length;
295
+ const end = source.indexOf("\n};", bodyStart);
296
+ if (end < 0) continue;
297
+ try {
298
+ // The meta block is a static literal (no function calls); evaluating
299
+ // the object literal in isolation is safe.
300
+ const meta = new Function("return (" + source.slice(bodyStart, end + 2) + ");")();
301
+ if (meta && Array.isArray(meta.steps)) {
302
+ const names = meta.steps.map((s) => s && s.name).filter((n) => typeof n === "string");
303
+ if (names.length > 0) return names;
304
+ }
305
+ } catch (e) { /* try the next candidate */ }
306
+ }
307
+ return null;
308
+ }
309
+
262
310
  // Minimal git-repo validation (ported from the dashboard's validateRepoPath):
263
311
  // accepts a repo root (.git dir) or a linked worktree (.git file).
264
312
  function isGitRepoPath(p) {
@@ -535,6 +583,17 @@ commands["claim-task"] = (db, args) => {
535
583
  if (!args.task_id) throw usageError("task_id is required.");
536
584
  const identity = (args.identity ?? "").trim();
537
585
  if (identity.length < 1 || identity.length > 80) throw usageError("identity is required (1-80 chars).");
586
+ // ATOMIC NEXT_PHASE CONSUMPTION (2026-09-15): expected_next_phase is the
587
+ // routing the dispatcher launched this run for. When the claim wins, the
588
+ // matching next_phase is cleared in the SAME transaction as the winning
589
+ // session insert — there is no claim→consume window for a platform death
590
+ // to replay the routing through. The separate consume-next-phase command
591
+ // is retained only as an idempotent public helper (out-of-band recovery);
592
+ // workflows no longer call it.
593
+ const rawExpected = args.expected_next_phase;
594
+ const expectedNextPhase = (rawExpected === undefined || rawExpected === null) ? null : String(rawExpected).trim();
595
+ if (expectedNextPhase !== null && (expectedNextPhase.length < 1 || expectedNextPhase.length > 120))
596
+ throw usageError("expected_next_phase must be 1-120 chars when provided.");
538
597
  const task = requireTask(db, args.task_id);
539
598
  const project = requireProject(db, task.project);
540
599
  if (project.quiesced) throw conflict("This project is paused. Resume it before claiming tasks.");
@@ -552,23 +611,41 @@ commands["claim-task"] = (db, args) => {
552
611
  notes: (args.notes ?? "").trim().slice(0, 3000), caveats: "[]",
553
612
  last_heartbeat: now(),
554
613
  };
555
- // Atomic: the partial unique index on (task_id) WHERE status='running'
556
- // turns a duplicate claim into a no-op insert.
557
- const info = db.prepare(
558
- `INSERT INTO agent_sessions (id, task_id, identity, step, status, started_at,
559
- ended_at, notes, caveats, last_heartbeat)
560
- VALUES (@id, @task_id, @identity, @step, @status, @started_at,
561
- @ended_at, @notes, @caveats, @last_heartbeat)
562
- ON CONFLICT DO NOTHING`).run(row);
563
- if (info.changes > 0) {
564
- return { ok: true, claimed: true, session_id: row.id, session: mapSession({ ...row, failure_reason: null }) };
565
- }
566
- const existing = db.prepare(
567
- `SELECT id FROM agent_sessions
568
- WHERE task_id = ? AND status = 'running'
569
- ORDER BY started_at DESC LIMIT 1`).get(args.task_id);
570
- if (!existing) throw new CrewError("claim_unresolved", "Task claim could not be resolved. Please retry.", 4);
571
- return { ok: true, claimed: false, reason: "already_claimed", existing_session_id: existing.id };
614
+ const timestamp = now();
615
+ db.exec("BEGIN");
616
+ try {
617
+ // Atomic: the partial unique index on (task_id) WHERE status='running'
618
+ // turns a duplicate claim into a no-op insert.
619
+ const info = db.prepare(
620
+ `INSERT INTO agent_sessions (id, task_id, identity, step, status, started_at,
621
+ ended_at, notes, caveats, last_heartbeat)
622
+ VALUES (@id, @task_id, @identity, @step, @status, @started_at,
623
+ @ended_at, @notes, @caveats, @last_heartbeat)
624
+ ON CONFLICT DO NOTHING`).run(row);
625
+ // Same transaction: only the claim winner consumes, and only when the
626
+ // stored routing still matches what the dispatcher launched on — a newer
627
+ // recover-task written in the race window survives untouched.
628
+ let nextPhaseConsumed = false;
629
+ if (info.changes > 0 && expectedNextPhase) {
630
+ const cleared = db.prepare(
631
+ "UPDATE tasks SET next_phase = NULL, updated_at = ? WHERE id = ? AND next_phase = ?"
632
+ ).run(timestamp, args.task_id, expectedNextPhase);
633
+ nextPhaseConsumed = cleared.changes > 0;
634
+ }
635
+ db.exec("COMMIT");
636
+ if (info.changes > 0) {
637
+ return { ok: true, claimed: true, session_id: row.id, session: mapSession({ ...row, failure_reason: null }),
638
+ next_phase_consumed: expectedNextPhase ? nextPhaseConsumed : undefined };
639
+ }
640
+ const existing = db.prepare(
641
+ `SELECT id FROM agent_sessions
642
+ WHERE task_id = ? AND status = 'running'
643
+ ORDER BY started_at DESC LIMIT 1`).get(args.task_id);
644
+ if (!existing) throw new CrewError("claim_unresolved", "Task claim could not be resolved. Please retry.", 4);
645
+ // Lost claim: the winner (or an earlier winner) owns next_phase
646
+ // consumption — a loser must never clear it.
647
+ return { ok: true, claimed: false, reason: "already_claimed", existing_session_id: existing.id };
648
+ } catch (e) { db.exec("ROLLBACK"); throw e; }
572
649
  };
573
650
 
574
651
  // Dispatch reservations: the cron worker creates one after launching a
@@ -844,6 +921,9 @@ commands["retry-platform-failure"] = (db, args) => {
844
921
  if (failure.retry_count >= maxRetries) {
845
922
  // Park the task — transient failures are not resolving.
846
923
  const task = db.prepare("SELECT * FROM tasks WHERE id = ?").get(failure.task_id);
924
+ if (task && task.state === "done") {
925
+ return { ok: true, action: "skipped", reason: "task already done — a platform retry never parks a done task", retry_count: failure.retry_count };
926
+ }
847
927
  if (task && task.state !== "parked") {
848
928
  const msg = "Platform workflow failed " + (failure.retry_count + 1) + "x: " +
849
929
  failure.error_message.substring(0, 200);
@@ -874,19 +954,47 @@ commands["retry-platform-failure"] = (db, args) => {
874
954
  }
875
955
  return { ok: true, action: "parked", reason: "max retries exceeded", retry_count: failure.retry_count, settled_sessions: 0 };
876
956
  }
877
- // Clear reservation and re-queue for retry.
957
+ // Clear the reservation and re-queue for retry — preserving the task's
958
+ // phase. Canary 2026-09-15 (task 1d692d91): the old code forced
959
+ // state='todo', so a platform failure during QA restarted the entire
960
+ // workflow at Triage; Build reimplemented an already-merged and
961
+ // already-verified task and produced an unrequested variant. The
962
+ // dispatcher already resumes a failed/timed_out/stalled session at that
963
+ // session's own step — so the fix is to leave the task's state alone and
964
+ // settle the ghost session as 'stalled', letting the dispatcher resume at
965
+ // the failed phase instead of restarting at Triage. 'done' is never
966
+ // resurrected and 'parked' is never unparked: parked is the designed
967
+ // human decision point.
968
+ const retryTask = db.prepare("SELECT id, state FROM tasks WHERE id = ?").get(failure.task_id);
969
+ if (!retryTask) {
970
+ return { ok: true, action: "skipped", reason: "task not found" };
971
+ }
972
+ if (retryTask.state === "done") {
973
+ return { ok: true, action: "skipped", reason: "task already done — never resurrected by a platform retry" };
974
+ }
975
+ if (retryTask.state === "parked") {
976
+ return { ok: true, action: "skipped", reason: "task parked — the human decision point, never unparked by a platform retry" };
977
+ }
978
+ // The failed phase is the ghost session's step: the dispatcher resumes
979
+ // failed/timed_out/stalled sessions at that step. Capture it before
980
+ // settling so the requeue reports where the run will resume.
981
+ const ghost = db.prepare(
982
+ `SELECT step FROM agent_sessions WHERE task_id = ? AND status = 'running'
983
+ ORDER BY started_at DESC LIMIT 1`
984
+ ).get(failure.task_id);
878
985
  db.exec("BEGIN");
879
986
  try {
880
987
  db.prepare("DELETE FROM dispatch_reservations WHERE task_id = ?").run(failure.task_id);
881
988
  db.prepare(
882
- "UPDATE tasks SET state = 'todo', updated_at = datetime('now') WHERE id = ? AND state != 'done'"
989
+ "UPDATE tasks SET updated_at = datetime('now') WHERE id = ?"
883
990
  ).run(failure.task_id);
884
991
  // Settle the ghost session atomically with the requeue: the platform
885
992
  // workflow is dead, but its agent_sessions row still says 'running' —
886
993
  // and the dispatcher skips any task whose latest session is running
887
994
  // (crew-dispatch.js "work in flight"). Without this, the requeued task
888
995
  // would sit until the 1-hour zombie-session sweep. 'stalled' is what
889
- // the dispatcher already treats as a retry candidate.
996
+ // the dispatcher already treats as a retry candidate, resuming at the
997
+ // session's own step.
890
998
  const ts = new Date().toISOString();
891
999
  const settled = db.prepare(
892
1000
  `UPDATE agent_sessions
@@ -904,6 +1012,8 @@ commands["retry-platform-failure"] = (db, args) => {
904
1012
  ok: true,
905
1013
  action: "requeued",
906
1014
  task_id: failure.task_id,
1015
+ state: retryTask.state,
1016
+ resume_step: ghost ? ghost.step : null,
907
1017
  retry_count: failure.retry_count + 1,
908
1018
  settled_sessions: Number(settled.changes),
909
1019
  };
@@ -950,41 +1060,77 @@ commands["recover-task"] = (db, args, ctx) => {
950
1060
  const targetPhase = (args.target_phase ?? "").trim();
951
1061
  if (targetPhase.length < 1 || targetPhase.length > 120) throw usageError("target_phase is required (1-120 chars).");
952
1062
  const task = requireTask(db, args.task_id);
1063
+ if (!task.workflow) throw conflict("This task does not have a workflow with selectable phases.");
1064
+ if (!workflowExists(ctx.crewHome, task.workflow)) throw notFound("Workflow not found.");
1065
+ // Validate the target against the workflow's own phase registry — an
1066
+ // unknown phase fails closed here, never as a dispatcher skip or a
1067
+ // workflow launched at the wrong step.
1068
+ const stepNames = workflowStepNames(ctx.crewHome, task.workflow);
1069
+ if (!stepNames) throw conflict(`Could not read the phase registry for workflow "${task.workflow}".`);
1070
+ if (!stepNames.includes(targetPhase)) {
1071
+ throw conflict(`Unknown phase "${targetPhase}" for workflow "${task.workflow}". Valid phases: ${stepNames.join(", ")}.`);
1072
+ }
1073
+ // Recovery intent, not a verdict: the latest session may be failed,
1074
+ // timed_out, stalled — or absent entirely (the platform died before the
1075
+ // first claim). A running session means work is in flight; completed,
1076
+ // rejected, passed, and superseded sessions have their own normal paths
1077
+ // (next step, rework) and are not recovery targets.
953
1078
  const latest = db.prepare(
954
1079
  `SELECT * FROM agent_sessions WHERE task_id = ? ORDER BY started_at DESC LIMIT 1`
955
1080
  ).get(args.task_id);
956
- if (!latest || !["failed", "timed_out"].includes(latest.status)) {
957
- throw conflict("The latest session is not failed or timed out.");
1081
+ if (latest && !["failed", "timed_out", "stalled"].includes(latest.status)) {
1082
+ throw conflict(`The latest session is ${latest.status}; recover-task accepts only failed, timed_out, stalled, or no session.`);
958
1083
  }
959
- if (!task.workflow) throw conflict("This task does not have a workflow with selectable phases.");
960
- if (!workflowExists(ctx.crewHome, task.workflow)) throw notFound("Workflow not found.");
1084
+ if (["done", "cancelled"].includes(task.state)) {
1085
+ throw conflict(`Task is ${task.state}; recovery applies to todo, in_progress, or parked tasks.`);
1086
+ }
1087
+ // No synthetic session: recovery intent must not consume retry budget.
1088
+ // The dispatcher routes on tasks.next_phase directly (one-shot, consumed
1089
+ // by the claiming workflow), so no failed-session marker is needed to
1090
+ // make the task eligible.
961
1091
  const timestamp = now();
962
- const queued = {
963
- id: uuid(), task_id: task.id,
964
- identity: latest.identity, step: targetPhase, status: "failed",
965
- started_at: timestamp, ended_at: timestamp,
966
- notes: (args.updated_description ?? "").trim().slice(0, 5000),
967
- caveats: "[]", failure_reason: null, last_heartbeat: timestamp,
968
- };
1092
+ const newState = (task.state === "parked" || task.state === "todo") ? "in_progress" : task.state;
1093
+ const note = (args.updated_description ?? "").trim().slice(0, 5000);
969
1094
  db.exec("BEGIN");
970
1095
  try {
971
- db.prepare("UPDATE tasks SET next_phase = ?, updated_at = ? WHERE id = ?")
972
- .run(targetPhase, timestamp, task.id);
973
- db.prepare("UPDATE agent_sessions SET ended_at = ? WHERE id = ?").run(timestamp, latest.id);
1096
+ db.prepare("UPDATE tasks SET next_phase = ?, state = ?, updated_at = ? WHERE id = ?")
1097
+ .run(targetPhase, newState, timestamp, task.id);
974
1098
  db.prepare(
975
- `INSERT INTO agent_sessions (id, task_id, identity, step, status, started_at,
976
- ended_at, notes, caveats, failure_reason, last_heartbeat)
977
- VALUES (@id, @task_id, @identity, @step, @status, @started_at,
978
- @ended_at, @notes, @caveats, @failure_reason, @last_heartbeat)`).run(queued);
1099
+ `INSERT INTO events (id, type, task_id, identity, message, timestamp)
1100
+ VALUES (?, 'note', ?, 'recover-task', ?, ?)`
1101
+ ).run(uuid(), task.id,
1102
+ `Recovery: next phase set to ${targetPhase}` +
1103
+ (task.state !== newState ? ` (state ${task.state} -> ${newState})` : "") +
1104
+ (note ? `. ${note}` : ""),
1105
+ timestamp);
979
1106
  db.exec("COMMIT");
980
1107
  } catch (e) { db.exec("ROLLBACK"); throw e; }
981
1108
  const taskStates = new Map(db.prepare("SELECT id, state FROM tasks").all().map((r) => [r.id, r.state]));
982
1109
  return {
983
- task: mapTask({ ...task, next_phase: targetPhase, updated_at: timestamp }, taskStates),
984
- session: mapSession(queued),
1110
+ task: mapTask({ ...task, next_phase: targetPhase, state: newState, updated_at: timestamp }, taskStates),
1111
+ valid_phases: stepNames,
985
1112
  };
986
1113
  };
987
1114
 
1115
+ // Retained as an idempotent public helper for out-of-band recovery tooling.
1116
+ // Workflows no longer call this: claim-task with expected_next_phase clears
1117
+ // the routed next_phase in the same transaction as the winning session
1118
+ // insert, so the old claim→consume death window is closed. The conditional
1119
+ // UPDATE keeps this exactly-once for any other caller: only the claim winner
1120
+ // clears, and only when the value still matches what the dispatcher routed
1121
+ // on — a newer recover-task written in the race window survives untouched.
1122
+ commands["consume-next-phase"] = (db, args) => {
1123
+ if (!args.task_id) throw usageError("task_id is required.");
1124
+ const expected = (args.expected ?? "").trim();
1125
+ if (expected.length < 1 || expected.length > 120) throw usageError("expected is required (1-120 chars).");
1126
+ requireTask(db, args.task_id);
1127
+ const timestamp = now();
1128
+ const info = db.prepare(
1129
+ "UPDATE tasks SET next_phase = NULL, updated_at = ? WHERE id = ? AND next_phase = ?"
1130
+ ).run(timestamp, args.task_id, expected);
1131
+ return { consumed: info.changes > 0, task_id: args.task_id };
1132
+ };
1133
+
988
1134
  commands["upsert-session"] = (db, args) => {
989
1135
  if (!args.task_id) throw usageError("task_id is required.");
990
1136
  requireTask(db, args.task_id);
@@ -1063,6 +1209,284 @@ commands["get-events"] = (db, args) => {
1063
1209
  return { events: rows.map(mapEvent) };
1064
1210
  };
1065
1211
 
1212
+ // Parent publish verification: scan + atomic claim + reconcile.
1213
+ //
1214
+ // Finds parked tasks whose latest parent publish note is
1215
+ // "publish: verification-requested" (the publisher parks instead of
1216
+ // stamping, per docs/publish-verification.md) with no terminal parent
1217
+ // verdict and no unexpired verification claim. Each returned task is
1218
+ // ATOMICALLY claimed by logging a "publish: verification-claimed"
1219
+ // note-event with a lease expiry, so two concurrent scanners (or a
1220
+ // retrying tick) cannot start a second verification for the same task.
1221
+ // The claim is a note, not a state change: the task stays parked, so the
1222
+ // dispatcher never dispatches QA mid-verification.
1223
+ //
1224
+ // Also reconciles the verified-but-still-parked gap: a parked task whose
1225
+ // latest parent verdict is "publish: verified" is re-queued to in_progress
1226
+ // (the verdict was recorded but the re-queue was lost to a crash).
1227
+ // Failure verdicts are left parked for human attention (fail-closed).
1228
+ //
1229
+ // Returns { to_verify: [...], reconciled: [...] }.
1230
+ // Each to_verify entry: { task_id, commit, build_agent_id, project_id,
1231
+ // repo_path, deploy_slug, crew_release, claim_expires_at }.
1232
+ //
1233
+ // crew_release is resolved through the crew home's `current` symlink (the
1234
+ // immutable active release the release manager maintains) and cross-checked
1235
+ // against this code's own realpath (the release it is actually running
1236
+ // from). Unresolvable or disputed => throw fail-closed BEFORE the claim
1237
+ // transaction: provenance is never stamped with an unknown release.
1238
+ function resolveActiveRelease(crewHome) {
1239
+ let linkTarget;
1240
+ try {
1241
+ linkTarget = readlinkSync(join(crewHome, "current"));
1242
+ } catch (e) {
1243
+ throw usageError(
1244
+ `scan-verification-pending: cannot resolve the active release: ${join(crewHome, "current")} is not a readable symlink (${e.message}). Refusing to verify.`
1245
+ );
1246
+ }
1247
+ const activeRelease = basename(resolve(crewHome, linkTarget));
1248
+ const selfPath = new URL(import.meta.url).pathname;
1249
+ const selfMatch = selfPath.match(/releases\/([^/]+)\/lib\/crew-api\.js$/);
1250
+ if (!selfMatch) {
1251
+ throw usageError(
1252
+ `scan-verification-pending: running code is not inside a releases/<name>/lib path (${selfPath}). Refusing to verify.`
1253
+ );
1254
+ }
1255
+ if (selfMatch[1] !== activeRelease) {
1256
+ throw usageError(
1257
+ `scan-verification-pending: running release ${selfMatch[1]} != active release ${activeRelease} (stale cron body or mid-deploy). Refusing to verify.`
1258
+ );
1259
+ }
1260
+ return activeRelease;
1261
+ }
1262
+ commands["scan-verification-pending"] = (db, args, ctx) => {
1263
+ const CLAIM_LEASE_MS = 60 * 60 * 1000; // 1 hour
1264
+ const nowMs = Date.now();
1265
+ const claimExpiry = new Date(nowMs + CLAIM_LEASE_MS).toISOString();
1266
+ const releaseName = resolveActiveRelease(ctx.crewHome);
1267
+
1268
+ const parked = db.prepare(
1269
+ `SELECT t.id AS task_id, t.project AS project_id,
1270
+ p.repo_path AS repo_path, p.deploy_slug AS deploy_slug
1271
+ FROM tasks t LEFT JOIN projects p ON p.id = t.project
1272
+ WHERE t.state = 'parked'`
1273
+ ).all();
1274
+
1275
+ const toVerify = [];
1276
+ const reconciled = [];
1277
+
1278
+ db.exec("BEGIN");
1279
+ try {
1280
+ for (const row of parked) {
1281
+ const notes = db.prepare(
1282
+ `SELECT message, timestamp FROM events
1283
+ WHERE task_id = ? AND type = 'note' AND message LIKE '%publish:%'
1284
+ ORDER BY timestamp DESC LIMIT 20`
1285
+ ).all(row.task_id);
1286
+ if (notes.length === 0) continue;
1287
+ const latest = notes[0].message;
1288
+
1289
+ // Reconcile first: verdict recorded but the re-queue was lost to a
1290
+ // crash (verified-but-still-parked). Failure verdicts stay parked.
1291
+ // Contained-string matching: the workflow's parkTask prepends
1292
+ // "Parked: " to the reason, so a verification request is stored as
1293
+ // "Parked: publish: verification-requested ..." (observed 2026-09-14).
1294
+ // Prefix anchors would never match; see tests/verify-publish.test.js.
1295
+ if (latest.includes("publish: verified")) {
1296
+ db.prepare("UPDATE tasks SET state = 'in_progress', updated_at = ? WHERE id = ?")
1297
+ .run(now(), row.task_id);
1298
+ db.prepare(
1299
+ `INSERT INTO events (id, type, task_id, identity, message, timestamp)
1300
+ VALUES (?, 'note', ?, NULL, ?, ?)`
1301
+ ).run(uuid(), row.task_id,
1302
+ "publish: reconciled verified-but-parked -> in_progress (re-queue lost to a crash)",
1303
+ now());
1304
+ reconciled.push({ task_id: row.task_id });
1305
+ continue;
1306
+ }
1307
+
1308
+ // Terminal parent verdicts (anything the parent decided, except a
1309
+ // fresh request, a procedural error, or an active claim) — never re-verify.
1310
+ const isRequest = latest.includes("publish: verification-requested");
1311
+ const isClaim = latest.includes("publish: verification-claimed");
1312
+ // Procedural errors (worker botched the procedure, not a content
1313
+ // failure) are retryable — treat like a fresh request. Content
1314
+ // failures (verification-failed) stay terminal.
1315
+ const isProceduralError = latest.includes("publish: verification-procedural-error");
1316
+ if (!isRequest && !isClaim && !isProceduralError) continue; // failed / blocked / mismatch: terminal, fail-closed
1317
+
1318
+ if (isClaim) {
1319
+ // Unexpired claim -> another verifier owns it. Expired -> re-claim below.
1320
+ const m = latest.match(/publish: verification-claimed (\S+)/);
1321
+ if (m && Date.parse(m[1]) > nowMs) continue;
1322
+ }
1323
+
1324
+ // Candidate: find the original verification-requested note to extract
1325
+ // the commit and build agent id (the latest note may be an expired claim
1326
+ // or a procedural error).
1327
+ const req = notes.find((n) => n.message.includes("publish: verification-requested"));
1328
+ if (!req) continue;
1329
+ const commitMatch = req.message.match(/verification-requested ([0-9a-f]{40})/);
1330
+ // Fallback: procedural error note carries the commit too.
1331
+ const procMatch = !commitMatch ? latest.match(/verification-procedural-error ([0-9a-f]{40})/) : null;
1332
+ const commit = commitMatch ? commitMatch[1] : (procMatch ? procMatch[1] : null);
1333
+ const buildMatch = req.message.match(/\(build ([0-9a-f-]{36})\)/);
1334
+ if (!commit) continue; // malformed request: cannot verify mechanically, leave parked
1335
+ db.prepare(
1336
+ `INSERT INTO events (id, type, task_id, identity, message, timestamp)
1337
+ VALUES (?, 'note', ?, NULL, ?, ?)`
1338
+ ).run(uuid(), row.task_id, `publish: verification-claimed ${claimExpiry}`, now());
1339
+ toVerify.push({
1340
+ task_id: row.task_id,
1341
+ commit: commit,
1342
+ build_agent_id: buildMatch ? buildMatch[1] : null,
1343
+ project_id: row.project_id,
1344
+ repo_path: row.repo_path,
1345
+ deploy_slug: row.deploy_slug,
1346
+ crew_release: releaseName,
1347
+ claim_expires_at: claimExpiry,
1348
+ });
1349
+ }
1350
+ db.exec("COMMIT");
1351
+ } catch (e) { db.exec("ROLLBACK"); throw e; }
1352
+ return { to_verify: toVerify, reconciled };
1353
+ };
1354
+
1355
+ // Recovery contract for publish attempts parked with an UNKNOWN outcome
1356
+ // (2026-09-14). The structured-output fallback parked the task because no
1357
+ // build could be observed — but the edit may still have gone through (the
1358
+ // in-flight-only fallback could not see a completed build; fixed
1359
+ // in-workflow the same day with the durable audit-dir check). When durable
1360
+ // evidence later shows the build DID complete — a platform audit directory
1361
+ // created inside the publish window (between Integrate completion and the
1362
+ // park event) — this action routes the task to the parent's independent
1363
+ // content verification WITHOUT re-issuing the edit and WITHOUT stamping
1364
+ // provenance. The original unknown ledger entry and park event are
1365
+ // preserved; the resolution is appended to the ledger and the event log.
1366
+ // The parent's read-back (docs/publish-verification.md) remains the real
1367
+ // verification and can still fail terminally. Never: blind retry, manual
1368
+ // stamp, silent state change.
1369
+ commands["resolve-publish-unknown"] = (db, args, ctx) => {
1370
+ const taskId = args.task_id;
1371
+ if (!taskId) throw usageError("task_id is required.");
1372
+ requireTask(db, taskId);
1373
+ const task = db.prepare("SELECT id, state, project FROM tasks WHERE id = ?").get(taskId);
1374
+ if (task.state !== "parked") {
1375
+ return { resolved: false, reason: `task is not parked (state=${task.state})` };
1376
+ }
1377
+ const project = db.prepare("SELECT deploy_slug FROM projects WHERE id = ?").get(task.project);
1378
+ const slug = project && project.deploy_slug;
1379
+ if (!slug) return { resolved: false, reason: "project has no deploy_slug" };
1380
+
1381
+ // The latest publish ledger entry for this task must be outcome "unknown".
1382
+ const ledgerPath = join(ctx.crewHome, ".publish-ledger", slug + ".jsonl");
1383
+ let entries = [];
1384
+ try {
1385
+ entries = readFileSync(ledgerPath, "utf8").split("\n")
1386
+ .filter((l) => l.trim().length > 0)
1387
+ .map((l) => JSON.parse(l))
1388
+ .filter((e) => e.task_id === taskId);
1389
+ } catch (e) {
1390
+ return { resolved: false, reason: `cannot read publish ledger: ${e.message}` };
1391
+ }
1392
+ if (entries.length === 0) return { resolved: false, reason: "no publish ledger entries for this task" };
1393
+ const latest = entries[entries.length - 1];
1394
+ if (latest.outcome === "unknown-resolved") {
1395
+ return { resolved: false, reason: "already resolved (unknown-resolved ledger entry present)" };
1396
+ }
1397
+ if (latest.outcome !== "unknown") {
1398
+ return { resolved: false, reason: `latest ledger outcome is '${latest.outcome}', not unknown` };
1399
+ }
1400
+ if (!latest.commit || !/^[0-9a-f]{40}$/.test(latest.commit)) {
1401
+ return { resolved: false, reason: "unknown ledger entry carries no usable commit hash" };
1402
+ }
1403
+
1404
+ // The publish window: Integrate completion -> the unknown-outcome park.
1405
+ // Nothing else touches the artifact during Publish, so an audit dir
1406
+ // created inside this window belongs to this publish attempt.
1407
+ const events = db.prepare(
1408
+ "SELECT type, message, timestamp FROM events WHERE task_id = ? ORDER BY timestamp ASC"
1409
+ ).all(taskId);
1410
+ const integrateDone = [...events].reverse().find((e) =>
1411
+ e.type === "completed" && /integrate completed/i.test(e.message || ""));
1412
+ const parkEvent = [...events].reverse().find((e) =>
1413
+ e.type === "note" && (e.message || "").includes("Publish outcome unknown"));
1414
+ if (!integrateDone || !parkEvent) {
1415
+ return { resolved: false, reason: "cannot determine publish window (missing integrate-completion or unknown-outcome park event)" };
1416
+ }
1417
+ const winStart = Date.parse(integrateDone.timestamp);
1418
+ const winEnd = Date.parse(parkEvent.timestamp);
1419
+ if (!(winStart < winEnd)) {
1420
+ return { resolved: false, reason: "publish window is empty or inverted" };
1421
+ }
1422
+
1423
+ // Audit dirs are named <UTC-timestamp>-<id>; the timestamp is the build's.
1424
+ const auditsDir = join(homedir(), "workspace", "ts-spaces", slug, "audits");
1425
+ let evidenceDirs = [];
1426
+ try {
1427
+ evidenceDirs = readdirSync(auditsDir).filter((name) => {
1428
+ const m = /^(\d{4})-(\d{2})-(\d{2})T(\d{2})-(\d{2})-(\d{2})Z-/.exec(name);
1429
+ if (!m) return false;
1430
+ const t = Date.parse(`${m[1]}-${m[2]}-${m[3]}T${m[4]}:${m[5]}:${m[6]}Z`);
1431
+ return t > winStart && t <= winEnd;
1432
+ }).sort();
1433
+ } catch (e) {
1434
+ return { resolved: false, reason: `cannot list audit dirs: ${e.message}` };
1435
+ }
1436
+ if (evidenceDirs.length === 0) {
1437
+ return { resolved: false, reason: "no audit build completed inside the publish window — still unknown" };
1438
+ }
1439
+ const evidenceDir = evidenceDirs[0];
1440
+
1441
+ // Corroboration, observation only: the audit harness's own verdict.
1442
+ let auditOk = null;
1443
+ try {
1444
+ auditOk = JSON.parse(readFileSync(join(auditsDir, evidenceDir, "report.json"), "utf8")).ok === true;
1445
+ } catch (e) { auditOk = null; }
1446
+
1447
+ const ts = now();
1448
+ // The verification-requested note must sort strictly after the
1449
+ // unknown-resolved note: scan-verification-pending reads the LATEST
1450
+ // publish: note, and equal timestamps would leave the order undefined.
1451
+ const tsLater = new Date(Date.parse(ts) + 1000).toISOString();
1452
+ db.exec("BEGIN");
1453
+ try {
1454
+ db.prepare(
1455
+ "INSERT INTO events (id, type, task_id, identity, message, timestamp) VALUES (?, 'note', ?, NULL, ?, ?)"
1456
+ ).run(uuid(), taskId,
1457
+ `publish: unknown-resolved ${latest.commit} (audit ${evidenceDir}, report ok=${auditOk}) — durable build evidence inside the publish window; routing to parent content verification. No re-trigger issued, no provenance stamped, original unknown outcome preserved.`,
1458
+ ts);
1459
+ // Mirror the workflow's verification-requested park note so
1460
+ // scan-verification-pending claims it on the next tick.
1461
+ db.prepare(
1462
+ "INSERT INTO events (id, type, task_id, identity, message, timestamp) VALUES (?, 'note', ?, NULL, ?, ?)"
1463
+ ).run(uuid(), taskId,
1464
+ `Parked: publish: verification-requested ${latest.commit} (build agent_id unobserved) — recovered from publish-unknown via resolve-publish-unknown; durable evidence ${evidenceDir}. Parent: run docs/publish-verification.md.`,
1465
+ tsLater);
1466
+ db.prepare("UPDATE tasks SET state = 'parked', updated_at = ? WHERE id = ?").run(tsLater, taskId);
1467
+ db.exec("COMMIT");
1468
+ } catch (e) { db.exec("ROLLBACK"); throw e; }
1469
+
1470
+ // Append-only audit trail: the resolution lands in the ledger next to the
1471
+ // original unknown entry (which is never rewritten).
1472
+ try {
1473
+ appendFileSync(ledgerPath, JSON.stringify({
1474
+ ts: new Date(ts).toISOString(),
1475
+ task_id: taskId,
1476
+ workflow: latest.workflow || "unknown",
1477
+ slug,
1478
+ commit: latest.commit,
1479
+ attempt: latest.attempt,
1480
+ agent_id: null,
1481
+ applied_report: null,
1482
+ outcome: "unknown-resolved",
1483
+ detail: `durable build evidence ${evidenceDir} (audit report ok=${auditOk}) inside publish window; routed to parent verification without re-trigger`,
1484
+ }) + "\n");
1485
+ } catch (e) { /* best-effort observability */ }
1486
+
1487
+ return { resolved: true, task_id: taskId, commit: latest.commit, evidence_dir: evidenceDir, audit_report_ok: auditOk };
1488
+ };
1489
+
1066
1490
  // Composite: record a phase's session verdict AND its event in one
1067
1491
  // transaction. This is the operation that used to crash the workflow when the
1068
1492
  // two writes went to different contracts: the session stored, the event