muse-crew 0.7.10 → 0.7.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -27,6 +27,15 @@ const startStepIndex = inputs.start_step_index || 0;
27
27
  // resolution back via updatetask in the self-claim below.
28
28
  const RESOLVED_WORKFLOW = inputs.resolved_workflow || null;
29
29
  const WORKFLOW_WAS_NULL = inputs.workflow_was_null === true;
30
+ // One-shot recovery routing: the dispatcher sets inputs.next_phase when it
31
+ // routes this run via an explicit recover-task redirect. The value is
32
+ // consumed (cleared) atomically by the successful self-claim below:
33
+ // claim-task takes expected_next_phase and clears the matching next_phase in
34
+ // the same transaction as the winning session insert, so no platform death
35
+ // can slip between claim and consumption and replay the routing. A stale or
36
+ // superseded routing survives — only an exact match clears.
37
+ // what the dispatcher routed on.
38
+ const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_phase.length > 0) ? inputs.next_phase : null;
30
39
  const CLAIM_WORKFLOW_PERSIST = (WORKFLOW_WAS_NULL && RESOLVED_WORKFLOW) ? ", \"workflow\": \"" + RESOLVED_WORKFLOW + "\"" : "";
31
40
 
32
41
  // Visual verdict protocol availability — the workflow parks for parent-run
@@ -472,52 +481,23 @@ function verifyAppliedChanges(expected, applied) {
472
481
  return { ok: true };
473
482
  }
474
483
 
475
- // Publish read-back request builder: the verbatim_request the workflow hands
476
- // to artifact_inspect (via a child) after the artifact build lands. Pure
477
- // function — no I/O, no clock. The request carries the merged diff as the
478
- // expected change and asks for an independent read of the artifact's actual
479
- // source: for each file, the exact current text of the changed regions plus
480
- // a per-line present/absent finding. The parent (docs/publish-verification.md)
481
- // compares these findings against the diff mechanically and stamps provenance
482
- // only on a match. This breaks the circularity that hollowed canary run 8
483
- // (2026-09-11): verifyAppliedChanges compares the builder's applied-report
484
- // against the diff the report was derived from — a fabricated report passes
485
- // by construction. Independent read-back cannot be fabricated from the diff;
484
+ // Publish read-back request (currently unavailable): the verbatim_request
485
+ // the parent protocol (docs/publish-verification.md) would hand to an
486
+ // independent read-back tool after the artifact build lands. artifact_inspect
487
+ // was removed by the platform (2026-09-14); artifact.inspect is malfunction
488
+ // diagnosis, not a substitute — so no agent-callable read-back tool exists
489
+ // and this request cannot currently be issued. Pure function — no I/O, no
490
+ // clock. The request carries the merged diff as the expected change and asks
491
+ // for an independent read of the artifact's actual source: for each file, the
492
+ // exact current text of the changed regions plus a per-line present/absent
493
+ // finding. Until a read-back path exists, the parent cannot independently
494
+ // confirm content and verification parks at "publish: verification-requested"
495
+ // (see docs/publish-verification.md). This preserves the circularity break
496
+ // that hollowed canary run 8 (2026-09-11): verifyAppliedChanges compares the
497
+ // builder's applied-report against the diff the report was derived from — a
498
+ // fabricated report passes by construction. Independent read-back cannot be
499
+ // fabricated from the diff;
486
500
  // it must match the artifact's real content.
487
- function buildPublishReadbackRequest(taskId, commit, diff, buildAgentId) {
488
- // Build-ID correlation (2026-09-12): buildAgentId is the build.agent_id the
489
- // workflow observed for the publish attempt (the artifact system's in-flight
490
- // build correlation ID — not a durable post-completion identifier). The read-back request carries it so the parent can
491
- // prove the read-back inspected the live build of THIS attempt — not a
492
- // different build's output. Null/empty means the edit was accepted but
493
- // never correlated to a builder run. Pure function of inputs — no I/O,
494
- // no clock.
495
- var buildIdLine = (typeof buildAgentId === "string" && buildAgentId.length > 0)
496
- ? "Expected builder build agent_id: " + buildAgentId + " (the artifact system's in-flight correlation ID for this publish attempt — not a durable post-completion identifier).\n"
497
- : "No build agent_id was observed for this publish attempt (the edit was accepted but never correlated to a builder run) — say so explicitly in your report.\n";
498
- return (
499
- "Publish content read-back for task " + taskId + ", merge commit " + commit + ".\n" +
500
- "The unified diff below was supposed to be applied to this artifact's source tree and deployed. Do NOT modify anything.\n" +
501
- "Do NOT rely on the builder's applied-changes report — it is derived from this same diff, so it cannot confirm the content. Read the artifact's CURRENT source directly.\n" +
502
- "\n" +
503
- buildIdLine +
504
- "Report the live build's agent_id as seen in artifact_status (or state explicitly that no build/agent_id is visible). If an expected agent_id is given above and the live one differs, say so exactly — the read-back may be inspecting a different build's output.\n" +
505
- "\n" +
506
- "UNIFIED DIFF (expected change):\n" +
507
- "```diff\n" + diff + "\n```\n" +
508
- "\n" +
509
- "For each file in the diff:\n" +
510
- "1. Read the file's CURRENT content in the artifact source tree.\n" +
511
- "2. Quote the exact current text of the regions around the changed lines.\n" +
512
- "3. For every added (+) line in the diff, state whether that exact line is PRESENT in the current source.\n" +
513
- "4. For every removed (-) line in the diff, state whether that exact line is ABSENT from the current source.\n" +
514
- "5. Report build/deploy health and the console error count.\n" +
515
- "\n" +
516
- "Return the per-file present/absent findings with the quoted observed lines. Do not modify anything.\n" +
517
- "This read-back feeds the parent content-verification protocol (docs/publish-verification.md): the parent stamps provenance only when every added line is present and every removed line is absent."
518
- );
519
- }
520
-
521
501
  // Pre-publish base observation (diagnostic, 2026-09-12): instruction fragment
522
502
  // for the builder's edit request, asking it to report the sha256 of each
523
503
  // touched file's CURRENT content BEFORE applying the diff. Pure function —
@@ -611,6 +591,18 @@ function extractMarkerLines(workerText) {
611
591
  return markers.join("\n");
612
592
  }
613
593
 
594
+ // Already-merged idempotency (canary 2026-09-15, task 1d692d91): when the
595
+ // builder correctly makes no commit because the deliverable is already on
596
+ // main (a prior merge or hand-repair landed it), it declares
597
+ // `repo_diff: none (already-merged: <sha>)` naming the main commit that
598
+ // carries the work. The sha is hex-only (7-40 chars) so the workflow can
599
+ // interpolate it into the mechanical ancestor check without injection
600
+ // risk. Pure — pinned byte-identical across standard/bugfix/chore.
601
+ function extractAlreadyMerged(workerText) {
602
+ var m = /^repo_diff:\s*none\s*\(already-merged:\s*([0-9a-f]{7,40})\)/im.exec(workerText || "");
603
+ return m ? { sha: m[1].toLowerCase() } : { sha: null };
604
+ }
605
+
614
606
  // Worktree confinement: the Build agent must declare the exact worktree
615
607
  // path it built in on a `worktree:` marker line. The workflow compares it
616
608
  // against WORKTREE_HINT mechanically (exact string match) — never by
@@ -751,44 +743,6 @@ async function baselineStatus() {
751
743
  return { baseline_found: false, baseline_kind: "", baseline_refs: "", requested_count: 0, evidence_count: 0 };
752
744
  }
753
745
  }
754
- // Visual verdict status: the parent records the visual verdict as a note
755
- // event after the rendered post-change inspection results arrive. Finds the
756
- // latest step "QA" session's started_at, then note events NEWER than it
757
- // whose message starts exactly "visual_verdict: PASS" / "visual_verdict:
758
- // FAIL". Returns { found, verdict, detail }. A fresh key per call: the
759
- // verdict lands while this run is parked.
760
- let visualVerdictCallCount = 0;
761
- async function visualVerdictStatus() {
762
- visualVerdictCallCount++;
763
- try {
764
- var vv = await agent(
765
- "Read this task's QA session and note events.\n" +
766
- "Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
767
- "Find the LATEST session with task_id \"" + taskId + "\" and step \"QA\" in the returned sessions array and note its started_at timestamp (call it QA_START; use empty string if there is no QA session).\n" +
768
- "Then run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId }) + "\n" +
769
- "Consider only events with type \"note\" whose timestamp is newer than QA_START. Among them, find messages starting exactly with \"visual_verdict: PASS\" or \"visual_verdict: FAIL\" (exact prefix, case-sensitive); use the latest such message.\n" +
770
- "Return JSON { \"found\": <true if such a message exists>, \"verdict\": \"<\"PASS\" or \"FAIL\" from that message, or empty string>\", \"detail\": \"<the text after the prefix in that message, or empty string>\" } and nothing else.",
771
- {
772
- key: "visual-verdict-status-" + taskId + "-" + visualVerdictCallCount,
773
- label: "Reading visual verdict status",
774
- schema: {
775
- type: "object",
776
- properties: {
777
- found: { type: "boolean" },
778
- verdict: { type: "string" },
779
- detail: { type: "string" }
780
- },
781
- required: ["found", "verdict", "detail"]
782
- }
783
- }
784
- );
785
- return { found: !!(vv && vv.found), verdict: (vv && vv.verdict) || "", detail: (vv && vv.detail) || "" };
786
- } catch (e) {
787
- log("visualVerdictStatus: agent call failed (" + (e && e.message ? e.message : e) + ") — treating as not found");
788
- return { found: false, verdict: "", detail: "" };
789
- }
790
- }
791
-
792
746
  // STEPS inline — export const meta is parsed as metadata, not a runtime binding
793
747
  const STEPS = [
794
748
  { name: "Triage", identity: "sage" },
@@ -827,6 +781,12 @@ let mapGateBounceCount = 0;
827
781
  // rationalized a skip against explicit instruction text — text alone did not
828
782
  // hold, so the decision now lives in workflow code, not agent judgment.
829
783
  let releaseDecision = null; // { release: "yes"|"no", version_bump: "patch"|"minor"|"major"|null }
784
+ // Already-merged idempotency: the verified sha from the builder's
785
+ // `repo_diff: none (already-merged: <sha>)` declaration (null when the
786
+ // builder made commits or declared a runtime-state deliverable). The
787
+ // workflow verifies the sha is an ancestor of main at Build closeout;
788
+ // Review's no-diff branch reads this, never the builder's prose.
789
+ let alreadyMergedSha = null;
830
790
  // Deterministic publish target — computed by the workflow (registry base +
831
791
  // bumpVersion), never by the Publish agent.
832
792
  let publishTarget = null; // { base, scope, target }
@@ -1058,7 +1018,7 @@ while (i < STEPS.length) {
1058
1018
  const claimResult = await agent(
1059
1019
  "Claim this task for the " + step.name + " step.\n" +
1060
1020
  "Run in shell and return the stdout verbatim:\n" + crewCmd("update-task", firstClaimUpdateArgs) + "\n" +
1061
- "Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started" }) + "\n" +
1021
+ "Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started", ...(NEXT_PHASE_ROUTED ? { expected_next_phase: NEXT_PHASE_ROUTED } : {}) }) + "\n" +
1062
1022
  "If the claim response has claimed=true, then run in shell and return the stdout verbatim:\n" + crewCmd("clear-reservation", { task_id: taskId }) + "\n" +
1063
1023
  "Do not interpret the claim response. It already contains an explicit \"claimed\" field — copy it verbatim.\n" +
1064
1024
  "Return { claimed: <verbatim>, session_id: \"<...>\" }. If claimed is false there is no session_id; return { claimed: false, session_id: \"\" }.",
@@ -1227,12 +1187,10 @@ while (i < STEPS.length) {
1227
1187
  }
1228
1188
  }
1229
1189
 
1230
- // Visual verdict routing: experiential artifact tasks get their visual
1231
- // verdict from the parent AFTER the rendered post-change inspection
1232
- // results arrive (this script cannot receive the async handoff). The QA
1233
- // work agent covers mechanical checks only; the visual-verdict gate below
1234
- // parks for the parent protocol when no visual_verdict: note event exists
1235
- // yet.
1190
+ // Experiential routing: experiential artifact tasks run the Hazel
1191
+ // experiential QA prompt (built in the PUBLISH_TYPE === "artifact" branch
1192
+ // of the QA step) — Hazel drives the artifact herself and owns the visual
1193
+ // verdict through her OODA report and write-ooda-verdict ledger.
1236
1194
  var qaVisual = false;
1237
1195
  if (step.name === "QA") {
1238
1196
  qaVisual = (await resolveExperiential()) === "yes" && PUBLISH_TYPE === "artifact";
@@ -1243,7 +1201,7 @@ while (i < STEPS.length) {
1243
1201
  var instructions = "";
1244
1202
 
1245
1203
  if (step.name === "Triage") {
1246
- instructions = "Validate the task, check clarity, note dependencies, confirm the standard workflow assignment.\nWrite a brief triage assessment as notes for the next step.\nReport back in plain prose — what you found.\nEXPERIENTIAL FLAG: does this task change anything rendered and visible in the project's user-facing artifact (pages, components, styles, layout, copy, visual states)? If yes it is experiential and gets baseline captures (plus a visual verdict where the workflow has a QA phase). End your report with exactly one line on its own, lowercase, unrephrased: experiential: yes — or experiential: no. This line is machine-read.";
1204
+ instructions = "Validate the task against the project's repo at " + REPO_PATH + " — that exact checkout, not any other copy of the project on disk. If you run git commands, cd " + REPO_PATH + " first.\nCheck clarity, note dependencies, confirm the standard workflow assignment.\nWrite a brief triage assessment as notes for the next step.\nReport back in plain prose — what you found.\nEXPERIENTIAL FLAG: does this task change anything rendered and visible in the project's user-facing artifact (pages, components, styles, layout, copy, visual states)? If yes it is experiential and gets baseline captures (plus a visual verdict where the workflow has a QA phase). End your report with exactly one line on its own, lowercase, unrephrased: experiential: yes — or experiential: no. This line is machine-read.";
1247
1205
 
1248
1206
  } else if (step.name === "Map") {
1249
1207
  var mapGatePara = "";
@@ -1279,12 +1237,33 @@ while (i < STEPS.length) {
1279
1237
  "cd " + WORKTREE_HINT + "\n" +
1280
1238
  "git add -A\n" +
1281
1239
  "git commit -m \"" + safeTitle + "\"\n\n" +
1282
- "If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of main and declare `repo_diff: none` in your report, naming the runtime-state deliverable. Otherwise commit your changes normally.\n\n" +
1240
+ "If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of main and declare `repo_diff: none` in your report, naming the runtime-state deliverable. If you verified the deliverable is already on main (a prior merge or hand-repair landed it — do NOT re-implement working code), make no commit and declare `repo_diff: none (already-merged: <sha>)` naming the main commit that carries the work; the workflow verifies the sha is an ancestor of main, and a false declaration fails the phase. Otherwise commit your changes normally.\n\n" +
1283
1241
  (rejectionNotes ? "This is REWORK after rejection. Address these specific issues:\n" + rejectionNotes + "\n\n" : "") +
1284
1242
  "Report back in plain prose: what you built and the outcome." +
1285
1243
  (PUBLISH_TYPE === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
1286
1244
 
1287
1245
  } else if (step.name === "Review") {
1246
+ // Already-merged hydration: when this run did not execute Build itself
1247
+ // (dispatcher resume at Review after a platform death between phases),
1248
+ // recover the workflow-attested verification from the latest completed
1249
+ // Build session notes. The `already_merged_verified:` line was written
1250
+ // by the workflow after a mechanical ancestor check — it is trusted;
1251
+ // the builder's bare declaration never is. Absent the line, the
1252
+ // mechanical fact below reads "none declared" and Cass fails closed.
1253
+ if (!alreadyMergedSha) {
1254
+ var hydNotes = await agent(
1255
+ "Read the latest completed Build session notes for task " + taskId + ".\n" +
1256
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
1257
+ "In the returned sessions array, find the most recent session (by started_at) with task_id \"" + taskId + "\", step \"Build\", and status \"completed\". Return ONLY its notes field, verbatim, with no commentary.",
1258
+ { key: "hydrate-already-merged" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Hydrating already-merged verification" }
1259
+ );
1260
+ var hydStr = (typeof hydNotes === "string") ? hydNotes : JSON.stringify(hydNotes);
1261
+ var hvm = /already_merged_verified:\s*([0-9a-f]{7,40})/i.exec(hydStr);
1262
+ if (hvm) {
1263
+ alreadyMergedSha = hvm[1].toLowerCase();
1264
+ log("Hydrated already-merged verification from Build session notes: " + alreadyMergedSha);
1265
+ }
1266
+ }
1288
1267
  instructions = "Review independently and cold. You have NOT seen any reasoning from the builder.\nDo NOT access the task dashboard, event log, or any comments. Your review is based solely on the spec and the code.\n\n" +
1289
1268
  (mapperSpec ? "MAPPER'S SPEC (the builder was asked to implement exactly this):\n" + mapperSpec + "\n\n" : "Read the spec (from the task description or spec files under " + crewHome + "/).\n\n") +
1290
1269
  "Examine the code changes by running:\n" +
@@ -1294,7 +1273,7 @@ while (i < STEPS.length) {
1294
1273
  WORKTREE_HINT + "/\n\n" +
1295
1274
  "Check quality, correctness, and spec compliance.\n" +
1296
1275
  "Check that public-affecting changes have matching public doc updates (API.md or the published API contract). If the docs are missing or inaccurate, report what is stale, then end your report with exactly this line: VERDICT: FAIL.\n" +
1297
- "If the branch has no commits ahead of main (inspect shows an empty commit log), approve ONLY if the Build summary declares `repo_diff: none` with a plausible runtime-state deliverable (e.g. a cron created via the cron tool). Otherwise report 'no commits ahead of main and no repo_diff: none declaration — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n" +
1276
+ "If the branch has no commits ahead of main (inspect shows an empty commit log), approve ONLY if the Build summary declares `repo_diff: none` with (a) a plausible runtime-state deliverable (e.g. a cron created via the cron tool), or (b) an already-merged declaration `repo_diff: none (already-merged: <sha>)` AND the mechanical fact below confirms the sha verified. MECHANICAL FACT (computed by the workflow, never by the builder): already_merged sha = " + (alreadyMergedSha ? alreadyMergedSha + " (verified ancestor of main: YES)" : "none declared") + ". Otherwise report 'no commits ahead of main and no valid repo_diff: none declaration — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n" +
1298
1277
  (PUBLISH_TYPE === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry, and versions are assigned at publish time — never in branches. Two checks:\n" +
1299
1278
  "(a) The task branch must NOT have changed package.json's `version` field. Check: cd " + REPO_PATH + " && git diff main..." + TASK_BRANCH + " -- package.json. If the branch touched `version` in any way, report 'versions are assigned at publish time, never in branches — remove the version change' in your notes, then end your report with exactly this line: VERDICT: FAIL.\n" +
1300
1279
  "(b) The accepted Build report declares: " + releaseDecisionText() + ". " +
@@ -1308,7 +1287,7 @@ while (i < STEPS.length) {
1308
1287
  "Run: "+ LIFECYCLE_ENV + "WORKFLOW_RUN_ID=" + lockHolder + " integrate " + taskId + " \"merge: " + safeTitle + "\"\n\n" +
1309
1288
  "Read the output:\n" +
1310
1289
  "- If it contains MERGED, integration succeeded. Report the merged commit hash.\n" +
1311
- "- If it contains MERGED_EMPTY, the branch had no commits ahead of main (a runtime-state deliverable, declared by Build as repo_diff: none). Integration succeeded vacuously: the merge lock was NOT taken and there is no new commit. Report 'merged empty: no repo changes — deliverable was runtime state', then end your report with exactly this line: VERDICT: PASS. SKIP STEP 2 (push): there is no new commit to push.\n" +
1290
+ "- If it contains MERGED_EMPTY, the branch had no commits ahead of main (declared by Build as repo_diff: none — either a runtime-state deliverable or an already-merged sha the workflow verified). Integration succeeded vacuously: the merge lock was NOT taken and there is no new commit. Report 'merged empty: no repo changes — deliverable was runtime state or already on main', then end your report with exactly this line: VERDICT: PASS. SKIP STEP 2 (push): there is no new commit to push.\n" +
1312
1291
  "- If it contains LOCK_HELD, another task holds the merge lock (mid Integrate/Publish) and the 10-minute bounded backoff is exhausted. Report 'merge lock held after bounded backoff', then end your report with exactly this line: VERDICT: FAIL.\n" +
1313
1292
  "- If it contains CONFLICT, the plain merge failed — the merge was aborted, main is clean, and your task still holds the merge lock. Do NOT fail yet. Resolve it:\n" +
1314
1293
  "RESOLUTION:\n" +
@@ -1371,10 +1350,11 @@ while (i < STEPS.length) {
1371
1350
  // (fail-closed). There is deliberately NO workflow-side provenance
1372
1351
  // stamp: the builder's applied-report is circular (canary run 8,
1373
1352
  // 2026-09-11), so the stamp moved to the parent — after the build
1374
- // lands, the workflow triggers an independent artifact_inspect
1375
- // read-back, records the session completed, and parks with
1376
- // "publish: verification-requested". The parent stamps provenance only
1377
- // after the read-back confirms the content (docs/publish-verification.md);
1353
+ // lands, the workflow records the session completed and parks with
1354
+ // "publish: verification-requested". The parent owns verification
1355
+ // (docs/publish-verification.md); the independent read-back step is
1356
+ // currently unavailable (no agent-callable read-back tool exists —
1357
+ // artifact_inspect was removed by the platform 2026-09-14).
1378
1358
  // QA's provenance check enforces the stamp mechanically.
1379
1359
  var artifactPublish = null;
1380
1360
  var publishLockRefreshed = false;
@@ -1423,14 +1403,45 @@ while (i < STEPS.length) {
1423
1403
  // changes — no prose claim to trust. If the artifact tool namespace
1424
1404
  // is missing from this child it reports honestly and the workflow
1425
1405
  // retries once with a fresh key (bounded); anything else parks.
1406
+ // Publish diff base (2026-09-14, task 0c53af4e): the carried diff is
1407
+ // BASE..HEAD where BASE is the previously-stamped provenance
1408
+ // source_commit — NOT HEAD^1. A push-time reconcile merge puts the
1409
+ // task's own changes behind an intermediate merge, so HEAD^1..HEAD
1410
+ // silently drops the task's fix while the artifact builds without
1411
+ // it. The stamped base is the artifact's actual content; BASE..HEAD
1412
+ // is the complete unpublished delta. Empty tree only for a genuine
1413
+ // first publish (no provenance stamped yet).
1414
+ var EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
1415
+ var provResult = await agent(
1416
+ crewCmd("get-provenance", {}) + "\n" +
1417
+ "Return JSON { \"provenance\": <the CLI's provenance object, or null when nothing is stamped> } and nothing else. Do not interpret it.",
1418
+ { key: attemptKey("publish-provenance-base-" + taskId, totalReworkCount), label: "Reading stamped publish base",
1419
+ schema: { type: "object", properties: { provenance: { type: ["object", "null"] } }, required: ["provenance"] } }
1420
+ );
1421
+ var publishBase = (provResult.provenance && provResult.provenance.source_commit) || "";
1422
+ publishBase = String(publishBase).trim();
1423
+ if (!publishBase) {
1424
+ publishBase = EMPTY_TREE_SHA;
1425
+ log("Publish base for task " + taskId + ": no provenance stamped yet — using empty tree (first publish)");
1426
+ } else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
1427
+ return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
1428
+ }
1426
1429
  var diffResult = await agent(
1427
- "Run: cd " + REPO_PATH + " && git rev-parse HEAD && echo '---PARENT---' && git rev-parse HEAD^1 && echo '---DIFF---' && git diff HEAD^1 HEAD && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r HEAD\n" +
1428
- "Return JSON { \"commit\": \"<HEAD trimmed>\", \"parent\": \"<HEAD^1 trimmed>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
1429
- { key: attemptKey("publish-artifact-diff-" + taskId, totalReworkCount), label: "Computing merged diff for publish",
1430
- schema: { type: "object", properties: { commit: { type: "string" }, parent: { type: "string" }, diff: { type: "string" }, files: { type: "string" } }, required: ["commit", "diff"] } }
1430
+ "Run: cd " + REPO_PATH + " && BASE='" + publishBase + "' && HEAD=$(git rev-parse HEAD) && " +
1431
+ "if [ \"$BASE\" = '" + EMPTY_TREE_SHA + "' ]; then ANCESTOR=yes; else git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no; fi && " +
1432
+ "echo '---COMMIT---' && echo \"$HEAD\" && echo '---BASE---' && echo \"$BASE\" && echo '---ANCESTOR---' && echo \"$ANCESTOR\" && " +
1433
+ "if [ \"$ANCESTOR\" = yes ]; then echo '---DIFF---' && git diff \"$BASE\" \"$HEAD\" && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r \"$BASE\" \"$HEAD\"; fi\n" +
1434
+ "Return JSON { \"commit\": \"<HEAD trimmed>\", \"base\": \"<BASE trimmed>\", \"ancestor\": \"<yes|no>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
1435
+ { key: attemptKey("publish-artifact-diff-" + taskId, totalReworkCount), label: "Computing publish diff from stamped base",
1436
+ schema: { type: "object", properties: { commit: { type: "string" }, base: { type: "string" }, ancestor: { type: "string" }, diff: { type: "string" }, files: { type: "string" } }, required: ["commit", "base", "ancestor", "diff"] } }
1431
1437
  );
1438
+ if ((diffResult.ancestor || "").trim() !== "yes") {
1439
+ return await parkTask("Publish base " + publishBase.slice(0, 12) + " is not an ancestor of HEAD " + (diffResult.commit || "").trim().slice(0, 12) + " — the stamped provenance does not lead to the integrated commit. Human attention needed.");
1440
+ }
1441
+ if ((diffResult.base || "").trim() !== publishBase) {
1442
+ return await parkTask("Publish diff base mismatch: agent reported '" + (diffResult.base || "").trim().slice(0, 12) + "' but the stamped base is '" + publishBase.slice(0, 12) + "'. Human attention needed.");
1443
+ }
1432
1444
  var mergeCommitForPublish = (diffResult.commit || "").trim();
1433
- var mergeParentForPublish = (diffResult.parent || "").trim();
1434
1445
  var mergeDiff = diffResult.diff || "";
1435
1446
  if (!mergeDiff.trim()) {
1436
1447
  return await parkTask("Publish diff is empty for commit " + (mergeCommitForPublish || "unknown") + " — a merge lock was held but there is no change to carry. Human attention needed.");
@@ -1462,12 +1473,16 @@ while (i < STEPS.length) {
1462
1473
  // but never parks. The observation tells us what the publish actually
1463
1474
  // reads, so the subsequent fix can require the right base.
1464
1475
  var expectedBaseHashes = {};
1465
- try {
1476
+ if (publishBase === EMPTY_TREE_SHA) {
1477
+ // First publish: every file in the diff is new to the artifact.
1478
+ expectedChanges.forEach(function (f) { expectedBaseHashes[f.path] = "NEW-FILE"; });
1479
+ log("Publish expected base hashes for task " + taskId + ": empty tree (first publish) — all " + expectedChanges.length + " file(s) new");
1480
+ } else try {
1466
1481
  // Shell-quote helper (no regex-with-quote: the test parser does not
1467
1482
  // understand regex literals containing quotes).
1468
1483
  var sq = function(s) { return "'" + String(s).split("'").join("'\\''") + "'"; };
1469
1484
  var baseHashResult = await agent(
1470
- "Run: cd " + REPO_PATH + " && parent=" + sq(mergeParentForPublish) + " && for f in " + expectedChanges.map(function(f) { return sq(f.path); }).join(" ") + "; do printf '%s:' \"$f\"; git show \"$parent:$f\" 2>/dev/null | sha256sum | cut -d' ' -f1; done\n" +
1485
+ "Run: cd " + REPO_PATH + " && parent=" + sq(publishBase) + " && for f in " + expectedChanges.map(function(f) { return sq(f.path); }).join(" ") + "; do printf '%s:' \"$f\"; git show \"$parent:$f\" 2>/dev/null | sha256sum | cut -d' ' -f1; done\n" +
1471
1486
  "Return JSON { \"hashes\": \"<newline-separated <path>:<sha256> lines, empty hash means the file is new in this diff>\" } and nothing else.",
1472
1487
  { key: attemptKey("publish-base-hashes-" + taskId, totalReworkCount), label: "Computing expected base content hashes",
1473
1488
  schema: { type: "object", properties: { hashes: { type: "string" } }, required: ["hashes"] } }
@@ -1476,7 +1491,7 @@ while (i < STEPS.length) {
1476
1491
  var m = /^([^:]+):([0-9a-f]*)$/.exec(line.trim());
1477
1492
  if (m) expectedBaseHashes[m[1]] = m[2] || "NEW-FILE";
1478
1493
  });
1479
- log("Publish expected base hashes for task " + taskId + " (merge parent " + (mergeParentForPublish || "unknown").slice(0, 12) + "): " + JSON.stringify(expectedBaseHashes));
1494
+ log("Publish expected base hashes for task " + taskId + " (stamped base " + publishBase.slice(0, 12) + "): " + JSON.stringify(expectedBaseHashes));
1480
1495
  } catch (e) {
1481
1496
  log("Publish expected base hash computation failed for task " + taskId + " (non-fatal, observation degraded): " + (e && e.message ? e.message : e));
1482
1497
  }
@@ -1534,6 +1549,7 @@ while (i < STEPS.length) {
1534
1549
  required: ["edit_started", "build_agent_id", "applied"] };
1535
1550
  var rebuildTrigger = null;
1536
1551
  var rebuildReportMissing = false; // true if the edit went through but the agent returned no applied report (structured-output failure) — the smoke-check is skipped; the parent's independent read-back is the verification
1552
+ var rebuildEvidenceNote = null; // human-readable evidence line for the ledger when the edit is confirmed via fallback evidence (in-flight poll or durable audit dir) rather than the trigger's own report
1537
1553
  // The trigger key of the attempt that last ran, for the publish ledger.
1538
1554
  // Minted once here (not re-minted per use site) so the ledger always
1539
1555
  // records the exact key that was issued — and so a re-minted duplicate
@@ -1552,6 +1568,32 @@ while (i < STEPS.length) {
1552
1568
  // (2026-09-12, task 23ca8f3f): computed once the trigger outcome is
1553
1569
  // known, logged loudly, never a park.
1554
1570
  var publishAppliedObservation = null; // "match" | "mismatch: <reason>" | "missing-report" — observation only, never a park
1571
+ // Durable-evidence snapshot (2026-09-14): the structured-output
1572
+ // fallback below only observes IN-FLIGHT builds. A build that
1573
+ // finished before the poll leaves no in-flight trace — but the
1574
+ // platform's audit harness leaves a durable one:
1575
+ // ~/workspace/ts-spaces/<slug>/audits/<timestamp>-<id>/ per
1576
+ // completed build. Snapshot the listing BEFORE the trigger so the
1577
+ // fallback can diff before/after: a directory appearing during the
1578
+ // trigger window is positive evidence the edit went through and
1579
+ // the build completed. Best-effort and non-gating: if the snapshot
1580
+ // fails, the durable check is skipped and the fallback behaves as
1581
+ // before. No wall-clock in-script (deterministic replay) — the
1582
+ // comparison is a pure before/after set diff.
1583
+ var auditDirsBeforeTrigger = [];
1584
+ try {
1585
+ var auditBefore = await agent(
1586
+ "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
1587
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1588
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1589
+ { key: attemptKey("publish-audit-before-" + taskId, totalReworkCount), label: "Snapshotting audit dirs before rebuild trigger",
1590
+ schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1591
+ );
1592
+ auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1593
+ log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
1594
+ } catch (auditBeforeErr) {
1595
+ log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1596
+ }
1555
1597
  try {
1556
1598
  rebuildTrigger = await agent(rebuildPrompt,
1557
1599
  { key: rebuildAttemptKey, label: "Triggering artifact rebuild", schema: rebuildSchema });
@@ -1610,7 +1652,46 @@ while (i < STEPS.length) {
1610
1652
  rebuildTrigger = { edit_started: true, error: "", applied: null };
1611
1653
  rebuildReportMissing = true;
1612
1654
  rebuildAgentId = acceptedAgentId;
1655
+ rebuildEvidenceNote = "edit confirmed via build-state poll after structured-output failure (build " + acceptedAgentId + "); builder applied-report missing";
1613
1656
  } else {
1657
+ // Durable completion check (2026-09-14): the in-flight poll
1658
+ // above only sees RUNNING builds. Attempt 7 (2026-09-14) proved
1659
+ // the gap: the trigger child applied the edit, the build ran
1660
+ // and completed — the platform's audit harness captured it
1661
+ // mid-window — then the child failed to return JSON. The
1662
+ // fallback poll saw no in-flight build, so a successful publish
1663
+ // parked as "unknown". Diff the audit-dir listing against the
1664
+ // pre-trigger snapshot: a timestamped directory that appeared
1665
+ // during the trigger window is positive evidence the edit went
1666
+ // through and the build completed. This never re-issues the
1667
+ // edit and never stamps provenance — it only routes to the
1668
+ // parent's independent content read-back, which remains the
1669
+ // real verification.
1670
+ var newAuditDirs = [];
1671
+ try {
1672
+ var auditAfter = await agent(
1673
+ "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
1674
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1675
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1676
+ { key: attemptKey("publish-audit-after-" + taskId, totalReworkCount), label: "Re-listing audit dirs after trigger failure",
1677
+ schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1678
+ );
1679
+ var auditDirsAfterTrigger = String((auditAfter && auditAfter.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1680
+ // Only timestamped build dirs count — the "latest" symlink
1681
+ // and anything else are not builds.
1682
+ newAuditDirs = auditDirsAfterTrigger.filter(function (d) {
1683
+ return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
1684
+ });
1685
+ } catch (auditAfterErr) {
1686
+ log("Publish audit-dir re-list after trigger failure failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
1687
+ }
1688
+ if (newAuditDirs.length > 0) {
1689
+ log("Publish rebuild trigger: new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and the build completed despite the structured-output failure. Skipping applied-report smoke-check; parent read-back is the verification.");
1690
+ rebuildTrigger = { edit_started: true, error: "", applied: null };
1691
+ rebuildReportMissing = true;
1692
+ rebuildAgentId = null;
1693
+ rebuildEvidenceNote = "edit confirmed via durable audit evidence after structured-output failure (new audit dir " + newAuditDirs[0] + "); builder applied-report missing";
1694
+ } else {
1614
1695
  // No build observed — but that proves nothing (a fast-completing
1615
1696
  // build can finish between polls, or the check itself failed). The
1616
1697
  // outcome is UNKNOWN. No retry: re-issuing the edit here duplicated
@@ -1626,6 +1707,7 @@ while (i < STEPS.length) {
1626
1707
  detail: "structured-output failure on rebuild trigger; build-state poll saw no build (or the check itself failed); edit may have been accepted as pending_init"
1627
1708
  }, totalReworkCount);
1628
1709
  return await parkTask("Publish outcome unknown: the rebuild trigger's child did not return JSON, and the follow-up build-state poll could not observe a build for slug " + PUBLISH_SLUG + ". The edit may have been accepted as pending_init, so no retry was issued — a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion before re-driving Publish. Fail-closed.");
1710
+ }
1629
1711
  }
1630
1712
  }
1631
1713
  if (!rebuildReportMissing && !rebuildTrigger.edit_started && rebuildTrigger.error === "artifact_tools missing after load") {
@@ -1695,7 +1777,7 @@ while (i < STEPS.length) {
1695
1777
  applied_report: publishAppliedObservation,
1696
1778
  outcome: "submitted",
1697
1779
  detail: rebuildReportMissing
1698
- ? "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing"
1780
+ ? (rebuildEvidenceNote || "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing")
1699
1781
  : "edit accepted; builder applied-report received"
1700
1782
  }, totalReworkCount);
1701
1783
  } else if (rebuildTrigger) {
@@ -1752,6 +1834,15 @@ while (i < STEPS.length) {
1752
1834
  // lock was lost: stop the run and park the task — never continue to
1753
1835
  // a provenance stamp or version assignment without holding the lock.
1754
1836
  var buildPoll = null;
1837
+ // STEP 1b poll-signal accumulators (2026-09-15, task aadeccc3):
1838
+ // the durable audit-dir fallback below needs the poll's own
1839
+ // observations, not just its final verdict — whether our build was
1840
+ // ever seen, whether a stranger's build was ever in flight, and
1841
+ // what the last check observed. OR-ed across all three chunks so
1842
+ // a signal seen in any chunk survives the chunk boundary.
1843
+ var pollSawOurBuild = false;
1844
+ var pollSawStranger = false;
1845
+ var lastObservedAgentId = null;
1755
1846
  for (var chunk = 1; chunk <= 3; chunk++) {
1756
1847
  if (chunk > 1) {
1757
1848
  var refreshPoll = await agent(
@@ -1774,36 +1865,152 @@ while (i < STEPS.length) {
1774
1865
  : attemptKey("publish-artifact-poll-" + taskId + "-c" + chunk, totalReworkCount);
1775
1866
  buildPoll = await agent(
1776
1867
  "First call tool_search.load_tool_namespace with paths [\"artifact\"]. Then poll artifact_status for slug \"" + PUBLISH_SLUG + "\" \u2014 for OUR build only, the one whose agent_id is \"" + rebuildAgentId + "\" (the receipt captured when the edit was accepted; the agent_id is the artifact system's in-flight build correlation ID, stable across polls while the build runs). Check every 30 seconds, up to 7 checks (3.5 minutes max). On each check, read the raw build object:\n" +
1777
- "- If no build is running (build is null): OUR build finished. Stop and report done.\n" +
1868
+ "On every check, record whether you have positively OBSERVED our build: a running build whose agent_id equals \"" + rebuildAgentId + "\", or a completed-build record whose agent_id equals \"" + rebuildAgentId + "\" (if the tool surfaces one \u2014 match it mechanically, never assume).\n" +
1869
+ "- If no build is running (build is null) and you have NOT observed our build: our build's completion is UNPROVEN. Absence of a running build is not evidence our build ran. Do NOT report done.\n" +
1870
+ "- If no build is running (build is null) and you previously observed our build running: our build finished. Stop and report done.\n" +
1778
1871
  "- If the running build's agent_id equals \"" + rebuildAgentId + "\": still ours \u2014 keep waiting.\n" +
1779
- "- If the running build's agent_id is present but DIFFERENT: our build is gone (it finished before this one started). Do NOT wait on the stranger's build and do NOT attribute its completion to our attempt \u2014 stop and report done.\n" +
1780
- "Return JSON { \"build_done\": <true if our build is no longer running within budget, false on timeout>, \"status\": \"<final status or timeout note>\", \"observed_agent_id\": \"<the agent_id seen on the last check, or null when no build was running>\" } and nothing else.",
1872
+ "- If the running build's agent_id is present but DIFFERENT: that is a stranger's build. Do NOT attribute its completion to our attempt and do NOT wait on it \u2014 keep checking within budget; if the budget expires without observing our build, report done=false. Record it in saw_stranger regardless of what else you observe.\n" +
1873
+ "Return JSON { \"build_done\": <true ONLY when you positively observed our build and it is no longer running, false otherwise>, \"saw_our_build\": <true if you observed our build at any check, false if never>, \"saw_stranger\": true if at ANY check a running build had an agent_id different from ours (\"" + rebuildAgentId + "\"), false otherwise, \"status\": \"<final status or timeout note>\", \"observed_agent_id\": \"<the agent_id seen on the last check, or null when no build was running>\" } and nothing else.",
1781
1874
  { key: pollKey, label: "Waiting for artifact build to complete (chunk " + chunk + " of 3)",
1782
- schema: { type: "object", properties: { build_done: { type: "boolean" }, status: { type: "string" }, observed_agent_id: { type: ["string", "null"] } }, required: ["build_done"] },
1875
+ schema: { type: "object", properties: { build_done: { type: "boolean" }, saw_our_build: { type: "boolean" }, saw_stranger: { type: "boolean" }, status: { type: "string" }, observed_agent_id: { type: ["string", "null"] } }, required: ["build_done"] },
1783
1876
  timeoutMs: 270000 }
1784
1877
  );
1878
+ pollSawOurBuild = pollSawOurBuild || (buildPoll && buildPoll.saw_our_build === true);
1879
+ pollSawStranger = pollSawStranger || (buildPoll && buildPoll.saw_stranger === true);
1880
+ lastObservedAgentId = (buildPoll && buildPoll.observed_agent_id) || null;
1785
1881
  if (buildPoll && buildPoll.build_done) { break; }
1786
1882
  }
1787
1883
  if (!buildPoll || !buildPoll.build_done) {
1788
1884
  buildPoll = { build_done: false, status: (buildPoll && buildPoll.status) || "build still running after the 10.5-minute bounded poll" };
1789
1885
  }
1790
- if (buildPoll.build_done) {
1886
+ if (buildPoll.build_done && pollSawOurBuild) {
1791
1887
  // STEP 1c (mechanical): NO provenance stamp here. Canary run 8
1792
1888
  // (2026-09-11) proved the stamp cannot certify content: the
1793
1889
  // builder's applied-report is derived from the carried diff, so
1794
1890
  // verifyAppliedChanges above is circular — a fabricated report
1795
1891
  // passes by construction, and every phase went green on a hollow
1796
- // build. The stamp moves to the parent (docs/publish-verification.md):
1797
- // after an independent artifact_inspect read-back confirms the
1798
- // artifact's actual content matches the merged diff, the parent
1799
- // stamps provenance and re-queues; QA's provenance check then
1800
- // enforces the stamp mechanically, so an unverified publish fails
1801
- // loudly in QA instead of passing silently here.
1892
+ // build. The stamp moves to the parent (docs/publish-verification.md);
1893
+ // the independent read-back step is currently unavailable (no
1894
+ // agent-callable read-back tool exists — artifact_inspect was
1895
+ // removed by the platform 2026-09-14), so the parent cannot
1896
+ // confirm content and the task parks for verification.
1897
+ // QA's provenance check enforces the stamp mechanically.
1898
+ // An unverified publish fails loudly in QA instead of passing
1899
+ // silently here.
1802
1900
  publishBuildLanded = true;
1803
1901
  artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
1804
1902
  log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
1805
1903
  } else {
1806
- publishFailure = "Artifact build did not complete within budget: " + (buildPoll.status || "timeout") + ". The publish may or may not have landed — provenance was not stamped.";
1904
+ // STEP 1b durable audit-dir fallback (2026-09-15, task aadeccc3):
1905
+ // the poll above only observes IN-FLIGHT builds. A build that
1906
+ // finished between the receipt capture and the poll's first check
1907
+ // leaves no in-flight trace — but the platform's audit harness
1908
+ // leaves a durable one (~/workspace/ts-spaces/<slug>/audits/
1909
+ // <timestamp>-<id>/ per completed build). Diff the audit-dir
1910
+ // listing against the pre-trigger snapshot: a timestamped dir
1911
+ // that appeared during the attempt window is evidence a build
1912
+ // completed. Attribution is by window, not by build identity:
1913
+ // the poll's saw_stranger signal only catches stranger builds in
1914
+ // flight AT a check — a stranger that finished entirely inside
1915
+ // the window is indistinguishable, so any observed stranger
1916
+ // blocks attribution and the outcome stays unknown. This never
1917
+ // re-issues the edit and never stamps provenance — ok=true only
1918
+ // routes to the parent's independent content read-back, which
1919
+ // remains the real verification.
1920
+ //
1921
+ // The poll end-state is read from the poll's own observations,
1922
+ // not from build_done alone: a build in flight at the last check
1923
+ // means the budget was shorter than the latency (or the build is
1924
+ // stuck) — NOT that no build ever started; nothing observed at
1925
+ // any check is the never-started signal.
1926
+ var pollEndState = lastObservedAgentId ? "build-still-running-at-poll-end"
1927
+ : (pollSawOurBuild ? "our-build-observed-then-unconfirmed" : "no-build-observed-in-window");
1928
+ var strangerObserved = pollSawStranger;
1929
+ var newAuditDirsAfterPoll = [];
1930
+ try {
1931
+ var auditAfterPoll = await agent(
1932
+ "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
1933
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1934
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1935
+ { key: attemptKey("publish-audit-after-poll-" + taskId, totalReworkCount), label: "Re-listing audit dirs after build poll",
1936
+ schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1937
+ );
1938
+ var auditDirsAfterPollList = String((auditAfterPoll && auditAfterPoll.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1939
+ newAuditDirsAfterPoll = auditDirsAfterPollList.filter(function (d) {
1940
+ return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
1941
+ });
1942
+ log("Publish audit-dir re-list after build poll for task " + taskId + ": " + newAuditDirsAfterPoll.length + " new timestamped dir(s)");
1943
+ } catch (auditAfterPollErr) {
1944
+ log("Publish audit-dir re-list after build poll failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterPollErr && auditAfterPollErr.message ? auditAfterPollErr.message : auditAfterPollErr));
1945
+ }
1946
+ // auditReportOk: pure tri-state read of a report.json body —
1947
+ // true (build ok), false (build failed), null (missing or
1948
+ // unreadable — not evidence either way). The child returns the
1949
+ // raw body verbatim; interpretation lives here, never in prose.
1950
+ var auditReportOk = function (raw) {
1951
+ if (typeof raw !== "string") return null;
1952
+ var trimmed = raw.trim();
1953
+ if (trimmed === "" || trimmed === "MISSING") return null;
1954
+ var parsed;
1955
+ try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
1956
+ if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
1957
+ return null;
1958
+ };
1959
+ var auditOkAfterPoll = null;
1960
+ var newestAuditDirAfterPoll = null;
1961
+ if (newAuditDirsAfterPoll.length > 0 && !strangerObserved) {
1962
+ newAuditDirsAfterPoll.sort();
1963
+ newestAuditDirAfterPoll = newAuditDirsAfterPoll[newAuditDirsAfterPoll.length - 1];
1964
+ try {
1965
+ var auditOkRead = await agent(
1966
+ "Read the build report for artifact slug \"" + PUBLISH_SLUG + "\", audit dir \"" + newestAuditDirAfterPoll + "\" (verbatim read, never interpreted, never a gate).\n" +
1967
+ "Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestAuditDirAfterPoll + "/report.json 2>/dev/null || echo MISSING\n" +
1968
+ "Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
1969
+ { key: attemptKey("publish-audit-ok-after-poll-" + taskId, totalReworkCount), label: "Reading build report after build poll",
1970
+ schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
1971
+ );
1972
+ auditOkAfterPoll = auditReportOk(auditOkRead && auditOkRead.raw);
1973
+ } catch (auditOkReadErr) {
1974
+ log("Publish build-report read after build poll failed for task " + taskId + " (non-fatal, treated as unknown): " + (auditOkReadErr && auditOkReadErr.message ? auditOkReadErr.message : auditOkReadErr));
1975
+ auditOkAfterPoll = null;
1976
+ }
1977
+ }
1978
+ if (auditOkAfterPoll === true) {
1979
+ publishBuildLanded = true;
1980
+ artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
1981
+ log("Publish build landed for task " + taskId + " via durable audit evidence — provenance stamp deferred to parent content verification");
1982
+ await recordPublishLedger({
1983
+ commit: mergeCommitForPublish,
1984
+ attempt: rebuildAttemptKey,
1985
+ agent_id: rebuildAgentId,
1986
+ applied_report: publishAppliedObservation,
1987
+ outcome: "submitted",
1988
+ detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
1989
+ }, totalReworkCount);
1990
+ } else if (auditOkAfterPoll === false) {
1991
+ publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestAuditDirAfterPoll + ", report ok=false). Explicit negative evidence: a build ran and failed (attribution by window, not by build identity — no stranger build was observed in flight during the poll). The publish did not land — provenance was not stamped. Fail-closed.";
1992
+ await recordPublishLedger({
1993
+ commit: mergeCommitForPublish,
1994
+ attempt: rebuildAttemptKey,
1995
+ agent_id: rebuildAgentId,
1996
+ applied_report: publishAppliedObservation,
1997
+ outcome: "failed",
1998
+ detail: "a build ran and failed (attribution by window, not by build identity): audit dir " + newestAuditDirAfterPoll + " report ok=false; no stranger build observed in flight during the poll"
1999
+ }, totalReworkCount);
2000
+ } else {
2001
+ var unattributableReason = strangerObserved ? "stranger-build-observed-during-poll"
2002
+ : (pollEndState === "build-still-running-at-poll-end" ? "build-still-running-at-poll-end"
2003
+ : (newAuditDirsAfterPoll.length === 0 ? "no-new-audit-dir-in-window" : "audit-report-unreadable-or-missing"));
2004
+ publishFailure = "Artifact build completion unproven (fail-closed, no provenance stamped): unattributable_reason=" + unattributableReason + "; poll_end_state=" + pollEndState + "; " + "saw_our_build=" + pollSawOurBuild + "; new_audit_dirs=" + newAuditDirsAfterPoll.length + ". Attribution is by window, not by build identity. The publish may or may not have landed. Fail-closed.";
2005
+ await recordPublishLedger({
2006
+ commit: mergeCommitForPublish,
2007
+ attempt: rebuildAttemptKey,
2008
+ agent_id: rebuildAgentId,
2009
+ applied_report: publishAppliedObservation,
2010
+ outcome: "unknown",
2011
+ detail: "durable audit-dir fallback could not attribute a completed build to this attempt (unattributable_reason=" + unattributableReason + ", poll_end_state=" + pollEndState + ")"
2012
+ }, totalReworkCount);
2013
+ }
1807
2014
  }
1808
2015
  } else {
1809
2016
  publishFailure = "Artifact rebuild trigger failed: " + (rebuildTrigger.error || "artifact_edit not accepted") + ". The publish did not land.";
@@ -1896,12 +2103,19 @@ while (i < STEPS.length) {
1896
2103
  var safeDesc = taskDescription.replace(/"/g, "'").replace(/\\/g, "\\\\").slice(0, 500);
1897
2104
  instructions = "You are code-blind QA. You NEVER read source files.\n" +
1898
2105
  "Public docs (API.md, README, published action schemas) are NOT source code — read them freely, exactly as a user would.\n\n" +
1899
- "STEP 1: Trigger a visual inspection of the published artifact.\n" +
1900
- "Call artifact_inspect with:\n" +
1901
- " slug: \"" + PUBLISH_SLUG + "\"\n" +
1902
- " repair_authorized: false\n" +
1903
- " verbatim_request: \"Verify task: " + safeTitle + ". " + safeDesc + "\"\n\n" +
1904
- "This call is asynchronous — it fires the inspection but results arrive outside this workflow. That is expected and correct.\n\n" +
2106
+ "STEP 1: Experiential visual inspection — drive the artifact as a user would, one browser step at a time.\n" +
2107
+ "You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
2108
+ "Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
2109
+ "a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and continue with the mechanical checks.\n" +
2110
+ "b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
2111
+ "c. Bounded see-act loop, at most 8 steps: drive the artifact as a user would, one browser step at a time. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
2112
+ "c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
2113
+ "c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
2114
+ "c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
2115
+ "d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the task's change needs such a sequence, report NOT POSSIBLE for that part and judge what you can.\n" +
2116
+ "e. Judge as a user against the task description: does the change render correctly? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
2117
+ "f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
2118
+ "Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
1905
2119
  "STEP 2: Verify data integrity via the crew API.\n" +
1906
2120
  "Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
1907
2121
  "Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
@@ -1919,7 +2133,8 @@ while (i < STEPS.length) {
1919
2133
  "For each issue, run in shell:\n" +
1920
2134
  "node " + CREW_API + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + LAUNCH_PROJECT_ID + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
1921
2135
  "(replace <issue title> and <issue details> with the real values).\n\n" +
1922
- "Report back in plain prose — what checks you ran and their results. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
2136
+ "BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
2137
+ "Report back in plain prose — what checks you ran and their results. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<what the task required>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten).";
1923
2138
  } else {
1924
2139
  instructions = "Test from a user's perspective. You are CODE-BLIND — do NOT read source code.\n" +
1925
2140
  "Public docs (API.md, README) are NOT source code — read them freely, exactly as a user would.\n" +
@@ -1931,37 +2146,10 @@ while (i < STEPS.length) {
1931
2146
  npmPublishCheck +
1932
2147
  "Report back in plain prose — what you tested and found. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
1933
2148
  }
1934
- // Visual verdict ownership (experiential artifact tasks only): the QA
1935
- // work agent covers the MECHANICAL CHECKS below — the artifact_inspect
1936
- // trigger prose is replaced, because the visual verdict is produced by
1937
- // the parent after the rendered post-change inspection results arrive.
1938
- // Mechanical checks are kept verbatim from the artifact path above.
1939
- if (qaVisual) {
1940
- instructions = "You are code-blind QA. You NEVER read source files. Public docs are not source — read them as a user would.\n" +
1941
- "VISUAL VERDICT OWNERSHIP: this task is experiential. The visual verdict is NOT yours to issue — it is produced after the rendered post-change inspection results arrive, by Hazel with the baseline and post-change evidence in hand.\n" +
1942
- "Your VERDICT below covers the MECHANICAL CHECKS only. Do NOT call artifact_inspect (async; the parent triggers the post-change capture after your step).\n\n" +
1943
- "MECHANICAL CHECKS:\n" +
1944
- "STEP 2: Verify data integrity via the crew API.\n" +
1945
- "Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
1946
- "Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
1947
- "DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale for a public-affecting change, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n\n" +
1948
- "PROVENANCE CHECK: Run in shell and return the stdout verbatim:\n" + crewCmd("get-provenance", {}) + "\n" +
1949
- "If provenance is null, report 'provenance missing — publish did not stamp source/crew release', then end your report with exactly this line: VERDICT: FAIL.\n" +
1950
- "Run: cd " + REPO_PATH + " && git rev-parse HEAD — call this LIVE_HEAD.\n" +
1951
- "Run: test -d " + crewHome + "/releases/<provenance.crew_release> (substitute the real stamped hash; do not run the literal placeholder). If the directory does not exist, FAIL: { \"passed\": false, \"summary\": \"provenance mismatch: crew_release [value from get-provenance] not found in release registry\" }.\n" +
1952
- "If provenance.source_commit equals LIVE_HEAD, the source check passes — continue to STEP 3.\n" +
1953
- "Otherwise the check is NOT failed yet: the parent stamps provenance AFTER post-deploy (docs/publish-verification.md), and post-deploy may commit an artifact-builder staging commit (\"rebuild: <task_id>\"), so LIVE_HEAD may sit ahead of the stamped commit ONLY IF every commit in between is such a rebuild marker. Verify exactly:\n" +
1954
- "1. Run: cd " + REPO_PATH + " && git merge-base --is-ancestor <provenance.source_commit> LIVE_HEAD && echo ANCESTOR_OK (substitute the real stamped hash and LIVE_HEAD; do not run the literal placeholders). If this command fails, FAIL: { \"passed\": false, \"summary\": \"provenance mismatch: stamped source_commit is not an ancestor of live HEAD\" }.\n" +
1955
- "2. Run: cd " + REPO_PATH + " && git log --format=%s <provenance.source_commit>..LIVE_HEAD (substitute real values). Every subject line MUST start with \"rebuild: \". If any line does not, report 'provenance mismatch: live HEAD moved past the stamped commit with non-rebuild source commits: [paste the offending subject lines]', then end your report with exactly this line: VERDICT: FAIL.\n" +
1956
- "If both pass, the source check passes — the only drift since the stamp is builder staging output committed by post-deploy. Continue to STEP 3.\n\n" +
1957
- "STEP 3: File follow-up tasks for any related issues you discover.\n" +
1958
- "For each issue, run in shell:\n" +
1959
- "node " + CREW_API + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + LAUNCH_PROJECT_ID + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
1960
- "(replace <issue title> and <issue details> with the real values).\n\n" +
1961
- "BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
1962
- "File follow-up tasks as today.\n" +
1963
- "Report back in plain prose — what checks you ran and their results. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL on the mechanical checks.";
1964
- }
2149
+ // qaVisual (experiential artifact tasks): Hazel owns the visual verdict
2150
+ // through the experiential QA prompt built in the PUBLISH_TYPE ===
2151
+ // "artifact" branch above (STEP 1 see-act loop + OODA report). There is
2152
+ // no parent visual-verdict protocol anymore — no override here.
1965
2153
  }
1966
2154
 
1967
2155
  // Task event history — Review and Publish are excluded. Review is cold by
@@ -2142,6 +2330,52 @@ while (i < STEPS.length) {
2142
2330
  };
2143
2331
  }
2144
2332
  log("Build worktree confinement passed: " + wt.path);
2333
+
2334
+ // Already-merged idempotency: a `repo_diff: none (already-merged:
2335
+ // <sha>)` declaration is verified mechanically — <sha> must resolve
2336
+ // and be an ancestor of main in the configured repo. A fabricated or
2337
+ // mistaken declaration fails the phase here (the dispatcher retries
2338
+ // Build under its consecutive-failure cap); a verified declaration is
2339
+ // recorded in alreadyMergedSha for Review's no-diff branch. Without
2340
+ // this guard, Build correctly doing nothing left Review with no
2341
+ // mechanical way to accept an empty diff, and Cass rejected for "no
2342
+ // commits ahead of main — the builder likely forgot to commit" while
2343
+ // the deliverable sat on main (canary 2026-09-15, task 1d692d91).
2344
+ // The sha is hex-only by construction (extractAlreadyMerged), so
2345
+ // interpolating it into the shell command cannot inject.
2346
+ var am = extractAlreadyMerged(workerText);
2347
+ if (am.sha) {
2348
+ var amCheck = await agent(
2349
+ "Verify the builder's already-merged declaration.\n" +
2350
+ "Run in shell and return the stdout verbatim:\n" +
2351
+ "cd " + REPO_PATH + " && git rev-parse --verify --quiet " + am.sha + " >/dev/null && git merge-base --is-ancestor " + am.sha + " main && echo ALREADY_MERGED_YES || echo ALREADY_MERGED_NO",
2352
+ { key: "verify-already-merged" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Verifying already-merged declaration" }
2353
+ );
2354
+ var amOut = (typeof amCheck === "string") ? amCheck : JSON.stringify(amCheck);
2355
+ if (!/ALREADY_MERGED_YES/.test(amOut)) {
2356
+ log("Build already-merged declaration failed verification — " + am.sha + " is not an ancestor of main — marking failed for retry");
2357
+ await agent(
2358
+ "Record already-merged verification failure.\n" +
2359
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
2360
+ task_id: taskId,
2361
+ session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: "failed",
2362
+ notes: "Build declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of main in the configured repo. The declaration is fabricated or mistaken; the work is not on main. Phase failed for retry" },
2363
+ event: { task_id: taskId, type: "failed", message: "Build already-merged declaration failed verification — " + am.sha + " not an ancestor of main, phase failed, dispatcher will retry" }
2364
+ }),
2365
+ { key: "record-already-merged-fail-" + step.name, label: "Recording already-merged verification failure" }
2366
+ );
2367
+ return {
2368
+ __hatchWorkflowControl: "blocked",
2369
+ result: {
2370
+ blocked_reason: "Build already-merged declaration failed verification",
2371
+ message: "The builder declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of main. The work is not on main; the phase is marked failed and the dispatcher will retry Build.",
2372
+ task_id: taskId
2373
+ }
2374
+ };
2375
+ }
2376
+ alreadyMergedSha = am.sha;
2377
+ log("Build already-merged declaration verified: " + am.sha + " is an ancestor of main");
2378
+ }
2145
2379
  }
2146
2380
 
2147
2381
  // Deterministic closeout: no formatter agent. The verdict is mechanical
@@ -2270,9 +2504,10 @@ while (i < STEPS.length) {
2270
2504
  // it to HEAD: that verifies the stamp, not the content. Canary run 8
2271
2505
  // (2026-09-11) passed it with a hollow build — the stamp was honest, the
2272
2506
  // artifact was stale, all eight phases green. The stamp now moves to the
2273
- // parent: trigger an independent artifact_inspect read-back of the changed
2274
- // regions here; the parent stamps provenance only after mechanically
2275
- // confirming the artifact's actual content matches the merged diff. QA's
2507
+ // parent (docs/publish-verification.md); the independent read-back step
2508
+ // is currently unavailable (no agent-callable read-back tool exists —
2509
+ // artifact_inspect was removed by the platform 2026-09-14), so the parent
2510
+ // cannot confirm content and the task parks for verification. QA's
2276
2511
  // provenance check enforces the stamp — an unverified publish fails loudly
2277
2512
  // there instead of passing silently here.
2278
2513
  // Skip-aware (park 2026-09-11): an empty-diff Integrate takes no merge
@@ -2280,37 +2515,13 @@ while (i < STEPS.length) {
2280
2515
  // there is no new content to verify, so verification is vacuous.
2281
2516
  // publishSkippedNoLock is workflow-computed state from the explicit
2282
2517
  // lock-status read in STEP 0, not agent prose.
2283
- var publishVerifyInspect = { triggered: false, inspection_id: "", error: "" };
2284
- if (step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG) {
2285
- if (publishSkippedNoLock) {
2286
- log("Publish skipped for task " + taskId + " (no merge lock held — empty-diff Integrate): content verification vacuous, nothing was shipped");
2287
- } else if (!publishBuildLanded) {
2288
- log("Publish build did not land for task " + taskId + " — no content to verify (the failure park above already fired)");
2289
- } else {
2290
- try {
2291
- var inspectResult = await agent(
2292
- ARTIFACT_LOAD_PREAMBLE +
2293
- "Call artifact_inspect with slug \"" + PUBLISH_SLUG + "\", repair_authorized false, and verbatim_request exactly as follows:\n" +
2294
- "<<<READBACK_REQUEST\n" + buildPublishReadbackRequest(taskId, mergeCommitForPublish, mergeDiff, rebuildAgentId) + "\nREADBACK_REQUEST\n" +
2295
- "If artifact_inspect is still not available after the load, do NOT improvise — return { \"triggered\": false, \"inspection_id\": \"\", \"error\": \"artifact_tools missing after load\" } and nothing else.\n" +
2296
- "Return JSON { \"triggered\": <true if the inspection started, false otherwise>, \"inspection_id\": \"<the inspection id, or empty string>\", \"error\": \"<details or empty string>\" } and nothing else.",
2297
- { key: attemptKey("publish-verify-inspect-" + taskId, totalReworkCount), label: "Triggering publish content read-back",
2298
- schema: { type: "object", properties: { triggered: { type: "boolean" }, inspection_id: { type: "string" }, error: { type: "string" } }, required: ["triggered"] } }
2299
- );
2300
- publishVerifyInspect.triggered = !!(inspectResult && inspectResult.triggered);
2301
- publishVerifyInspect.inspection_id = (inspectResult && inspectResult.inspection_id) || "";
2302
- publishVerifyInspect.error = (inspectResult && inspectResult.error) || "";
2303
- if (publishVerifyInspect.triggered) {
2304
- log("Publish content read-back inspection triggered for task " + taskId + ": " + publishVerifyInspect.inspection_id);
2305
- } else {
2306
- log("Publish content read-back inspect trigger failed for task " + taskId + ": " + (publishVerifyInspect.error || "not started") + " — the park below asks the parent to trigger it manually");
2307
- }
2308
- } catch (e) {
2309
- publishVerifyInspect.error = (e && e.message ? e.message : String(e)).slice(0, 200);
2310
- log("Publish content read-back inspect trigger threw for task " + taskId + ": " + publishVerifyInspect.error + " — the park below asks the parent to trigger it manually");
2311
- }
2312
- } // end: !publishSkippedNoLock && publishBuildLanded — a skipped or failed publish has nothing to verify
2313
- }
2518
+ // The parent (tick worker) triggers the ONE read-back inspection it can
2519
+ // actually receive (async results go to the root agent, never into a
2520
+ // workflow run — a workflow-side trigger would be an orphan). The workflow
2521
+ // only parks; the parent's scan builds the request deterministically via
2522
+ // lib/build-readback-request.js and ferries the inspection.
2523
+ // publishBuildLanded and publishSkippedNoLock are workflow-computed state;
2524
+ // a skipped or failed publish has nothing to verify.
2314
2525
 
2315
2526
  // Session notes. Machine-readable marker lines are extracted from the full
2316
2527
  // worker report and appended AFTER the slice so a long report can never
@@ -2329,15 +2540,20 @@ while (i < STEPS.length) {
2329
2540
  } else {
2330
2541
  summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
2331
2542
  }
2543
+ // Already-merged attestation: when the Build gate verified the builder's
2544
+ // already-merged declaration, the workflow records its own marker line in
2545
+ // the session notes (like the builder markers above, it is appended after
2546
+ // the slice so it can never be amputated). A later run resumed at Review
2547
+ // hydrates alreadyMergedSha from this workflow-attested line — never from
2548
+ // the builder's declaration alone.
2549
+ if (step.name === "Build" && alreadyMergedSha) {
2550
+ summary += "\nalready_merged_verified: " + alreadyMergedSha;
2551
+ }
2332
2552
 
2333
2553
  // Visual verdict evidence: for experiential artifact tasks, append the
2334
2554
  // deterministic post-change capture plan to the QA session notes. The
2335
2555
  // parent protocol (docs/visual-verdict.md) triggers the inspection with
2336
- // this plan; "visual: pending" marks the owed verdict.
2337
- if (step.name === "QA" && passed && qaVisual) {
2338
- summary = (summary + "\nvisual: pending\ncapture_plan: " +
2339
- buildVisualCapturePlan(taskTitle, taskDescription, "postchange", captureTargets).replace(/\s+/g, " ")).slice(0, 2900);
2340
- }
2556
+
2341
2557
 
2342
2558
  // Capture mapper's spec for Build and Review
2343
2559
  if (step.name === "Map" && passed) {
@@ -2372,41 +2588,6 @@ while (i < STEPS.length) {
2372
2588
  }
2373
2589
  );
2374
2590
 
2375
- // Visual-verdict gate: for experiential artifact tasks the task is done
2376
- // only when the parent has recorded a visual_verdict: note event. The QA
2377
- // work agent above covered the mechanical checks only. No recorded
2378
- // verdict → park for the parent protocol (never mark done on a pending
2379
- // visual verdict). A recorded FAIL with budget left → rework at Build.
2380
- if (step.name === "QA" && passed && qaVisual) {
2381
- var vvStatus = await visualVerdictStatus();
2382
- if (vvStatus.found && vvStatus.verdict === "PASS") {
2383
- log("Visual verdict PASS recorded for task " + taskId + (vvStatus.detail ? " — " + vvStatus.detail : ""));
2384
- } else if (vvStatus.found && vvStatus.verdict === "FAIL") {
2385
- // A FAIL whose reason begins exactly "rendering impossible:" is not
2386
- // reworkable — there is no rendered evidence to fix against. Park for
2387
- // human attention instead of bouncing to Build.
2388
- if (vvStatus.detail.indexOf("rendering impossible:") === 0) {
2389
- log("Visual verdict FAIL (rendering impossible) for task " + taskId + " — parking for human attention");
2390
- return await parkTask("Visual verdict FAIL — rendering impossible, human attention required: " + vvStatus.detail);
2391
- }
2392
- totalReworkCount++;
2393
- if (totalReworkCount > MAX_TOTAL_REWORK) {
2394
- log("Shared rework budget exhausted for task " + taskId + " — parking after visual verdict FAIL");
2395
- return await parkTask("Exceeded shared rework budget (" + MAX_TOTAL_REWORK + " total rework attempts across Review and QA) after visual verdict FAIL: " + vvStatus.detail);
2396
- }
2397
- rejectionNotes = "Visual verdict FAIL: " + vvStatus.detail;
2398
- i = BUILD_INDEX;
2399
- log("Visual verdict FAIL — bouncing to Build (rework #" + totalReworkCount + " of " + MAX_TOTAL_REWORK + ")");
2400
- continue;
2401
- } else {
2402
- if (!VISUAL_PROTOCOL_AVAILABLE) {
2403
- log("Visual verdict protocol not available (VISUAL_PROTOCOL_AVAILABLE=false) for task " + taskId + " — skipping visual gate, QA mechanical checks already passed");
2404
- } else {
2405
- return await parkTask("Visual verdict pending — parent: run the post-change capture + visual verdict protocol in docs/visual-verdict.md (capture plan is in the QA session notes)");
2406
- }
2407
- }
2408
- }
2409
-
2410
2591
  // Handle rejection — bounce back to Build
2411
2592
  if (!passed && (step.name === "Review" || step.name === "QA")) {
2412
2593
  totalReworkCount++;
@@ -2446,10 +2627,7 @@ while (i < STEPS.length) {
2446
2627
  if (passed && step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG && !publishSkippedNoLock && publishBuildLanded) {
2447
2628
  return await parkTask("publish: verification-requested " + mergeCommitForPublish +
2448
2629
  " (build " + (rebuildAgentId || "agent_id unobserved") + ")" +
2449
- " — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md" +
2450
- (publishVerifyInspect.triggered
2451
- ? " (content read-back inspection " + publishVerifyInspect.inspection_id + " already triggered)."
2452
- : " (read-back inspect trigger failed: " + (publishVerifyInspect.error || "not started") + " — parent: trigger artifact_inspect manually)."));
2630
+ " — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md.");
2453
2631
  }
2454
2632
 
2455
2633
  i++;