muse-crew 0.7.9 → 0.7.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -472,52 +472,22 @@ function verifyAppliedChanges(expected, applied) {
472
472
  return { ok: true };
473
473
  }
474
474
 
475
- // Publish read-back request builder: the verbatim_request the workflow hands
476
- // to artifact_inspect (via a child) after the artifact build lands. Pure
477
- // function — no I/O, no clock. The request carries the merged diff as the
478
- // expected change and asks for an independent read of the artifact's actual
479
- // source: for each file, the exact current text of the changed regions plus
480
- // a per-line present/absent finding. The parent (docs/publish-verification.md)
481
- // compares these findings against the diff mechanically and stamps provenance
482
- // only on a match. This breaks the circularity that hollowed canary run 8
483
- // (2026-09-11): verifyAppliedChanges compares the builder's applied-report
484
- // against the diff the report was derived from — a fabricated report passes
485
- // by construction. Independent read-back cannot be fabricated from the diff;
486
- // it must match the artifact's real content.
487
- function buildPublishReadbackRequest(taskId, commit, diff, buildAgentId) {
488
- // Build-ID correlation (2026-09-12): buildAgentId is the build.agent_id the
489
- // workflow observed for the publish attempt (the artifact system's in-flight
490
- // build correlation ID — not a durable post-completion identifier). The read-back request carries it so the parent can
491
- // prove the read-back inspected the live build of THIS attempt — not a
492
- // different build's output. Null/empty means the edit was accepted but
493
- // never correlated to a builder run. Pure function of inputs — no I/O,
494
- // no clock.
495
- var buildIdLine = (typeof buildAgentId === "string" && buildAgentId.length > 0)
496
- ? "Expected builder build agent_id: " + buildAgentId + " (the artifact system's in-flight correlation ID for this publish attempt — not a durable post-completion identifier).\n"
497
- : "No build agent_id was observed for this publish attempt (the edit was accepted but never correlated to a builder run) — say so explicitly in your report.\n";
498
- return (
499
- "Publish content read-back for task " + taskId + ", merge commit " + commit + ".\n" +
500
- "The unified diff below was supposed to be applied to this artifact's source tree and deployed. Do NOT modify anything.\n" +
501
- "Do NOT rely on the builder's applied-changes report — it is derived from this same diff, so it cannot confirm the content. Read the artifact's CURRENT source directly.\n" +
502
- "\n" +
503
- buildIdLine +
504
- "Report the live build's agent_id as seen in artifact_status (or state explicitly that no build/agent_id is visible). If an expected agent_id is given above and the live one differs, say so exactly — the read-back may be inspecting a different build's output.\n" +
505
- "\n" +
506
- "UNIFIED DIFF (expected change):\n" +
507
- "```diff\n" + diff + "\n```\n" +
508
- "\n" +
509
- "For each file in the diff:\n" +
510
- "1. Read the file's CURRENT content in the artifact source tree.\n" +
511
- "2. Quote the exact current text of the regions around the changed lines.\n" +
512
- "3. For every added (+) line in the diff, state whether that exact line is PRESENT in the current source.\n" +
513
- "4. For every removed (-) line in the diff, state whether that exact line is ABSENT from the current source.\n" +
514
- "5. Report build/deploy health and the console error count.\n" +
515
- "\n" +
516
- "Return the per-file present/absent findings with the quoted observed lines. Do not modify anything.\n" +
517
- "This read-back feeds the parent content-verification protocol (docs/publish-verification.md): the parent stamps provenance only when every added line is present and every removed line is absent."
518
- );
519
- }
520
-
475
+ // Publish read-back request (currently unavailable): the verbatim_request
476
+ // the parent protocol (docs/publish-verification.md) would hand to an
477
+ // independent read-back tool after the artifact build lands. artifact_inspect
478
+ // was removed by the platform (2026-09-14); artifact.inspect is malfunction
479
+ // diagnosis, not a substitute — so no agent-callable read-back tool exists
480
+ // and this request cannot currently be issued. Pure function — no I/O, no
481
+ // clock. The request carries the merged diff as the expected change and asks
482
+ // for an independent read of the artifact's actual source: for each file, the
483
+ // exact current text of the changed regions plus a per-line present/absent
484
+ // finding. Until a read-back path exists, the parent cannot independently
485
+ // confirm content and verification parks at "publish: verification-requested"
486
+ // (see docs/publish-verification.md). This preserves the circularity break
487
+ // that hollowed canary run 8 (2026-09-11): verifyAppliedChanges compares the
488
+ // builder's applied-report against the diff the report was derived from — a
489
+ // fabricated report passes by construction. Independent read-back cannot be
490
+ // fabricated from the diff; it must match the artifact's real content.
521
491
  // Pre-publish base observation (diagnostic, 2026-09-12): instruction fragment
522
492
  // for the builder's edit request, asking it to report the sha256 of each
523
493
  // touched file's CURRENT content BEFORE applying the diff. Pure function —
@@ -751,43 +721,14 @@ async function baselineStatus() {
751
721
  return { baseline_found: false, baseline_kind: "", baseline_refs: "", requested_count: 0, evidence_count: 0 };
752
722
  }
753
723
  }
754
- // Visual verdict status: the parent records the visual verdict as a note
755
- // event after the rendered post-change inspection results arrive. Finds the
756
- // latest step "QA" session's started_at, then note events NEWER than it
757
- // whose message starts exactly "visual_verdict: PASS" / "visual_verdict:
758
- // FAIL". Returns { found, verdict, detail }. A fresh key per call: the
759
- // verdict lands while this run is parked.
760
- let visualVerdictCallCount = 0;
761
- async function visualVerdictStatus() {
762
- visualVerdictCallCount++;
763
- try {
764
- var vv = await agent(
765
- "Read this task's QA session and note events from the crew API.\n" +
766
- "Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
767
- "Find the LATEST session with task_id \"" + taskId + "\" and step \"QA\" in the returned sessions array and note its started_at timestamp (call it QA_START; use empty string if there is no QA session).\n" +
768
- "Then run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId }) + "\n" +
769
- "Consider only events with type \"note\" whose timestamp is newer than QA_START. Among them, find messages starting exactly with \"visual_verdict: PASS\" or \"visual_verdict: FAIL\" (exact prefix, case-sensitive); use the latest such message.\n" +
770
- "Return JSON { \"found\": <true if such a message exists>, \"verdict\": \"<\"PASS\" or \"FAIL\" from that message, or empty string>\", \"detail\": \"<the text after the prefix in that message, or empty string>\" } and nothing else.",
771
- {
772
- key: "visual-verdict-status-" + taskId + "-" + visualVerdictCallCount,
773
- label: "Reading visual verdict status",
774
- schema: {
775
- type: "object",
776
- properties: {
777
- found: { type: "boolean" },
778
- verdict: { type: "string" },
779
- detail: { type: "string" }
780
- },
781
- required: ["found", "verdict", "detail"]
782
- }
783
- }
784
- );
785
- return { found: !!(vv && vv.found), verdict: (vv && vv.verdict) || "", detail: (vv && vv.detail) || "" };
786
- } catch (e) {
787
- log("visualVerdictStatus: agent call failed (" + (e && e.message ? e.message : e) + ") — treating as not found");
788
- return { found: false, verdict: "", detail: "" };
789
- }
790
- }
724
+ // The parent visual-verdict reader was removed 2026-09-15: Hazel (the QA
725
+ // work agent) now owns the visual verdict experientially — she drives the
726
+ // now owns the visual verdict experientially — she drives the see-act loop
727
+ // herself and records verdict.json + the append-only verdicts.jsonl. The old
728
+ // see-act loop herself and records verdict.json + the append-only
729
+ // verdicts.jsonl. The old parent note gate is obsolete and has been
730
+ // deleted from the QA closeout below. (See docs/visual-verdict.md.)
731
+ // from the QA closeout below. (See docs/visual-verdict.md.)
791
732
 
792
733
  // STEPS inline — export const meta is parsed as metadata, not a runtime binding
793
734
  const STEPS = [
@@ -1245,12 +1186,10 @@ while (i < STEPS.length) {
1245
1186
  }
1246
1187
  }
1247
1188
 
1248
- // Visual verdict routing: experiential artifact tasks get their visual
1249
- // verdict from the parent AFTER the rendered post-change inspection
1250
- // results arrive (this script cannot receive the async handoff). The QA
1251
- // work agent covers mechanical checks only; the visual-verdict gate below
1252
- // parks for the parent protocol when no visual_verdict: note event exists
1253
- // yet.
1189
+ // Visual verdict routing (2026-09-15): experiential artifact tasks route
1190
+ // to Hazel's own experiential QA instructions below — she drives the
1191
+ // see-act loop herself and owns the visual verdict (verdict.json +
1192
+ // verdicts.jsonl). No parent verdict gate remains.
1254
1193
  var qaVisual = false;
1255
1194
  if (step.name === "QA") {
1256
1195
  qaVisual = (await resolveExperiential()) === "yes" && PUBLISH_TYPE === "artifact";
@@ -1260,18 +1199,25 @@ while (i < STEPS.length) {
1260
1199
  var instructions = "";
1261
1200
 
1262
1201
  if (step.name === "Triage") {
1263
- instructions = "Validate the task, check clarity, note dependencies, confirm the bugfix workflow assignment.\nIf the task needs decomposition, note that in your assessment.\nReport back in plain prose — what you found.\nEXPERIENTIAL FLAG: does this task change anything rendered and visible in the project's user-facing artifact (pages, components, styles, layout, copy, visual states)? If yes it is experiential and gets baseline captures (plus a visual verdict where the workflow has a QA phase). End your report with exactly one line on its own, lowercase, unrephrased: experiential: yes — or experiential: no. This line is machine-read.";
1202
+ instructions = "Validate the task against the project's repo at " + REPO_PATH + " — that exact checkout, not any other copy of the project on disk. If you run git commands, cd " + REPO_PATH + " first.\nCheck clarity, note dependencies, confirm the bugfix workflow assignment.\nIf the task needs decomposition, note that in your assessment.\nReport back in plain prose — what you found.\nEXPERIENTIAL FLAG: does this task change anything rendered and visible in the project's user-facing artifact (pages, components, styles, layout, copy, visual states)? If yes it is experiential and gets baseline captures (plus a visual verdict where the workflow has a QA phase). End your report with exactly one line on its own, lowercase, unrephrased: experiential: yes — or experiential: no. This line is machine-read.";
1264
1203
 
1265
1204
  } else if (step.name === "Reproduce") {
1266
- instructions = "Reproduce the bug from a user's perspective. You are CODE-BLIND — do NOT read source code.\n" +
1267
- "To investigate, run in shell and read the stdout JSON:\n" + crewCmd("get-state", {}) + "\n" +
1268
- "This returns current sessions, events, and tasks.\n" +
1269
- "Look at session notes in the returned data — check whether multiline content has newlines preserved or runs together.\n" +
1270
- "You can also check a specific task's events with: node " + CREW_API + " --crew-home " + crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
1271
- "Do NOT use artifact_inspect — it is async and will not return results inline.\n" +
1272
- "Capture concrete evidence from the data you retrieve.\n" +
1273
- "Report your reproduction evidence and steps as plain prose.\n" +
1274
- "End your report with exactly one line: VERDICT: PASS if reproduction succeeded, VERDICT: FAIL if it failed.";
1205
+ instructions = "Reproduce the bug from a user's perspective — by USING the artifact, not by reading data. You are CODE-BLIND — do NOT read source code.\n" +
1206
+ "You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
1207
+ "Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and fall back to the data-level investigation at the end.\n" +
1208
+ "a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and fall back to the data-level investigation.\n" +
1209
+ "b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-repro > /tmp/repro-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/repro-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/repro-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/repro-server-" + taskId + ".log) and fall back to the data-level investigation.\n" +
1210
+ "c. Bounded see-act loop, at most 8 steps: drive the artifact to TRIGGER the reported bug. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/repro/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
1211
+ "c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/repro/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you. If the server stops responding mid-loop (curl --max-time 3 http://localhost:<N>/ fails), restart it per (b).\n" +
1212
+ "c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
1213
+ "c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/repro/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/repro/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
1214
+ "d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the bug needs such a sequence, report NOT POSSIBLE for that part and reproduce what you can.\n" +
1215
+ "e. Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/repro/ and indexed in ooda-log.jsonl — that log plus the frames is your OODA report for this reproduction. A frame you did not read is not evidence. Loading, error, or blank frames never reproduce anything.\n" +
1216
+ "f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-repro' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
1217
+ "Data-level fallback (only if the browser loop above is NOT POSSIBLE): run in shell and read the stdout JSON:\n" + crewCmd("get-state", {}) + "\n" +
1218
+ "This returns current sessions, events, and tasks — capture concrete evidence from the data you retrieve.\n" +
1219
+ "Report your reproduction steps and evidence as plain prose.\n" +
1220
+ "End your report with exactly one line: VERDICT: PASS if you reproduced the reported bug (your frames show the reported misbehavior), VERDICT: FAIL if you could not. --expected names the bug as reported (its visible manifestation); --actual names what your frames actually showed. Checks you could not run are evidence gaps, not silent drops: name every one in --missing. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/repro/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/repro/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the bug as reported — its visible manifestation>\" --actual \"<what your frames actually showed>\" --missing '[\"honest evidence gap, if any\"]' — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten).";
1275
1221
 
1276
1222
  } else if (step.name === "Map") {
1277
1223
  var mapGatePara = "";
@@ -1399,10 +1345,11 @@ while (i < STEPS.length) {
1399
1345
  // (fail-closed). There is deliberately NO workflow-side provenance
1400
1346
  // stamp: the builder's applied-report is circular (canary run 8,
1401
1347
  // 2026-09-11), so the stamp moved to the parent — after the build
1402
- // lands, the workflow triggers an independent artifact_inspect
1403
- // read-back, records the session completed, and parks with
1404
- // "publish: verification-requested". The parent stamps provenance only
1405
- // after the read-back confirms the content (docs/publish-verification.md);
1348
+ // lands, the workflow records the session completed and parks with
1349
+ // "publish: verification-requested". The parent owns verification
1350
+ // (docs/publish-verification.md); the independent read-back step is
1351
+ // currently unavailable (no agent-callable read-back tool exists —
1352
+ // artifact_inspect was removed by the platform 2026-09-14).
1406
1353
  // QA's provenance check enforces the stamp mechanically.
1407
1354
  var artifactPublish = null;
1408
1355
  var publishLockRefreshed = false;
@@ -1451,14 +1398,45 @@ while (i < STEPS.length) {
1451
1398
  // changes — no prose claim to trust. If the artifact tool namespace
1452
1399
  // is missing from this child it reports honestly and the workflow
1453
1400
  // retries once with a fresh key (bounded); anything else parks.
1401
+ // Publish diff base (2026-09-14, task 0c53af4e): the carried diff is
1402
+ // BASE..HEAD where BASE is the previously-stamped provenance
1403
+ // source_commit — NOT HEAD^1. A push-time reconcile merge puts the
1404
+ // task's own changes behind an intermediate merge, so HEAD^1..HEAD
1405
+ // silently drops the task's fix while the artifact builds without
1406
+ // it. The stamped base is the artifact's actual content; BASE..HEAD
1407
+ // is the complete unpublished delta. Empty tree only for a genuine
1408
+ // first publish (no provenance stamped yet).
1409
+ var EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
1410
+ var provResult = await agent(
1411
+ crewCmd("get-provenance", {}) + "\n" +
1412
+ "Return JSON { \"provenance\": <the CLI's provenance object, or null when nothing is stamped> } and nothing else. Do not interpret it.",
1413
+ { key: attemptKey("publish-provenance-base-" + taskId, totalReworkCount), label: "Reading stamped publish base",
1414
+ schema: { type: "object", properties: { provenance: { type: ["object", "null"] } }, required: ["provenance"] } }
1415
+ );
1416
+ var publishBase = (provResult.provenance && provResult.provenance.source_commit) || "";
1417
+ publishBase = String(publishBase).trim();
1418
+ if (!publishBase) {
1419
+ publishBase = EMPTY_TREE_SHA;
1420
+ log("Publish base for task " + taskId + ": no provenance stamped yet — using empty tree (first publish)");
1421
+ } else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
1422
+ return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
1423
+ }
1454
1424
  var diffResult = await agent(
1455
- "Run: cd " + REPO_PATH + " && git rev-parse HEAD && echo '---PARENT---' && git rev-parse HEAD^1 && echo '---DIFF---' && git diff HEAD^1 HEAD && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r HEAD\n" +
1456
- "Return JSON { \"commit\": \"<HEAD trimmed>\", \"parent\": \"<HEAD^1 trimmed>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
1457
- { key: attemptKey("publish-artifact-diff-" + taskId, totalReworkCount), label: "Computing merged diff for publish",
1458
- schema: { type: "object", properties: { commit: { type: "string" }, parent: { type: "string" }, diff: { type: "string" }, files: { type: "string" } }, required: ["commit", "diff"] } }
1425
+ "Run: cd " + REPO_PATH + " && BASE='" + publishBase + "' && HEAD=$(git rev-parse HEAD) && " +
1426
+ "if [ \"$BASE\" = '" + EMPTY_TREE_SHA + "' ]; then ANCESTOR=yes; else git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no; fi && " +
1427
+ "echo '---COMMIT---' && echo \"$HEAD\" && echo '---BASE---' && echo \"$BASE\" && echo '---ANCESTOR---' && echo \"$ANCESTOR\" && " +
1428
+ "if [ \"$ANCESTOR\" = yes ]; then echo '---DIFF---' && git diff \"$BASE\" \"$HEAD\" && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r \"$BASE\" \"$HEAD\"; fi\n" +
1429
+ "Return JSON { \"commit\": \"<HEAD trimmed>\", \"base\": \"<BASE trimmed>\", \"ancestor\": \"<yes|no>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
1430
+ { key: attemptKey("publish-artifact-diff-" + taskId, totalReworkCount), label: "Computing publish diff from stamped base",
1431
+ schema: { type: "object", properties: { commit: { type: "string" }, base: { type: "string" }, ancestor: { type: "string" }, diff: { type: "string" }, files: { type: "string" } }, required: ["commit", "base", "ancestor", "diff"] } }
1459
1432
  );
1433
+ if ((diffResult.ancestor || "").trim() !== "yes") {
1434
+ return await parkTask("Publish base " + publishBase.slice(0, 12) + " is not an ancestor of HEAD " + (diffResult.commit || "").trim().slice(0, 12) + " — the stamped provenance does not lead to the integrated commit. Human attention needed.");
1435
+ }
1436
+ if ((diffResult.base || "").trim() !== publishBase) {
1437
+ return await parkTask("Publish diff base mismatch: agent reported '" + (diffResult.base || "").trim().slice(0, 12) + "' but the stamped base is '" + publishBase.slice(0, 12) + "'. Human attention needed.");
1438
+ }
1460
1439
  var mergeCommitForPublish = (diffResult.commit || "").trim();
1461
- var mergeParentForPublish = (diffResult.parent || "").trim();
1462
1440
  var mergeDiff = diffResult.diff || "";
1463
1441
  if (!mergeDiff.trim()) {
1464
1442
  return await parkTask("Publish diff is empty for commit " + (mergeCommitForPublish || "unknown") + " — a merge lock was held but there is no change to carry. Human attention needed.");
@@ -1469,11 +1447,15 @@ while (i < STEPS.length) {
1469
1447
  if (/^rename from /m.test(mergeDiff)) {
1470
1448
  return await parkTask("Publish diff contains a rename — the diff transport cannot carry renames. Human attention needed.");
1471
1449
  }
1472
- var mergeDiffLines = mergeDiff.split("\n").length;
1473
- if (mergeDiffLines > 200) {
1474
- return await parkTask("Publish diff is " + mergeDiffLines + " lines (budget 200) — too large for the diff transport. Human attention needed.");
1475
- }
1476
1450
  var expectedChanges = parseUnifiedDiff(mergeDiff);
1451
+ // Budget counts CHANGED lines (added + removed), not raw unified-diff
1452
+ // output lines: context lines and file headers inflated the old
1453
+ // split("\n").length count ~2x, parking a 95-line change against a
1454
+ // 200-line budget (Gate 1 Journey 3 attempt 3, 2026-09-13).
1455
+ var mergeDiffChangedLines = expectedChanges.reduce(function (n, f) { return n + f.added.length + f.removed.length; }, 0);
1456
+ if (mergeDiffChangedLines > 200) {
1457
+ return await parkTask("Publish diff changes " + mergeDiffChangedLines + " lines (budget 200) — too large for the diff transport. Human attention needed.");
1458
+ }
1477
1459
  if (expectedChanges.length === 0) {
1478
1460
  return await parkTask("Publish diff parsed to zero files for commit " + (mergeCommitForPublish || "unknown") + " — cannot verify application. Human attention needed.");
1479
1461
  }
@@ -1486,12 +1468,16 @@ while (i < STEPS.length) {
1486
1468
  // but never parks. The observation tells us what the publish actually
1487
1469
  // reads, so the subsequent fix can require the right base.
1488
1470
  var expectedBaseHashes = {};
1489
- try {
1471
+ if (publishBase === EMPTY_TREE_SHA) {
1472
+ // First publish: every file in the diff is new to the artifact.
1473
+ expectedChanges.forEach(function (f) { expectedBaseHashes[f.path] = "NEW-FILE"; });
1474
+ log("Publish expected base hashes for task " + taskId + ": empty tree (first publish) — all " + expectedChanges.length + " file(s) new");
1475
+ } else try {
1490
1476
  // Shell-quote helper (no regex-with-quote: the test parser does not
1491
1477
  // understand regex literals containing quotes).
1492
1478
  var sq = function(s) { return "'" + String(s).split("'").join("'\\''") + "'"; };
1493
1479
  var baseHashResult = await agent(
1494
- "Run: cd " + REPO_PATH + " && parent=" + sq(mergeParentForPublish) + " && for f in " + expectedChanges.map(function(f) { return sq(f.path); }).join(" ") + "; do printf '%s:' \"$f\"; git show \"$parent:$f\" 2>/dev/null | sha256sum | cut -d' ' -f1; done\n" +
1480
+ "Run: cd " + REPO_PATH + " && parent=" + sq(publishBase) + " && for f in " + expectedChanges.map(function(f) { return sq(f.path); }).join(" ") + "; do printf '%s:' \"$f\"; git show \"$parent:$f\" 2>/dev/null | sha256sum | cut -d' ' -f1; done\n" +
1495
1481
  "Return JSON { \"hashes\": \"<newline-separated <path>:<sha256> lines, empty hash means the file is new in this diff>\" } and nothing else.",
1496
1482
  { key: attemptKey("publish-base-hashes-" + taskId, totalReworkCount), label: "Computing expected base content hashes",
1497
1483
  schema: { type: "object", properties: { hashes: { type: "string" } }, required: ["hashes"] } }
@@ -1500,7 +1486,7 @@ while (i < STEPS.length) {
1500
1486
  var m = /^([^:]+):([0-9a-f]*)$/.exec(line.trim());
1501
1487
  if (m) expectedBaseHashes[m[1]] = m[2] || "NEW-FILE";
1502
1488
  });
1503
- log("Publish expected base hashes for task " + taskId + " (merge parent " + (mergeParentForPublish || "unknown").slice(0, 12) + "): " + JSON.stringify(expectedBaseHashes));
1489
+ log("Publish expected base hashes for task " + taskId + " (stamped base " + publishBase.slice(0, 12) + "): " + JSON.stringify(expectedBaseHashes));
1504
1490
  } catch (e) {
1505
1491
  log("Publish expected base hash computation failed for task " + taskId + " (non-fatal, observation degraded): " + (e && e.message ? e.message : e));
1506
1492
  }
@@ -1558,6 +1544,7 @@ while (i < STEPS.length) {
1558
1544
  required: ["edit_started", "build_agent_id", "applied"] };
1559
1545
  var rebuildTrigger = null;
1560
1546
  var rebuildReportMissing = false; // true if the edit went through but the agent returned no applied report (structured-output failure) — the smoke-check is skipped; the parent's independent read-back is the verification
1547
+ var rebuildEvidenceNote = null; // human-readable evidence line for the ledger when the edit is confirmed via fallback evidence (in-flight poll or durable audit dir) rather than the trigger's own report
1561
1548
  // The trigger key of the attempt that last ran, for the publish ledger.
1562
1549
  // Minted once here (not re-minted per use site) so the ledger always
1563
1550
  // records the exact key that was issued — and so a re-minted duplicate
@@ -1576,6 +1563,32 @@ while (i < STEPS.length) {
1576
1563
  // (2026-09-12, task 23ca8f3f): computed once the trigger outcome is
1577
1564
  // known, logged loudly, never a park.
1578
1565
  var publishAppliedObservation = null; // "match" | "mismatch: <reason>" | "missing-report" — observation only, never a park
1566
+ // Durable-evidence snapshot (2026-09-14): the structured-output
1567
+ // fallback below only observes IN-FLIGHT builds. A build that
1568
+ // finished before the poll leaves no in-flight trace — but the
1569
+ // platform's audit harness leaves a durable one:
1570
+ // ~/workspace/ts-spaces/<slug>/audits/<timestamp>-<id>/ per
1571
+ // completed build. Snapshot the listing BEFORE the trigger so the
1572
+ // fallback can diff before/after: a directory appearing during the
1573
+ // trigger window is positive evidence the edit went through and
1574
+ // the build completed. Best-effort and non-gating: if the snapshot
1575
+ // fails, the durable check is skipped and the fallback behaves as
1576
+ // before. No wall-clock in-script (deterministic replay) — the
1577
+ // comparison is a pure before/after set diff.
1578
+ var auditDirsBeforeTrigger = [];
1579
+ try {
1580
+ var auditBefore = await agent(
1581
+ "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
1582
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1583
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1584
+ { key: attemptKey("publish-audit-before-" + taskId, totalReworkCount), label: "Snapshotting audit dirs before rebuild trigger",
1585
+ schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1586
+ );
1587
+ auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1588
+ log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
1589
+ } catch (auditBeforeErr) {
1590
+ log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1591
+ }
1579
1592
  try {
1580
1593
  rebuildTrigger = await agent(rebuildPrompt,
1581
1594
  { key: rebuildAttemptKey, label: "Triggering artifact rebuild", schema: rebuildSchema });
@@ -1634,7 +1647,46 @@ while (i < STEPS.length) {
1634
1647
  rebuildTrigger = { edit_started: true, error: "", applied: null };
1635
1648
  rebuildReportMissing = true;
1636
1649
  rebuildAgentId = acceptedAgentId;
1650
+ rebuildEvidenceNote = "edit confirmed via build-state poll after structured-output failure (build " + acceptedAgentId + "); builder applied-report missing";
1637
1651
  } else {
1652
+ // Durable completion check (2026-09-14): the in-flight poll
1653
+ // above only sees RUNNING builds. Attempt 7 (2026-09-14) proved
1654
+ // the gap: the trigger child applied the edit, the build ran
1655
+ // and completed — the platform's audit harness captured it
1656
+ // mid-window — then the child failed to return JSON. The
1657
+ // fallback poll saw no in-flight build, so a successful publish
1658
+ // parked as "unknown". Diff the audit-dir listing against the
1659
+ // pre-trigger snapshot: a timestamped directory that appeared
1660
+ // during the trigger window is positive evidence the edit went
1661
+ // through and the build completed. This never re-issues the
1662
+ // edit and never stamps provenance — it only routes to the
1663
+ // parent's independent content read-back, which remains the
1664
+ // real verification.
1665
+ var newAuditDirs = [];
1666
+ try {
1667
+ var auditAfter = await agent(
1668
+ "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
1669
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1670
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1671
+ { key: attemptKey("publish-audit-after-" + taskId, totalReworkCount), label: "Re-listing audit dirs after trigger failure",
1672
+ schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1673
+ );
1674
+ var auditDirsAfterTrigger = String((auditAfter && auditAfter.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1675
+ // Only timestamped build dirs count — the "latest" symlink
1676
+ // and anything else are not builds.
1677
+ newAuditDirs = auditDirsAfterTrigger.filter(function (d) {
1678
+ return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
1679
+ });
1680
+ } catch (auditAfterErr) {
1681
+ log("Publish audit-dir re-list after trigger failure failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
1682
+ }
1683
+ if (newAuditDirs.length > 0) {
1684
+ log("Publish rebuild trigger: new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and the build completed despite the structured-output failure. Skipping applied-report smoke-check; parent read-back is the verification.");
1685
+ rebuildTrigger = { edit_started: true, error: "", applied: null };
1686
+ rebuildReportMissing = true;
1687
+ rebuildAgentId = null;
1688
+ rebuildEvidenceNote = "edit confirmed via durable audit evidence after structured-output failure (new audit dir " + newAuditDirs[0] + "); builder applied-report missing";
1689
+ } else {
1638
1690
  // No build observed — but that proves nothing (a fast-completing
1639
1691
  // build can finish between polls, or the check itself failed). The
1640
1692
  // outcome is UNKNOWN. No retry: re-issuing the edit here duplicated
@@ -1650,6 +1702,7 @@ while (i < STEPS.length) {
1650
1702
  detail: "structured-output failure on rebuild trigger; build-state poll saw no build (or the check itself failed); edit may have been accepted as pending_init"
1651
1703
  }, totalReworkCount);
1652
1704
  return await parkTask("Publish outcome unknown: the rebuild trigger's child did not return JSON, and the follow-up build-state poll could not observe a build for slug " + PUBLISH_SLUG + ". The edit may have been accepted as pending_init, so no retry was issued — a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion before re-driving Publish. Fail-closed.");
1705
+ }
1653
1706
  }
1654
1707
  }
1655
1708
  if (!rebuildReportMissing && !rebuildTrigger.edit_started && rebuildTrigger.error === "artifact_tools missing after load") {
@@ -1719,7 +1772,7 @@ while (i < STEPS.length) {
1719
1772
  applied_report: publishAppliedObservation,
1720
1773
  outcome: "submitted",
1721
1774
  detail: rebuildReportMissing
1722
- ? "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing"
1775
+ ? (rebuildEvidenceNote || "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing")
1723
1776
  : "edit accepted; builder applied-report received"
1724
1777
  }, totalReworkCount);
1725
1778
  } else if (rebuildTrigger) {
@@ -1817,12 +1870,14 @@ while (i < STEPS.length) {
1817
1870
  // builder's applied-report is derived from the carried diff, so
1818
1871
  // verifyAppliedChanges above is circular — a fabricated report
1819
1872
  // passes by construction, and every phase went green on a hollow
1820
- // build. The stamp moves to the parent (docs/publish-verification.md):
1821
- // after an independent artifact_inspect read-back confirms the
1822
- // artifact's actual content matches the merged diff, the parent
1823
- // stamps provenance and re-queues; QA's provenance check then
1824
- // enforces the stamp mechanically, so an unverified publish fails
1825
- // loudly in QA instead of passing silently here.
1873
+ // build. The stamp moves to the parent (docs/publish-verification.md);
1874
+ // the independent read-back step is currently unavailable (no
1875
+ // agent-callable read-back tool exists — artifact_inspect was
1876
+ // removed by the platform 2026-09-14), so the parent cannot
1877
+ // confirm content and the task parks for verification.
1878
+ // QA's provenance check enforces the stamp mechanically.
1879
+ // An unverified publish fails loudly in QA instead of passing
1880
+ // silently here.
1826
1881
  publishBuildLanded = true;
1827
1882
  artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
1828
1883
  log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
@@ -1920,32 +1975,40 @@ while (i < STEPS.length) {
1920
1975
  "Public docs (API.md, README, published action schemas) are NOT source code — read them freely, exactly as a user would.\n" +
1921
1976
  "DOCS GATE: If the fix is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n" +
1922
1977
  "To test, run in shell and read the stdout JSON:\n" + crewCmd("get-state", {}) + "\nThis returns current sessions, events, and tasks.\n" +
1923
- "Verify the fix by checking that session notes in the returned data now handle newlines correctly.\n" +
1978
+ "Verify the fix against the task description in the returned data: the state the bug corrupted should now read correctly, and the task's own expected behavior should hold.\n" +
1924
1979
  "You can also check specific data with: node " + CREW_API + " --crew-home " + crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
1925
- "Do NOT use artifact_inspect — it is async and will not return results inline.\n" +
1980
+ "Do NOT use artifact_inspect — it was removed by the platform (2026-09-14) and does not exist; do not substitute artifact.inspect (malfunction diagnosis, not an inspection tool).\n" +
1926
1981
  "File follow-up tasks by running in shell:\n" + crewCmd("create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n" +
1927
1982
  npmPublishCheck +
1928
1983
  "Report your test results as plain prose.\n" +
1929
1984
  "End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails.";
1930
- // Visual verdict ownership (experiential artifact tasks only): the QA
1931
- // work agent covers the MECHANICAL CHECKS below — it does not call
1932
- // artifact_inspect (async; the parent triggers the post-change capture
1933
- // after this step). The visual verdict is produced by the parent after
1934
- // the rendered post-change inspection results arrive.
1985
+ // qaVisual: Hazel runs the experiential see-act loop herself (STEP 1
1986
+ // below) and owns the visual verdict — no parent capture protocol.
1935
1987
  if (qaVisual) {
1936
1988
  instructions = "You are code-blind QA. You NEVER read source files. Public docs are not source — read them as a user would.\n" +
1937
- "VISUAL VERDICT OWNERSHIP: this task is experiential. The visual verdict is NOT yours to issue — it is produced after the rendered post-change inspection results arrive, by Hazel with the baseline and post-change evidence in hand.\n" +
1938
- "Your VERDICT below covers the MECHANICAL CHECKS only. Do NOT call artifact_inspect.\n\n" +
1989
+ "STEP 1: Experiential visual inspection — drive the fixed artifact as a user would, one browser step at a time, and verify the reported bug is actually fixed.\n" +
1990
+ "You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
1991
+ "Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
1992
+ "a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and continue with the mechanical checks.\n" +
1993
+ "b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
1994
+ "c. Bounded see-act loop, at most 8 steps: re-run the reproduction steps for the reported bug — does it still occur? Then check the surrounding views for regressions. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
1995
+ "c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
1996
+ "c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
1997
+ "c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
1998
+ "d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the fix needs such a sequence to verify, report NOT POSSIBLE for that part and judge what you can.\n" +
1999
+ "e. Judge as a user against the task description: is the reported bug fixed AND is nothing else visibly broken? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
2000
+ "f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
2001
+ "Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
1939
2002
  "MECHANICAL CHECKS:\n" +
1940
2003
  "DOCS GATE: If the fix is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n" +
1941
2004
  "To test, run in shell and read the stdout JSON:\n" + crewCmd("get-state", {}) + "\nThis returns current sessions, events, and tasks.\n" +
1942
- "Verify the fix by checking that session notes in the returned data now handle newlines correctly.\n" +
2005
+ "Verify the fix against the task description in the returned data: the state the bug corrupted should now read correctly, and the task's own expected behavior should hold.\n" +
1943
2006
  "You can also check specific data with: node " + CREW_API + " --crew-home " + crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
1944
2007
  npmPublishCheck +
1945
2008
  "File follow-up tasks by running in shell:\n" + crewCmd("create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n\n" +
1946
2009
  "BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
1947
2010
  "Report your test results as plain prose.\n" +
1948
- "End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails on the mechanical checks.";
2011
+ "End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails on the visual or the mechanical checks. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the reported bug, fixed>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten).";
1949
2012
  }
1950
2013
  if (PUBLISH_TYPE === "artifact") {
1951
2014
  instructions = "PROVENANCE CHECK (this project publishes to a dashboard artifact).\n" +
@@ -2272,9 +2335,10 @@ while (i < STEPS.length) {
2272
2335
  // it to HEAD: that verifies the stamp, not the content. Canary run 8
2273
2336
  // (2026-09-11) passed it with a hollow build — the stamp was honest, the
2274
2337
  // artifact was stale, all eight phases green. The stamp now moves to the
2275
- // parent: trigger an independent artifact_inspect read-back of the changed
2276
- // regions here; the parent stamps provenance only after mechanically
2277
- // confirming the artifact's actual content matches the merged diff. QA's
2338
+ // parent (docs/publish-verification.md); the independent read-back step
2339
+ // is currently unavailable (no agent-callable read-back tool exists —
2340
+ // artifact_inspect was removed by the platform 2026-09-14), so the parent
2341
+ // cannot confirm content and the task parks for verification. QA's
2278
2342
  // provenance check enforces the stamp — an unverified publish fails loudly
2279
2343
  // there instead of passing silently here.
2280
2344
  // Skip-aware (park 2026-09-11): an empty-diff Integrate takes no merge
@@ -2282,38 +2346,13 @@ while (i < STEPS.length) {
2282
2346
  // there is no new content to verify, so verification is vacuous.
2283
2347
  // publishSkippedNoLock is workflow-computed state from the explicit
2284
2348
  // lock-status read in STEP 0, not agent prose.
2285
- var publishVerifyInspect = { triggered: false, inspection_id: "", error: "" };
2286
- if (step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG) {
2287
- if (publishSkippedNoLock) {
2288
- log("Publish skipped for task " + taskId + " (no merge lock held — empty-diff Integrate): content verification vacuous, nothing was shipped");
2289
- } else if (!publishBuildLanded) {
2290
- log("Publish build did not land for task " + taskId + " — no content to verify (the failure park above already fired)");
2291
- } else {
2292
- try {
2293
- var inspectResult = await agent(
2294
- ARTIFACT_LOAD_PREAMBLE +
2295
- "Call artifact_inspect with slug \"" + PUBLISH_SLUG + "\", repair_authorized false, and verbatim_request exactly as follows:\n" +
2296
- "<<<READBACK_REQUEST\n" + buildPublishReadbackRequest(taskId, mergeCommitForPublish, mergeDiff, rebuildAgentId) + "\nREADBACK_REQUEST\n" +
2297
- "If artifact_inspect is still not available after the load, do NOT improvise — return { \"triggered\": false, \"inspection_id\": \"\", \"error\": \"artifact_tools missing after load\" } and nothing else.\n" +
2298
- "Return JSON { \"triggered\": <true if the inspection started, false otherwise>, \"inspection_id\": \"<the inspection id, or empty string>\", \"error\": \"<details or empty string>\" } and nothing else.",
2299
- { key: attemptKey("publish-verify-inspect-" + taskId, totalReworkCount), label: "Triggering publish content read-back",
2300
- schema: { type: "object", properties: { triggered: { type: "boolean" }, inspection_id: { type: "string" }, error: { type: "string" } }, required: ["triggered"] } }
2301
- );
2302
- publishVerifyInspect.triggered = !!(inspectResult && inspectResult.triggered);
2303
- publishVerifyInspect.inspection_id = (inspectResult && inspectResult.inspection_id) || "";
2304
- publishVerifyInspect.error = (inspectResult && inspectResult.error) || "";
2305
- if (publishVerifyInspect.triggered) {
2306
- log("Publish content read-back inspection triggered for task " + taskId + ": " + publishVerifyInspect.inspection_id);
2307
- } else {
2308
- log("Publish content read-back inspect trigger failed for task " + taskId + ": " + (publishVerifyInspect.error || "not started") + " — the park below asks the parent to trigger it manually");
2309
- }
2310
- } catch (e) {
2311
- publishVerifyInspect.error = (e && e.message ? e.message : String(e)).slice(0, 200);
2312
- log("Publish content read-back inspect trigger threw for task " + taskId + ": " + publishVerifyInspect.error + " — the park below asks the parent to trigger it manually");
2313
- }
2314
- } // end: !publishSkippedNoLock && publishBuildLanded — a skipped or failed publish has nothing to verify
2315
- }
2316
-
2349
+ // The parent (tick worker) triggers the ONE read-back inspection it can
2350
+ // actually receive (async results go to the root agent, never into a
2351
+ // workflow run — a workflow-side trigger would be an orphan). The workflow
2352
+ // only parks; the parent's scan builds the request deterministically via
2353
+ // lib/build-readback-request.js and ferries the inspection.
2354
+ // publishBuildLanded and publishSkippedNoLock are workflow-computed state;
2355
+ // a skipped or failed publish has nothing to verify.
2317
2356
  // Session notes. Machine-readable marker lines are extracted from the full
2318
2357
  // worker report and appended AFTER the slice so a long report can never
2319
2358
  // amputate them; later phases (Review reading repo_diff:, QA backstop
@@ -2332,14 +2371,12 @@ while (i < STEPS.length) {
2332
2371
  summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
2333
2372
  }
2334
2373
 
2335
- // Visual verdict evidence: for experiential artifact tasks, append the
2336
- // deterministic post-change capture plan to the QA session notes. The
2337
- // parent protocol (docs/visual-verdict.md) triggers the inspection with
2338
- // this plan; "visual: pending" marks the owed verdict.
2339
- if (step.name === "QA" && passed && qaVisual) {
2340
- summary = (summary + "\nvisual: pending\ncapture_plan: " +
2341
- buildVisualCapturePlan(taskTitle, taskDescription, "postchange", captureTargets).replace(/\s+/g, " ")).slice(0, 2900);
2342
- }
2374
+ // Visual verdict evidence (2026-09-15): no parent marker is appended. Hazel
2375
+ // records her own verdict.json + the append-only verdicts.jsonl in the
2376
+ // task-evidence dir during the QA loop itself — the QA session notes carry
2377
+ // her prose report, and the OODA report carries the machine-readable state.
2378
+ // The old parent-protocol marker for the QA closeout is deleted.
2379
+ // is deleted.
2343
2380
 
2344
2381
  // Capture mapper's spec for Build and Review
2345
2382
  if (step.name === "Map" && passed) {
@@ -2383,40 +2420,12 @@ while (i < STEPS.length) {
2383
2420
  return await parkTask("Reproduction failed — needs PM attention");
2384
2421
  }
2385
2422
 
2386
- // Visual-verdict gate: for experiential artifact tasks the task is done
2387
- // only when the parent has recorded a visual_verdict: note event. The QA
2388
- // work agent above covered the mechanical checks only. No recorded
2389
- // verdict → park for the parent protocol (never mark done on a pending
2390
- // visual verdict). A recorded FAIL with budget left → rework at Build.
2391
- if (step.name === "QA" && passed && qaVisual) {
2392
- var vvStatus = await visualVerdictStatus();
2393
- if (vvStatus.found && vvStatus.verdict === "PASS") {
2394
- log("Visual verdict PASS recorded for task " + taskId + (vvStatus.detail ? " — " + vvStatus.detail : ""));
2395
- } else if (vvStatus.found && vvStatus.verdict === "FAIL") {
2396
- // A FAIL whose reason begins exactly "rendering impossible:" is not
2397
- // reworkable — there is no rendered evidence to fix against. Park for
2398
- // human attention instead of bouncing to Build.
2399
- if (vvStatus.detail.indexOf("rendering impossible:") === 0) {
2400
- log("Visual verdict FAIL (rendering impossible) for task " + taskId + " — parking for human attention");
2401
- return await parkTask("Visual verdict FAIL — rendering impossible, human attention required: " + vvStatus.detail);
2402
- }
2403
- totalReworkCount++;
2404
- if (totalReworkCount > MAX_TOTAL_REWORK) {
2405
- log("Shared rework budget exhausted for task " + taskId + " — parking after visual verdict FAIL");
2406
- return await parkTask("Exceeded shared rework budget (" + MAX_TOTAL_REWORK + " total rework attempts across Review and QA) after visual verdict FAIL: " + vvStatus.detail);
2407
- }
2408
- rejectionNotes = "Visual verdict FAIL: " + vvStatus.detail;
2409
- i = BUILD_INDEX;
2410
- log("Visual verdict FAIL — bouncing to Build (rework #" + totalReworkCount + " of " + MAX_TOTAL_REWORK + ")");
2411
- continue;
2412
- } else {
2413
- if (!VISUAL_PROTOCOL_AVAILABLE) {
2414
- log("Visual verdict protocol not available (VISUAL_PROTOCOL_AVAILABLE=false) for task " + taskId + " — skipping visual gate, QA mechanical checks already passed");
2415
- } else {
2416
- return await parkTask("Visual verdict pending — parent: run the post-change capture + visual verdict protocol in docs/visual-verdict.md (capture plan is in the QA session notes)");
2417
- }
2418
- }
2419
- }
2423
+ // Visual verdict ownership (2026-09-15): Hazel owns the visual verdict
2424
+ // experientially — the QA instructions above have her drive the see-act
2425
+ // loop herself and record verdict.json + the append-only verdicts.jsonl.
2426
+ // Her prose VERDICT: line drives `passed` via extractVerdict; a FAIL
2427
+ // bounces to Build through the standard Review/QA rejection path below.
2428
+ // The old parent-recorded note gate is deleted.
2420
2429
 
2421
2430
  // Review/QA rejection bounces to Build
2422
2431
  if (!passed && (step.name === "Review" || step.name === "QA")) {
@@ -2457,10 +2466,7 @@ while (i < STEPS.length) {
2457
2466
  if (passed && step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG && !publishSkippedNoLock && publishBuildLanded) {
2458
2467
  return await parkTask("publish: verification-requested " + mergeCommitForPublish +
2459
2468
  " (build " + (rebuildAgentId || "agent_id unobserved") + ")" +
2460
- " — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md" +
2461
- (publishVerifyInspect.triggered
2462
- ? " (content read-back inspection " + publishVerifyInspect.inspection_id + " already triggered)."
2463
- : " (read-back inspect trigger failed: " + (publishVerifyInspect.error || "not started") + " — parent: trigger artifact_inspect manually)."));
2469
+ " — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md.");
2464
2470
  }
2465
2471
 
2466
2472
  i++;