muse-crew 0.13.2 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -28,26 +28,11 @@ const startStepIndex = inputs.start_step_index || 0;
28
28
  // resolution back via updatetask in the self-claim below.
29
29
  const RESOLVED_WORKFLOW = inputs.resolved_workflow || null;
30
30
  const WORKFLOW_WAS_NULL = inputs.workflow_was_null === true;
31
- // One-shot recovery routing: the dispatcher sets inputs.next_phase when it
32
- // routes this run via an explicit recover-task redirect. The value is
33
- // consumed (cleared) atomically by the successful self-claim below:
34
- // claim-task takes expected_next_phase and clears the matching next_phase in
35
- // the same transaction as the winning session insert, so no platform death
36
- // can slip between claim and consumption and replay the routing. A stale or
37
- // superseded routing survives — only an exact match clears.
38
- // what the dispatcher routed on.
31
+ // See docs/decisions/workflow-core.md#oneshot-recovery: the dispatcher sets input for one-shot recovery routing.
39
32
  const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_phase.length > 0) ? inputs.next_phase : null;
40
33
  const CLAIM_WORKFLOW_PERSIST = (WORKFLOW_WAS_NULL && RESOLVED_WORKFLOW) ? ", \"workflow\": \"" + RESOLVED_WORKFLOW + "\"" : "";
41
34
 
42
- // Visual verdict protocol availability — the workflow parks for parent-run
43
- // baseline capture and visual verdict ONLY when the protocol is fully
44
- // shipped. The protocol requires docs/visual-verdict.md in the release AND
45
- // the parent-side capture tooling (task b309a97d, "QA owns the visual
46
- // verdict"). Until both exist, the parks would deadlock waiting for a
47
- // parent who cannot fulfill them.
48
- // Effective value for this run, resolved by the dispatcher from the
49
- // project's visual_protocol setting (null=inherits crew default=off).
50
- // Manual launches without the arg default to off (previous behavior).
35
+ // See docs/decisions/qa-reproduce.md#visual-protocol-avail: the workflow parks if the visual protocol is unavailable.
51
36
  var VISUAL_PROTOCOL_AVAILABLE = inputs.visual_protocol === true;
52
37
 
53
38
  // crewHome is required — the dispatcher always passes it (crew-dispatch.js
@@ -85,15 +70,7 @@ const COMPUTE_DIFF_SRC = crewHome + "/current/lib/compute-publish-diff.js";
85
70
  const COMPUTE_DIFF = RUN_LIB + "/compute-publish-diff.js";
86
71
  const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
87
72
  const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
88
- // The seven basenames the pin step must materialize — asserted mechanically
89
- // by workflow code from the verbatim listing, never from agent prose.
90
- // COMPUTE_DIFF is the deterministic publish-diff computer (room #14,
91
- // 2026-09-17): the diff is computed by this script, never ferried as an
92
- // agent JSON string. Pinned like the other publish-critical modules so a
93
- // mid-run release swap cannot change it under the workflow.
94
- // CLASSIFY_SURFACE is the surface classifier (room #15, 2026-09-18):
95
- // crew-api.js statically imports it, so the pin must carry it — a pin
96
- // without it kills every claim with ERR_MODULE_NOT_FOUND.
73
+ // See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
97
74
  const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE].map(function (p) { return p.split("/").pop(); });
98
75
 
99
76
  // Project config — passed by dispatcher, falls back to dashboard defaults
@@ -126,14 +103,7 @@ const WORKTREE_HINT = REPO_PATH + "/.worktrees/" + taskId;
126
103
  const WORKTREE_PRESERVED_HINT = ".worktrees/" + taskId;
127
104
 
128
105
  const PUBLISH_TYPE = projectConfig.deploy_type || "";
129
- // User-facing surface for experiential QA routing: 'artifact' (a rendered
130
- // web UI Hazel drives with the see-act browser loop) | 'terminal' (a CLI
131
- // Hazel drives herself, keeping attempt-scoped transcripts) | null
132
- // (unclassified — no experiential QA). environment_type is the canonical
133
- // UX-surface axis; deploy_type names the deployment target, but
134
- // deploy_type === "artifact" remains a legacy artifact-surface signal so
135
- // pre-field projects keep today's experiential QA (the migration does not
136
- // backfill the column).
106
+ // See docs/decisions/qa-reproduce.md#surface-routing: the user-facing surface determines experiential QA routing.
137
107
  const ENV_TYPE = projectConfig.environment_type || null;
138
108
  // Surface resolution: artifact wins on contradictory config (the deployed
139
109
  // artifact is what users see). Unclassified surface => Capture skips, QA
@@ -148,12 +118,7 @@ const SURFACE_TRIAGE_DESC = SURFACE_ARTIFACT
148
118
  : SURFACE_TERMINAL
149
119
  ? "This project's user-facing surface is terminal: a command-line interface."
150
120
  : "This project's user-facing surface is unclassified (environment_type not set): judge by what a user would directly observe.";
151
- // UX doctrine page: the shared UX bar for this run's surface, resolved
152
- // mechanically — every phase prompt reads UX_DOCTRINE_PATH, never a
153
- // hardcoded filename. Canonical map: lib/ux-doctrine.js (mirrored here as a
154
- // one-liner because the workflow runtime's relative-import support is
155
- // unverified; tests pin the mirror). Null on unclassified surfaces: no
156
- // shared page, and prompts say so instead of naming the wrong one.
121
+ // See docs/decisions/qa-reproduce.md#ux-doctrine-page: the shared UX bar for this run's surface.
157
122
  const UX_DOCTRINE_PAGE = SURFACE_TERMINAL ? "terminal-ux.md" : (SURFACE_ARTIFACT ? "artifact-ux.md" : null);
158
123
  const UX_DOCTRINE_PATH = UX_DOCTRINE_PAGE ? crewHome + "/current/docs/" + UX_DOCTRINE_PAGE : null;
159
124
  const PUBLISH_SLUG = projectConfig.deploy_slug || "";
@@ -164,21 +129,7 @@ if (!taskId) {
164
129
  throw new Error("task_id is required in args");
165
130
  }
166
131
 
167
- // Closeout is deterministic: the work agent returns the runtime's native
168
- // transport envelope {"status": "ok", "result": "<prose report>"} with no
169
- // schema, so the workflow receives the report as a plain string. There is no
170
- // {"report"} wrapper: that invented shape invited agents to improvise sibling
171
- // keys (notably "status"), which the runtime duck-types as its own envelope
172
- // and fatally misparses. The envelope is the runtime's own documented shape
173
- // — not a demand for machine-structured reasoning.
174
- // The verdict is extracted mechanically by extractVerdict below — never by an
175
- // agent. The summary is the worker's report truncated. The release decision
176
- // comes from extractReleaseDecision. No formatter agent: it added a failure
177
- // mode while contributing nothing the workflow doesn't compute itself.
178
- // Steps whose passed=false drives a control-flow branch (rework bounce,
179
- // block) declare their verdict explicitly on a VERDICT: line. The verdict is
180
- // extracted DETERMINISTICALLY by workflow code (extractVerdict) — never by
181
- // an agent. Missing, malformed, or contradictory lines fail the phase (never silently pass).
132
+ // See docs/decisions/workflow-core.md#closeout-envelope: the work agent returns the runtime's native envelope; verdict extracted mechanically.
182
133
  const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
183
134
  function extractVerdict(workerText) {
184
135
  // The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
@@ -202,16 +153,7 @@ function extractVerdict(workerText) {
202
153
  if (uniq.length !== 1) return { ok: false, count: matches.length };
203
154
  return { ok: true, passed: last.value === "PASS" };
204
155
  }
205
- // Verdict re-ask (bug cd18ccc2): a verdict-step report that fails
206
- // extractVerdict is not failed immediately. Stochastic verdict-line
207
- // non-compliance (the agent did the work but omitted or garbled the VERDICT
208
- // line) gets up to two bounded follow-up agent() calls whose only job is to
209
- // read the preserved report and emit exactly one VERDICT line. The verdict
210
- // is still extracted mechanically by extractVerdict — the re-ask agent
211
- // transcribes, never decides the phase outcome. Each attempt uses a fresh
212
- // stable-key suffix so a cached failure can never replay deterministically.
213
- // Exhaustion keeps the existing fail-closed behavior. This is structure, not
214
- // prompt hardening: no instruction text was stern-ified to get here.
156
+ // See docs/decisions/workflow-core.md#verdict-reask: a report that fails extractVerdict gets bounded re-ask calls.
215
157
  function verdictReaskKey(stepName, reworkSuffix, attempt) {
216
158
  return "verdict-reask-" + stepName + reworkSuffix + "-a" + attempt;
217
159
  }
@@ -269,12 +211,7 @@ function workRetryKey(stepName, reworkSuffix, attempt) {
269
211
  function attemptKey(base, reworkCount) {
270
212
  return base + (reworkCount > 0 ? "-r" + reworkCount : "");
271
213
  }
272
- // pinLifecycle(key) — snapshot the lifecycle scripts into RUN_LIB and return
273
- // the verbatim `ls -1` listing so WORKFLOW CODE asserts the six pinned
274
- // basenames; the agent cannot self-certify. (The pin step was the one place
275
- // the workflows trusted agent prose: task 24be1cd6 walked to Publish on an
276
- // empty pin dir.) Byte-identical across standard/bugfix/chore — pinned by
277
- // tests/pin-location.test.js.
214
+ // See docs/decisions/qa-reproduce.md#pin-lifecycle: snapshot the lifecycle scripts in the pin.
278
215
  function pinLifecycle(key) {
279
216
  return agent(
280
217
  "Snapshot lifecycle scripts for version pinning.\n" +
@@ -478,24 +415,7 @@ function hydrateReleaseDecision(rec) {
478
415
  // read-back cannot be
479
416
  // fabricated from the diff; it must match the artifact's real content.
480
417
 
481
- // Durable publish-attempt ledger (2026-09-12): every artifact publish
482
- // attempt is recorded append-only at $CREW_HOME/.publish-ledger/<slug>.jsonl
483
- // on persistent disk (NOT /tmp). The ledger is the correlation record for
484
- // publish attempts whose outcome is UNKNOWN. When the rebuild trigger's
485
- // child returns prose instead of JSON (structured-output failure), the edit
486
- // may already have been accepted as pending_init — and artifact_status
487
- // cannot see pending_init (diagnostic canary 2026-09-12: an edit accepted
488
- // as pending_init was immediately followed by an all-false status check,
489
- // and the old retry issued a DUPLICATE edit). "No build visible" is NOT
490
- // evidence the edit did not go through, so the workflow never blind-retries
491
- // on an unknown outcome: it records the attempt and parks fail-closed. A
492
- // human or a later run correlates the accepted edit via the ledger (commit
493
- // hash + attempt key + the artifact build's agent_id when one was observed)
494
- // instead of guessing from a blind status poll.
495
- // Best-effort observability: a failed write is logged loudly but never
496
- // throws — the caller's park/proceed decision never depends on the ledger.
497
- // Byte-identical across standard/bugfix/chore — pinned by
498
- // tests/publish-ledger.test.js.
418
+ // See docs/decisions/publish-path.md#publish-attempt-ledger: every trigger outcome is recorded in the durable ledger.
499
419
  async function recordPublishLedger(entry, rework) {
500
420
  try {
501
421
  var ledgerDir = crewHome + "/.publish-ledger";
@@ -508,6 +428,7 @@ async function recordPublishLedger(entry, rework) {
508
428
  attempt: entry.attempt || null,
509
429
  agent_id: entry.agent_id || null,
510
430
  applied_report: entry.applied_report || null,
431
+ manifest_before: entry.manifest_before || null,
511
432
  outcome: entry.outcome,
512
433
  detail: entry.detail || ""
513
434
  });
@@ -545,45 +466,19 @@ function extractMarkerLines(workerText) {
545
466
  return markers.join("\n");
546
467
  }
547
468
 
548
- // Already-merged idempotency (canary 2026-09-15, task 1d692d91): when the
549
- // builder correctly makes no commit because the deliverable is already on
550
- // main (a prior merge or hand-repair landed it), it declares
551
- // `repo_diff: none (already-merged: <sha>)` naming the main commit that
552
- // carries the work. Room #16 blocker 11 (2026-09-18): the line anchor
553
- // missed Wren's mid-paragraph declaration, and the persisted notes truncated
554
- // the tail — so the anchor is gone and a sha followed by `)`, whitespace, or
555
- // end-of-string (truncation) is accepted. The sha is hex-only (7-40 chars)
556
- // so the workflow can interpolate it into the mechanical ancestor check
557
- // without injection risk; a over-long hex run never matches (the lookahead
558
- // fails on the extra hex char). Pure — pinned byte-identical across
559
- // standard/bugfix/chore.
469
+ // See docs/decisions/publish-path.md#already-merged-idem2: idempotency for already-merged tasks.
560
470
  function extractAlreadyMerged(workerText) {
561
471
  var m = /repo_diff:\s*none\s*\(already-merged:\s*([0-9a-f]{7,40})(?=[\s)]|$)/i.exec(workerText || "");
562
472
  return m ? { sha: m[1].toLowerCase() } : { sha: null };
563
473
  }
564
474
 
565
- // Explicit artifact refusal (room #16 blocker 10, 2026-09-18): the rebuild
566
- // trigger child ends its turn with `ARTIFACT_EDIT_REFUSED: <text>` when
567
- // artifact_edit explicitly refuses the edit (e.g. the artifact does not
568
- // exist). A refusal is conclusive negative evidence — the edit provably did
569
- // NOT go through — distinct from an unconsumed trigger return (unknown).
570
- // Pure — pinned byte-identical across standard/bugfix/chore.
571
- // The signal must be the ENTIRE trimmed turn output (not a line within prose):
572
- // the trigger child is instructed to end its turn with exactly this line and
573
- // nothing else. A confused child quoting the instructions back in prose must
574
- // NOT produce a conclusive negative — that degrades to unknown (fail-closed).
475
+ // See docs/decisions/publish-path.md#refusal-signal: the refusal signal must be the ENTIRE trimmed turn output.
575
476
  function extractRefusal(workerText) {
576
477
  var m = /^ARTIFACT_EDIT_REFUSED:\s*(.+?)\s*$/.exec(String(workerText || "").trim());
577
478
  return m ? m[1].slice(0, 300) : null;
578
479
  }
579
480
 
580
- // Worktree confinement: the Build agent must declare the exact worktree
581
- // path it built in on a `worktree:` marker line. The workflow compares it
582
- // against WORKTREE_HINT mechanically (exact string match) — never by
583
- // reading agent prose. This closes the hole where a builder whose prepare
584
- // failed freelanced into a different checkout (canary, 2026-09-11): the
585
- // honest-but-confused case fails here, and a fabricated path is caught one
586
- // phase later when Review's inspect finds no commits in the configured repo.
481
+ // See docs/decisions/qa-reproduce.md#worktree-confinement: the Build agent must declare its worktree.
587
482
  function extractWorktree(workerText) {
588
483
  var lines = (workerText || "").split("\n");
589
484
  var found = null;
@@ -623,26 +518,7 @@ function extractLayer(workerText) {
623
518
  if (!r) return null;
624
519
  return r[1].toLowerCase();
625
520
  }
626
- function buildVisualCapturePlan(taskTitle, taskDescription, kind, captureTargets) {
627
- // Deterministic visual-capture frame. kind: "baseline" | "postchange".
628
- // This string IS the capture script: fixed viewport matrix, scroll
629
- // positions, and interaction states — the inspection agent executes it
630
- // verbatim, nothing is improvised. Task-specific targets fill the slots.
631
- var title = String(taskTitle || "").replace(/"/g, "'").slice(0, 120);
632
- var targets = String(captureTargets || "").trim() ||
633
- String(taskDescription || "").replace(/"/g, "'").slice(0, 300);
634
- return "VISUAL CAPTURE — " + kind.toUpperCase() + " — task: " + title + ". " +
635
- "Target views/controls: " + targets + ". " +
636
- "For EACH target, capture exactly: " +
637
- "(1) desktop 1440x900, full view, scrolled to top; " +
638
- "(2) desktop 1440x900, scrolled so the target is vertically centered; " +
639
- "(3) mobile 390x844, scrolled so the target is vertically centered; " +
640
- "(4) desktop 1440x900, hover state on the target control; " +
641
- "(5) desktop 1440x900, keyboard-focus state on the target control; " +
642
- "(6) desktop 1440x900, active/pressed state if the target is a button or control. " +
643
- "Also record: console error count, the ARIA tree of the target region, any horizontal overflow. " +
644
- "Name captures " + kind + "-<n>-<viewport>-<state>. Return the captures, not a summary.";
645
- }
521
+
646
522
  // Experiential flag resolution: the task is experiential when Sage's Triage
647
523
  // report ends with the machine-read marker "experiential: yes". The flag is
648
524
  // opt-in — a missing or garbled line degrades to "unknown", which callers
@@ -680,12 +556,7 @@ async function resolveExperiential() {
680
556
  else experientialResolved = "unknown";
681
557
  return experientialResolved;
682
558
  }
683
- // Layer resolution: the task's layer is "artifact", "engine", or "docs" —
684
- // the machine-read "layer:" marker Sage's Triage report ends with. Unknown
685
- // (missing/garbled line, failed lookup) degrades to "artifact": today's
686
- // single-strategy behavior, never a park. Mirrors resolveExperiential()
687
- // (same cache shape, same Triage-notes re-read); the layer flag is captured
688
- // at Triage closeout (bugLayer) so the common path needs no extra agent call.
559
+ // See docs/decisions/qa-reproduce.md#layer-resolution: the task's layer determines the QA surface.
689
560
  async function resolveLayer() {
690
561
  // Triage notes are immutable within a run: cache the resolved layer so
691
562
  // the dashboard lookup runs at most once per run.
@@ -762,14 +633,7 @@ async function baselineStatus() {
762
633
  return { baseline_found: false, baseline_kind: "", baseline_refs: "", requested_count: 0, evidence_count: 0 };
763
634
  }
764
635
  }
765
- // The parent visual-verdict reader was removed 2026-09-15: Hazel (the QA
766
- // work agent) now owns the visual verdict experientially — she drives the
767
- // now owns the visual verdict experientially — she drives the see-act loop
768
- // herself and records verdict.json + the append-only verdicts.jsonl. The old
769
- // see-act loop herself and records verdict.json + the append-only
770
- // verdicts.jsonl. The old parent note gate is obsolete and has been
771
- // deleted from the QA closeout below. (See docs/visual-verdict.md.)
772
- // from the QA closeout below. (See docs/visual-verdict.md.)
636
+ // See docs/decisions/qa-reproduce.md#parent-reader-removed: the parent visual-verdict reader was removed; Hazel owns visual QA.
773
637
 
774
638
  // STEPS inline — export const meta is parsed as metadata, not a runtime binding
775
639
  const STEPS = [
@@ -839,23 +703,9 @@ function releaseDecisionText() {
839
703
  if (!releaseDecision) return "no machine-readable release decision from the Build report";
840
704
  return "release: " + releaseDecision.release + (releaseDecision.version_bump ? ", version_bump: " + releaseDecision.version_bump : " (no version_bump line)");
841
705
  }
842
- // Park the task for human attention and end the run. "blocked" is never
843
- // manually authored — the dashboard derives it mechanically from unmet
844
- // dependencies — so a workflow outcome that needs a human parks the task
845
- // instead. Parking is one atomic dashboard action (parktask): the parked
846
- // state and the explanatory note land in one transaction, never half.
847
- // The dispatcher skips parked tasks; a human moving parked→todo
848
- // mechanically resets the retry counters. Returns the workflow result
849
- // envelope the launcher sees. If the park call itself fails, the run
850
- // reports "failed" (retryable) so the next tick re-attempts the park —
851
- // a lost park is never reported as parked.
852
- // Terminal cleanup: the run's last act at every park/fail boundary. A run
853
- // that parks or fails must not leak its worktree, branch, or merge lock.
854
- // The lifecycle's terminal-cleanup releases the lock unconditionally and
855
- // reclaims the worktree+branch ONLY when the task branch is fully merged
856
- // into main (then it is redundant); unmerged work is preserved for the
857
- // human by design. Fire-and-forget with one bounded retry — the merge-lock
858
- // lease expiry and the orphan sweep are the backstop for a dead transport.
706
+ // Terminal cleanup: the run's last act — release the lock unconditionally;
707
+ // reclaim worktree+branch only when fully merged (unmerged work is preserved
708
+ // for the human by design). See `docs/decisions/workflow-core.md#park-contract`.
859
709
  async function terminalCleanup() {
860
710
  for (var attempt = 1; attempt <= 2; attempt++) {
861
711
  try {
@@ -1106,14 +956,7 @@ while (i < STEPS.length) {
1106
956
  lockHolder = taskId + "/" + activeSessionId;
1107
957
  }
1108
958
 
1109
- // ── Capture: baseline evidence for experiential tasks ─────────────
1110
- // Hazel's QA capture pass runs right after Triage, before Map, for tasks
1111
- // Sage flagged experiential. The capture itself is parent-driven (the
1112
- // inspection handoff arrives at the root agent, outside this script), so
1113
- // when no baseline evidence is recorded yet the script logs a note event
1114
- // and parks with the exact parent protocol + resume path. Never fails the
1115
- // task over missing evidence: after two requests, baseline:none is
1116
- // recorded and final QA judges on the rubric alone.
959
+ // See docs/decisions/qa-reproduce.md#capture-baseline: baseline evidence for experiential tasks is captured before work begins.
1117
960
  if (step.name === "Capture") {
1118
961
  var capExp = await resolveExperiential();
1119
962
  var bounceSuffix = (mapGateBounceCount > 0 ? "-g" + mapGateBounceCount : "");
@@ -1160,12 +1003,7 @@ while (i < STEPS.length) {
1160
1003
  continue;
1161
1004
  }
1162
1005
  var capStatus = await baselineStatus();
1163
- // Stale-decision guard: a "baseline: none (visual protocol unavailable)"
1164
- // note is only durable while the protocol is unavailable. When
1165
- // VISUAL_PROTOCOL_AVAILABLE is true, that old decision no longer
1166
- // stands — fall through to the request path for a fresh capture
1167
- // attempt. Exact-string trim comparison against the workflow's own
1168
- // written message (explicit state, never English matching).
1006
+ // See docs/decisions/qa-reproduce.md#stale-decision-guard: a stale baseline decision parks fail-closed.
1169
1007
  var baselineLatestMessage = ("baseline: " + capStatus.baseline_kind + capStatus.baseline_refs).trim();
1170
1008
  var baselineStale = VISUAL_PROTOCOL_AVAILABLE && baselineLatestMessage === "baseline: none (visual protocol unavailable)";
1171
1009
  if (capStatus.baseline_found && !baselineStale) {
@@ -1396,17 +1234,7 @@ while (i < STEPS.length) {
1396
1234
  (PUBLISH_TYPE === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
1397
1235
 
1398
1236
  } else if (step.name === "Review") {
1399
- // Already-merged hydration: when this run did not execute Build itself
1400
- // (dispatcher resume at Review after a platform death between phases),
1401
- // recover the workflow-verified sha. Room #16 blocker 11: the structured
1402
- // session field is read FIRST — the `already_merged_verified:` notes line
1403
- // is only a fallback, because session notes are hard-capped at 3000
1404
- // chars and a truthful declaration at the report's tail was silently
1405
- // truncated. The structured value was written by the workflow after a
1406
- // mechanical ancestor check — it is trusted; the builder's bare
1407
- // declaration never is. Absent both, the mechanical fact below reads
1408
- // "none declared" and Cass fails closed. The hydration read is best-effort:
1409
- // a transport throw degrades to "none declared" rather than crashing Review.
1237
+ // See docs/decisions/publish-path.md#already-merged-hydra: when this run did not execute, hydration uses the existing merge.
1410
1238
  if (!alreadyMergedSha) {
1411
1239
  var hydResult = null;
1412
1240
  try {
@@ -1514,25 +1342,7 @@ while (i < STEPS.length) {
1514
1342
  "published: muse-crew@" + publishTarget.target + "\n" +
1515
1343
  "VERDICT: PASS\n\n";
1516
1344
  } else if (PUBLISH_TYPE === "artifact") {
1517
- // Deterministic artifact publish (canary b5efd1b1, 2026-09-10): the work
1518
- // agent claimed "Rebuilt and deployed" while no build ran and no
1519
- // provenance was stamped — prose-trusted side effects, the same failure
1520
- // class as the npm double-skip (bb739316). The npm path already runs one
1521
- // deterministic script; the artifact path now has the same shape. Lock
1522
- // refresh, rebuild trigger, build-completion poll, and post-deploy are
1523
- // narrow schema'd bookkeeping calls owned by the workflow — the work
1524
- // agent reports on the mechanical outcome and cannot skip what it never
1525
- // owned. Any step failing parks with an honest, step-specific reason
1526
- // (fail-closed). There is deliberately NO workflow-side provenance
1527
- // stamp: the builder's applied-report is circular (canary run 8,
1528
- // 2026-09-11), so the stamp moved to the parent — after the build
1529
- // lands, the workflow records the session completed and parks with
1530
- // "publish: verification-requested". The parent owns verification
1531
- // (docs/publish-verification.md); the primary sensor is the
1532
- // deterministic lib/readback-disk.js (the agent-callable read-back
1533
- // tool is unavailable — artifact_inspect was removed by the platform
1534
- // 2026-09-14 — so the LLM-inspector path is manual-fallback only).
1535
- // QA's provenance check enforces the stamp mechanically.
1345
+ // See docs/decisions/publish-path.md#deterministic-artifact-publish: the work agent never publishes; the parent runs the deterministic publish script.
1536
1346
  var artifactPublish = null;
1537
1347
  var publishLockRefreshed = false;
1538
1348
  var publishSkippedNoLock = false;
@@ -1559,20 +1369,7 @@ while (i < STEPS.length) {
1559
1369
  }
1560
1370
  publishLockRefreshed = true;
1561
1371
  }
1562
- // STEP 1 (mechanical): carry the merged change to the artifact
1563
- // builder. The builder's source tree is NOT the crew's repo —
1564
- // canary run 4 (2026-09-11) proved it: Publish asked for "rebuild
1565
- // from current source. Do not modify any source files" and the
1566
- // builder rebuilt a stale copy predating the canary's changes, then
1567
- // the workflow stamped the new commit hash on the stale build.
1568
- // Provenance fiction; all eight phases passed. The merge diff is
1569
- // embedded in the edit request; the builder applies it to its own
1570
- // tree and reports the applied changes; the workflow verifies the
1571
- // report matches the diff BEFORE stamping provenance. A mismatch
1572
- // parks without stamping — the stamp must never certify a build
1573
- // whose content was not verified.
1574
- // Skipped entirely when no lock was held — nothing merged, nothing
1575
- // to ship.
1372
+ // See docs/decisions/publish-path.md#step1-builder-source: the builder's source tree is NOT the crew's repo; verify report before stamping.
1576
1373
  if (!publishSkippedNoLock) {
1577
1374
  // The trigger key of the attempt that last ran, for the publish ledger.
1578
1375
  // Minted once here (not re-minted per use site) so the ledger always
@@ -1653,20 +1450,7 @@ while (i < STEPS.length) {
1653
1450
  return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact target preflight was inconclusive (no parsable signal). Target existence is unproven, so no edit was issued and nothing was retried blindly. Human attention needed.");
1654
1451
  }
1655
1452
  log("Publish artifact preflight for task " + taskId + ": target " + artifactTargetDir + " present");
1656
- // (below) the diff computation, rebuild trigger, application
1657
- // verification, bounded poll, and provenance stamp. The builder
1658
- // only makes the artifact_edit call and reports the applied
1659
- // changes — no prose claim to trust. If the artifact tool namespace
1660
- // is missing from this child it reports honestly and the workflow
1661
- // retries once with a fresh key (bounded); anything else parks.
1662
- // Publish diff base (2026-09-14, task 0c53af4e): the carried diff is
1663
- // BASE..HEAD where BASE is the previously-stamped provenance
1664
- // source_commit — NOT HEAD^1. A push-time reconcile merge puts the
1665
- // task's own changes behind an intermediate merge, so HEAD^1..HEAD
1666
- // silently drops the task's fix while the artifact builds without
1667
- // it. The stamped base is the artifact's actual content; BASE..HEAD
1668
- // is the complete unpublished delta. Empty tree only for a genuine
1669
- // first publish (no provenance stamped yet).
1453
+ // See docs/decisions/publish-path.md#diff-computation: the diff is computed, the rebuild is triggered, and the report is verified.
1670
1454
  var EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
1671
1455
  var provResult = await agent(
1672
1456
  crewCmd("get-provenance", { project_id: LAUNCH_PROJECT_ID }) + "\n" +
@@ -1682,16 +1466,7 @@ while (i < STEPS.length) {
1682
1466
  } else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
1683
1467
  return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
1684
1468
  }
1685
- // Publish diff transport (room #14, 2026-09-17): the diff used to be
1686
- // ferried as a JSON string field in the agent's response — the agent
1687
- // produced a valid 700-line diff on disk but the JSON ferry dropped
1688
- // it, and the parse saw zero files ("Publish diff parsed to zero
1689
- // files"). The diff now travels git -> file -> deterministic script
1690
- // summary; the LLM never carries diff bytes. The agent is pure hands:
1691
- // it runs exactly one command (the PINNED compute-publish-diff.js —
1692
- // a mid-run release swap cannot change it under the workflow) and
1693
- // returns the small JSON summary verbatim. All fail-closed parks
1694
- // below are unchanged in meaning.
1469
+ // See docs/decisions/publish-path.md#diff-transport: the diff travels via file, not the LLM.
1695
1470
  var publishDiffFile = crewHome + "/.publish-diffs/" + taskId + (totalReworkCount > 0 ? "-r" + totalReworkCount : "") + ".diff";
1696
1471
  var diffResult = await agent(
1697
1472
  "Run exactly one command and nothing else:\n" +
@@ -1716,6 +1491,7 @@ while (i < STEPS.length) {
1716
1491
  return await parkTask("Publish diff base mismatch: script reported '" + String(diffSummary.base || "").slice(0, 12) + "' but the stamped base is '" + publishBase.slice(0, 12) + "'. Human attention needed.");
1717
1492
  }
1718
1493
  var mergeCommitForPublish = String(diffSummary.commit || "").trim();
1494
+ var mergeCommitShortForPublish = String(mergeCommitForPublish).substring(0, 7) || "unknown";
1719
1495
  var publishDiffSha256 = String(diffSummary.sha256 || "");
1720
1496
  if (!diffSummary.bytes) {
1721
1497
  return await parkTask("Publish diff is empty for commit " + (mergeCommitForPublish || "unknown") + " — a merge lock was held but there is no change to carry. Human attention needed.");
@@ -1767,83 +1543,44 @@ while (i < STEPS.length) {
1767
1543
  "- If artifact_edit is not available after the load, do NOT improvise — end your turn.\n" +
1768
1544
  "- You do NOT call setprovenance, artifact_inspect, or post-deploy yourself.\n" +
1769
1545
  "No report is needed: do not return JSON, do not summarize what you did, do not echo the diff. End your turn after the artifact_edit call.\n";
1770
- // The artifact build's agent_id, attributed to this edit by the
1771
- // workflow-owned observation below. The agent_id is the artifact
1772
- // system's in-flight correlation ID (research 2026-09-12):
1773
- // artifact.edit returns pending_init with NO agent_id, but
1774
- // artifact_status exposes build.agent_id immediately after
1775
- // acceptance, stable across polls. Recorded in the ledger so an
1776
- // attempt correlates to the exact builder run; null when no build
1777
- // was ever observed.
1546
+ // See docs/decisions/publish-path.md#agent-id-attribution: the artifact build's agent_id is attributed to the edit call.
1778
1547
  var rebuildAgentId = null;
1779
- // The builder's applied report is gone (2026-09-16): it rode on the
1780
- // trigger's JSON closeout contract, which is removed below. The
1781
- // parent's independent read-back (docs/publish-verification.md) is
1782
- // the verification — this field stays "missing-report" on ledger
1783
- // lines for issued triggers; pre-trigger parks (toolcheck
1784
- // rejected/inconclusive) and unattributed-unknown parks write null
1785
- // (no trigger was observed, so there is nothing to report).
1548
+ // See docs/decisions/publish-path.md#applied-report-gone: the builder's applied report is gone; the workflow verifies differently.
1786
1549
  var publishAppliedObservation = "missing-report";
1787
- // Durable-evidence snapshot (2026-09-14): the observation below only
1788
- // detects IN-FLIGHT builds. A build that finished before the
1789
- // observation leaves no in-flight trace — but the platform's audit
1790
- // harness leaves a durable one:
1791
- // ~/workspace/ts-spaces/<slug>/audits/<timestamp>-<id>/ per
1792
- // completed build. Snapshot the listing BEFORE the trigger so the
1793
- // fallback can diff before/after: a directory appearing during the
1794
- // trigger window is positive evidence the edit went through and
1795
- // the build completed. Best-effort and non-gating: if the snapshot
1796
- // fails, auditBeforeOk stays false and BOTH fallback comparisons
1797
- // are disabled (2026-09-16, critic finding 4) — without a baseline,
1798
- // an empty before-list would make every historical audit dir look
1799
- // "new". No wall-clock in-script (deterministic replay) — the
1800
- // comparison is a pure before/after set diff.
1550
+ // See docs/decisions/publish-path.md#durable-evidence-snapshot: snapshot the audit-dir listing BEFORE the trigger; fallback diffs before/after.
1801
1551
  var auditDirsBeforeTrigger = [];
1802
1552
  var auditBeforeOk = false;
1553
+ // Pre-trigger baselines (design §1.9): the audit-dir listing and the
1554
+ // manifest snapshot. The verified-path freshness check compares the
1555
+ // post-trigger manifest against the baseline (built_at advance +
1556
+ // content_sha256 change). Best-effort, never gates — a missing
1557
+ // manifest baseline fails the verified path closed.
1558
+ var preTriggerManifest = null;
1803
1559
  try {
1804
1560
  var auditBefore = await agent(
1805
- "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
1806
- "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1807
- "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1808
- { key: attemptKey("publish-audit-before-" + taskId, totalReworkCount), label: "Snapshotting audit dirs before rebuild trigger",
1809
- schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1561
+ "Capture pre-trigger baselines for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshots, never gates).\n" +
1562
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null; echo ---MANIFEST---; cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/.space-build/manifest.json 2>/dev/null\n" +
1563
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when missing>\", \"manifest\": \"<the manifest's full text, or empty string when missing/unreadable>\" } and nothing else.",
1564
+ { key: attemptKey("publish-baseline-before-" + taskId, totalReworkCount), label: "Snapshotting baselines before rebuild trigger",
1565
+ schema: { type: "object", properties: { dirs: { type: "string" }, manifest: { type: "string" } }, required: ["dirs", "manifest"] } }
1810
1566
  );
1811
1567
  auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1568
+ var manifestText = String((auditBefore && auditBefore.manifest) || "").trim();
1569
+ if (manifestText) {
1570
+ var manifestJson = JSON.parse(manifestText);
1571
+ preTriggerManifest = {
1572
+ built_at: typeof manifestJson.built_at === "string" ? manifestJson.built_at : null,
1573
+ content_sha256: typeof manifestJson.content_sha256 === "string" ? manifestJson.content_sha256 : null,
1574
+ };
1575
+ }
1576
+ // Arm only after both baselines parse.
1812
1577
  auditBeforeOk = true;
1813
- log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
1578
+ log("Publish pre-trigger baselines for task " + taskId + ": " + auditDirsBeforeTrigger.length + " audit dirs, manifest " + (preTriggerManifest ? "built_at=" + preTriggerManifest.built_at : "none"));
1814
1579
  } catch (auditBeforeErr) {
1815
- log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1580
+ log("Publish pre-trigger baselines failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1816
1581
  }
1817
- // Fire-and-forget trigger + workflow-owned observation (2026-09-16,
1818
- // clean-room task e2a8d9f8): the trigger's JSON closeout contract
1819
- // traveled over the stochastic text channel, and the runtime's
1820
- // JSON-candidate heuristic misfired on it ("workflow agent output
1821
- // was not JSON: no JSON object or array found in final response"),
1822
- // parking a task whose edit may have gone through. The contract's
1823
- // content was already observation-only (the applied report never
1824
- // gated; the pre_hashes were diagnostic-only), so the contract is
1825
- // removed: the trigger carries NO schema and its return value is
1826
- // never consumed, which takes the extraction heuristic out of this
1827
- // call entirely. The workflow attributes the edit itself through
1828
- // the tiny schema'd reads below — no prose is parsed for the
1829
- // trigger outcome.
1830
- // (Probe, 2026-09-16: the workflow scope exposes only agent() —
1831
- // tool_search, artifact_edit and artifact_status are undefined
1832
- // there — so the workflow cannot call the artifact tools directly;
1833
- // observation still goes through minimal child calls with tiny
1834
- // schemas, never a broad JSON contract.)
1835
- //
1836
- // Pre-trigger toolcheck (tiny, schema'd): the artifact namespace is
1837
- // deferred for workflow children — the child self-loads it and emits
1838
- // one exact signal line, read mechanically (never English prose).
1839
- // Only a parsed ARTIFACT_TOOLS: missing signal is explicit negative
1840
- // evidence: it gets one bounded retry with a fresh key, then parks
1841
- // rejected — without the tools the edit provably did NOT go through,
1842
- // so this is the one safe retry on the publish path. A throw (or an
1843
- // unparseable signal) is INCONCLUSIVE transport noise, never
1844
- // evidence of missing tools (2026-09-16, critic finding 3): it is
1845
- // recorded, it retries once in case the flake clears, but it can
1846
- // never take the rejected path.
1582
+
1583
+ // See docs/decisions/publish-path.md#fire-and-forget-trigger: the trigger child returns immediately; the workflow owns observation and verdict.
1847
1584
  var publishToolsOk = false;
1848
1585
  var publishToolsMissing = false;
1849
1586
  for (var toolcheckAttempt = 1; toolcheckAttempt <= 2 && !publishToolsOk; toolcheckAttempt++) {
@@ -1891,14 +1628,7 @@ while (i < STEPS.length) {
1891
1628
  }, totalReworkCount);
1892
1629
  return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was explicitly missing (parsed signal — the edit provably did not go through, so no trigger was issued and nothing was retried blindly). Human attention needed.");
1893
1630
  }
1894
- // Pre-trigger build-state baseline (tiny, schema'd): one read of
1895
- // artifact_status. The post-trigger observation diffs against this
1896
- // baseline — a build whose agent_id was absent from (or differs
1897
- // from) the baseline is attributed to our edit; a build already in
1898
- // flight at baseline predates the trigger and is never attributed
1899
- // to it. If the baseline read itself fails, receipt attribution is
1900
- // skipped and the durable audit-dir evidence below is the only
1901
- // positive signal.
1631
+ // See docs/decisions/publish-path.md#pretrigger-baseline: a tiny schema'd baseline is captured before the trigger.
1902
1632
  var baselineAgentId = null;
1903
1633
  var baselineFailed = false;
1904
1634
  try {
@@ -1915,32 +1645,13 @@ while (i < STEPS.length) {
1915
1645
  baselineFailed = true;
1916
1646
  log("Publish pre-trigger baseline read failed for task " + taskId + " (" + (baselineErr && baselineErr.message ? baselineErr.message : baselineErr) + ") — receipt attribution skipped; durable audit-dir evidence is the only positive signal");
1917
1647
  }
1918
- // The trigger itself: the artifact_edit call is AWAITED (the workflow
1919
- // waits for it to complete) but its return value is intentionally
1920
- // UNCONSUMED — NO schema, so no schema validation can fail this
1921
- // call: a schema-less call resolves to the child's raw response as
1922
- // a plain string (probed live 2026-09-16 — never parsed, never
1923
- // throws on content). One caveat, also probed: the runtime still
1924
- // scans the response for a JSON candidate, and an unparseable
1925
- // {...}-looking substring in the child's prose throws ("response
1926
- // JSON candidate", probe P6). The prompt tells the child to end its
1927
- // turn with no prose at all, which keeps the common case clean —
1928
- // but the channel is stochastic, so any throw is possible and
1929
- // inconclusive: the edit may still have gone through, so the
1930
- // outcome stays unknown until the observation below confirms it —
1931
- // never inferred from the throw, and never blind-retried (a blind
1932
- // re-trigger duplicated the edit on 2026-09-12).
1648
+ // See docs/decisions/publish-path.md#trigger-await: the artifact_edit call is awaited.
1933
1649
  var rebuildTrigger = null;
1934
1650
  try {
1935
1651
  var triggerText = String(await agent(rebuildPrompt,
1936
1652
  { key: rebuildAttemptKey, label: "Triggering artifact rebuild" }) || "");
1937
1653
  log("Publish rebuild trigger for task " + taskId + " returned (" + triggerText.length + " chars; awaited; scanned only for the explicit refusal signal)");
1938
- // Explicit refusal (room #16 blocker 10): the child ends its turn
1939
- // with ARTIFACT_EDIT_REFUSED when artifact_edit explicitly refused.
1940
- // Conclusive negative evidence — the edit provably did NOT go
1941
- // through — so this parks rejected and skips observation polling.
1942
- // A missing/unparseable signal is NOT a refusal: it stays unknown
1943
- // and fail-closed below.
1654
+ // See docs/decisions/publish-path.md#explicit-refusal-16: the child must explicitly refuse artifact work.
1944
1655
  var refusalText = extractRefusal(triggerText);
1945
1656
  if (refusalText) {
1946
1657
  await recordPublishLedger({
@@ -1987,17 +1698,79 @@ while (i < STEPS.length) {
1987
1698
  log("Publish post-trigger build-state check failed for task " + taskId + " (" + (buildCheckErr && buildCheckErr.message ? buildCheckErr.message : buildCheckErr) + ") — this signal is unknown, not negative");
1988
1699
  }
1989
1700
  var observedAgentId = (buildState && buildState.build && typeof buildState.build.agent_id === "string" && buildState.build.agent_id) || null;
1990
- // Known limitation (failure-mode audit 2026-09-16): attribution
1991
- // is timing-based — any agent_id new relative to the baseline is
1992
- // treated as this edit's receipt. A stranger's build starting inside
1993
- // the trigger window is indistinguishable by timing and would be
1994
- // misattributed here. The consequence is bounded: the completion
1995
- // poll below tracks the recorded id, and the parent's mechanical
1996
- // content read-back (docs/publish-verification.md) certifies the
1997
- // exact commit's content — a wrong build's content fails closed as
1998
- // verification-failed, never stamped. Timing narrows the candidate;
1999
- // content decides.
1701
+ // See docs/decisions/publish-path.md#attribution-limitation: attribution is timing-based; content verification bounds the risk.
2000
1702
  var receiptAgentId = (!buildStateFailed && !baselineFailed && observedAgentId && observedAgentId !== baselineAgentId) ? observedAgentId : null;
1703
+ // auditReportOk: pure tri-state read of a report.json body —
1704
+ // true (build ok), false (build failed), null (missing or
1705
+ // unreadable — not evidence either way). The child returns the
1706
+ // raw body verbatim; interpretation lives here, never in prose.
1707
+ // Hoisted to Publish-step scope (before the receipt branch) so both
1708
+ // the immediate and post-poll audit fallbacks share it on every path —
1709
+ // the receipt path skips the else below, which must not leave these
1710
+ // undefined.
1711
+ var auditReportOk = function (raw) {
1712
+ if (typeof raw !== "string") return null;
1713
+ var trimmed = raw.trim();
1714
+ if (trimmed === "" || trimmed === "MISSING") return null;
1715
+ var parsed;
1716
+ try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
1717
+ if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
1718
+ return null;
1719
+ };
1720
+ // (2026-09-18, H2 verdict-first) Publish-verdict vocabulary. The
1721
+ // verdict is one of "landed" | "unknown". Decided once,
1722
+ // before the receipt poll, and dispatched on — never re-derived.
1723
+ // Pure and self-contained: unit-tested by
1724
+ // tests/publish-verdict-first.test.js.
1725
+ // See docs/decisions/publish-path.md#h2-verdict-dispatch.
1726
+ var decidePublishVerdict = function (opts) {
1727
+ var immediateReport = (opts && "immediateReport" in opts) ? opts.immediateReport : null;
1728
+ var newDirCount = (opts && typeof opts.newDirCount === "number") ? opts.newDirCount : 0;
1729
+ if (newDirCount === 1 && immediateReport === true) return { verdict: "landed", unattributableReason: null };
1730
+ if (newDirCount === 1 && immediateReport === false) return { verdict: "unknown", unattributableReason: "audit-report-ok-false" };
1731
+ if (newDirCount === 1) return { verdict: "unknown", unattributableReason: "audit-report-unreadable-or-missing" };
1732
+ if (newDirCount === 0) return { verdict: "unknown", unattributableReason: "no-new-audit-dir-in-window" };
1733
+ return { verdict: "unknown", unattributableReason: "audit-dir-ambiguity", ambiguousDirCount: newDirCount };
1734
+ };
1735
+ // (2026-09-18, H2 message contract, design §1.5) The human line is
1736
+ // the output of a mechanical field→template mapping: the trigger
1737
+ // commit short-sha, one plain clause per unattributable reason, the
1738
+ // no-republish warning, the ledger path with (outcome: unknown), and
1739
+ // the unknown-recovery clause. The appendix carries every machine
1740
+ // field. Pure and self-contained: unit-tested by
1741
+ // tests/publish-verdict-first.test.js.
1742
+ var composeUnattributedParkReason = function (fields) {
1743
+ var f = fields || {};
1744
+ var task = f.taskId || taskId;
1745
+ var reason = f.unattributableReason || "unknown-outcome";
1746
+ var shortSha = String(f.commitShortSha || "").substring(0, 7) || "unknown";
1747
+ var reasonClauses = {
1748
+ "no-new-audit-dir-in-window": "no build could be tied to this attempt",
1749
+ "audit-report-unreadable-or-missing": "the build's report is unreadable",
1750
+ "audit-dir-ambiguity": "more than one build appeared in the check window",
1751
+ "stranger-build-observed-during-poll": "a different build was running during the check",
1752
+ "status-read-errors-during-poll": "status reads kept failing"
1753
+ };
1754
+ var clause = reasonClauses[reason] || "no build could be tied to this attempt";
1755
+ return "Publish outcome unknown for task " + task +
1756
+ ": the trigger was sent for commit " + shortSha + " but the outcome could not be confirmed — " + clause + ". " +
1757
+ "Do NOT republish: if the trigger was accepted, a retry duplicates the build (2026-09-12). " +
1758
+ "The attempt is in the ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (outcome: unknown) " +
1759
+ "and the crew's unknown-recovery will re-examine it. " +
1760
+ "Appendix: unattributable_reason=" + reason +
1761
+ "; poll_end_state=" + (f.pollEndState || "not-polled") +
1762
+ "; saw_our_build=" + (f.sawOurBuild ? "true" : "false") +
1763
+ "; new_audit_dirs=" + (f.newAuditDirCount == null ? 0 : f.newAuditDirCount) +
1764
+ "; poll_chunks_failed=" + (f.chunkFailures == null ? 0 : f.chunkFailures) +
1765
+ "; poll_status_errors=" + (f.pollStatusErrors == null ? 0 : f.pollStatusErrors) + ".";
1766
+ };
1767
+ // publishFailure is declared here (per-Publish-pass scope) so the
1768
+ // post-poll explicit build failure survives to the final routing
1769
+ // below; publishUnknownFields carries the structured unknown fields
1770
+ // for the single composed fail-closed park. Both re-initialize on
1771
+ // every pass — a rework re-entry never leaks a stale verdict.
1772
+ var publishFailure = null;
1773
+ var publishUnknownFields = null;
2001
1774
  if (receiptAgentId) {
2002
1775
  // The edit went through — a build with a new agent_id appeared
2003
1776
  // after the trigger. The parent's independent read-back
@@ -2011,36 +1784,12 @@ while (i < STEPS.length) {
2011
1784
  attempt: rebuildAttemptKey,
2012
1785
  agent_id: rebuildAgentId,
2013
1786
  applied_report: publishAppliedObservation,
1787
+ manifest_before: preTriggerManifest,
2014
1788
  outcome: "submitted",
2015
1789
  detail: "fire-and-forget trigger; build receipt captured by workflow-owned build-state observation (pre/post-trigger diff)"
2016
1790
  }, totalReworkCount);
2017
1791
  } else {
2018
1792
  var newAuditDirs = [];
2019
- // auditReportOk: pure tri-state read of a report.json body —
2020
- // true (build ok), false (build failed), null (missing or
2021
- // unreadable — not evidence either way). The child returns the
2022
- // raw body verbatim; interpretation lives here, never in prose.
2023
- // Defined here so both the immediate and post-poll audit
2024
- // fallbacks share it.
2025
- var auditReportOk = function (raw) {
2026
- if (typeof raw !== "string") return null;
2027
- var trimmed = raw.trim();
2028
- if (trimmed === "" || trimmed === "MISSING") return null;
2029
- var parsed;
2030
- try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
2031
- if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
2032
- return null;
2033
- };
2034
- // (2026-09-16, critic finding 2) When durable audit evidence
2035
- // confirms (or refutes) the build, there is no receipt agent_id
2036
- // to chain the completion poll to — skipReceiptPoll bypasses the
2037
- // poll below, which with a null receipt could only observe
2038
- // strangers or nothing.
2039
- var skipReceiptPoll = false;
2040
- // publishFailure is declared here (moved up from below) so the
2041
- // immediate audit fallback can record an explicit build failure
2042
- // without the later declaration resetting it.
2043
- var publishFailure = null;
2044
1793
  try {
2045
1794
  var auditAfter = await agent(
2046
1795
  "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
@@ -2060,88 +1809,110 @@ while (i < STEPS.length) {
2060
1809
  log("Publish audit-dir re-list after trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
2061
1810
  }
2062
1811
  if (newAuditDirs.length > 0) {
2063
- rebuildTrigger = { edit_started: true };
2064
1812
  rebuildAgentId = null;
2065
1813
  newAuditDirs.sort();
2066
1814
  var newestImmediateDir = newAuditDirs[newAuditDirs.length - 1];
2067
1815
  log("Publish rebuild trigger for task " + taskId + ": new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and a build completed; no in-flight receipt was observed.");
2068
- // (2026-09-16, critic finding 2) Durable audit evidence exists,
1816
+ // (2026-09-18, H2 verdict-first) Durable audit evidence exists,
2069
1817
  // but there is no receipt agent_id to chain the completion poll
2070
1818
  // to — polling with a null receipt can only observe strangers
2071
1819
  // (any running build differs from "null") or nothing, burning
2072
- // 10.5 minutes to park unknown. Read the build report now
2073
- // instead of polling: ok=true confirms completion and routes
2074
- // directly to parent verification (the poll is skipped);
2075
- // ok=false is explicit failure; unreadable is unknown.
1820
+ // 10.5 minutes to park unknown. Read the build report now instead
1821
+ // of polling, then decide the verdict ONCE via
1822
+ // decidePublishVerdict: exactly one new dir with ok=true lands
1823
+ // (attribution by window, not identity — never poll blind);
1824
+ // ok=false is UNKNOWN with the failure evidence preserved in the
1825
+ // ledger detail (the evidence is explicit, the attribution is
1826
+ // not); unreadable / zero / ambiguous dirs are UNKNOWN.
1827
+ // Verdict-first: landed bypasses the poll below, unknown falls
1828
+ // through to post-deploy. STEP 2 runs on every path.
1829
+ // See docs/decisions/publish-path.md#h2-verdict-dispatch.
2076
1830
  var immediateReportOk = null;
2077
- try {
2078
- var immediateOkRead = await agent(
2079
- "Read the artifact build report for slug \"" + PUBLISH_SLUG + "\".\n" +
2080
- "Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestImmediateDir + "/report.json 2>/dev/null || echo MISSING\n" +
2081
- "Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
2082
- { key: attemptKey("publish-audit-ok-immediate-" + taskId, totalReworkCount), label: "Reading build report for audit-confirmed build",
2083
- schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
2084
- );
2085
- immediateReportOk = auditReportOk(immediateOkRead && immediateOkRead.raw);
2086
- } catch (immediateOkErr) {
2087
- log("Publish build-report read for audit-confirmed dir failed for task " + taskId + " (treated as unknown): " + (immediateOkErr && immediateOkErr.message ? immediateOkErr.message : immediateOkErr));
2088
- immediateReportOk = null;
1831
+ // (N1) Read the report only when exactly one new dir exists:
1832
+ // ambiguity (>1) forces UNKNOWN regardless — don't shell out to
1833
+ // read a report that will be discarded.
1834
+ if (newAuditDirs.length === 1) {
1835
+ try {
1836
+ var immediateOkRead = await agent(
1837
+ "Read the artifact build report for slug \"" + PUBLISH_SLUG + "\".\n" +
1838
+ "Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestImmediateDir + "/report.json 2>/dev/null || echo MISSING\n" +
1839
+ "Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
1840
+ { key: attemptKey("publish-audit-ok-immediate-" + taskId, totalReworkCount), label: "Reading build report for audit-confirmed build",
1841
+ schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
1842
+ );
1843
+ immediateReportOk = auditReportOk(immediateOkRead && immediateOkRead.raw);
1844
+ } catch (immediateOkErr) {
1845
+ log("Publish build-report read for audit-confirmed dir failed for task " + taskId + " (treated as unknown): " + (immediateOkErr && immediateOkErr.message ? immediateOkErr.message : immediateOkErr));
1846
+ immediateReportOk = null;
1847
+ }
2089
1848
  }
2090
- if (immediateReportOk === true) {
1849
+ var immediateVerdict = decidePublishVerdict({ editStarted: true, immediateReport: immediateReportOk, newDirCount: newAuditDirs.length });
1850
+ if (immediateVerdict.verdict === "landed") {
2091
1851
  publishBuildLanded = true;
2092
1852
  artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
2093
- skipReceiptPoll = true;
2094
- log("Publish build landed for task " + taskId + " via immediate durable audit evidence (audit dir " + newestImmediateDir + ", report ok=true) — receipt poll skipped (no receipt to chain to), routing directly to parent verification");
2095
1853
  await recordPublishLedger({
2096
1854
  commit: mergeCommitForPublish,
2097
1855
  attempt: rebuildAttemptKey,
2098
1856
  agent_id: null,
2099
1857
  applied_report: publishAppliedObservation,
2100
- outcome: "submitted",
2101
- detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestImmediateDir + ", report ok=true); receipt poll skipped (no receipt agent_id), routed to parent verification"
2102
- }, totalReworkCount);
2103
- } else if (immediateReportOk === false) {
2104
- skipReceiptPoll = true;
2105
- publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestImmediateDir + ", report ok=false — immediate audit evidence, no receipt observed). Explicit negative evidence: a build ran and failed. The publish did not land — provenance was not stamped. Fail-closed.";
2106
- await recordPublishLedger({
2107
- commit: mergeCommitForPublish,
2108
- attempt: rebuildAttemptKey,
2109
- agent_id: null,
2110
- applied_report: publishAppliedObservation,
2111
- outcome: "failed",
2112
- detail: "a build ran and failed: audit dir " + newestImmediateDir + " report ok=false (immediate audit evidence, no receipt)"
1858
+ manifest_before: preTriggerManifest,
1859
+ outcome: "build-observed",
1860
+ detail: "durable audit evidence shows a build completed during the attempt window (no receipt agent_id — attribution by window, not identity; receipt poll bypassed (verdict decided), routed to parent verification)"
2113
1861
  }, totalReworkCount);
1862
+ log("Publish verdict LANDED for task " + taskId + ": a completed build was observed during the attempt window — receipt poll bypassed (verdict decided, no receipt to chain to), routing directly to parent verification.");
2114
1863
  } else {
1864
+ // Verdict UNKNOWN on the immediate path. ok=false is explicit
1865
+ // failure evidence but not an attributable failure — the ledger
1866
+ // records unknown with the evidence preserved in the detail,
1867
+ // and the flow continues to post-deploy (never parks early).
1868
+ publishUnknownFields = {
1869
+ taskId: taskId,
1870
+ unattributableReason: immediateVerdict.unattributableReason,
1871
+ pollEndState: "not-polled",
1872
+ sawOurBuild: false,
1873
+ newAuditDirCount: newAuditDirs.length,
1874
+ chunkFailures: 0,
1875
+ pollStatusErrors: 0
1876
+ };
1877
+ var immediateDetail = "durable audit-dir fallback could not prove a completed build for this attempt (unattributable_reason=" + immediateVerdict.unattributableReason + ", no receipt agent_id)";
1878
+ if (immediateReportOk === false) {
1879
+ immediateDetail += "; explicit failure evidence preserved: report ok=false for audit dir " + newestImmediateDir;
1880
+ }
1881
+ if (immediateVerdict.unattributableReason === "audit-dir-ambiguity") {
1882
+ immediateDetail += "; audit-dir ambiguity: " + newAuditDirs.length + " new dirs in window";
1883
+ }
2115
1884
  await recordPublishLedger({
2116
1885
  commit: mergeCommitForPublish,
2117
1886
  attempt: rebuildAttemptKey,
2118
1887
  agent_id: null,
2119
- applied_report: null,
1888
+ applied_report: publishAppliedObservation,
2120
1889
  outcome: "unknown",
2121
- detail: "new audit dir " + newestImmediateDir + " appeared during the trigger window but its build report is unreadable/missing; no receipt agent_id to poll — outcome unknown, fail-closed with no blind retry"
1890
+ detail: immediateDetail
2122
1891
  }, totalReworkCount);
2123
- return await parkTask("Publish outcome unknown for task " + taskId + ": a new audit dir (" + newestImmediateDir + ") appeared during the trigger window but its build report is unreadable, and no in-flight receipt was observed to poll. The edit may have completed. Correlate the accepted edit via the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl — do NOT reissue the edit blindly: if the trigger was accepted, a retry duplicates it (2026-09-12). Verify independently whether the build completed (audit dir + report, or the parent's content read-back) before deciding the next step. Fail-closed.");
1892
+ log("Publish verdict UNKNOWN for task " + taskId + ": " + immediateDetail + " — continuing to post-deploy; never polling blind and never parking early.");
2124
1893
  }
2125
1894
  } else {
2126
- // No attributable build and no durable evidence — but that
2127
- // proves nothing (a fast-completing build can finish between
2128
- // polls, or the checks themselves failed). The outcome is
2129
- // UNKNOWN. No retry: re-issuing the edit here duplicated it on
2130
- // 2026-09-12. Record the attempt durably and park fail-closed;
2131
- // correlate via the ledger, never by guessing from a blind poll.
2132
- log("Publish rebuild trigger for task " + taskId + ": no attributable build observed and no new audit dir — outcome UNKNOWN. Recording the attempt and parking fail-closed; no blind retry.");
1895
+ // See docs/decisions/publish-path.md#no-attributable-build: no attributable build and no durable evidence means no publish.
1896
+ publishUnknownFields = {
1897
+ taskId: taskId,
1898
+ unattributableReason: decidePublishVerdict({ editStarted: true, immediateReport: null, newDirCount: 0 }).unattributableReason,
1899
+ pollEndState: "not-polled",
1900
+ sawOurBuild: false,
1901
+ newAuditDirCount: 0,
1902
+ chunkFailures: 0,
1903
+ pollStatusErrors: 0
1904
+ };
2133
1905
  await recordPublishLedger({
2134
1906
  commit: mergeCommitForPublish,
2135
1907
  attempt: rebuildAttemptKey,
2136
1908
  agent_id: null,
2137
- applied_report: null,
1909
+ applied_report: publishAppliedObservation,
2138
1910
  outcome: "unknown",
2139
- detail: "fire-and-forget trigger; post-trigger build-state poll saw no attributable build (or the check failed) and the audit-dir diff found no new dir; the edit may have been accepted as pending_init"
1911
+ detail: "no new audit dir appeared in the trigger window and no receipt agent_id was observed — the edit was issued fire-and-forget, so completion is unproven; never poll blind on a null receipt"
2140
1912
  }, totalReworkCount);
2141
- return await parkTask("Publish outcome unknown for task " + taskId + ": the rebuild trigger was issued fire-and-forget (no schema, so no validation failure mode; a candidate-parse throw stays possible and is inconclusive), and the follow-up observation could not attribute a build to the edit for slug " + PUBLISH_SLUG + " — no in-flight build with a new agent_id appeared in the poll window and no new audit dir landed. The edit may have been accepted as pending_init, so no retry was issued: a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion — do NOT reissue the edit blindly. Verify independently whether the build completed before deciding the next step. Fail-closed.");
1913
+ log("Publish verdict UNKNOWN for task " + taskId + ": no new audit dir in window and no receipt — continuing to post-deploy; never polling blind and never parking early.");
2142
1914
  }
2143
1915
  }
2144
-
2145
1916
  // Durable publish-attempt ledger: record the trigger outcome while the
2146
1917
  // attempt key and commit are in scope. Every attempt lands here with
2147
1918
  // its outcome — submitted, rejected, or unknown (unknown is recorded
@@ -2151,9 +1922,45 @@ while (i < STEPS.length) {
2151
1922
  // already recorded the ledger's submitted line on both positive paths
2152
1923
  // and parked on unknown — there is no applied report to observe and
2153
1924
  // no rejection signal to record.
2154
- // (publishFailure is declared with the immediate audit fallback
2155
- // above so an explicit build failure there survives to here.)
2156
- if (rebuildTrigger.edit_started && !skipReceiptPoll) {
1925
+ // (2026-09-18, H2 verdict-first) The verdict was decided exactly
1926
+ // once above; dispatch on it. landed bypasses the receipt poll
1927
+ // (the audit evidence already proved completion); an explicit
1928
+ // publishFailure is preserved verbatim through post-deploy.
1929
+ // Otherwise the verdict is open — but the poll below is only
1930
+ // legitimate against a real receipt: the null-safe assertion records
1931
+ // UNKNOWN and continues to STEP 2 instead of polling blind.
1932
+ // See docs/decisions/publish-path.md#h2-verdict-dispatch.
1933
+ if (publishBuildLanded) {
1934
+ log("Publish verdict already LANDED for task " + taskId + " — bypassing receipt poll.");
1935
+ } else if (publishFailure) {
1936
+ // design §1.2 pre-poll branch — currently unassigned; kept for the converged dispatch shape.
1937
+ log("Publish verdict already FAILED for task " + taskId + " — preserved verbatim through post-deploy.");
1938
+ } else if (!rebuildTrigger || !rebuildTrigger.edit_started) {
1939
+ // Loud defensive assertion (replaces the old lying "Unreachable"
1940
+ // else): with no receipt state the poll would observe strangers or
1941
+ // nothing — record UNKNOWN and continue to STEP 2. Never park
1942
+ // early here; never poll blind.
1943
+ if (!publishUnknownFields) {
1944
+ publishUnknownFields = {
1945
+ taskId: taskId,
1946
+ unattributableReason: "no-receipt-state",
1947
+ pollEndState: "not-polled",
1948
+ sawOurBuild: false,
1949
+ newAuditDirCount: 0,
1950
+ chunkFailures: 0,
1951
+ pollStatusErrors: 0
1952
+ };
1953
+ await recordPublishLedger({
1954
+ commit: mergeCommitForPublish,
1955
+ attempt: rebuildAttemptKey,
1956
+ agent_id: null,
1957
+ applied_report: publishAppliedObservation,
1958
+ outcome: "unknown",
1959
+ detail: "no receipt state was recorded for this attempt — never polling blind; continuing to post-deploy"
1960
+ }, totalReworkCount);
1961
+ }
1962
+ log("Publish verdict UNKNOWN for task " + taskId + ": no receipt state recorded — never polling blind, continuing to post-deploy.");
1963
+ } else {
2157
1964
  // (2026-09-16) There is no builder report: the fire-and-forget
2158
1965
  // trigger carries no JSON contract, so there is nothing to
2159
1966
  // compare and no pre-hash diagnostic. The builder's old
@@ -2226,13 +2033,7 @@ while (i < STEPS.length) {
2226
2033
  timeoutMs: 270000 }
2227
2034
  );
2228
2035
  } catch (chunkErr) {
2229
- // A hung or failed chunk is inconclusive, never terminal:
2230
- // record it and continue to the next chunk. (2026-09-16,
2231
- // clean-room task 1febe8eb: the platform's 270s agent
2232
- // timeout killed chunk 2, which threw out of this loop —
2233
- // skipping chunk 3 AND the STEP 1b audit-dir fallback and
2234
- // parking on the exception path.) Fail-closed still applies
2235
- // after chunk 3 and the fallback are exhausted.
2036
+ // See docs/decisions/publish-path.md#hung-chunk: a hung or failed chunk is inconclusive, never terminal.
2236
2037
  chunkFailures.push("chunk " + chunk + ": " + (chunkErr && chunkErr.message ? chunkErr.message : chunkErr));
2237
2038
  log("Artifact build poll chunk " + chunk + " of 3 failed (" + (chunkErr && chunkErr.message ? chunkErr.message : chunkErr) + ") \u2014 continuing to the next chunk; build completion still unproven.");
2238
2039
  }
@@ -2248,20 +2049,7 @@ while (i < STEPS.length) {
2248
2049
  buildPoll = { build_done: false, status: (buildPoll && buildPoll.status) || "build still running after the 10.5-minute bounded poll" };
2249
2050
  }
2250
2051
  if (buildPoll.build_done && pollSawOurBuild) {
2251
- // STEP 1c (mechanical): NO provenance stamp here. Canary run 8
2252
- // (2026-09-11) proved the stamp cannot certify content: the
2253
- // builder's applied-report was derived from the carried diff, so
2254
- // the old report check was circular — a fabricated report
2255
- // passed by construction, and every phase went green on a hollow
2256
- // build. The stamp moves to the parent (docs/publish-verification.md);
2257
- // the deterministic lib/readback-disk.js is the primary sensor
2258
- // (the agent-callable read-back tool is unavailable —
2259
- // artifact_inspect was removed by the platform 2026-09-14 — so
2260
- // the LLM-inspector path is manual-fallback only), and the task
2261
- // parks for parent verification.
2262
- // QA's provenance check enforces the stamp mechanically.
2263
- // An unverified publish fails loudly in QA instead of passing
2264
- // silently here.
2052
+ // See docs/decisions/publish-path.md#step-1c-no-stamp: no provenance stamp in STEP 1c.
2265
2053
  publishBuildLanded = true;
2266
2054
  artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
2267
2055
  log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
@@ -2340,11 +2128,16 @@ while (i < STEPS.length) {
2340
2128
  attempt: rebuildAttemptKey,
2341
2129
  agent_id: rebuildAgentId,
2342
2130
  applied_report: publishAppliedObservation,
2343
- outcome: "submitted",
2131
+ manifest_before: preTriggerManifest,
2132
+ outcome: "build-observed",
2344
2133
  detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
2345
2134
  }, totalReworkCount);
2346
2135
  } else if (auditOkAfterPoll === false) {
2347
- publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestAuditDirAfterPoll + ", report ok=false). Explicit negative evidence: a build ran and failed (attribution by window, not by build identity — no stranger build was observed in flight during the poll). The publish did not land — provenance was not stamped. Fail-closed.";
2136
+ // Explicit negative evidence under the stranger guard: a build
2137
+ // ran and failed during the poll window with no stranger in
2138
+ // flight. This is an attributable failure — preserved verbatim
2139
+ // through post-deploy. Message contract unchanged.
2140
+ publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestAuditDirAfterPoll + ", report ok=false). Explicit negative evidence: a build ran and failed (attribution by window, not by build identity — no stranger build was observed in flight during the poll). The publish did not land — provenance was not stamped — your change was NOT published. Build log: ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestAuditDirAfterPoll + "/. This explicit failure is not auto-retried. Appendix: attribution=window; report=ok=false; provenance=unstamped.";
2348
2141
  await recordPublishLedger({
2349
2142
  commit: mergeCommitForPublish,
2350
2143
  attempt: rebuildAttemptKey,
@@ -2358,7 +2151,19 @@ while (i < STEPS.length) {
2358
2151
  : ((pollStatusErrors > 0 && !pollSawOurBuild) ? "status-read-errors-during-poll"
2359
2152
  : (pollEndState === "build-still-running-at-poll-end" ? "build-still-running-at-poll-end"
2360
2153
  : (newAuditDirsAfterPoll.length === 0 ? "no-new-audit-dir-in-window" : "audit-report-unreadable-or-missing")));
2361
- publishFailure = "Artifact build completion unproven (fail-closed, no provenance stamped): unattributable_reason=" + unattributableReason + "; poll_end_state=" + pollEndState + "; " + "saw_our_build=" + pollSawOurBuild + "; new_audit_dirs=" + newAuditDirsAfterPoll.length + "; poll_chunks_failed=" + chunkFailures.length + "; poll_status_errors=" + pollStatusErrors + ". Attribution is by window, not by build identity. The publish may or may not have landed. Fail-closed.";
2154
+ // Verdict UNKNOWN (poll path): the build cannot be attributed
2155
+ // to this attempt. The fields feed the single composed
2156
+ // fail-closed park reason after post-deploy — never a
2157
+ // fabricated verdict, never a silent pass.
2158
+ publishUnknownFields = {
2159
+ taskId: taskId,
2160
+ unattributableReason: unattributableReason,
2161
+ pollEndState: pollEndState,
2162
+ sawOurBuild: pollSawOurBuild,
2163
+ newAuditDirCount: newAuditDirsAfterPoll.length,
2164
+ chunkFailures: chunkFailures.length,
2165
+ pollStatusErrors: pollStatusErrors
2166
+ };
2362
2167
  await recordPublishLedger({
2363
2168
  commit: mergeCommitForPublish,
2364
2169
  attempt: rebuildAttemptKey,
@@ -2369,11 +2174,8 @@ while (i < STEPS.length) {
2369
2174
  }, totalReworkCount);
2370
2175
  }
2371
2176
  }
2372
- } else {
2373
- // Unreachable: the observation above either attributes the edit
2374
- // (edit_started) or parks. Defensive only — never a silent pass.
2375
- publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish is unattributed (not proven landed, not proven failed) — provenance was not stamped. Fail-closed.";
2376
2177
  }
2178
+
2377
2179
  } // end: publishSkippedNoLock — no rebuild, no stamp, nothing to ship
2378
2180
  // STEP 2 (mechanical, always — skip path included): post-deploy
2379
2181
  // commits builder leftovers if any, removes the worktree, and releases
@@ -2397,6 +2199,19 @@ while (i < STEPS.length) {
2397
2199
  ? " Post-deploy finalized cleanup."
2398
2200
  : " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
2399
2201
  }
2202
+ if (!publishBuildLanded) {
2203
+ // Verdict UNKNOWN (single fail-closed park — post-deploy always
2204
+ // runs first; no early parks anywhere above). Machine contract:
2205
+ // "unattributable_reason=" and "poll_end_state=" are always
2206
+ // present; attribution is by window, not identity.
2207
+ // (2026-09-18, H2 message contract) Thread the commit short-sha
2208
+ // through the unknown fields so the human line names the trigger
2209
+ // commit (design §1.5: "the trigger was sent for commit <short-sha>").
2210
+ if (publishUnknownFields) publishUnknownFields.commitShortSha = mergeCommitShortForPublish;
2211
+ return await parkTask(composeUnattributedParkReason(publishUnknownFields) + (postDeploy.deployed
2212
+ ? " Post-deploy finalized cleanup."
2213
+ : " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
2214
+ }
2400
2215
  if (!postDeploy.deployed) {
2401
2216
  return await parkTask("Post-deploy failed after the artifact build landed: " + (postDeploy.output || "no output") + ". The build may have landed but worktree cleanup and lock release are unknown — human attention needed.");
2402
2217
  }
@@ -2498,12 +2313,7 @@ while (i < STEPS.length) {
2498
2313
  "Report your test results as plain prose.\n" +
2499
2314
  "End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails on the visual or the mechanical checks. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the reported bug, fixed>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL verdict must carry a machine-readable reason: the workflow closeout cross-checks verdict.json against your prose VERDICT line, and an unreasoned or contradictory verdict fails the phase (never routes to rework).";
2500
2315
  } else if (qaTerminal) {
2501
- // Terminal-surface experiential QA: Hazel drives the CLI herself —
2502
- // the terminal counterpart to the see-act loop above. Same OODA
2503
- // discipline (append-ooda-step with --action terminal and a transcript
2504
- // per step; verdict via write-ooda-verdict), judged against the
2505
- // shared bar resolved via UX_DOCTRINE_PATH (the terminal doctrine page here). Transcripts are delivered;
2506
- // screenshots are never invented for terminal work.
2316
+ // See docs/decisions/qa-reproduce.md#terminal-qa: Hazel drives CLI transcripts for terminal surfaces.
2507
2317
  var btermEvidence = crewHome + "/task-evidence/" + taskId + "/postchange";
2508
2318
  var btermTargetsLine = terminalTargets || "not declared — derive from --help and the task description";
2509
2319
  instructions = "You are code-blind QA. You NEVER read source files.\n" +
@@ -2578,14 +2388,7 @@ while (i < STEPS.length) {
2578
2388
  "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your report here\"}. " +
2579
2389
  "The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
2580
2390
  var workKeyBase = "work-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "");
2581
- // Reproduce wrong-layer guard baseline (2026-09-17): snapshot the repro
2582
- // evidence dir listing BEFORE the Reproduce work agent runs. The guard
2583
- // after the run diffs against this baseline — only frames created by the
2584
- // current attempt count as a wrong-layer violation; stale frames from
2585
- // pre-fix attempts appear in both snapshots and are excluded (set diff,
2586
- // no wall-clock, deterministic). One snapshot per phase attempt covers
2587
- // all transport retries inside the dispatch loop. null = baseline
2588
- // unavailable; the guard then fails closed for retry.
2391
+ // See docs/decisions/qa-reproduce.md#wrong-layer-guard: reproduce must run at the correct layer.
2589
2392
  var reproEvidenceDir = crewHome + "/task-evidence/" + taskId + "/repro";
2590
2393
  var reproFramesBefore = null; // null = baseline unavailable (guard must fail closed)
2591
2394
  if (step.name === "Reproduce" && (reproLayer === "engine" || reproLayer === "docs")) {
@@ -2727,18 +2530,9 @@ while (i < STEPS.length) {
2727
2530
  }
2728
2531
 
2729
2532
  // QA verdict.json closeout gate (task 30dceb78): the prose VERDICT: line is
2730
- // the routing signal, but verdict.json is the reason-carrying record the
2731
- // workflow actually reads. A FAIL verdict must carry a machine-readable
2732
- // reason; the workflow refuses to route to rework on an unreasoned or
2733
- // contradictory verdict. The deterministic cross-checker
2734
- // (lib/read-ooda-verdict.js) runs against the prose verdict: a missing,
2735
- // corrupt, contradictory, or reason-less record fails the phase for retry
2736
- // — the dispatcher re-runs QA at the same step under its
2737
- // consecutive-failure cap — instead of routing to rework. Without this
2738
- // gate, a bare VERDICT: FAIL with an all-positive report (canary
2739
- // 2026-09-15, task 1d692d91) rebuilt nothing and parked at Publish on an
2740
- // unobserved artifact build. The gate applies only to the experiential QA
2741
- // path (qaArtifact / qaTerminal), the paths that write verdict.json.
2533
+ // the routing signal; verdict.json is cross-checked before any rework
2534
+ // routing (experiential QA paths only).
2535
+ // See docs/decisions/qa-reproduce.md#qa-verdict-gate for the full decision history.
2742
2536
  if (step.name === "QA" && (qaArtifact || qaTerminal) && verdictPassed !== null) {
2743
2537
  var qaVerdictDir = crewHome + "/task-evidence/" + taskId + "/postchange";
2744
2538
  var qaVerdictExpect = verdictPassed ? "PASS" : "FAIL";
@@ -2789,17 +2583,7 @@ while (i < STEPS.length) {
2789
2583
  }
2790
2584
  log("QA verdict.json closeout gate passed: verdict.json agrees with prose VERDICT: " + qaVerdictExpect);
2791
2585
 
2792
- // QA experiential-loop guard (clean-room defect 2026-09-16): Hazel's
2793
- // verdict.json is honest about missing experiential evidence, but the
2794
- // closeout treated a PASS as terminal done even when the experiential
2795
- // loop never ran. A PASS verdict with missing experiential evidence must
2796
- // never be terminal: the task parks fail-closed with
2797
- // unattributable_reason=qa-visual-loop-unavailable (artifact surface)
2798
- // or qa-terminal-loop-unavailable (terminal surface) instead of
2799
- // transitioning to done. Code, not prompt text: the loop-availability
2800
- // flags come from lib/read-ooda-verdict.js (the OODA log's NOT POSSIBLE
2801
- // steps and the verdict's missing_evidence tool-unavailability notes),
2802
- // already parsed into qaVerdictGate above.
2586
+ // See docs/decisions/qa-reproduce.md#experiential-loop-guard: a PASS with missing experiential evidence parks fail-closed.
2803
2587
  var qaLoopSurface = SURFACE_TERMINAL ? "terminal" : "visual";
2804
2588
  var qaLoopReason = SURFACE_TERMINAL ? "qa-terminal-loop-unavailable" : "qa-visual-loop-unavailable";
2805
2589
  var qaLoopUnavailable = SURFACE_TERMINAL ? qaVerdictGate.terminal_loop_unavailable : qaVerdictGate.visual_loop_unavailable;
@@ -2930,18 +2714,7 @@ while (i < STEPS.length) {
2930
2714
  }
2931
2715
  log("Build worktree confinement passed: " + wt.path);
2932
2716
 
2933
- // Already-merged idempotency: a `repo_diff: none (already-merged:
2934
- // <sha>)` declaration is verified mechanically — <sha> must resolve
2935
- // and be an ancestor of main in the configured repo. A fabricated or
2936
- // mistaken declaration fails the phase here (the dispatcher retries
2937
- // Build under its consecutive-failure cap); a verified declaration is
2938
- // recorded in alreadyMergedSha for Review's no-diff branch. Without
2939
- // this guard, Build correctly doing nothing left Review with no
2940
- // mechanical way to accept an empty diff, and Cass rejected for "no
2941
- // commits ahead of main — the builder likely forgot to commit" while
2942
- // the deliverable sat on main (canary 2026-09-15, task 1d692d91).
2943
- // The sha is hex-only by construction (extractAlreadyMerged), so
2944
- // interpolating it into the shell command cannot inject.
2717
+ // See docs/decisions/publish-path.md#already-merged-idem: an already-merged repo_diff is idempotent; no rebuild.
2945
2718
  var am = extractAlreadyMerged(workerText);
2946
2719
  if (am.sha) {
2947
2720
  var amCheck = await agent(
@@ -2987,20 +2760,7 @@ while (i < STEPS.length) {
2987
2760
  };
2988
2761
  let passed = stepResult.passed === true;
2989
2762
 
2990
- // Deterministic integrate verification: the agent cannot self-certify a
2991
- // merge. After the Integrate agent claims success, the workflow confirms
2992
- // mechanically that the task branch tip is an ancestor of main via the
2993
- // lifecycle script's verify-merge command (which resolves the branch
2994
- // through the crew registry — never by reconstructing "task/"+taskId — so
2995
- // the check cannot verify the wrong branch). The VERIFIED marker is matched
2996
- // by regex on the script's own stdout; agent prose is never read. This
2997
- // closes the hole where an agent reported "merged empty" while approved
2998
- // commits were still stranded on the task branch (bug b1b1f919). A genuine
2999
- // empty-diff Integrate (MERGED_EMPTY: no commits ahead of main) verifies
3000
- // vacuously — the tip is then an ancestor of main. Verification failure is
3001
- // an operational step failure, not a park: the dispatcher retries Integrate
3002
- // under its consecutive-failure cap, and the retry finds the commits still
3003
- // on the branch and performs the real merge — self-healing.
2763
+ // See docs/decisions/publish-path.md#integrate-verify: the agent cannot verify integrate mechanically; the workflow checks the diff.
3004
2764
  if (step.name === "Integrate" && passed) {
3005
2765
  var integrateVerifyOut = "";
3006
2766
  try {
@@ -3032,17 +2792,7 @@ while (i < STEPS.length) {
3032
2792
  // dispatcher's consecutive-failure cap.
3033
2793
  const status = passed ? "completed" : (step.name === "Integrate" || step.name === "Publish" ? "failed" : "rejected");
3034
2794
 
3035
- // Deterministic publish verification: the agent cannot self-certify a publish.
3036
- // Skip-aware (park 2026-09-11): when the deterministic publish script found
3037
- // no merge lock held (empty-diff Integrate), it skips the publish path
3038
- // gracefully and emits the machine-readable PUBLISH_SKIPPED=no-lock-held
3039
- // marker. The preflight (bugfix 2026-09-17) emits
3040
- // PUBLISH_SKIPPED=no-npm-publish when npm publish is not configured on
3041
- // this machine (helper or credential absent) — also before any mutation.
3042
- // Verification is then vacuous — nothing was shipped, and the
3043
- // registry must NOT have moved. The marker is script-emitted explicit state
3044
- // (pasted verbatim per the Publish agent instructions), not agent prose; a
3045
- // report without the marker still runs the full verification fail-closed.
2795
+ // See docs/decisions/publish-path.md#publish-verify: the agent cannot verify publish; the parent does.
3046
2796
  var publishVerified = false;
3047
2797
  var npmPublishSkipped = false;
3048
2798
  if (step.name === "Publish" && PUBLISH_TYPE === "npm" && publishTarget) {
@@ -3068,15 +2818,7 @@ while (i < STEPS.length) {
3068
2818
  publishTarget.target + " (" + publishTarget.base + " + " + publishTarget.scope + "). The publish did not land.");
3069
2819
  }
3070
2820
  publishVerified = true;
3071
- // Provenance refresh for self-publishes (task 7946d2a4): the npm path
3072
- // installs and activates a new immutable release (crew-release.sh deploy
3073
- // swaps the `current` symlink inside publish-npm.sh) but never stamped
3074
- // the dashboard's provenance record — every crew release left
3075
- // crew_release pointing at a pruned release. After a verified landed
3076
- // publish, refresh the record's crew_release to the now-live release
3077
- // identity, preserving the existing source_commit (the dashboard
3078
- // artifact's build source — a crew-repo commit here would fail the
3079
- // dashboard QA source check).
2821
+ // See docs/decisions/publish-path.md#provenance-refresh: refresh crew_release after landed publish, preserving source_commit.
3080
2822
  try {
3081
2823
  var provRefresh = await agent(
3082
2824
  "Run in shell and read the stdout JSON:\n" + crewCmd("get-provenance", { project_id: LAUNCH_PROJECT_ID }) + "\n" +
@@ -3106,33 +2848,10 @@ while (i < STEPS.length) {
3106
2848
  } // end: !npmPublishSkipped — a skipped publish has nothing to verify
3107
2849
  }
3108
2850
 
3109
- // Publish content verification — parent-owned (docs/publish-verification.md).
3110
- // The old block read back the workflow's OWN provenance stamp and compared
3111
- // it to HEAD: that verifies the stamp, not the content. Canary run 8
3112
- // (2026-09-11) passed it with a hollow build — the stamp was honest, the
3113
- // artifact was stale, all eight phases green. The stamp now moves to the
3114
- // parent (docs/publish-verification.md); the independent read-back step
3115
- // is currently unavailable (no agent-callable read-back tool exists —
3116
- // artifact_inspect was removed by the platform 2026-09-14), so the parent
3117
- // cannot confirm content and the task parks for verification. QA's
3118
- // provenance check enforces the stamp — an unverified publish fails loudly
3119
- // there instead of passing silently here.
3120
- // Skip-aware (park 2026-09-11): an empty-diff Integrate takes no merge
3121
- // lock, and the deterministic publish path skips rebuild/stamp entirely —
3122
- // there is no new content to verify, so verification is vacuous.
3123
- // publishSkippedNoLock is workflow-computed state from the explicit
3124
- // lock-status read in STEP 0, not agent prose.
3125
- // The parent (tick worker) triggers the ONE read-back inspection it can
3126
- // actually receive (async results go to the root agent, never into a
3127
- // workflow run — a workflow-side trigger would be an orphan). The workflow
3128
- // only parks; the parent's scan builds the request deterministically via
3129
- // lib/build-readback-request.js and ferries the inspection.
3130
- // publishBuildLanded and publishSkippedNoLock are workflow-computed state;
3131
- // a skipped or failed publish has nothing to verify.
3132
- // Session notes. Machine-readable marker lines are extracted from the full
3133
- // worker report and appended AFTER the slice so a long report can never
3134
- // amputate them; later phases (Review reading repo_diff:, QA backstop
3135
- // reading TARGET_VERSION=) depend on them.
2851
+ // Parent-owned verification (docs/publish-verification.md): the parent's
2852
+ // scan builds the request via lib/build-readback-request.js and ferries
2853
+ // the inspection; a skipped or failed publish has nothing to verify.
2854
+ // See docs/decisions/publish-path.md#parent-owned-verification for history.
3136
2855
  let summary;
3137
2856
  var workerMarkers = extractMarkerLines(workerText);
3138
2857
  if (step.name === "Publish" && PUBLISH_TYPE === "npm" && publishTarget && (publishVerified || npmPublishSkipped)) {
@@ -3146,12 +2865,7 @@ while (i < STEPS.length) {
3146
2865
  } else {
3147
2866
  summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
3148
2867
  }
3149
- // Already-merged attestation: when the Build gate verified the builder's
3150
- // already-merged declaration, the workflow records its own marker line in
3151
- // the session notes (like the builder markers above, it is appended after
3152
- // the slice so it can never be amputated). A later run resumed at Review
3153
- // hydrates alreadyMergedSha from this workflow-attested line — never from
3154
- // the builder's declaration alone.
2868
+ // See docs/decisions/publish-path.md#already-merged: when the Build gate verifies already-merged, attestation is recorded.
3155
2869
  if (step.name === "Build" && alreadyMergedSha) {
3156
2870
  summary += "\nalready_merged_verified: " + alreadyMergedSha;
3157
2871
  }
@@ -3215,12 +2929,7 @@ while (i < STEPS.length) {
3215
2929
  return await parkTask("Reproduction failed — needs PM attention");
3216
2930
  }
3217
2931
 
3218
- // Visual verdict ownership (2026-09-15): Hazel owns the visual verdict
3219
- // experientially — the QA instructions above have her drive the see-act
3220
- // loop herself and record verdict.json + the append-only verdicts.jsonl.
3221
- // Her prose VERDICT: line drives `passed` via extractVerdict; a FAIL
3222
- // bounces to Build through the standard Review/QA rejection path below.
3223
- // The old parent-recorded note gate is deleted.
2932
+ // See docs/decisions/qa-reproduce.md#visual-ownership: Hazel owns the visual verdict.
3224
2933
 
3225
2934
  // Review/QA rejection bounces to Build
3226
2935
  if (!passed && (step.name === "Review" || step.name === "QA")) {
@@ -3259,18 +2968,18 @@ while (i < STEPS.length) {
3259
2968
  return { status: "failed", task_id: taskId, reason: "Publish failed: " + summary };
3260
2969
  }
3261
2970
 
3262
- // Publish verification park: the build landed and post-deploy finalized,
3263
- // but provenance is UNSTAMPED until the parent's independent read-back
3264
- // (docs/publish-verification.md) confirms the artifact's actual content
3265
- // matches the merged diff. The parent stamps provenance, then re-queues;
3266
- // the dispatcher resumes at QA, whose provenance check enforces the stamp
3267
- // mechanically. A failed Publish never reaches this park — it returned
3268
- // failed above and retries under the dispatcher's cap. The merge lock is
3269
- // already released (post-deploy), so the parked task holds no resources.
2971
+ // See docs/decisions/publish-path.md#verification-park: the build landed but provenance is unstamped until parent verification.
3270
2972
  if (passed && step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG && !publishSkippedNoLock && publishBuildLanded) {
2973
+ // Success contract (2026-09-18, H2): the parent asked for exactly
2974
+ // one build for this publish. The build landed — do NOT republish: a
2975
+ // duplicate build would re-publish the same change. content_check=pending
2976
+ // means parent read-back is pending; this
2977
+ // park is NOT proof the content is correct.
3271
2978
  return await parkTask("publish: verification-requested " + mergeCommitForPublish +
3272
- " (build " + (rebuildAgentId || "agent_id unobserved") + ")" +
3273
- " — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md.");
2979
+ " attempt=" + rebuildAttemptKey +
2980
+ " Do NOT republish: a duplicate build would re-publish the same change. " +
2981
+ "Artifact build landed, post-deploy finalized, provenance not stamped — the crew has not yet independently confirmed the live artifact contains exactly the change; waiting on the manual read-back in docs/publish-verification.md. " +
2982
+ "Appendix: build=" + (rebuildAgentId || "agent_id unobserved") + "; provenance=unstamped; content_check=pending.");
3274
2983
  }
3275
2984
 
3276
2985
  i++;