muse-crew 0.13.2 → 0.13.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -27,26 +27,11 @@ const startStepIndex = inputs.start_step_index || 0;
27
27
  // resolution back via updatetask in the self-claim below.
28
28
  const RESOLVED_WORKFLOW = inputs.resolved_workflow || null;
29
29
  const WORKFLOW_WAS_NULL = inputs.workflow_was_null === true;
30
- // One-shot recovery routing: the dispatcher sets inputs.next_phase when it
31
- // routes this run via an explicit recover-task redirect. The value is
32
- // consumed (cleared) atomically by the successful self-claim below:
33
- // claim-task takes expected_next_phase and clears the matching next_phase in
34
- // the same transaction as the winning session insert, so no platform death
35
- // can slip between claim and consumption and replay the routing. A stale or
36
- // superseded routing survives — only an exact match clears.
37
- // what the dispatcher routed on.
30
+ // See docs/decisions/workflow-core.md#oneshot-recovery: the dispatcher sets input for one-shot recovery routing.
38
31
  const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_phase.length > 0) ? inputs.next_phase : null;
39
32
  const CLAIM_WORKFLOW_PERSIST = (WORKFLOW_WAS_NULL && RESOLVED_WORKFLOW) ? ", \"workflow\": \"" + RESOLVED_WORKFLOW + "\"" : "";
40
33
 
41
- // Visual verdict protocol availability — the workflow parks for parent-run
42
- // baseline capture and visual verdict ONLY when the protocol is fully
43
- // shipped. The protocol requires docs/visual-verdict.md in the release AND
44
- // the parent-side capture tooling (task b309a97d, "QA owns the visual
45
- // verdict"). Until both exist, the parks would deadlock waiting for a
46
- // parent who cannot fulfill them.
47
- // Effective value for this run, resolved by the dispatcher from the
48
- // project's visual_protocol setting (null=inherits crew default=off).
49
- // Manual launches without the arg default to off (previous behavior).
34
+ // See docs/decisions/qa-reproduce.md#visual-protocol-avail: the workflow parks if the visual protocol is unavailable.
50
35
  var VISUAL_PROTOCOL_AVAILABLE = inputs.visual_protocol === true;
51
36
 
52
37
  // crewHome is required — the dispatcher always passes it (crew-dispatch.js
@@ -93,15 +78,7 @@ const COMPUTE_DIFF_SRC = crewHome + "/current/lib/compute-publish-diff.js";
93
78
  const COMPUTE_DIFF = RUN_LIB + "/compute-publish-diff.js";
94
79
  const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
95
80
  const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
96
- // The seven basenames the pin step must materialize — asserted mechanically
97
- // by workflow code from the verbatim listing, never from agent prose.
98
- // COMPUTE_DIFF is the deterministic publish-diff computer (room #14,
99
- // 2026-09-17): the diff is computed by this script, never ferried as an
100
- // agent JSON string. Pinned like the other publish-critical modules so a
101
- // mid-run release swap cannot change it under the workflow.
102
- // CLASSIFY_SURFACE is the surface classifier (room #15, 2026-09-18):
103
- // crew-api.js statically imports it, so the pin must carry it — a pin
104
- // without it kills every claim with ERR_MODULE_NOT_FOUND.
81
+ // See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
105
82
  const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE].map(function (p) { return p.split("/").pop(); });
106
83
 
107
84
  // Project config — passed by dispatcher, falls back to dashboard defaults
@@ -134,14 +111,7 @@ const WORKTREE_HINT = REPO_PATH + "/.worktrees/" + taskId;
134
111
  const WORKTREE_PRESERVED_HINT = ".worktrees/" + taskId;
135
112
 
136
113
  const PUBLISH_TYPE = projectConfig.deploy_type || "";
137
- // User-facing surface for experiential QA routing: 'artifact' (a rendered
138
- // web UI Hazel drives with the see-act browser loop) | 'terminal' (a CLI
139
- // Hazel drives herself, keeping attempt-scoped transcripts) | null
140
- // (unclassified — no experiential QA). environment_type is the canonical
141
- // UX-surface axis; deploy_type names the deployment target, but
142
- // deploy_type === "artifact" remains a legacy artifact-surface signal so
143
- // pre-field projects keep today's experiential QA (the migration does not
144
- // backfill the column).
114
+ // See docs/decisions/qa-reproduce.md#surface-routing: the user-facing surface determines experiential QA routing.
145
115
  const ENV_TYPE = projectConfig.environment_type || null;
146
116
  // Surface resolution: artifact wins on contradictory config (the deployed
147
117
  // artifact is what users see). Unclassified surface => Capture skips, QA
@@ -156,12 +126,7 @@ const SURFACE_TRIAGE_DESC = SURFACE_ARTIFACT
156
126
  : SURFACE_TERMINAL
157
127
  ? "This project's user-facing surface is terminal: a command-line interface."
158
128
  : "This project's user-facing surface is unclassified (environment_type not set): judge by what a user would directly observe.";
159
- // UX doctrine page: the shared UX bar for this run's surface, resolved
160
- // mechanically — every phase prompt reads UX_DOCTRINE_PATH, never a
161
- // hardcoded filename. Canonical map: lib/ux-doctrine.js (mirrored here as a
162
- // one-liner because the workflow runtime's relative-import support is
163
- // unverified; tests pin the mirror). Null on unclassified surfaces: no
164
- // shared page, and prompts say so instead of naming the wrong one.
129
+ // See docs/decisions/qa-reproduce.md#ux-doctrine-page: the shared UX bar for this run's surface.
165
130
  const UX_DOCTRINE_PAGE = SURFACE_TERMINAL ? "terminal-ux.md" : (SURFACE_ARTIFACT ? "artifact-ux.md" : null);
166
131
  const UX_DOCTRINE_PATH = UX_DOCTRINE_PAGE ? crewHome + "/current/docs/" + UX_DOCTRINE_PAGE : null;
167
132
  const PUBLISH_SLUG = projectConfig.deploy_slug || "";
@@ -172,21 +137,7 @@ if (!taskId) {
172
137
  throw new Error("task_id is required in args");
173
138
  }
174
139
 
175
- // Closeout is deterministic: the work agent returns the runtime's native
176
- // transport envelope {"status": "ok", "result": "<prose report>"} with no
177
- // schema, so the workflow receives the report as a plain string. There is no
178
- // {"report"} wrapper: that invented shape invited agents to improvise sibling
179
- // keys (notably "status"), which the runtime duck-types as its own envelope
180
- // and fatally misparses. The envelope is the runtime's own documented shape
181
- // — not a demand for machine-structured reasoning.
182
- // The verdict is extracted mechanically by extractVerdict below — never by an
183
- // agent. The summary is the worker's report truncated. The release decision
184
- // comes from extractReleaseDecision. No formatter agent: it added a failure
185
- // mode while contributing nothing the workflow doesn't compute itself.
186
- // Steps whose passed=false drives a control-flow branch (rework bounce,
187
- // block) declare their verdict explicitly on a VERDICT: line. The verdict is
188
- // extracted DETERMINISTICALLY by workflow code (extractVerdict) — never by
189
- // an agent. Missing, malformed, or contradictory lines fail the phase (never silently pass).
140
+ // See docs/decisions/workflow-core.md#closeout-envelope: the work agent returns the runtime's native envelope; verdict extracted mechanically.
190
141
  const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
191
142
  function extractVerdict(workerText) {
192
143
  // The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
@@ -210,16 +161,7 @@ function extractVerdict(workerText) {
210
161
  if (uniq.length !== 1) return { ok: false, count: matches.length };
211
162
  return { ok: true, passed: last.value === "PASS" };
212
163
  }
213
- // Verdict re-ask (bug cd18ccc2): a verdict-step report that fails
214
- // extractVerdict is not failed immediately. Stochastic verdict-line
215
- // non-compliance (the agent did the work but omitted or garbled the VERDICT
216
- // line) gets up to two bounded follow-up agent() calls whose only job is to
217
- // read the preserved report and emit exactly one VERDICT line. The verdict
218
- // is still extracted mechanically by extractVerdict — the re-ask agent
219
- // transcribes, never decides the phase outcome. Each attempt uses a fresh
220
- // stable-key suffix so a cached failure can never replay deterministically.
221
- // Exhaustion keeps the existing fail-closed behavior. This is structure, not
222
- // prompt hardening: no instruction text was stern-ified to get here.
164
+ // See docs/decisions/workflow-core.md#verdict-reask: a report that fails extractVerdict gets bounded re-ask calls.
223
165
  function verdictReaskKey(stepName, reworkSuffix, attempt) {
224
166
  return "verdict-reask-" + stepName + reworkSuffix + "-a" + attempt;
225
167
  }
@@ -277,12 +219,7 @@ function workRetryKey(stepName, reworkSuffix, attempt) {
277
219
  function attemptKey(base, reworkCount) {
278
220
  return base + (reworkCount > 0 ? "-r" + reworkCount : "");
279
221
  }
280
- // pinLifecycle(key) — snapshot the lifecycle scripts into RUN_LIB and return
281
- // the verbatim `ls -1` listing so WORKFLOW CODE asserts the six pinned
282
- // basenames; the agent cannot self-certify. (The pin step was the one place
283
- // the workflows trusted agent prose: task 24be1cd6 walked to Publish on an
284
- // empty pin dir.) Byte-identical across standard/bugfix/chore — pinned by
285
- // tests/pin-location.test.js.
222
+ // See docs/decisions/qa-reproduce.md#pin-lifecycle: snapshot the lifecycle scripts in the pin.
286
223
  function pinLifecycle(key) {
287
224
  return agent(
288
225
  "Snapshot lifecycle scripts for version pinning.\n" +
@@ -490,24 +427,7 @@ function hydrateReleaseDecision(rec) {
490
427
  // fabricated from the diff;
491
428
  // it must match the artifact's real content.
492
429
 
493
- // Durable publish-attempt ledger (2026-09-12): every artifact publish
494
- // attempt is recorded append-only at $CREW_HOME/.publish-ledger/<slug>.jsonl
495
- // on persistent disk (NOT /tmp). The ledger is the correlation record for
496
- // publish attempts whose outcome is UNKNOWN. When the rebuild trigger's
497
- // child returns prose instead of JSON (structured-output failure), the edit
498
- // may already have been accepted as pending_init — and artifact_status
499
- // cannot see pending_init (diagnostic canary 2026-09-12: an edit accepted
500
- // as pending_init was immediately followed by an all-false status check,
501
- // and the old retry issued a DUPLICATE edit). "No build visible" is NOT
502
- // evidence the edit did not go through, so the workflow never blind-retries
503
- // on an unknown outcome: it records the attempt and parks fail-closed. A
504
- // human or a later run correlates the accepted edit via the ledger (commit
505
- // hash + attempt key + the artifact build's agent_id when one was observed)
506
- // instead of guessing from a blind status poll.
507
- // Best-effort observability: a failed write is logged loudly but never
508
- // throws — the caller's park/proceed decision never depends on the ledger.
509
- // Byte-identical across standard/bugfix/chore — pinned by
510
- // tests/publish-ledger.test.js.
430
+ // See docs/decisions/publish-path.md#publish-attempt-ledger: every trigger outcome is recorded in the durable ledger.
511
431
  async function recordPublishLedger(entry, rework) {
512
432
  try {
513
433
  var ledgerDir = crewHome + "/.publish-ledger";
@@ -557,45 +477,19 @@ function extractMarkerLines(workerText) {
557
477
  return markers.join("\n");
558
478
  }
559
479
 
560
- // Already-merged idempotency (canary 2026-09-15, task 1d692d91): when the
561
- // builder correctly makes no commit because the deliverable is already on
562
- // main (a prior merge or hand-repair landed it), it declares
563
- // `repo_diff: none (already-merged: <sha>)` naming the main commit that
564
- // carries the work. Room #16 blocker 11 (2026-09-18): the line anchor
565
- // missed Wren's mid-paragraph declaration, and the persisted notes truncated
566
- // the tail — so the anchor is gone and a sha followed by `)`, whitespace, or
567
- // end-of-string (truncation) is accepted. The sha is hex-only (7-40 chars)
568
- // so the workflow can interpolate it into the mechanical ancestor check
569
- // without injection risk; a over-long hex run never matches (the lookahead
570
- // fails on the extra hex char). Pure — pinned byte-identical across
571
- // standard/bugfix/chore.
480
+ // See docs/decisions/publish-path.md#already-merged-idem2: idempotency for already-merged tasks.
572
481
  function extractAlreadyMerged(workerText) {
573
482
  var m = /repo_diff:\s*none\s*\(already-merged:\s*([0-9a-f]{7,40})(?=[\s)]|$)/i.exec(workerText || "");
574
483
  return m ? { sha: m[1].toLowerCase() } : { sha: null };
575
484
  }
576
485
 
577
- // Explicit artifact refusal (room #16 blocker 10, 2026-09-18): the rebuild
578
- // trigger child ends its turn with `ARTIFACT_EDIT_REFUSED: <text>` when
579
- // artifact_edit explicitly refuses the edit (e.g. the artifact does not
580
- // exist). A refusal is conclusive negative evidence — the edit provably did
581
- // NOT go through — distinct from an unconsumed trigger return (unknown).
582
- // Pure — pinned byte-identical across standard/bugfix/chore.
583
- // The signal must be the ENTIRE trimmed turn output (not a line within prose):
584
- // the trigger child is instructed to end its turn with exactly this line and
585
- // nothing else. A confused child quoting the instructions back in prose must
586
- // NOT produce a conclusive negative — that degrades to unknown (fail-closed).
486
+ // See docs/decisions/publish-path.md#refusal-signal: the refusal signal must be the ENTIRE trimmed turn output.
587
487
  function extractRefusal(workerText) {
588
488
  var m = /^ARTIFACT_EDIT_REFUSED:\s*(.+?)\s*$/.exec(String(workerText || "").trim());
589
489
  return m ? m[1].slice(0, 300) : null;
590
490
  }
591
491
 
592
- // Worktree confinement: the Build agent must declare the exact worktree
593
- // path it built in on a `worktree:` marker line. The workflow compares it
594
- // against WORKTREE_HINT mechanically (exact string match) — never by
595
- // reading agent prose. This closes the hole where a builder whose prepare
596
- // failed freelanced into a different checkout (canary, 2026-09-11): the
597
- // honest-but-confused case fails here, and a fabricated path is caught one
598
- // phase later when Review's inspect finds no commits in the configured repo.
492
+ // See docs/decisions/qa-reproduce.md#worktree-confinement: the Build agent must declare its worktree.
599
493
  function extractWorktree(workerText) {
600
494
  var lines = (workerText || "").split("\n");
601
495
  var found = null;
@@ -625,26 +519,7 @@ function extractExperiential(workerText) {
625
519
  if (!r) return null;
626
520
  return r[1].toLowerCase() === "yes";
627
521
  }
628
- function buildVisualCapturePlan(taskTitle, taskDescription, kind, captureTargets) {
629
- // Deterministic visual-capture frame. kind: "baseline" | "postchange".
630
- // This string IS the capture script: fixed viewport matrix, scroll
631
- // positions, and interaction states — the inspection agent executes it
632
- // verbatim, nothing is improvised. Task-specific targets fill the slots.
633
- var title = String(taskTitle || "").replace(/"/g, "'").slice(0, 120);
634
- var targets = String(captureTargets || "").trim() ||
635
- String(taskDescription || "").replace(/"/g, "'").slice(0, 300);
636
- return "VISUAL CAPTURE — " + kind.toUpperCase() + " — task: " + title + ". " +
637
- "Target views/controls: " + targets + ". " +
638
- "For EACH target, capture exactly: " +
639
- "(1) desktop 1440x900, full view, scrolled to top; " +
640
- "(2) desktop 1440x900, scrolled so the target is vertically centered; " +
641
- "(3) mobile 390x844, scrolled so the target is vertically centered; " +
642
- "(4) desktop 1440x900, hover state on the target control; " +
643
- "(5) desktop 1440x900, keyboard-focus state on the target control; " +
644
- "(6) desktop 1440x900, active/pressed state if the target is a button or control. " +
645
- "Also record: console error count, the ARIA tree of the target region, any horizontal overflow. " +
646
- "Name captures " + kind + "-<n>-<viewport>-<state>. Return the captures, not a summary.";
647
- }
522
+
648
523
  // Experiential flag resolution: the task is experiential when Sage's Triage
649
524
  // report ends with the machine-read marker "experiential: yes". The flag is
650
525
  // opt-in — a missing or garbled line degrades to "unknown", which callers
@@ -790,23 +665,9 @@ function releaseDecisionText() {
790
665
  if (!releaseDecision) return "no machine-readable release decision from the Build report";
791
666
  return "release: " + releaseDecision.release + (releaseDecision.version_bump ? ", version_bump: " + releaseDecision.version_bump : " (no version_bump line)");
792
667
  }
793
- // Park the task for human attention and end the run. "blocked" is never
794
- // manually authored — the dashboard derives it mechanically from unmet
795
- // dependencies — so a workflow outcome that needs a human parks the task
796
- // instead. Parking is one atomic dashboard action (parktask): the parked
797
- // state and the explanatory note land in one transaction, never half.
798
- // The dispatcher skips parked tasks; a human moving parked→todo
799
- // mechanically resets the retry counters. Returns the workflow result
800
- // envelope the launcher sees. If the park call itself fails, the run
801
- // reports "failed" (retryable) so the next tick re-attempts the park —
802
- // a lost park is never reported as parked.
803
- // Terminal cleanup: the run's last act at every park/fail boundary. A run
804
- // that parks or fails must not leak its worktree, branch, or merge lock.
805
- // The lifecycle's terminal-cleanup releases the lock unconditionally and
806
- // reclaims the worktree+branch ONLY when the task branch is fully merged
807
- // into main (then it is redundant); unmerged work is preserved for the
808
- // human by design. Fire-and-forget with one bounded retry — the merge-lock
809
- // lease expiry and the orphan sweep are the backstop for a dead transport.
668
+ // Terminal cleanup: the run's last act — release the lock unconditionally;
669
+ // reclaim worktree+branch only when fully merged (unmerged work is preserved
670
+ // for the human by design). See `docs/decisions/workflow-core.md#park-contract`.
810
671
  async function terminalCleanup() {
811
672
  for (var attempt = 1; attempt <= 2; attempt++) {
812
673
  try {
@@ -1057,14 +918,7 @@ while (i < STEPS.length) {
1057
918
  lockHolder = taskId + "/" + activeSessionId;
1058
919
  }
1059
920
 
1060
- // ── Capture: baseline evidence for experiential tasks ─────────────
1061
- // Hazel's QA capture pass runs right after Triage, before Map, for tasks
1062
- // Sage flagged experiential. The capture itself is parent-driven (the
1063
- // inspection handoff arrives at the root agent, outside this script), so
1064
- // when no baseline evidence is recorded yet the script logs a note event
1065
- // and parks with the exact parent protocol + resume path. Never fails the
1066
- // task over missing evidence: after two requests, baseline:none is
1067
- // recorded and final QA judges on the rubric alone.
921
+ // See docs/decisions/qa-reproduce.md#capture-baseline: baseline evidence for experiential tasks is captured before work begins.
1068
922
  if (step.name === "Capture") {
1069
923
  var capExp = await resolveExperiential();
1070
924
  var bounceSuffix = (mapGateBounceCount > 0 ? "-g" + mapGateBounceCount : "");
@@ -1111,12 +965,7 @@ while (i < STEPS.length) {
1111
965
  continue;
1112
966
  }
1113
967
  var capStatus = await baselineStatus();
1114
- // Stale-decision guard: a "baseline: none (visual protocol unavailable)"
1115
- // note is only durable while the protocol is unavailable. When
1116
- // VISUAL_PROTOCOL_AVAILABLE is true, that old decision no longer
1117
- // stands — fall through to the request path for a fresh capture
1118
- // attempt. Exact-string trim comparison against the workflow's own
1119
- // written message (explicit state, never English matching).
968
+ // See docs/decisions/qa-reproduce.md#stale-decision-guard: a stale baseline decision parks fail-closed.
1120
969
  var baselineLatestMessage = ("baseline: " + capStatus.baseline_kind + capStatus.baseline_refs).trim();
1121
970
  var baselineStale = VISUAL_PROTOCOL_AVAILABLE && baselineLatestMessage === "baseline: none (visual protocol unavailable)";
1122
971
  if (capStatus.baseline_found && !baselineStale) {
@@ -1277,17 +1126,7 @@ while (i < STEPS.length) {
1277
1126
  (PUBLISH_TYPE === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
1278
1127
 
1279
1128
  } else if (step.name === "Review") {
1280
- // Already-merged hydration: when this run did not execute Build itself
1281
- // (dispatcher resume at Review after a platform death between phases),
1282
- // recover the workflow-verified sha. Room #16 blocker 11: the structured
1283
- // session field is read FIRST — the `already_merged_verified:` notes line
1284
- // is only a fallback, because session notes are hard-capped at 3000
1285
- // chars and a truthful declaration at the report's tail was silently
1286
- // truncated. The structured value was written by the workflow after a
1287
- // mechanical ancestor check — it is trusted; the builder's bare
1288
- // declaration never is. Absent both, the mechanical fact below reads
1289
- // "none declared" and Cass fails closed. The hydration read is best-effort:
1290
- // a transport throw degrades to "none declared" rather than crashing Review.
1129
+ // See docs/decisions/publish-path.md#already-merged-hydra: when this run did not execute, hydration uses the existing merge.
1291
1130
  if (!alreadyMergedSha) {
1292
1131
  var hydResult = null;
1293
1132
  try {
@@ -1395,25 +1234,7 @@ while (i < STEPS.length) {
1395
1234
  "published: muse-crew@" + publishTarget.target + "\n" +
1396
1235
  "VERDICT: PASS\n\n";
1397
1236
  } else if (PUBLISH_TYPE === "artifact") {
1398
- // Deterministic artifact publish (canary b5efd1b1, 2026-09-10): the work
1399
- // agent claimed "Rebuilt and deployed" while no build ran and no
1400
- // provenance was stamped — prose-trusted side effects, the same failure
1401
- // class as the npm double-skip (bb739316). The npm path already runs one
1402
- // deterministic script; the artifact path now has the same shape. Lock
1403
- // refresh, rebuild trigger, build-completion poll, and post-deploy are
1404
- // narrow schema'd bookkeeping calls owned by the workflow — the work
1405
- // agent reports on the mechanical outcome and cannot skip what it never
1406
- // owned. Any step failing parks with an honest, step-specific reason
1407
- // (fail-closed). There is deliberately NO workflow-side provenance
1408
- // stamp: the builder's applied-report is circular (canary run 8,
1409
- // 2026-09-11), so the stamp moved to the parent — after the build
1410
- // lands, the workflow records the session completed and parks with
1411
- // "publish: verification-requested". The parent owns verification
1412
- // (docs/publish-verification.md); the primary sensor is the
1413
- // deterministic lib/readback-disk.js (the agent-callable read-back
1414
- // tool is unavailable — artifact_inspect was removed by the platform
1415
- // 2026-09-14 — so the LLM-inspector path is manual-fallback only).
1416
- // QA's provenance check enforces the stamp mechanically.
1237
+ // See docs/decisions/publish-path.md#deterministic-artifact-publish: the work agent never publishes; the parent runs the deterministic publish script.
1417
1238
  var artifactPublish = null;
1418
1239
  var publishLockRefreshed = false;
1419
1240
  var publishSkippedNoLock = false;
@@ -1440,20 +1261,7 @@ while (i < STEPS.length) {
1440
1261
  }
1441
1262
  publishLockRefreshed = true;
1442
1263
  }
1443
- // STEP 1 (mechanical): carry the merged change to the artifact
1444
- // builder. The builder's source tree is NOT the crew's repo —
1445
- // canary run 4 (2026-09-11) proved it: Publish asked for "rebuild
1446
- // from current source. Do not modify any source files" and the
1447
- // builder rebuilt a stale copy predating the canary's changes, then
1448
- // the workflow stamped the new commit hash on the stale build.
1449
- // Provenance fiction; all eight phases passed. The merge diff is
1450
- // embedded in the edit request; the builder applies it to its own
1451
- // tree and reports the applied changes; the workflow verifies the
1452
- // report matches the diff BEFORE stamping provenance. A mismatch
1453
- // parks without stamping — the stamp must never certify a build
1454
- // whose content was not verified.
1455
- // Skipped entirely when no lock was held — nothing merged, nothing
1456
- // to ship.
1264
+ // See docs/decisions/publish-path.md#step1-builder-source: the builder's source tree is NOT the crew's repo; verify report before stamping.
1457
1265
  if (!publishSkippedNoLock) {
1458
1266
  // The trigger key of the attempt that last ran, for the publish ledger.
1459
1267
  // Minted once here (not re-minted per use site) so the ledger always
@@ -1534,20 +1342,7 @@ while (i < STEPS.length) {
1534
1342
  return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact target preflight was inconclusive (no parsable signal). Target existence is unproven, so no edit was issued and nothing was retried blindly. Human attention needed.");
1535
1343
  }
1536
1344
  log("Publish artifact preflight for task " + taskId + ": target " + artifactTargetDir + " present");
1537
- // (below) the diff computation, rebuild trigger, application
1538
- // verification, bounded poll, and provenance stamp. The builder
1539
- // only makes the artifact_edit call and reports the applied
1540
- // changes — no prose claim to trust. If the artifact tool namespace
1541
- // is missing from this child it reports honestly and the workflow
1542
- // retries once with a fresh key (bounded); anything else parks.
1543
- // Publish diff base (2026-09-14, task 0c53af4e): the carried diff is
1544
- // BASE..HEAD where BASE is the previously-stamped provenance
1545
- // source_commit — NOT HEAD^1. A push-time reconcile merge puts the
1546
- // task's own changes behind an intermediate merge, so HEAD^1..HEAD
1547
- // silently drops the task's fix while the artifact builds without
1548
- // it. The stamped base is the artifact's actual content; BASE..HEAD
1549
- // is the complete unpublished delta. Empty tree only for a genuine
1550
- // first publish (no provenance stamped yet).
1345
+ // See docs/decisions/publish-path.md#diff-computation: the diff is computed, the rebuild is triggered, and the report is verified.
1551
1346
  var EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
1552
1347
  var provResult = await agent(
1553
1348
  crewCmd("get-provenance", { project_id: LAUNCH_PROJECT_ID }) + "\n" +
@@ -1563,16 +1358,7 @@ while (i < STEPS.length) {
1563
1358
  } else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
1564
1359
  return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
1565
1360
  }
1566
- // Publish diff transport (room #14, 2026-09-17): the diff used to be
1567
- // ferried as a JSON string field in the agent's response — the agent
1568
- // produced a valid 700-line diff on disk but the JSON ferry dropped
1569
- // it, and the parse saw zero files ("Publish diff parsed to zero
1570
- // files"). The diff now travels git -> file -> deterministic script
1571
- // summary; the LLM never carries diff bytes. The agent is pure hands:
1572
- // it runs exactly one command (the PINNED compute-publish-diff.js —
1573
- // a mid-run release swap cannot change it under the workflow) and
1574
- // returns the small JSON summary verbatim. All fail-closed parks
1575
- // below are unchanged in meaning.
1361
+ // See docs/decisions/publish-path.md#diff-transport: the diff travels via file, not the LLM.
1576
1362
  var publishDiffFile = crewHome + "/.publish-diffs/" + taskId + (totalReworkCount > 0 ? "-r" + totalReworkCount : "") + ".diff";
1577
1363
  var diffResult = await agent(
1578
1364
  "Run exactly one command and nothing else:\n" +
@@ -1597,6 +1383,7 @@ while (i < STEPS.length) {
1597
1383
  return await parkTask("Publish diff base mismatch: script reported '" + String(diffSummary.base || "").slice(0, 12) + "' but the stamped base is '" + publishBase.slice(0, 12) + "'. Human attention needed.");
1598
1384
  }
1599
1385
  var mergeCommitForPublish = String(diffSummary.commit || "").trim();
1386
+ var mergeCommitShortForPublish = String(mergeCommitForPublish).substring(0, 7) || "unknown";
1600
1387
  var publishDiffSha256 = String(diffSummary.sha256 || "");
1601
1388
  if (!diffSummary.bytes) {
1602
1389
  return await parkTask("Publish diff is empty for commit " + (mergeCommitForPublish || "unknown") + " — a merge lock was held but there is no change to carry. Human attention needed.");
@@ -1648,37 +1435,11 @@ while (i < STEPS.length) {
1648
1435
  "- If artifact_edit is not available after the load, do NOT improvise — end your turn.\n" +
1649
1436
  "- You do NOT call setprovenance, artifact_inspect, or post-deploy yourself.\n" +
1650
1437
  "No report is needed: do not return JSON, do not summarize what you did, do not echo the diff. End your turn after the artifact_edit call.\n";
1651
- // The artifact build's agent_id, attributed to this edit by the
1652
- // workflow-owned observation below. The agent_id is the artifact
1653
- // system's in-flight correlation ID (research 2026-09-12):
1654
- // artifact.edit returns pending_init with NO agent_id, but
1655
- // artifact_status exposes build.agent_id immediately after
1656
- // acceptance, stable across polls. Recorded in the ledger so an
1657
- // attempt correlates to the exact builder run; null when no build
1658
- // was ever observed.
1438
+ // See docs/decisions/publish-path.md#agent-id-attribution: the artifact build's agent_id is attributed to the edit call.
1659
1439
  var rebuildAgentId = null;
1660
- // The builder's applied report is gone (2026-09-16): it rode on the
1661
- // trigger's JSON closeout contract, which is removed below. The
1662
- // parent's independent read-back (docs/publish-verification.md) is
1663
- // the verification — this field stays "missing-report" on ledger
1664
- // lines for issued triggers; pre-trigger parks (toolcheck
1665
- // rejected/inconclusive) and unattributed-unknown parks write null
1666
- // (no trigger was observed, so there is nothing to report).
1440
+ // See docs/decisions/publish-path.md#applied-report-gone: the builder's applied report is gone; the workflow verifies differently.
1667
1441
  var publishAppliedObservation = "missing-report";
1668
- // Durable-evidence snapshot (2026-09-14): the observation below only
1669
- // detects IN-FLIGHT builds. A build that finished before the
1670
- // observation leaves no in-flight trace — but the platform's audit
1671
- // harness leaves a durable one:
1672
- // ~/workspace/ts-spaces/<slug>/audits/<timestamp>-<id>/ per
1673
- // completed build. Snapshot the listing BEFORE the trigger so the
1674
- // fallback can diff before/after: a directory appearing during the
1675
- // trigger window is positive evidence the edit went through and
1676
- // the build completed. Best-effort and non-gating: if the snapshot
1677
- // fails, auditBeforeOk stays false and BOTH fallback comparisons
1678
- // are disabled (2026-09-16, critic finding 4) — without a baseline,
1679
- // an empty before-list would make every historical audit dir look
1680
- // "new". No wall-clock in-script (deterministic replay) — the
1681
- // comparison is a pure before/after set diff.
1442
+ // See docs/decisions/publish-path.md#durable-evidence-snapshot: snapshot the audit-dir listing BEFORE the trigger; fallback diffs before/after.
1682
1443
  var auditDirsBeforeTrigger = [];
1683
1444
  var auditBeforeOk = false;
1684
1445
  try {
@@ -1695,36 +1456,7 @@ while (i < STEPS.length) {
1695
1456
  } catch (auditBeforeErr) {
1696
1457
  log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1697
1458
  }
1698
- // Fire-and-forget trigger + workflow-owned observation (2026-09-16,
1699
- // clean-room task e2a8d9f8): the trigger's JSON closeout contract
1700
- // traveled over the stochastic text channel, and the runtime's
1701
- // JSON-candidate heuristic misfired on it ("workflow agent output
1702
- // was not JSON: no JSON object or array found in final response"),
1703
- // parking a task whose edit may have gone through. The contract's
1704
- // content was already observation-only (the applied report never
1705
- // gated; the pre_hashes were diagnostic-only), so the contract is
1706
- // removed: the trigger carries NO schema and its return value is
1707
- // never consumed, which takes the extraction heuristic out of this
1708
- // call entirely. The workflow attributes the edit itself through
1709
- // the tiny schema'd reads below — no prose is parsed for the
1710
- // trigger outcome.
1711
- // (Probe, 2026-09-16: the workflow scope exposes only agent() —
1712
- // tool_search, artifact_edit and artifact_status are undefined
1713
- // there — so the workflow cannot call the artifact tools directly;
1714
- // observation still goes through minimal child calls with tiny
1715
- // schemas, never a broad JSON contract.)
1716
- //
1717
- // Pre-trigger toolcheck (tiny, schema'd): the artifact namespace is
1718
- // deferred for workflow children — the child self-loads it and emits
1719
- // one exact signal line, read mechanically (never English prose).
1720
- // Only a parsed ARTIFACT_TOOLS: missing signal is explicit negative
1721
- // evidence: it gets one bounded retry with a fresh key, then parks
1722
- // rejected — without the tools the edit provably did NOT go through,
1723
- // so this is the one safe retry on the publish path. A throw (or an
1724
- // unparseable signal) is INCONCLUSIVE transport noise, never
1725
- // evidence of missing tools (2026-09-16, critic finding 3): it is
1726
- // recorded, it retries once in case the flake clears, but it can
1727
- // never take the rejected path.
1459
+ // See docs/decisions/publish-path.md#fire-and-forget-trigger: the trigger child returns immediately; the workflow owns observation and verdict.
1728
1460
  var publishToolsOk = false;
1729
1461
  var publishToolsMissing = false;
1730
1462
  for (var toolcheckAttempt = 1; toolcheckAttempt <= 2 && !publishToolsOk; toolcheckAttempt++) {
@@ -1772,14 +1504,7 @@ while (i < STEPS.length) {
1772
1504
  }, totalReworkCount);
1773
1505
  return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was explicitly missing (parsed signal — the edit provably did not go through, so no trigger was issued and nothing was retried blindly). Human attention needed.");
1774
1506
  }
1775
- // Pre-trigger build-state baseline (tiny, schema'd): one read of
1776
- // artifact_status. The post-trigger observation diffs against this
1777
- // baseline — a build whose agent_id was absent from (or differs
1778
- // from) the baseline is attributed to our edit; a build already in
1779
- // flight at baseline predates the trigger and is never attributed
1780
- // to it. If the baseline read itself fails, receipt attribution is
1781
- // skipped and the durable audit-dir evidence below is the only
1782
- // positive signal.
1507
+ // See docs/decisions/publish-path.md#pretrigger-baseline: a tiny schema'd baseline is captured before the trigger.
1783
1508
  var baselineAgentId = null;
1784
1509
  var baselineFailed = false;
1785
1510
  try {
@@ -1796,32 +1521,13 @@ while (i < STEPS.length) {
1796
1521
  baselineFailed = true;
1797
1522
  log("Publish pre-trigger baseline read failed for task " + taskId + " (" + (baselineErr && baselineErr.message ? baselineErr.message : baselineErr) + ") — receipt attribution skipped; durable audit-dir evidence is the only positive signal");
1798
1523
  }
1799
- // The trigger itself: the artifact_edit call is AWAITED (the workflow
1800
- // waits for it to complete) but its return value is intentionally
1801
- // UNCONSUMED — NO schema, so no schema validation can fail this
1802
- // call: a schema-less call resolves to the child's raw response as
1803
- // a plain string (probed live 2026-09-16 — never parsed, never
1804
- // throws on content). One caveat, also probed: the runtime still
1805
- // scans the response for a JSON candidate, and an unparseable
1806
- // {...}-looking substring in the child's prose throws ("response
1807
- // JSON candidate", probe P6). The prompt tells the child to end its
1808
- // turn with no prose at all, which keeps the common case clean —
1809
- // but the channel is stochastic, so any throw is possible and
1810
- // inconclusive: the edit may still have gone through, so the
1811
- // outcome stays unknown until the observation below confirms it —
1812
- // never inferred from the throw, and never blind-retried (a blind
1813
- // re-trigger duplicated the edit on 2026-09-12).
1524
+ // See docs/decisions/publish-path.md#trigger-await: the artifact_edit call is awaited.
1814
1525
  var rebuildTrigger = null;
1815
1526
  try {
1816
1527
  var triggerText = String(await agent(rebuildPrompt,
1817
1528
  { key: rebuildAttemptKey, label: "Triggering artifact rebuild" }) || "");
1818
1529
  log("Publish rebuild trigger for task " + taskId + " returned (" + triggerText.length + " chars; awaited; scanned only for the explicit refusal signal)");
1819
- // Explicit refusal (room #16 blocker 10): the child ends its turn
1820
- // with ARTIFACT_EDIT_REFUSED when artifact_edit explicitly refused.
1821
- // Conclusive negative evidence — the edit provably did NOT go
1822
- // through — so this parks rejected and skips observation polling.
1823
- // A missing/unparseable signal is NOT a refusal: it stays unknown
1824
- // and fail-closed below.
1530
+ // See docs/decisions/publish-path.md#explicit-refusal-16: the child must explicitly refuse artifact work.
1825
1531
  var refusalText = extractRefusal(triggerText);
1826
1532
  if (refusalText) {
1827
1533
  await recordPublishLedger({
@@ -1868,17 +1574,79 @@ while (i < STEPS.length) {
1868
1574
  log("Publish post-trigger build-state check failed for task " + taskId + " (" + (buildCheckErr && buildCheckErr.message ? buildCheckErr.message : buildCheckErr) + ") — this signal is unknown, not negative");
1869
1575
  }
1870
1576
  var observedAgentId = (buildState && buildState.build && typeof buildState.build.agent_id === "string" && buildState.build.agent_id) || null;
1871
- // Known limitation (failure-mode audit 2026-09-16): attribution
1872
- // is timing-based — any agent_id new relative to the baseline is
1873
- // treated as this edit's receipt. A stranger's build starting inside
1874
- // the trigger window is indistinguishable by timing and would be
1875
- // misattributed here. The consequence is bounded: the completion
1876
- // poll below tracks the recorded id, and the parent's mechanical
1877
- // content read-back (docs/publish-verification.md) certifies the
1878
- // exact commit's content — a wrong build's content fails closed as
1879
- // verification-failed, never stamped. Timing narrows the candidate;
1880
- // content decides.
1577
+ // See docs/decisions/publish-path.md#attribution-limitation: attribution is timing-based; content verification bounds the risk.
1881
1578
  var receiptAgentId = (!buildStateFailed && !baselineFailed && observedAgentId && observedAgentId !== baselineAgentId) ? observedAgentId : null;
1579
+ // auditReportOk: pure tri-state read of a report.json body —
1580
+ // true (build ok), false (build failed), null (missing or
1581
+ // unreadable — not evidence either way). The child returns the
1582
+ // raw body verbatim; interpretation lives here, never in prose.
1583
+ // Hoisted to Publish-step scope (before the receipt branch) so both
1584
+ // the immediate and post-poll audit fallbacks share it on every path —
1585
+ // the receipt path skips the else below, which must not leave these
1586
+ // undefined.
1587
+ var auditReportOk = function (raw) {
1588
+ if (typeof raw !== "string") return null;
1589
+ var trimmed = raw.trim();
1590
+ if (trimmed === "" || trimmed === "MISSING") return null;
1591
+ var parsed;
1592
+ try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
1593
+ if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
1594
+ return null;
1595
+ };
1596
+ // (2026-09-18, H2 verdict-first) Publish-verdict vocabulary. The
1597
+ // verdict is one of "landed" | "unknown". Decided once,
1598
+ // before the receipt poll, and dispatched on — never re-derived.
1599
+ // Pure and self-contained: unit-tested by
1600
+ // tests/publish-verdict-first.test.js.
1601
+ // See docs/decisions/publish-path.md#h2-verdict-dispatch.
1602
+ var decidePublishVerdict = function (opts) {
1603
+ var immediateReport = (opts && "immediateReport" in opts) ? opts.immediateReport : null;
1604
+ var newDirCount = (opts && typeof opts.newDirCount === "number") ? opts.newDirCount : 0;
1605
+ if (newDirCount === 1 && immediateReport === true) return { verdict: "landed", unattributableReason: null };
1606
+ if (newDirCount === 1 && immediateReport === false) return { verdict: "unknown", unattributableReason: "audit-report-ok-false" };
1607
+ if (newDirCount === 1) return { verdict: "unknown", unattributableReason: "audit-report-unreadable-or-missing" };
1608
+ if (newDirCount === 0) return { verdict: "unknown", unattributableReason: "no-new-audit-dir-in-window" };
1609
+ return { verdict: "unknown", unattributableReason: "audit-dir-ambiguity", ambiguousDirCount: newDirCount };
1610
+ };
1611
+ // (2026-09-18, H2 message contract, design §1.5) The human line is
1612
+ // the output of a mechanical field→template mapping: the trigger
1613
+ // commit short-sha, one plain clause per unattributable reason, the
1614
+ // no-republish warning, the ledger path with (outcome: unknown), and
1615
+ // the unknown-recovery clause. The appendix carries every machine
1616
+ // field. Pure and self-contained: unit-tested by
1617
+ // tests/publish-verdict-first.test.js.
1618
+ var composeUnattributedParkReason = function (fields) {
1619
+ var f = fields || {};
1620
+ var task = f.taskId || taskId;
1621
+ var reason = f.unattributableReason || "unknown-outcome";
1622
+ var shortSha = String(f.commitShortSha || "").substring(0, 7) || "unknown";
1623
+ var reasonClauses = {
1624
+ "no-new-audit-dir-in-window": "no build could be tied to this attempt",
1625
+ "audit-report-unreadable-or-missing": "the build's report is unreadable",
1626
+ "audit-dir-ambiguity": "more than one build appeared in the check window",
1627
+ "stranger-build-observed-during-poll": "a different build was running during the check",
1628
+ "status-read-errors-during-poll": "status reads kept failing"
1629
+ };
1630
+ var clause = reasonClauses[reason] || "no build could be tied to this attempt";
1631
+ return "Publish outcome unknown for task " + task +
1632
+ ": the trigger was sent for commit " + shortSha + " but the outcome could not be confirmed — " + clause + ". " +
1633
+ "Do NOT republish: if the trigger was accepted, a retry duplicates the build (2026-09-12). " +
1634
+ "The attempt is in the ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (outcome: unknown) " +
1635
+ "and the crew's unknown-recovery will re-examine it. " +
1636
+ "Appendix: unattributable_reason=" + reason +
1637
+ "; poll_end_state=" + (f.pollEndState || "not-polled") +
1638
+ "; saw_our_build=" + (f.sawOurBuild ? "true" : "false") +
1639
+ "; new_audit_dirs=" + (f.newAuditDirCount == null ? 0 : f.newAuditDirCount) +
1640
+ "; poll_chunks_failed=" + (f.chunkFailures == null ? 0 : f.chunkFailures) +
1641
+ "; poll_status_errors=" + (f.pollStatusErrors == null ? 0 : f.pollStatusErrors) + ".";
1642
+ };
1643
+ // publishFailure is declared here (per-Publish-pass scope) so the
1644
+ // post-poll explicit build failure survives to the final routing
1645
+ // below; publishUnknownFields carries the structured unknown fields
1646
+ // for the single composed fail-closed park. Both re-initialize on
1647
+ // every pass — a rework re-entry never leaks a stale verdict.
1648
+ var publishFailure = null;
1649
+ var publishUnknownFields = null;
1882
1650
  if (receiptAgentId) {
1883
1651
  // The edit went through — a build with a new agent_id appeared
1884
1652
  // after the trigger. The parent's independent read-back
@@ -1897,31 +1665,6 @@ while (i < STEPS.length) {
1897
1665
  }, totalReworkCount);
1898
1666
  } else {
1899
1667
  var newAuditDirs = [];
1900
- // auditReportOk: pure tri-state read of a report.json body —
1901
- // true (build ok), false (build failed), null (missing or
1902
- // unreadable — not evidence either way). The child returns the
1903
- // raw body verbatim; interpretation lives here, never in prose.
1904
- // Defined here so both the immediate and post-poll audit
1905
- // fallbacks share it.
1906
- var auditReportOk = function (raw) {
1907
- if (typeof raw !== "string") return null;
1908
- var trimmed = raw.trim();
1909
- if (trimmed === "" || trimmed === "MISSING") return null;
1910
- var parsed;
1911
- try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
1912
- if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
1913
- return null;
1914
- };
1915
- // (2026-09-16, critic finding 2) When durable audit evidence
1916
- // confirms (or refutes) the build, there is no receipt agent_id
1917
- // to chain the completion poll to — skipReceiptPoll bypasses the
1918
- // poll below, which with a null receipt could only observe
1919
- // strangers or nothing.
1920
- var skipReceiptPoll = false;
1921
- // publishFailure is declared here (moved up from below) so the
1922
- // immediate audit fallback can record an explicit build failure
1923
- // without the later declaration resetting it.
1924
- var publishFailure = null;
1925
1668
  try {
1926
1669
  var auditAfter = await agent(
1927
1670
  "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
@@ -1941,88 +1684,109 @@ while (i < STEPS.length) {
1941
1684
  log("Publish audit-dir re-list after trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
1942
1685
  }
1943
1686
  if (newAuditDirs.length > 0) {
1944
- rebuildTrigger = { edit_started: true };
1945
1687
  rebuildAgentId = null;
1946
1688
  newAuditDirs.sort();
1947
1689
  var newestImmediateDir = newAuditDirs[newAuditDirs.length - 1];
1948
1690
  log("Publish rebuild trigger for task " + taskId + ": new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and a build completed; no in-flight receipt was observed.");
1949
- // (2026-09-16, critic finding 2) Durable audit evidence exists,
1691
+ // (2026-09-18, H2 verdict-first) Durable audit evidence exists,
1950
1692
  // but there is no receipt agent_id to chain the completion poll
1951
1693
  // to — polling with a null receipt can only observe strangers
1952
1694
  // (any running build differs from "null") or nothing, burning
1953
- // 10.5 minutes to park unknown. Read the build report now
1954
- // instead of polling: ok=true confirms completion and routes
1955
- // directly to parent verification (the poll is skipped);
1956
- // ok=false is explicit failure; unreadable is unknown.
1695
+ // 10.5 minutes to park unknown. Read the build report now instead
1696
+ // of polling, then decide the verdict ONCE via
1697
+ // decidePublishVerdict: exactly one new dir with ok=true lands
1698
+ // (attribution by window, not identity — never poll blind);
1699
+ // ok=false is UNKNOWN with the failure evidence preserved in the
1700
+ // ledger detail (the evidence is explicit, the attribution is
1701
+ // not); unreadable / zero / ambiguous dirs are UNKNOWN.
1702
+ // Verdict-first: landed bypasses the poll below, unknown falls
1703
+ // through to post-deploy. STEP 2 runs on every path.
1704
+ // See docs/decisions/publish-path.md#h2-verdict-dispatch.
1957
1705
  var immediateReportOk = null;
1958
- try {
1959
- var immediateOkRead = await agent(
1960
- "Read the artifact build report for slug \"" + PUBLISH_SLUG + "\".\n" +
1961
- "Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestImmediateDir + "/report.json 2>/dev/null || echo MISSING\n" +
1962
- "Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
1963
- { key: attemptKey("publish-audit-ok-immediate-" + taskId, totalReworkCount), label: "Reading build report for audit-confirmed build",
1964
- schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
1965
- );
1966
- immediateReportOk = auditReportOk(immediateOkRead && immediateOkRead.raw);
1967
- } catch (immediateOkErr) {
1968
- log("Publish build-report read for audit-confirmed dir failed for task " + taskId + " (treated as unknown): " + (immediateOkErr && immediateOkErr.message ? immediateOkErr.message : immediateOkErr));
1969
- immediateReportOk = null;
1706
+ // (N1) Read the report only when exactly one new dir exists:
1707
+ // ambiguity (>1) forces UNKNOWN regardless — don't shell out to
1708
+ // read a report that will be discarded.
1709
+ if (newAuditDirs.length === 1) {
1710
+ try {
1711
+ var immediateOkRead = await agent(
1712
+ "Read the artifact build report for slug \"" + PUBLISH_SLUG + "\".\n" +
1713
+ "Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestImmediateDir + "/report.json 2>/dev/null || echo MISSING\n" +
1714
+ "Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
1715
+ { key: attemptKey("publish-audit-ok-immediate-" + taskId, totalReworkCount), label: "Reading build report for audit-confirmed build",
1716
+ schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
1717
+ );
1718
+ immediateReportOk = auditReportOk(immediateOkRead && immediateOkRead.raw);
1719
+ } catch (immediateOkErr) {
1720
+ log("Publish build-report read for audit-confirmed dir failed for task " + taskId + " (treated as unknown): " + (immediateOkErr && immediateOkErr.message ? immediateOkErr.message : immediateOkErr));
1721
+ immediateReportOk = null;
1722
+ }
1970
1723
  }
1971
- if (immediateReportOk === true) {
1724
+ var immediateVerdict = decidePublishVerdict({ editStarted: true, immediateReport: immediateReportOk, newDirCount: newAuditDirs.length });
1725
+ if (immediateVerdict.verdict === "landed") {
1972
1726
  publishBuildLanded = true;
1973
1727
  artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
1974
- skipReceiptPoll = true;
1975
- log("Publish build landed for task " + taskId + " via immediate durable audit evidence (audit dir " + newestImmediateDir + ", report ok=true) — receipt poll skipped (no receipt to chain to), routing directly to parent verification");
1976
1728
  await recordPublishLedger({
1977
1729
  commit: mergeCommitForPublish,
1978
1730
  attempt: rebuildAttemptKey,
1979
1731
  agent_id: null,
1980
1732
  applied_report: publishAppliedObservation,
1981
1733
  outcome: "submitted",
1982
- detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestImmediateDir + ", report ok=true); receipt poll skipped (no receipt agent_id), routed to parent verification"
1983
- }, totalReworkCount);
1984
- } else if (immediateReportOk === false) {
1985
- skipReceiptPoll = true;
1986
- publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestImmediateDir + ", report ok=false — immediate audit evidence, no receipt observed). Explicit negative evidence: a build ran and failed. The publish did not land — provenance was not stamped. Fail-closed.";
1987
- await recordPublishLedger({
1988
- commit: mergeCommitForPublish,
1989
- attempt: rebuildAttemptKey,
1990
- agent_id: null,
1991
- applied_report: publishAppliedObservation,
1992
- outcome: "failed",
1993
- detail: "a build ran and failed: audit dir " + newestImmediateDir + " report ok=false (immediate audit evidence, no receipt)"
1734
+ detail: "durable audit evidence shows a build completed during the attempt window (no receipt agent_id — attribution by window, not identity; receipt poll bypassed (verdict decided), routed to parent verification)"
1994
1735
  }, totalReworkCount);
1736
+ log("Publish verdict LANDED for task " + taskId + ": a completed build was observed during the attempt window — receipt poll bypassed (verdict decided, no receipt to chain to), routing directly to parent verification.");
1995
1737
  } else {
1738
+ // Verdict UNKNOWN on the immediate path. ok=false is explicit
1739
+ // failure evidence but not an attributable failure — the ledger
1740
+ // records unknown with the evidence preserved in the detail,
1741
+ // and the flow continues to post-deploy (never parks early).
1742
+ publishUnknownFields = {
1743
+ taskId: taskId,
1744
+ unattributableReason: immediateVerdict.unattributableReason,
1745
+ pollEndState: "not-polled",
1746
+ sawOurBuild: false,
1747
+ newAuditDirCount: newAuditDirs.length,
1748
+ chunkFailures: 0,
1749
+ pollStatusErrors: 0
1750
+ };
1751
+ var immediateDetail = "durable audit-dir fallback could not prove a completed build for this attempt (unattributable_reason=" + immediateVerdict.unattributableReason + ", no receipt agent_id)";
1752
+ if (immediateReportOk === false) {
1753
+ immediateDetail += "; explicit failure evidence preserved: report ok=false for audit dir " + newestImmediateDir;
1754
+ }
1755
+ if (immediateVerdict.unattributableReason === "audit-dir-ambiguity") {
1756
+ immediateDetail += "; audit-dir ambiguity: " + newAuditDirs.length + " new dirs in window";
1757
+ }
1996
1758
  await recordPublishLedger({
1997
1759
  commit: mergeCommitForPublish,
1998
1760
  attempt: rebuildAttemptKey,
1999
1761
  agent_id: null,
2000
- applied_report: null,
1762
+ applied_report: publishAppliedObservation,
2001
1763
  outcome: "unknown",
2002
- detail: "new audit dir " + newestImmediateDir + " appeared during the trigger window but its build report is unreadable/missing; no receipt agent_id to poll — outcome unknown, fail-closed with no blind retry"
1764
+ detail: immediateDetail
2003
1765
  }, totalReworkCount);
2004
- return await parkTask("Publish outcome unknown for task " + taskId + ": a new audit dir (" + newestImmediateDir + ") appeared during the trigger window but its build report is unreadable, and no in-flight receipt was observed to poll. The edit may have completed. Correlate the accepted edit via the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl — do NOT reissue the edit blindly: if the trigger was accepted, a retry duplicates it (2026-09-12). Verify independently whether the build completed (audit dir + report, or the parent's content read-back) before deciding the next step. Fail-closed.");
1766
+ log("Publish verdict UNKNOWN for task " + taskId + ": " + immediateDetail + " — continuing to post-deploy; never polling blind and never parking early.");
2005
1767
  }
2006
1768
  } else {
2007
- // No attributable build and no durable evidence — but that
2008
- // proves nothing (a fast-completing build can finish between
2009
- // polls, or the checks themselves failed). The outcome is
2010
- // UNKNOWN. No retry: re-issuing the edit here duplicated it on
2011
- // 2026-09-12. Record the attempt durably and park fail-closed;
2012
- // correlate via the ledger, never by guessing from a blind poll.
2013
- log("Publish rebuild trigger for task " + taskId + ": no attributable build observed and no new audit dir — outcome UNKNOWN. Recording the attempt and parking fail-closed; no blind retry.");
1769
+ // See docs/decisions/publish-path.md#no-attributable-build: no attributable build and no durable evidence means no publish.
1770
+ publishUnknownFields = {
1771
+ taskId: taskId,
1772
+ unattributableReason: decidePublishVerdict({ editStarted: true, immediateReport: null, newDirCount: 0 }).unattributableReason,
1773
+ pollEndState: "not-polled",
1774
+ sawOurBuild: false,
1775
+ newAuditDirCount: 0,
1776
+ chunkFailures: 0,
1777
+ pollStatusErrors: 0
1778
+ };
2014
1779
  await recordPublishLedger({
2015
1780
  commit: mergeCommitForPublish,
2016
1781
  attempt: rebuildAttemptKey,
2017
1782
  agent_id: null,
2018
- applied_report: null,
1783
+ applied_report: publishAppliedObservation,
2019
1784
  outcome: "unknown",
2020
- detail: "fire-and-forget trigger; post-trigger build-state poll saw no attributable build (or the check failed) and the audit-dir diff found no new dir; the edit may have been accepted as pending_init"
1785
+ detail: "no new audit dir appeared in the trigger window and no receipt agent_id was observed — the edit was issued fire-and-forget, so completion is unproven; never poll blind on a null receipt"
2021
1786
  }, totalReworkCount);
2022
- return await parkTask("Publish outcome unknown for task " + taskId + ": the rebuild trigger was issued fire-and-forget (no schema, so no validation failure mode; a candidate-parse throw stays possible and is inconclusive), and the follow-up observation could not attribute a build to the edit for slug " + PUBLISH_SLUG + " — no in-flight build with a new agent_id appeared in the poll window and no new audit dir landed. The edit may have been accepted as pending_init, so no retry was issued: a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion — do NOT reissue the edit blindly. Verify independently whether the build completed before deciding the next step. Fail-closed.");
1787
+ log("Publish verdict UNKNOWN for task " + taskId + ": no new audit dir in window and no receipt — continuing to post-deploy; never polling blind and never parking early.");
2023
1788
  }
2024
1789
  }
2025
-
2026
1790
  // Durable publish-attempt ledger: record the trigger outcome while the
2027
1791
  // attempt key and commit are in scope. Every attempt lands here with
2028
1792
  // its outcome — submitted, rejected, or unknown (unknown is recorded
@@ -2032,9 +1796,45 @@ while (i < STEPS.length) {
2032
1796
  // already recorded the ledger's submitted line on both positive paths
2033
1797
  // and parked on unknown — there is no applied report to observe and
2034
1798
  // no rejection signal to record.
2035
- // (publishFailure is declared with the immediate audit fallback
2036
- // above so an explicit build failure there survives to here.)
2037
- if (rebuildTrigger.edit_started && !skipReceiptPoll) {
1799
+ // (2026-09-18, H2 verdict-first) The verdict was decided exactly
1800
+ // once above; dispatch on it. landed bypasses the receipt poll
1801
+ // (the audit evidence already proved completion); an explicit
1802
+ // publishFailure is preserved verbatim through post-deploy.
1803
+ // Otherwise the verdict is open — but the poll below is only
1804
+ // legitimate against a real receipt: the null-safe assertion records
1805
+ // UNKNOWN and continues to STEP 2 instead of polling blind.
1806
+ // See docs/decisions/publish-path.md#h2-verdict-dispatch.
1807
+ if (publishBuildLanded) {
1808
+ log("Publish verdict already LANDED for task " + taskId + " — bypassing receipt poll.");
1809
+ } else if (publishFailure) {
1810
+ // design §1.2 pre-poll branch — currently unassigned; kept for the converged dispatch shape.
1811
+ log("Publish verdict already FAILED for task " + taskId + " — preserved verbatim through post-deploy.");
1812
+ } else if (!rebuildTrigger || !rebuildTrigger.edit_started) {
1813
+ // Loud defensive assertion (replaces the old lying "Unreachable"
1814
+ // else): with no receipt state the poll would observe strangers or
1815
+ // nothing — record UNKNOWN and continue to STEP 2. Never park
1816
+ // early here; never poll blind.
1817
+ if (!publishUnknownFields) {
1818
+ publishUnknownFields = {
1819
+ taskId: taskId,
1820
+ unattributableReason: "no-receipt-state",
1821
+ pollEndState: "not-polled",
1822
+ sawOurBuild: false,
1823
+ newAuditDirCount: 0,
1824
+ chunkFailures: 0,
1825
+ pollStatusErrors: 0
1826
+ };
1827
+ await recordPublishLedger({
1828
+ commit: mergeCommitForPublish,
1829
+ attempt: rebuildAttemptKey,
1830
+ agent_id: null,
1831
+ applied_report: publishAppliedObservation,
1832
+ outcome: "unknown",
1833
+ detail: "no receipt state was recorded for this attempt — never polling blind; continuing to post-deploy"
1834
+ }, totalReworkCount);
1835
+ }
1836
+ log("Publish verdict UNKNOWN for task " + taskId + ": no receipt state recorded — never polling blind, continuing to post-deploy.");
1837
+ } else {
2038
1838
  // (2026-09-16) There is no builder report: the fire-and-forget
2039
1839
  // trigger carries no JSON contract, so there is nothing to
2040
1840
  // compare and no pre-hash diagnostic. The builder's old
@@ -2107,13 +1907,7 @@ while (i < STEPS.length) {
2107
1907
  timeoutMs: 270000 }
2108
1908
  );
2109
1909
  } catch (chunkErr) {
2110
- // A hung or failed chunk is inconclusive, never terminal:
2111
- // record it and continue to the next chunk. (2026-09-16,
2112
- // clean-room task 1febe8eb: the platform's 270s agent
2113
- // timeout killed chunk 2, which threw out of this loop —
2114
- // skipping chunk 3 AND the STEP 1b audit-dir fallback and
2115
- // parking on the exception path.) Fail-closed still applies
2116
- // after chunk 3 and the fallback are exhausted.
1910
+ // See docs/decisions/publish-path.md#hung-chunk: a hung or failed chunk is inconclusive, never terminal.
2117
1911
  chunkFailures.push("chunk " + chunk + ": " + (chunkErr && chunkErr.message ? chunkErr.message : chunkErr));
2118
1912
  log("Artifact build poll chunk " + chunk + " of 3 failed (" + (chunkErr && chunkErr.message ? chunkErr.message : chunkErr) + ") \u2014 continuing to the next chunk; build completion still unproven.");
2119
1913
  }
@@ -2129,20 +1923,7 @@ while (i < STEPS.length) {
2129
1923
  buildPoll = { build_done: false, status: (buildPoll && buildPoll.status) || "build still running after the 10.5-minute bounded poll" };
2130
1924
  }
2131
1925
  if (buildPoll.build_done && pollSawOurBuild) {
2132
- // STEP 1c (mechanical): NO provenance stamp here. Canary run 8
2133
- // (2026-09-11) proved the stamp cannot certify content: the
2134
- // builder's applied-report was derived from the carried diff, so
2135
- // the old report check was circular — a fabricated report
2136
- // passed by construction, and every phase went green on a hollow
2137
- // build. The stamp moves to the parent (docs/publish-verification.md);
2138
- // the deterministic lib/readback-disk.js is the primary sensor
2139
- // (the agent-callable read-back tool is unavailable —
2140
- // artifact_inspect was removed by the platform 2026-09-14 — so
2141
- // the LLM-inspector path is manual-fallback only), and the task
2142
- // parks for parent verification.
2143
- // QA's provenance check enforces the stamp mechanically.
2144
- // An unverified publish fails loudly in QA instead of passing
2145
- // silently here.
1926
+ // See docs/decisions/publish-path.md#step-1c-no-stamp: no provenance stamp in STEP 1c.
2146
1927
  publishBuildLanded = true;
2147
1928
  artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
2148
1929
  log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
@@ -2225,7 +2006,11 @@ while (i < STEPS.length) {
2225
2006
  detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
2226
2007
  }, totalReworkCount);
2227
2008
  } else if (auditOkAfterPoll === false) {
2228
- publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestAuditDirAfterPoll + ", report ok=false). Explicit negative evidence: a build ran and failed (attribution by window, not by build identity — no stranger build was observed in flight during the poll). The publish did not land — provenance was not stamped. Fail-closed.";
2009
+ // Explicit negative evidence under the stranger guard: a build
2010
+ // ran and failed during the poll window with no stranger in
2011
+ // flight. This is an attributable failure — preserved verbatim
2012
+ // through post-deploy. Message contract unchanged.
2013
+ publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestAuditDirAfterPoll + ", report ok=false). Explicit negative evidence: a build ran and failed (attribution by window, not by build identity — no stranger build was observed in flight during the poll). The publish did not land — provenance was not stamped — your change was NOT published. Build log: ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestAuditDirAfterPoll + "/. This explicit failure is not auto-retried. Appendix: attribution=window; report=ok=false; provenance=unstamped.";
2229
2014
  await recordPublishLedger({
2230
2015
  commit: mergeCommitForPublish,
2231
2016
  attempt: rebuildAttemptKey,
@@ -2239,7 +2024,19 @@ while (i < STEPS.length) {
2239
2024
  : ((pollStatusErrors > 0 && !pollSawOurBuild) ? "status-read-errors-during-poll"
2240
2025
  : (pollEndState === "build-still-running-at-poll-end" ? "build-still-running-at-poll-end"
2241
2026
  : (newAuditDirsAfterPoll.length === 0 ? "no-new-audit-dir-in-window" : "audit-report-unreadable-or-missing")));
2242
- publishFailure = "Artifact build completion unproven (fail-closed, no provenance stamped): unattributable_reason=" + unattributableReason + "; poll_end_state=" + pollEndState + "; " + "saw_our_build=" + pollSawOurBuild + "; new_audit_dirs=" + newAuditDirsAfterPoll.length + "; poll_chunks_failed=" + chunkFailures.length + "; poll_status_errors=" + pollStatusErrors + ". Attribution is by window, not by build identity. The publish may or may not have landed. Fail-closed.";
2027
+ // Verdict UNKNOWN (poll path): the build cannot be attributed
2028
+ // to this attempt. The fields feed the single composed
2029
+ // fail-closed park reason after post-deploy — never a
2030
+ // fabricated verdict, never a silent pass.
2031
+ publishUnknownFields = {
2032
+ taskId: taskId,
2033
+ unattributableReason: unattributableReason,
2034
+ pollEndState: pollEndState,
2035
+ sawOurBuild: pollSawOurBuild,
2036
+ newAuditDirCount: newAuditDirsAfterPoll.length,
2037
+ chunkFailures: chunkFailures.length,
2038
+ pollStatusErrors: pollStatusErrors
2039
+ };
2243
2040
  await recordPublishLedger({
2244
2041
  commit: mergeCommitForPublish,
2245
2042
  attempt: rebuildAttemptKey,
@@ -2250,11 +2047,8 @@ while (i < STEPS.length) {
2250
2047
  }, totalReworkCount);
2251
2048
  }
2252
2049
  }
2253
- } else {
2254
- // Unreachable: the observation above either attributes the edit
2255
- // (edit_started) or parks. Defensive only — never a silent pass.
2256
- publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish is unattributed (not proven landed, not proven failed) — provenance was not stamped. Fail-closed.";
2257
2050
  }
2051
+
2258
2052
  } // end: publishSkippedNoLock — no rebuild, no stamp, nothing to ship
2259
2053
  // STEP 2 (mechanical, always — skip path included): post-deploy
2260
2054
  // commits builder leftovers if any, removes the worktree, and releases
@@ -2278,6 +2072,19 @@ while (i < STEPS.length) {
2278
2072
  ? " Post-deploy finalized cleanup."
2279
2073
  : " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
2280
2074
  }
2075
+ if (!publishBuildLanded) {
2076
+ // Verdict UNKNOWN (single fail-closed park — post-deploy always
2077
+ // runs first; no early parks anywhere above). Machine contract:
2078
+ // "unattributable_reason=" and "poll_end_state=" are always
2079
+ // present; attribution is by window, not identity.
2080
+ // (2026-09-18, H2 message contract) Thread the commit short-sha
2081
+ // through the unknown fields so the human line names the trigger
2082
+ // commit (design §1.5: "the trigger was sent for commit <short-sha>").
2083
+ if (publishUnknownFields) publishUnknownFields.commitShortSha = mergeCommitShortForPublish;
2084
+ return await parkTask(composeUnattributedParkReason(publishUnknownFields) + (postDeploy.deployed
2085
+ ? " Post-deploy finalized cleanup."
2086
+ : " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
2087
+ }
2281
2088
  if (!postDeploy.deployed) {
2282
2089
  return await parkTask("Post-deploy failed after the artifact build landed: " + (postDeploy.output || "no output") + ". The build may have landed but worktree cleanup and lock release are unknown — human attention needed.");
2283
2090
  }
@@ -2377,12 +2184,7 @@ while (i < STEPS.length) {
2377
2184
  "BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
2378
2185
  "Report back in plain prose — what checks you ran and their results. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<what the task required>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten).";
2379
2186
  } else if (SURFACE_TERMINAL) {
2380
- // Terminal-surface experiential QA: Hazel drives the CLI herself —
2381
- // the terminal counterpart to the see-act loop above. Same OODA
2382
- // discipline (append-ooda-step with --action terminal and a transcript
2383
- // per step; verdict via write-ooda-verdict), judged against the
2384
- // shared bar resolved via UX_DOCTRINE_PATH (the terminal doctrine page here). Transcripts are delivered;
2385
- // screenshots are never invented for terminal work.
2187
+ // See docs/decisions/qa-reproduce.md#terminal-qa: Hazel drives CLI transcripts for terminal surfaces.
2386
2188
  var termEvidence = crewHome + "/task-evidence/" + taskId + "/postchange";
2387
2189
  var termTargetsLine = terminalTargets || "not declared — derive from --help and the task description";
2388
2190
  instructions = "You are code-blind QA. You NEVER read source files.\n" +
@@ -2606,18 +2408,7 @@ while (i < STEPS.length) {
2606
2408
  }
2607
2409
  log("Build worktree confinement passed: " + wt.path);
2608
2410
 
2609
- // Already-merged idempotency: a `repo_diff: none (already-merged:
2610
- // <sha>)` declaration is verified mechanically — <sha> must resolve
2611
- // and be an ancestor of main in the configured repo. A fabricated or
2612
- // mistaken declaration fails the phase here (the dispatcher retries
2613
- // Build under its consecutive-failure cap); a verified declaration is
2614
- // recorded in alreadyMergedSha for Review's no-diff branch. Without
2615
- // this guard, Build correctly doing nothing left Review with no
2616
- // mechanical way to accept an empty diff, and Cass rejected for "no
2617
- // commits ahead of main — the builder likely forgot to commit" while
2618
- // the deliverable sat on main (canary 2026-09-15, task 1d692d91).
2619
- // The sha is hex-only by construction (extractAlreadyMerged), so
2620
- // interpolating it into the shell command cannot inject.
2411
+ // See docs/decisions/publish-path.md#already-merged-idem: an already-merged repo_diff is idempotent; no rebuild.
2621
2412
  var am = extractAlreadyMerged(workerText);
2622
2413
  if (am.sha) {
2623
2414
  var amCheck = await agent(
@@ -2653,19 +2444,7 @@ while (i < STEPS.length) {
2653
2444
  }
2654
2445
  }
2655
2446
 
2656
- // QA experiential-loop guard (clean-room defect 2026-09-16): Hazel's verdict.json
2657
- // is honest about missing experiential evidence, but the closeout treated a PASS
2658
- // as terminal done even when the experiential loop never ran (playwright-core
2659
- // was unresolvable from the release layout — the dependency lived in the
2660
- // npm install dir, severed from the crew home). A PASS verdict with missing
2661
- // experiential evidence must never be terminal: the task parks fail-closed
2662
- // with unattributable_reason=qa-visual-loop-unavailable (artifact surface)
2663
- // or qa-terminal-loop-unavailable (terminal surface) instead of
2664
- // transitioning to done. Code, not prompt text: the check reads the
2665
- // machine-readable verdict via lib/read-ooda-verdict.js, which reports
2666
- // visual_loop_unavailable from the OODA log's NOT POSSIBLE browser steps
2667
- // and terminal_loop_unavailable from NOT POSSIBLE terminal steps, plus the
2668
- // verdict's missing_evidence tool-unavailability notes.
2447
+ // See docs/decisions/qa-reproduce.md#experiential-loop-guard: a PASS with missing experiential evidence parks fail-closed.
2669
2448
  if (step.name === "QA" && qaExperiential && verdictPassed === true) {
2670
2449
  var qaLoopDir = crewHome + "/task-evidence/" + taskId + "/postchange";
2671
2450
  var qaLoopSurface = SURFACE_TERMINAL ? "terminal" : "visual";
@@ -2717,20 +2496,7 @@ while (i < STEPS.length) {
2717
2496
  };
2718
2497
  let passed = stepResult.passed === true;
2719
2498
 
2720
- // Deterministic integrate verification: the agent cannot self-certify a
2721
- // merge. After the Integrate agent claims success, the workflow confirms
2722
- // mechanically that the task branch tip is an ancestor of main via the
2723
- // lifecycle script's verify-merge command (which resolves the branch
2724
- // through the crew registry — never by reconstructing "task/"+taskId — so
2725
- // the check cannot verify the wrong branch). The VERIFIED marker is matched
2726
- // by regex on the script's own stdout; agent prose is never read. This
2727
- // closes the hole where an agent reported "merged empty" while approved
2728
- // commits were still stranded on the task branch (bug b1b1f919). A genuine
2729
- // empty-diff Integrate (MERGED_EMPTY: no commits ahead of main) verifies
2730
- // vacuously — the tip is then an ancestor of main. Verification failure is
2731
- // an operational step failure, not a park: the dispatcher retries Integrate
2732
- // under its consecutive-failure cap, and the retry finds the commits still
2733
- // on the branch and performs the real merge — self-healing.
2499
+ // See docs/decisions/publish-path.md#integrate-verify: the agent cannot verify integrate mechanically; the workflow checks the diff.
2734
2500
  if (step.name === "Integrate" && passed) {
2735
2501
  var integrateVerifyOut = "";
2736
2502
  try {
@@ -2791,17 +2557,7 @@ while (i < STEPS.length) {
2791
2557
  // dispatcher's consecutive-failure cap.
2792
2558
  const status = passed ? "completed" : (step.name === "Integrate" || step.name === "Publish" || step.name === "Map" ? "failed" : "rejected");
2793
2559
 
2794
- // Deterministic publish verification: the agent cannot self-certify a publish.
2795
- // Skip-aware (park 2026-09-11): when the deterministic publish script found
2796
- // no merge lock held (empty-diff Integrate), it skips the publish path
2797
- // gracefully and emits the machine-readable PUBLISH_SKIPPED=no-lock-held
2798
- // marker. The preflight (bugfix 2026-09-17) emits
2799
- // PUBLISH_SKIPPED=no-npm-publish when npm publish is not configured on
2800
- // this machine (helper or credential absent) — also before any mutation.
2801
- // Verification is then vacuous — nothing was shipped, and the
2802
- // registry must NOT have moved. The marker is script-emitted explicit state
2803
- // (pasted verbatim per the Publish agent instructions), not agent prose; a
2804
- // report without the marker still runs the full verification fail-closed.
2560
+ // See docs/decisions/publish-path.md#publish-verify: the agent cannot verify publish; the parent does.
2805
2561
  var publishVerified = false;
2806
2562
  var npmPublishSkipped = false;
2807
2563
  if (step.name === "Publish" && PUBLISH_TYPE === "npm" && publishTarget) {
@@ -2827,15 +2583,7 @@ while (i < STEPS.length) {
2827
2583
  publishTarget.target + " (" + publishTarget.base + " + " + publishTarget.scope + "). The publish did not land.");
2828
2584
  }
2829
2585
  publishVerified = true;
2830
- // Provenance refresh for self-publishes (task 7946d2a4): the npm path
2831
- // installs and activates a new immutable release (crew-release.sh deploy
2832
- // swaps the `current` symlink inside publish-npm.sh) but never stamped
2833
- // the dashboard's provenance record — every crew release left
2834
- // crew_release pointing at a pruned release. After a verified landed
2835
- // publish, refresh the record's crew_release to the now-live release
2836
- // identity, preserving the existing source_commit (the dashboard
2837
- // artifact's build source — a crew-repo commit here would fail the
2838
- // dashboard QA source check).
2586
+ // See docs/decisions/publish-path.md#provenance-refresh: refresh crew_release after landed publish, preserving source_commit.
2839
2587
  try {
2840
2588
  var provRefresh = await agent(
2841
2589
  "Run in shell and return the stdout verbatim:\n" + crewCmd("get-provenance", { project_id: LAUNCH_PROJECT_ID }) + "\n" +
@@ -2865,29 +2613,10 @@ while (i < STEPS.length) {
2865
2613
  } // end: !npmPublishSkipped — a skipped publish has nothing to verify
2866
2614
  }
2867
2615
 
2868
- // Publish content verification — parent-owned (docs/publish-verification.md).
2869
- // The old block read back the workflow's OWN provenance stamp and compared
2870
- // it to HEAD: that verifies the stamp, not the content. Canary run 8
2871
- // (2026-09-11) passed it with a hollow build — the stamp was honest, the
2872
- // artifact was stale, all eight phases green. The stamp now moves to the
2873
- // parent (docs/publish-verification.md); the independent read-back step
2874
- // is currently unavailable (no agent-callable read-back tool exists —
2875
- // artifact_inspect was removed by the platform 2026-09-14), so the parent
2876
- // cannot confirm content and the task parks for verification. QA's
2877
- // provenance check enforces the stamp — an unverified publish fails loudly
2878
- // there instead of passing silently here.
2879
- // Skip-aware (park 2026-09-11): an empty-diff Integrate takes no merge
2880
- // lock, and the deterministic publish path skips rebuild/stamp entirely —
2881
- // there is no new content to verify, so verification is vacuous.
2882
- // publishSkippedNoLock is workflow-computed state from the explicit
2883
- // lock-status read in STEP 0, not agent prose.
2884
- // The parent (tick worker) triggers the ONE read-back inspection it can
2885
- // actually receive (async results go to the root agent, never into a
2886
- // workflow run — a workflow-side trigger would be an orphan). The workflow
2887
- // only parks; the parent's scan builds the request deterministically via
2888
- // lib/build-readback-request.js and ferries the inspection.
2889
- // publishBuildLanded and publishSkippedNoLock are workflow-computed state;
2890
- // a skipped or failed publish has nothing to verify.
2616
+ // Parent-owned verification (docs/publish-verification.md): the parent's
2617
+ // scan builds the request via lib/build-readback-request.js and ferries
2618
+ // the inspection; a skipped or failed publish has nothing to verify.
2619
+ // See docs/decisions/publish-path.md#parent-owned-verification for history.
2891
2620
 
2892
2621
  // Session notes. Machine-readable marker lines are extracted from the full
2893
2622
  // worker report and appended AFTER the slice so a long report can never
@@ -2906,12 +2635,7 @@ while (i < STEPS.length) {
2906
2635
  } else {
2907
2636
  summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
2908
2637
  }
2909
- // Already-merged attestation: when the Build gate verified the builder's
2910
- // already-merged declaration, the workflow records its own marker line in
2911
- // the session notes (like the builder markers above, it is appended after
2912
- // the slice so it can never be amputated). A later run resumed at Review
2913
- // hydrates alreadyMergedSha from this workflow-attested line — never from
2914
- // the builder's declaration alone.
2638
+ // See docs/decisions/publish-path.md#already-merged: when the Build gate verifies already-merged, attestation is recorded.
2915
2639
  if (step.name === "Build" && alreadyMergedSha) {
2916
2640
  summary += "\nalready_merged_verified: " + alreadyMergedSha;
2917
2641
  }
@@ -3008,18 +2732,17 @@ while (i < STEPS.length) {
3008
2732
  return { status: "failed", task_id: taskId, reason: "Map spec verification failed: no spec file at " + SPEC_PATH };
3009
2733
  }
3010
2734
 
3011
- // Publish verification park: the build landed and post-deploy finalized,
3012
- // but provenance is UNSTAMPED until the parent's independent read-back
3013
- // (docs/publish-verification.md) confirms the artifact's actual content
3014
- // matches the merged diff. The parent stamps provenance, then re-queues;
3015
- // the dispatcher resumes at QA, whose provenance check enforces the stamp
3016
- // mechanically. A failed Publish never reaches this park — it returned
3017
- // failed above and retries under the dispatcher's cap. The merge lock is
3018
- // already released (post-deploy), so the parked task holds no resources.
2735
+ // See docs/decisions/publish-path.md#verification-park: the build landed but provenance is unstamped until parent verification.
3019
2736
  if (passed && step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG && !publishSkippedNoLock && publishBuildLanded) {
2737
+ // Success contract (2026-09-18, H2): the parent asked for exactly
2738
+ // one build for this publish. The build landed — do NOT republish: a
2739
+ // duplicate build would re-publish the same change. content_check=pending
2740
+ // means the parent's independent read-back has not happened yet; this
2741
+ // park is NOT proof the content is correct.
3020
2742
  return await parkTask("publish: verification-requested " + mergeCommitForPublish +
3021
- " (build " + (rebuildAgentId || "agent_id unobserved") + ")" +
3022
- " — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md.");
2743
+ " Do NOT republish: a duplicate build would re-publish the same change. " +
2744
+ "Artifact build landed, post-deploy finalized, provenance not stamped — the crew has not yet independently confirmed the live artifact contains exactly the change; waiting on the manual read-back in docs/publish-verification.md. " +
2745
+ "Appendix: build=" + (rebuildAgentId || "agent_id unobserved") + "; provenance=unstamped; content_check=pending.");
3023
2746
  }
3024
2747
 
3025
2748
  i++;