muse-crew 0.14.6 → 0.14.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -155,6 +155,62 @@ function extractVerdict(workerText) {
155
155
  if (uniq.length !== 1) return { ok: false, count: matches.length };
156
156
  return { ok: true, passed: last.value === "PASS" };
157
157
  }
158
+ // Room #26 blocker 38 (2026-09-21): content-finding attribution. Hazel
159
+ // reports user-visible content observations in a machine-readable
160
+ // CONTENT-FINDINGS block; the workflow classifies each against the task's
161
+ // publish diff — mechanical set membership, never prose judgment. A finding
162
+ // is attributable to the task's change iff its subject file is in the
163
+ // publish diff. The special subject "live-data" (content in the artifact's
164
+ // runtime data stores — decks, cards, rows) is never in a git diff, so it
165
+ // always classifies environment-attributable: the crew has no live-DB write
166
+ // path (blocker 32: QA writes go to a fresh per-run temp DB), so live
167
+ // content is never the task's change. Findings are always recorded — never
168
+ // suppressed for looking audit-y — but environment-attributable findings
169
+ // never park the journey and never spawn an artifact bugfix. The
170
+ // platform-side half (audit sessions writing to production via the
171
+ // shared-concurrent path) is out of crew scope; see
172
+ // docs/decisions/qa-reproduce.md#content-finding-attribution.
173
+ function extractContentFindings(workerText) {
174
+ // The block is the LAST "CONTENT-FINDINGS:" line; the JSON array follows
175
+ // on the same line. Fail closed (ok:false) when missing or malformed —
176
+ // the caller blocks the phase for retry; unknown attribution must never
177
+ // degrade to "no findings" (the prose grounds stay preserved in the
178
+ // session notes).
179
+ var text = workerText || "";
180
+ var regex = /^CONTENT-FINDINGS:\s*(\[.*\])\s*$/gim;
181
+ var matches = [];
182
+ var m;
183
+ while ((m = regex.exec(text)) !== null) {
184
+ matches.push(m[1]);
185
+ }
186
+ if (matches.length === 0) return { ok: false, reason: "no CONTENT-FINDINGS block" };
187
+ var findings;
188
+ try {
189
+ findings = JSON.parse(matches[matches.length - 1]);
190
+ } catch (e) {
191
+ return { ok: false, reason: "CONTENT-FINDINGS block is not valid JSON" };
192
+ }
193
+ if (!Array.isArray(findings)) return { ok: false, reason: "CONTENT-FINDINGS is not an array" };
194
+ for (var i = 0; i < findings.length; i++) {
195
+ var f = findings[i];
196
+ if (!f || typeof f !== "object" || typeof f.subject !== "string" || typeof f.observation !== "string") {
197
+ return { ok: false, reason: "finding " + i + " needs string subject and observation" };
198
+ }
199
+ }
200
+ return { ok: true, findings: findings };
201
+ }
202
+ function classifyContentFinding(finding, diffFiles) {
203
+ // diffFiles: the task's publish-diff file set (repo-relative paths).
204
+ // Pure set membership — no prose, no judgment.
205
+ var subject = finding.subject;
206
+ if (subject === "live-data") {
207
+ return { attribution: "environment-attributable", reason: "live-data is never in a git publish diff — the crew has no live-DB write path (blocker 32)" };
208
+ }
209
+ if (diffFiles.indexOf(subject) !== -1) {
210
+ return { attribution: "task-change", reason: "subject is in the task's publish diff" };
211
+ }
212
+ return { attribution: "environment-attributable", reason: "subject is not in the task's publish diff" };
213
+ }
158
214
  // See docs/decisions/workflow-core.md#verdict-reask: a report that fails extractVerdict gets bounded re-ask calls.
159
215
  function verdictReaskKey(stepName, reworkSuffix, attempt) {
160
216
  return "verdict-reask-" + stepName + reworkSuffix + "-a" + attempt;
@@ -458,13 +514,6 @@ function extractMarkerLines(workerText) {
458
514
  return markers.join("\n");
459
515
  }
460
516
 
461
- // See docs/decisions/publish-path.md#already-merged-idem2: idempotency for already-merged tasks.
462
- function extractAlreadyMerged(workerText) {
463
- var m = /repo_diff:\s*none\s*\(already-merged:\s*([0-9a-f]{7,40})(?=[\s)]|$)/i.exec(workerText || "");
464
- return m ? { sha: m[1].toLowerCase() } : { sha: null };
465
- }
466
-
467
-
468
517
  // See docs/decisions/qa-reproduce.md#worktree-confinement: the Build agent must declare its worktree.
469
518
  function extractWorktree(workerText) {
470
519
  var lines = (workerText || "").split("\n");
@@ -671,12 +720,24 @@ let mapGateBounceCount = 0;
671
720
  // rationalized a skip against explicit instruction text — text alone did not
672
721
  // hold, so the decision now lives in workflow code, not agent judgment.
673
722
  let releaseDecision = null; // { release: "yes"|"no", version_bump: "patch"|"minor"|"major"|null }
674
- // Already-merged idempotency: the verified sha from the builder's
675
- // `repo_diff: none (already-merged: <sha>)` declaration (null when the
676
- // builder made commits or declared a runtime-state deliverable). The
677
- // workflow verifies the sha is an ancestor of the integration target at Build closeout;
678
- // Review's no-diff branch reads this, never the builder's prose.
723
+ // Already-merged attribution (blocker 34, 2026-09-21): the sha of the
724
+ // task-attributed merge when classify-branch reports already-merged.
725
+ // Set ONLY from the classifier's mechanical output — never from agent
726
+ // prose. The agent-authored sha-bearing repo_diff declaration path is
727
+ // deleted; attribution is identity-first from git, then the task's own
728
+ // durable merge records.
679
729
  let alreadyMergedSha = null;
730
+ // Room #26 blocker 34 (redesign, 2026-09-21): mechanical branch-state
731
+ // classification at Review entry. Git facts are never adjudicated by a
732
+ // reviewer — the workflow classifies the branch with `classify-branch`
733
+ // (pure git) before Cass is dispatched: "has-work",
734
+ // "already-merged:<sha>", or "empty-no-work". An empty branch with no
735
+ // verified sha and no `repo_diff: none` claim fails mechanically without
736
+ // dispatching Cass; reviewer prose is never parsed for git identity.
737
+ let branchState = null; // "has-work" | "already-merged:<sha>" | "empty-no-work"
738
+ let buildClaimedNoDiff = false; // Build declared plain `repo_diff: none` (no repo change — runtime-state deliverable, or believed already-merged). Same-process Build report only; there is no cross-process hydration — an empty branch on a fresh dispatch with no task-attributed merge fails mechanically.
739
+ let versionTouched = false; // npm: branch changed package.json's `version` (mechanical git fact, blocker-34 follow-up)
740
+ let branchStateErr = ""; // classifier failure detail (fail-closed grounds)
680
741
  // Deterministic publish target — computed by the workflow (registry base +
681
742
  // bumpVersion), never by the Publish agent.
682
743
  let publishTarget = null; // { base, scope, target }
@@ -1215,55 +1276,126 @@ while (i < STEPS.length) {
1215
1276
  "cd " + WORKTREE_HINT + "\n" +
1216
1277
  "git add -A\n" +
1217
1278
  "git commit -m \"fix: " + safeTitle + "\"\n\n" +
1218
- "If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of the integration target and declare `repo_diff: none` in your report, naming the runtime-state deliverable. If you verified the deliverable is already on the integration target (a prior merge or hand-repair landed it — do NOT re-implement working code), make no commit and declare `repo_diff: none (already-merged: <sha>)` naming the integration-target commit that carries the work; the workflow verifies the sha is an ancestor of the integration target, and a false declaration fails the phase. Otherwise commit your changes normally.\n\n" +
1279
+ "If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of the integration target and declare `repo_diff: none` in your report, naming the runtime-state deliverable. If you verified the deliverable is already on the integration target (a prior merge landed it — do NOT re-implement working code), make no commit and declare `repo_diff: none` in your report. The workflow verifies the claim mechanically from git — the task's own merge records, never your declaration, are the proof of delivery. Otherwise commit your changes normally.\n\n" +
1219
1280
  (rejectionNotes ? "This is REWORK after rejection. Address these specific issues:\n" + rejectionNotes + "\n\n" : "") +
1220
1281
  "Report back in plain prose: what you built and the outcome." +
1221
1282
  (PUBLISH_TYPE === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
1222
1283
 
1223
1284
  } else if (step.name === "Review") {
1224
- // See docs/decisions/publish-path.md#already-merged-hydra: when this run did not execute, hydration uses the existing merge.
1225
- if (!alreadyMergedSha) {
1226
- var hydResult = null;
1227
- try {
1228
- hydResult = await agent(
1229
- "Read the latest completed Build session for task " + taskId + ".\n" +
1230
- "Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
1231
- "In the returned sessions array, find the most recent session (by started_at) with task_id \"" + taskId + "\", step \"Build\", and status \"completed\". Return exactly two sections, verbatim, with no commentary:\n" +
1232
- "SHA: <the session's already_merged_sha field value, or the word null when it is null>\n" +
1233
- "NOTES:\n<the session's notes field, verbatim>",
1234
- { key: "hydrate-already-merged" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Hydrating already-merged verification" }
1235
- );
1236
- } catch (hydErr) {
1237
- log("Hydration read failed (" + String(hydErr && hydErr.message || hydErr) + "); treating as none declared.");
1285
+ // Room #26 blocker 34 (redesign, 2026-09-21): classify the branch
1286
+ // mechanically BEFORE Cass is dispatched. The classifier is pure git
1287
+ // (classify-branch in the lifecycle script): has-work |
1288
+ // already-merged:<sha> | empty-no-work. Attribution is identity-first
1289
+ // from git, then the task's own durable merge records — there is no
1290
+ // agent-authored declared-sha input and no cross-process hydration
1291
+ // ferry (the re-entry case it served is covered by the durable record
1292
+ // → classifier directly). An unparseable result fails closed as
1293
+ // empty-no-work — an empty branch never passes Review silently, and
1294
+ // the bounce to Build gives the next round a chance to classify.
1295
+ try {
1296
+ var bsResult = await agent(
1297
+ "Classify the task branch state. This is a mechanical git check, not a judgment call.\n" +
1298
+ "Run in shell and return the stdout verbatim, then the single stderr line starting with `DIAG:` verbatim:\n" +
1299
+ LIFECYCLE_ENV + LIFECYCLE + " classify-branch " + taskId,
1300
+ { key: "classify-branch" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Classifying branch state (mechanical)" }
1301
+ );
1302
+ var bsStr = (typeof bsResult === "string") ? bsResult : JSON.stringify(bsResult);
1303
+ var bsMatch = /^\s*BRANCH_STATE:\s*(has-work|already-merged:[0-9a-f]{40}|empty-no-work)\s*$/im.exec(bsStr);
1304
+ if (bsMatch) {
1305
+ branchState = bsMatch[1].toLowerCase();
1306
+ } else {
1307
+ branchStateErr = "unparseable classifier output: " + bsStr.slice(0, 120);
1238
1308
  }
1239
- var hydStr = hydResult ? ((typeof hydResult === "string") ? hydResult : JSON.stringify(hydResult)) : "";
1240
- var hydSha = /^SHA:\s*([0-9a-f]{7,40})\s*$/im.exec(hydStr);
1241
- if (hydSha) {
1242
- alreadyMergedSha = hydSha[1].toLowerCase();
1243
- log("Hydrated already-merged verification from structured session field: " + alreadyMergedSha);
1309
+ // DIAG pin (blocker 34 design v2.1): the merge-record scan is
1310
+ // load-bearing with no fallback — the classifier's stderr counter is
1311
+ // logged verbatim so silent record loss surfaces in the evidence.
1312
+ var bsDiagMatch = /^[ \t]*DIAG:\s*classify-branch\s+records=\d+\s+resolvable=\d+[ \t]*$/im.exec(bsStr);
1313
+ if (bsDiagMatch) {
1314
+ log(bsDiagMatch[0].trim());
1244
1315
  } else {
1245
- var hvm = /already_merged_verified:\s*([0-9a-f]{7,40})/i.exec(hydStr);
1246
- if (hvm) {
1247
- alreadyMergedSha = hvm[1].toLowerCase();
1248
- log("Hydrated already-merged verification from Build session notes (fallback): " + alreadyMergedSha);
1249
- }
1316
+ log("classify-branch: no DIAG: record-scan counter returned (scan unverifiable)");
1250
1317
  }
1318
+ } catch (bsErr) {
1319
+ branchStateErr = "classifier call failed: " + String(bsErr && bsErr.message || bsErr).slice(0, 120);
1320
+ }
1321
+ if (!branchState) {
1322
+ branchState = "empty-no-work";
1323
+ log("Branch-state classification failed (" + branchStateErr + ") — failing closed as empty-no-work");
1324
+ } else if (branchState.indexOf("already-merged:") === 0) {
1325
+ alreadyMergedSha = branchState.slice("already-merged:".length);
1326
+ log("Branch state already-merged: " + alreadyMergedSha + " (mechanically verified) — Cass reviews the frozen merge diff");
1327
+ } else {
1328
+ log("Branch state: " + branchState);
1329
+ }
1330
+ // buildClaimedNoDiff was set at the Build gate from the full Build
1331
+ // report (same process only). Only an empty branch WITH this claim
1332
+ // goes to Cass for a plausibility judgment; without it, empty-no-work
1333
+ // fails mechanically and Cass is never dispatched.
1334
+ // Blocker-34 follow-up: the package.json `version` check is a mechanical
1335
+ // git fact, not a reviewer judgment. For npm projects the workflow checks
1336
+ // it here, before Cass is dispatched — a touched `version` fails Review
1337
+ // mechanically (versions are assigned at publish time, never in
1338
+ // branches). Fail closed: anything but an explicit VERSION_CLEAN counts
1339
+ // as touched.
1340
+ if (PUBLISH_TYPE === "npm") {
1341
+ try {
1342
+ var vtResult = await agent(
1343
+ "Check whether the task branch changed package.json's `version` field. This is a mechanical git check, not a judgment call.\n" +
1344
+ "Run in shell and return the stdout verbatim:\n" +
1345
+ LIFECYCLE_ENV + LIFECYCLE + " version-check " + taskId,
1346
+ { key: "version-check" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Checking package.json version (mechanical)" }
1347
+ );
1348
+ var vtStr = (typeof vtResult === "string") ? vtResult : JSON.stringify(vtResult);
1349
+ var vtMatch = /^\s*VERSION_(TOUCHED|CLEAN)\s*$/im.exec(vtStr);
1350
+ versionTouched = !(vtMatch && vtMatch[1].toUpperCase() === "CLEAN");
1351
+ } catch (vtErr) {
1352
+ versionTouched = true;
1353
+ log("Version check failed (" + String(vtErr && vtErr.message || vtErr).slice(0, 120) + ") — failing closed as touched");
1354
+ }
1355
+ log("Review: package.json `version` " + (versionTouched ? "touched by the branch — mechanical FAIL, Cass not dispatched" : "untouched — mechanical check clean"));
1356
+ }
1357
+ // The empty-branch rule Cass used to adjudicate is gone. What she
1358
+ // examines and what rule she applies depend on the classified state —
1359
+ // never on her own judgment of git identity.
1360
+ var reviewChangeExam, reviewBranchRule;
1361
+ if (branchState === "has-work") {
1362
+ reviewChangeExam =
1363
+ "Examine the code changes by running:\n" +
1364
+ LIFECYCLE_ENV + LIFECYCLE + " inspect " + taskId + "\n\n" +
1365
+ "The inspect output is authoritative: it prints the task branch's actual tip commit (TIP) and every commit ahead of the integration target (inspect prints the target name). Base your review ONLY on this output — do NOT run git log yourself to pick commits, and do NOT discuss commit hashes from any other source (they may come from stale rework rounds or a different repo).\n\n";
1366
+ reviewBranchRule =
1367
+ "MECHANICAL FACT (computed by the workflow from git — never by a reviewer): branch state = has-work. The branch has commits ahead of the integration target. Review their content; branch emptiness is settled and is not yours to judge.\n";
1368
+ } else if (branchState.indexOf("already-merged:") === 0) {
1369
+ reviewChangeExam =
1370
+ "The change under review is the frozen merge " + alreadyMergedSha + " — it is already on the integration target (see MECHANICAL FACT below). Examine it by running:\n" +
1371
+ "cd " + REPO_PATH + " && git diff " + alreadyMergedSha + "^1 " + alreadyMergedSha + "\n\n" +
1372
+ "That first-parent diff is the frozen, attributable change. Base your review ONLY on this diff — do NOT run git log to pick commits, and do NOT discuss commit hashes from any other source (they may come from stale rework rounds or a different repo).\n\n";
1373
+ reviewBranchRule =
1374
+ "MECHANICAL FACT (computed by the workflow from git — never by a reviewer): branch state = already-merged:" + alreadyMergedSha + ". The deliverable landed on the integration target via this task's own prior merge " + alreadyMergedSha + " (verified ancestor of the integration target). The branch is empty by design — emptiness is settled fact, not a finding.\n";
1375
+ } else if (buildClaimedNoDiff) {
1376
+ reviewChangeExam =
1377
+ "The branch has no commits ahead of the integration target (see MECHANICAL FACT below) — there is no diff to examine. Judge the Build report's `repo_diff: none` claim on its plausibility.\n\n";
1378
+ reviewBranchRule =
1379
+ "MECHANICAL FACT (computed by the workflow from git — never by a reviewer): the branch has no commits ahead of the integration target and no task-attributed merge on the target was verified; the Build report declares `repo_diff: none` (no repo change). Approve ONLY if the task's deliverable is plausibly runtime state (e.g. a cron definition, scheduler change, or dashboard/config state created outside the repo). Otherwise report 'no commits ahead of the integration target and no plausible runtime-state deliverable — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n";
1380
+ } else {
1381
+ // Mechanical FAIL — Cass is never dispatched (see the dispatch gate
1382
+ // below). Instructions are unused; keep the shape.
1383
+ reviewChangeExam = "";
1384
+ reviewBranchRule = "";
1251
1385
  }
1252
1386
  instructions = "Review independently and cold. You have NOT seen any reasoning from the builder.\nDo NOT access the task dashboard, event log, or any comments. Your review is based solely on the spec and the code.\n\n" +
1253
1387
  (mapperSpec ? "MAPPER'S SPEC (the builder was asked to implement exactly this):\n" + mapperSpec + "\n\n" : "Read the spec (from the task description or spec files under " + crewHome + "/).\n\n") +
1254
- "Examine the code changes by running:\n" +
1255
- LIFECYCLE_ENV + LIFECYCLE + " inspect " + taskId + "\n\n" +
1256
- "The inspect output is authoritative: it prints the task branch's actual tip commit (TIP) and every commit ahead of the integration target (inspect prints the target name). Base your review ONLY on this output — do NOT run git log yourself to pick commits, and do NOT discuss commit hashes from any other source (they may come from stale rework rounds or a different repo).\n\n" +
1388
+ reviewChangeExam +
1257
1389
  "You can also read specific files in the worktree at:\n" +
1258
1390
  WORKTREE_HINT + "/\n\n" +
1259
1391
  "Check quality, correctness, and spec compliance.\n" +
1260
1392
  (SURFACE_TERMINAL ? "TERMINAL UX REVIEW: judge the CLI surface against " + UX_DOCTRINE_PATH + " — help accuracy, error quality, exit codes, output clarity. Reject when the bar is not met.\n" : "") +
1261
1393
  (SURFACE_ARTIFACT ? "ARTIFACT UX REVIEW: judge the rendered surface against " + UX_DOCTRINE_PATH + " — alignment, spacing, hierarchy, composition, balance, finish, correctness. Reject when the bar is not met.\n" : "") +
1262
1394
  "Check that public-affecting changes have matching public doc updates (API.md or the published API contract). If the docs are missing or inaccurate, report what is stale, then end your report with exactly this line: VERDICT: FAIL.\n" +
1263
- "If the branch has no commits ahead of the integration target (inspect shows an empty commit log), approve ONLY if the Build summary declares `repo_diff: none` with (a) a plausible runtime-state deliverable (e.g. a cron created via the cron tool), or (b) an already-merged declaration `repo_diff: none (already-merged: <sha>)` AND the mechanical fact below confirms the sha verified. MECHANICAL FACT (computed by the workflow, never by the builder): already_merged sha = " + (alreadyMergedSha ? alreadyMergedSha + " (verified ancestor of the integration target: YES)" : "none declared") + ". Otherwise report 'no commits ahead of the integration target and no valid repo_diff: none declaration — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n" +
1264
- (PUBLISH_TYPE === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry, and versions are assigned at publish time — never in branches. Two checks:\n" +
1265
- "(a) The task branch must NOT have changed package.json's `version` field. First resolve the integration target: " + LIFECYCLE_ENV + LIFECYCLE + " integration-target. Then check: cd " + REPO_PATH + " && git diff <target>..." + TASK_BRANCH + " -- package.json (substitute the exact token integration-target printed — when it prints HEAD, use the literal word HEAD: `git diff HEAD...<branch>` diffs the detached checkout against the branch). If the branch touched `version` in any way, report 'versions are assigned at publish time, never in branches — remove the version change' in your notes, then end your report with exactly this line: VERDICT: FAIL.\n" +
1266
- "(b) The accepted Build report declares: " + releaseDecisionText() + ". " +
1395
+ reviewBranchRule +
1396
+ (PUBLISH_TYPE === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry, and versions are assigned at publish time — never in branches.\n" +
1397
+ "MECHANICAL FACT (computed by the workflow from git — never by a reviewer): the task branch did not change package.json's `version` field. A branch that touches `version` fails Review mechanically before any reviewer is dispatched, so this is settled — do not re-check it.\n" +
1398
+ "The accepted Build report declares: " + releaseDecisionText() + ". " +
1267
1399
  (releaseDecision
1268
1400
  ? "Validate this decision against the change: release must be 'yes' when the change is consumer-observable and 'no' when internal-only; the version_bump scope must fit the change (patch for fixes, minor for new behavior, major for breaking changes). If the decision is wrong or mis-scoped, report your notes, then end with exactly this line: VERDICT: FAIL."
1269
1401
  : "The decision is missing or malformed — report 'Build report must end with release: yes|no and (when release is yes) version_bump: patch|minor|major lines', then end with exactly this line: VERDICT: FAIL.") + "\n" : "") +
@@ -1277,8 +1409,9 @@ while (i < STEPS.length) {
1277
1409
  "Then run: "+ LIFECYCLE_ENV + "WORKFLOW_RUN_ID=" + lockHolder + " CREW_STAGED_BASE=<provenance.source_commit, empty when null> integrate " + taskId + " \"merge: fix: " + safeTitle + "\\\"\\n" +
1278
1410
  "The script merges, reconciles, records, and pushes under the merge lock. Informational only: RECONCILED, NO_REMOTE, NO_REMOTE_RECONCILE. Read the output:\n" +
1279
1411
  "- STAGED_BASE_MISMATCH: the line left the staged base (stray checkout). VERDICT: FAIL.\n" +
1280
- "- MERGED_EMPTY — check FIRST (it contains the word MERGED): either a runtime-state deliverable or an already-merged sha the workflow verified (Build declared repo_diff: none). No lock taken, no new commit. Report 'merged empty: no repo changes'. VERDICT: PASS.\n" +
1412
+ "- MERGED_EMPTY — check FIRST (it contains the word MERGED): either a runtime-state deliverable or a task-attributed merge the workflow verified from the task's own merge records. No lock taken, no new commit. Report 'merged empty: no repo changes'. VERDICT: PASS.\n" +
1281
1413
  "- MERGED: the merge landed; the script pushed inline. Read the push line: PUSHED — report the hash (detached prints PUSHED: origin/main (refspec HEAD:main)), VERDICT: PASS; NO_REMOTE_PUSH — no remote configured, VERDICT: PASS; ERROR after MERGED — a retry recovers the push; report it, VERDICT: FAIL.\n" +
1414
+ "- STALE_MERGE: the recorded merge no longer matches the task branch (rework moved it after the merge) — the script refused to push stale state. Report the STALE_MERGE line, VERDICT: FAIL.\n" +
1282
1415
  "- LOCK_HELD: 10-minute backoff exhausted. VERDICT: FAIL.\n" +
1283
1416
  "- CONFLICT: the plain merge failed — aborted, the target is clean, and your task still holds the merge lock. Do NOT fail yet. Resolve it:\n" +
1284
1417
  "RESOLUTION:\n" +
@@ -1328,8 +1461,9 @@ while (i < STEPS.length) {
1328
1461
  // See docs/decisions/publish-path.md#deterministic-artifact-publish: the work agent never publishes; the parent runs the deterministic publish script.
1329
1462
  // Blocker 22 (one-party publish, room #24): verified re-entry guard.
1330
1463
  // The workflow parks at publish intent; the tick worker issues
1331
- // artifact_edit, verifies by read-back, stamps provenance, and
1332
- // re-queues. The dispatcher resumes this task at the Publish step
1464
+ // artifact_edit; the ack scan stamps provenance on exact per-attempt
1465
+ // version acknowledgement (sole positive completion criterion; no
1466
+ // read-back — docs/publish-verification.md) and re-queues. The dispatcher resumes this task at the Publish step
1333
1467
  // (the parked session's own step — the failed→retry path). If the
1334
1468
  // stamped provenance names THIS task and its source_commit is the
1335
1469
  // repo's current HEAD, the merge commit was already published and
@@ -1385,8 +1519,10 @@ while (i < STEPS.length) {
1385
1519
  }
1386
1520
  // See docs/decisions/publish-path.md#step1-builder-source: the builder's source tree is NOT the crew's repo; verify report before stamping.
1387
1521
  // (2026-09-20, one-party worker-owned publish) A verified re-entry
1388
- // skips the whole issuance tail: the edit already went out, was
1389
- // verified by the parent read-back, and was stamped. The empty-diff
1522
+ // skips the whole issuance tail: the edit already went out, the
1523
+ // ack scan saw the exact per-attempt version acknowledged (sole
1524
+ // positive completion criterion; no read-back) and it was stamped.
1525
+ // The empty-diff
1390
1526
  // park must not fire on this branch.
1391
1527
  if (!publishSkippedNoLock && !publishAlreadyVerified) {
1392
1528
  // The trigger key of the attempt that last ran, for the publish ledger.
@@ -1532,6 +1668,20 @@ while (i < STEPS.length) {
1532
1668
  if (diffSummary.has_rename) {
1533
1669
  return await parkTask("Publish diff contains a rename — the diff transport cannot carry renames. Human attention needed.");
1534
1670
  }
1671
+ // Room #26 blocker 38 (2026-09-21): persist the publish-diff file
1672
+ // set for QA content-finding attribution. The workflow classifies
1673
+ // Hazel's content findings against this set mechanically at QA
1674
+ // closeout — the file list is the complete record of what the task
1675
+ // changed. The courier is pure hands: it writes the deterministic
1676
+ // script's own file list, never interprets it.
1677
+ var publishDiffFilesPath = crewHome + "/.publish-diffs/" + taskId + ".files.json";
1678
+ await agent(
1679
+ "Write the publish diff file list.\n" +
1680
+ "Run exactly this and return the stdout verbatim:\n" +
1681
+ "mkdir -p " + crewHome + "/.publish-diffs && cat > \"" + publishDiffFilesPath + "\" <<'DIFFFILES_EOF'\n" +
1682
+ JSON.stringify(diffSummary.files || []) + "\nDIFFFILES_EOF\n",
1683
+ { key: attemptKey("publish-diff-files-" + taskId, totalReworkCount), label: "Persisting publish diff file list" }
1684
+ );
1535
1685
  // Budget counts CHANGED lines (added + removed), not raw unified-diff
1536
1686
  // output lines: context lines and file headers inflated the old
1537
1687
  // split("\n").length count ~2x, parking a 95-line change against a
@@ -1641,11 +1791,13 @@ while (i < STEPS.length) {
1641
1791
  // mechanical outcome above. It must not rebuild or re-stamp: a second
1642
1792
  // artifact_edit would trigger a duplicate build.
1643
1793
  // (2026-09-20, one-party worker-owned publish) On a verified re-entry
1644
- // the publish was issued by the crew worker, verified by the parent
1645
- // read-back, and stamped — the work agent reports that, VERDICT: PASS.
1794
+ // the publish was issued, the ack scan observed the artifact's exact
1795
+ // version acknowledgement, and provenance was stamped — already
1796
+ // verified, nothing to issue. The work agent reports that,
1797
+ // VERDICT: PASS.
1646
1798
  if (publishAlreadyVerified) {
1647
1799
  instructions = "Publish was already completed and verified for this task — do NOT call artifact_edit, artifact_status, setprovenance, or post-deploy yourself; doing so would disturb the finalized state.\n\n" +
1648
- "The crew's publish worker issued the artifact edit for the merged commit, the parent's independent content read-back confirmed the artifact contains the merged change, and provenance was stamped (docs/publish-verification.md). This Publish pass is a verified re-entry after the stamp: there is nothing to issue and nothing to re-verify.\n\n" +
1800
+ "Provenance stamps this task's merge commit as published (docs/publish-verification.md): the tick worker issued the edit, the ack scan observed the artifact's exact version acknowledgement — the sole positive completion criterion — and the publish was stamped. Already verified, nothing to issue and nothing to re-verify.\n\n" +
1649
1801
  "For the change summary, run: cd " + REPO_PATH + " && git log -1 --stat\n\n" +
1650
1802
  "Write plain prose describing what was published, then on its own line: VERDICT: PASS\n" +
1651
1803
  "The VERDICT line must be the last line of your report.";
@@ -1658,10 +1810,10 @@ while (i < STEPS.length) {
1658
1810
  instructions = "Publish the merged code to the live artifact.\n\n" +
1659
1811
  "The publish was performed deterministically by the workflow before your step — do NOT call artifact_edit, artifact_status, setprovenance, or post-deploy yourself; doing so would trigger a duplicate build or disturb the finalized state. You perform no publish actions.\n\n" +
1660
1812
  "For the change summary, run: cd " + REPO_PATH + " && git log -1 --stat\n\n" +
1661
- "Mechanical outcome (the workflow's mechanical steps; content verification is the parent's, still pending):\n" +
1813
+ "Mechanical outcome (the workflow's mechanical steps; the ack scan's version-acknowledgement check is still pending):\n" +
1662
1814
  "- merge lock refreshed: yes\n" +
1663
1815
  "- artifact rebuild triggered and completed: yes\n" +
1664
- "- provenance stamped: NO — not yet, and you must NOT stamp it. The stamp moved to the parent, which stamps it only after an independent content read-back confirms the artifact actually contains the merged change (docs/publish-verification.md). A stamped-but-hollow build is exactly how canary run 8 (2026-09-11) went green on stale content.\n" +
1816
+ "- provenance stamped: NO — not yet, and you must NOT stamp it. The ack scan stamps it when it observes the artifact's exact on-disk acknowledgement of this attempt's version — version acknowledgement is the sole positive completion criterion (docs/publish-verification.md). A stamped-but-hollow build is exactly how canary run 8 (2026-09-11) went green on stale content.\n" +
1665
1817
  "- post-deploy finalized: yes (worktree removed, merge lock released)\n\n" +
1666
1818
  "POLICY: The live artifact is rebuilt only in this phase, from the repo. Never use artifact_edit to change the artifact directly — fixes go through the repo and the loop. A source fix is not done until the artifact is rebuilt from it here.\n\n" +
1667
1819
  "The repo push already happened in Integrate — do NOT push to git in this phase.\n\n" +
@@ -1693,7 +1845,7 @@ while (i < STEPS.length) {
1693
1845
  "Verify the fix against the task description in the returned data: the state the bug corrupted should now read correctly, and the task's own expected behavior should hold.\n" +
1694
1846
  "You can also check specific data with: node " + CREW_API + " --crew-home " + crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
1695
1847
  "Do NOT use artifact_inspect — it was removed by the platform (2026-09-14) and does not exist; do not substitute artifact.inspect (malfunction diagnosis, not an inspection tool).\n" +
1696
- "File follow-up tasks by running in shell:\n" + crewCmd("create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n" +
1848
+ "File follow-up tasks by running in shell (EXCEPT content findings — user-visible content/data goes in the CONTENT-FINDINGS block at the end of these instructions, never through create-task):\n" + crewCmd("create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n" +
1697
1849
  npmPublishCheck +
1698
1850
  "Report your test results as plain prose.\n" +
1699
1851
  "End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails.";
@@ -1721,7 +1873,7 @@ while (i < STEPS.length) {
1721
1873
  "Verify the fix against the task description in the returned data: the state the bug corrupted should now read correctly, and the task's own expected behavior should hold.\n" +
1722
1874
  "You can also check specific data with: node " + CREW_API + " --crew-home " + crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
1723
1875
  npmPublishCheck +
1724
- "File follow-up tasks by running in shell:\n" + crewCmd("create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n\n" +
1876
+ "File follow-up tasks by running in shell (EXCEPT content findings — user-visible content/data goes in the CONTENT-FINDINGS block at the end of these instructions, never through create-task):\n" + crewCmd("create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n\n" +
1725
1877
  "BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
1726
1878
  "Report your test results as plain prose.\n" +
1727
1879
  "End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails on the visual or the mechanical checks. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the reported bug, fixed>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL verdict must carry a machine-readable reason: the workflow closeout cross-checks verdict.json against your prose VERDICT line, and an unreasoned or contradictory verdict fails the phase (never routes to rework).";
@@ -1748,7 +1900,7 @@ while (i < STEPS.length) {
1748
1900
  "Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
1749
1901
  "DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale for a public-affecting change, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n\n" +
1750
1902
  "STEP 3: File follow-up tasks for any related issues you discover.\n" +
1751
- "For each issue, run in shell:\n" +
1903
+ "For each issue (EXCEPT content findings — user-visible content/data goes in the CONTENT-FINDINGS block at the end of these instructions, never through create-task), run in shell:\n" +
1752
1904
  "node " + CREW_API + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + LAUNCH_PROJECT_ID + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
1753
1905
  "(replace <issue title> and <issue details> with the real values).\n\n" +
1754
1906
  "BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL. When the baseline is terminal transcripts, confirm every terminal target you judged has a baseline transcript: a target with no pre-change transcript is an evidence gap — name it in --missing, never invent the baseline.\n\n" +
@@ -1767,6 +1919,21 @@ while (i < STEPS.length) {
1767
1919
  "2. Run: cd " + REPO_PATH + " && git log --format=%s <provenance.source_commit>..LIVE_HEAD (substitute real values). Every subject line MUST start with \"rebuild: \". If any line does not, report 'provenance mismatch: live HEAD moved past the stamped commit with non-rebuild source commits: [paste the offending subject lines]', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
1768
1920
  instructions;
1769
1921
  }
1922
+ // Room #26 blocker 38 (2026-09-21): content-finding attribution
1923
+ // protocol. Hazel reports user-visible content/data observations in a
1924
+ // machine-readable CONTENT-FINDINGS block; the workflow classifies each
1925
+ // against the task's publish diff and files follow-ups itself — Hazel
1926
+ // never files bugfixes for content findings directly, and
1927
+ // environment-attributable content never fails her verdict alone.
1928
+ var contentFindingsProtocol =
1929
+ "CONTENT FINDINGS (machine-read — room #26 blocker 38): if you observe user-visible content or data that looks wrong, stale, or out of place (decks, cards, rows, text, files in the artifact — anything this task did not obviously produce), report it here — do NOT file a follow-up task for it yourself and do NOT fail your verdict for it alone. Emit exactly one line in this shape, immediately BEFORE your final VERDICT line (the verdict stays the last line of your report):\n" +
1930
+ "CONTENT-FINDINGS: [{\"subject\": \"<repo-relative file path, or the literal live-data for content in the artifact's runtime data stores>\", \"observation\": \"<what you saw, one line>\"}, ...]\n" +
1931
+ "Use subject \"live-data\" for anything you saw in the running artifact (UI content, database rows, uploaded files) — you are code-blind and cannot name its file. Use a repo-relative file path only when you know the content lives in a specific file (e.g. from the task description or public docs). When you saw no such content, emit the empty array: CONTENT-FINDINGS: [].\n" +
1932
+ "The workflow classifies each finding against the task's publish diff: a finding whose subject file is in the diff is attributable to this task's change and the workflow files a bugfix for it; anything else is recorded with its attribution (environment-attributable, or unknown when the diff is unavailable) — the finding itself never spawns an artifact bugfix. No finding is ever dropped for looking like test residue: it is classified by attribution and recorded with it. Your VERDICT judges this task's change; content you cannot attribute to it is not a failure of this task.\n";
1933
+ // The content-findings block is machine-read at closeout (blocker 38).
1934
+ // Hazel emits it immediately before the VERDICT line, so the verdict
1935
+ // keeps its trailing-window contract with extractVerdict.
1936
+ instructions += "\n" + contentFindingsProtocol;
1770
1937
  }
1771
1938
 
1772
1939
  // Task event history — Review and Publish are excluded. Review is cold by
@@ -1825,7 +1992,31 @@ while (i < STEPS.length) {
1825
1992
  }
1826
1993
  var workerResult = null;
1827
1994
  var workAttempts = [];
1828
- for (var workAttempt = 0; workAttempt <= 2; workAttempt++) {
1995
+ // Room #26 blocker 34 (redesign): an empty branch with no verified sha and
1996
+ // no repo_diff: none claim fails Review mechanically — Cass is never
1997
+ // dispatched, so no reviewer adjudicates the emptiness. The synthetic
1998
+ // report below is written by the workflow and flows through the same
1999
+ // verdict extraction, structured verdict recording, and rejection/bounce
2000
+ // path as a real FAIL.
2001
+ // Blocker-34 follow-up: the npm `version` check joins the mechanical gate.
2002
+ // Either mechanical fact fails Review without dispatching Cass.
2003
+ var mechanicalFailReason = (versionTouched ? "version-touched" : (branchState === "empty-no-work" && !buildClaimedNoDiff ? "empty-no-work" : null));
2004
+ var mechanicalReviewFail = (step.name === "Review" && mechanicalFailReason !== null);
2005
+ if (mechanicalReviewFail) {
2006
+ log("Review: " + mechanicalFailReason + " — mechanical FAIL, Cass not dispatched");
2007
+ workerResult =
2008
+ "MECHANICAL REVIEW VERDICT (written by the workflow — no reviewer was dispatched).\n" +
2009
+ (mechanicalFailReason === "version-touched"
2010
+ ? "Package.json `version` (checked by the workflow from git): the task branch changed the `version` field. Versions are assigned at publish time — never in branches. Remove the version change.\n"
2011
+ : "Branch state (classified by the workflow from git): empty-no-work — the task branch has no commits ahead of the integration target, " +
2012
+ "no task-attributed merge on the target was verified" + (branchStateErr ? " (branch classification itself failed: " + branchStateErr + ")" : "") + ", " +
2013
+ "and the accepted Build report declared no `repo_diff: none` deliverable. " +
2014
+ "No change was reviewed because there is no change to review. The builder likely forgot to commit.\n") +
2015
+ "Worktree: " + WORKTREE_HINT + "\n" +
2016
+ "Recovery: commit the deliverable on the task branch in the worktree above and re-run Review. If the deliverable is genuinely runtime state outside the repo, declare `repo_diff: none` in the Build report instead of leaving the branch empty without a declaration.\n" +
2017
+ "VERDICT: FAIL";
2018
+ }
2019
+ for (var workAttempt = 0; workAttempt <= 2 && !mechanicalReviewFail; workAttempt++) {
1829
2020
  var workKey = workAttempt === 0 ? workKeyBase : workRetryKey(step.name, (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), workAttempt);
1830
2021
  var prevAttempt = workAttempt === 0 ? null : workAttempts[workAttempt - 1];
1831
2022
  var retryReason = workAttempt === 0 ? null : (prevAttempt.threw ? "discarded" : (prevAttempt.outcome === "missing-artifact-tools" ? "no-tools" : (prevAttempt.outcome === "unavailable-shell-transport" ? "no-transport" : "empty")));
@@ -1926,7 +2117,18 @@ while (i < STEPS.length) {
1926
2117
  session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name,
1927
2118
  status: "failed", notes: "Worker report had no single unambiguous VERDICT: PASS/FAIL line (bounded re-ask exhausted)" },
1928
2119
  event: { task_id: taskId, type: "failed",
1929
- message: step.name + " verdict line missing or ambiguous, re-ask exhausted — phase failed, dispatcher will retry" }
2120
+ message: step.name + " verdict line missing or ambiguous, re-ask exhausted — phase failed, dispatcher will retry" },
2121
+ // Room #26 blocker 33: INDETERMINATE verdict record — the report
2122
+ // could not be read at all, but its full text is still preserved
2123
+ // as grounds. Review-scoped; other verdict steps keep the
2124
+ // existing failed-session behavior with no verdict row.
2125
+ verdict: (step.name === "Review" ? {
2126
+ step: "Review",
2127
+ attempt: totalReworkCount,
2128
+ reviewer: step.identity,
2129
+ verdict: "INDETERMINATE",
2130
+ grounds: workerText
2131
+ } : null)
1930
2132
  }),
1931
2133
  { key: "record-block-" + step.name, label: "Recording verdict failure" }
1932
2134
  );
@@ -2127,40 +2329,123 @@ while (i < STEPS.length) {
2127
2329
  }
2128
2330
  log("Build worktree confinement passed: " + wt.path);
2129
2331
 
2130
- // See docs/decisions/publish-path.md#already-merged-idem: an already-merged repo_diff is idempotent; no rebuild.
2131
- var am = extractAlreadyMerged(workerText);
2132
- if (am.sha) {
2133
- var amCheck = await agent(
2134
- "Verify the builder's already-merged declaration.\n" +
2135
- "Run in shell and return the stdout verbatim:\n" +
2136
- "TARGET=$(" + LIFECYCLE_ENV + LIFECYCLE + " integration-target) && cd " + REPO_PATH + " && git rev-parse --verify --quiet " + am.sha + " >/dev/null && git merge-base --is-ancestor " + am.sha + " $TARGET && echo ALREADY_MERGED_YES || echo ALREADY_MERGED_NO",
2137
- { key: "verify-already-merged" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Verifying already-merged declaration" }
2332
+ // Room #26 blocker 34 (redesign, 2026-09-21): capture the plain
2333
+ // `repo_diff: none` (no repo change) claim from the FULL Build report.
2334
+ // Each Build round sets this fresh from its own report; there is no
2335
+ // cross-process fallback. The agent-authored already-merged
2336
+ // declaration and its Build-closeout verification are deleted —
2337
+ // attribution is the classifier's job at Review entry.
2338
+ buildClaimedNoDiff = /repo_diff:\s*none/im.test(workerText || "");
2339
+ }
2340
+
2341
+ // Room #26 blocker 38 (2026-09-21): content-finding attribution at QA
2342
+ // closeout. Hazel's CONTENT-FINDINGS block is extracted deterministically
2343
+ // and each finding is classified against the task's publish-diff file set
2344
+ // (persisted at Publish). Attributable findings become bugfix tasks filed
2345
+ // by the workflow; environment-attributable findings are recorded as task
2346
+ // note events with their attribution — never suppressed, never a bugfix,
2347
+ // never a park. No automatic FAIL override: if the QA agent still
2348
+ // reports FAIL, it stands — finding attribution informs follow-up
2349
+ // filing only.
2350
+ if (step.name === "QA") {
2351
+ var contentFindings = extractContentFindings(workerText);
2352
+ if (!contentFindings.ok) {
2353
+ // Unknown attribution fails closed: a missing or malformed
2354
+ // CONTENT-FINDINGS block blocks the phase for retry — the way an
2355
+ // unreadable VERDICT line does (record-phase failed + halted run).
2356
+ // It must never degrade to "no findings" (fail-open): unknown
2357
+ // attribution converted to an empty set silently drops the
2358
+ // observations the prose grounds carry. Do NOT re-ask an LLM to
2359
+ // repair a machine-readable contract — enforce the contract.
2360
+ log("QA CONTENT-FINDINGS unreadable (" + contentFindings.reason + ") — marking failed for retry");
2361
+ await agent(
2362
+ "Record content-findings failure.\n" +
2363
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
2364
+ task_id: taskId,
2365
+ session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: "failed",
2366
+ notes: "QA report had no machine-readable CONTENT-FINDINGS block (" + contentFindings.reason + "); phase failed for retry" },
2367
+ event: { task_id: taskId, type: "failed", message: "QA CONTENT-FINDINGS block unreadable (" + contentFindings.reason + ") — phase failed, dispatcher will retry" }
2368
+ }),
2369
+ { key: "record-block-" + step.name, label: "Recording content-findings failure" }
2138
2370
  );
2139
- var amOut = (typeof amCheck === "string") ? amCheck : JSON.stringify(amCheck);
2140
- if (!/ALREADY_MERGED_YES/.test(amOut)) {
2141
- log("Build already-merged declaration failed verification — " + am.sha + " is not an ancestor of the integration target — marking failed for retry");
2371
+ return {
2372
+ __hatchWorkflowControl: "blocked",
2373
+ result: {
2374
+ blocked_reason: "QA worker report had no machine-readable CONTENT-FINDINGS block (" + contentFindings.reason + ")",
2375
+ message: "The " + step.identity + " agent's work may be valid — its report did not carry a CONTENT-FINDINGS block the workflow could read, and the attribution contract fails closed rather than degrading to \"no findings\". The report is preserved in the workflow log.",
2376
+ task_id: taskId
2377
+ }
2378
+ };
2379
+ }
2380
+ // The publish-diff file set, persisted at Publish. Unreadable →
2381
+ // attribution unknown: every finding records "unknown" with the
2382
+ // reason naming the missing diff (never environment-attributable,
2383
+ // never a bugfix). The QA verdict stands on its own.
2384
+ var qaDiffFiles = [];
2385
+ var qaDiffFilesOk = false;
2386
+ try {
2387
+ var qaDiffFilesOut = await agent(
2388
+ "Run exactly one command and return the stdout verbatim:\n" +
2389
+ "cat \"" + crewHome + "/.publish-diffs/" + taskId + ".files.json\" 2>/dev/null || echo DIFF_FILES_MISSING\n" +
2390
+ "Return JSON { \"output\": \"<the command's full stdout, trimmed>\" } and nothing else.",
2391
+ { key: attemptKey("qa-diff-files-" + taskId, totalReworkCount), label: "Loading publish diff file list",
2392
+ schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
2393
+ );
2394
+ var qaDiffFilesRaw = ((qaDiffFilesOut && qaDiffFilesOut.output) || "").trim();
2395
+ if (qaDiffFilesRaw && qaDiffFilesRaw !== "DIFF_FILES_MISSING") {
2396
+ var qaDiffParsed = JSON.parse(qaDiffFilesRaw);
2397
+ if (Array.isArray(qaDiffParsed)) {
2398
+ qaDiffFiles = qaDiffParsed;
2399
+ qaDiffFilesOk = true;
2400
+ }
2401
+ }
2402
+ } catch (e) {
2403
+ log("QA diff file list unreadable: " + ((e && e.message ? e.message : String(e)) || "").slice(0, 200));
2404
+ }
2405
+ if (!qaDiffFilesOk) {
2406
+ log("QA publish diff file list unavailable — content findings record attribution unknown");
2407
+ }
2408
+ for (var cfi = 0; cfi < contentFindings.findings.length; cfi++) {
2409
+ var cf = contentFindings.findings[cfi];
2410
+ // Unknown attribution fails closed: when the diff file set is
2411
+ // unavailable we cannot attribute, so the finding is recorded as
2412
+ // "unknown" (never environment-attributable, never a bugfix) and the
2413
+ // QA verdict stands — the journey does not continue on unproven
2414
+ // attribution.
2415
+ var cfc = qaDiffFilesOk ? classifyContentFinding(cf, qaDiffFiles)
2416
+ : { attribution: "unknown", reason: "publish diff file set unavailable — cannot attribute" };
2417
+ if (cfc.attribution === "task-change") {
2418
+ log("QA content finding attributable to task change — filing bugfix: " + cf.subject + " :: " + cf.observation.slice(0, 120));
2142
2419
  await agent(
2143
- "Record already-merged verification failure.\n" +
2144
- "Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
2145
- task_id: taskId,
2146
- session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: "failed",
2147
- notes: "Build declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of the integration target in the configured repo. The declaration is fabricated or mistaken; the work is not on the integration target. Phase failed for retry" },
2148
- event: { task_id: taskId, type: "failed", message: "Build already-merged declaration failed verification — " + am.sha + " not an ancestor of the integration target, phase failed, dispatcher will retry" }
2149
- }),
2150
- { key: "record-already-merged-fail-" + step.name, label: "Recording already-merged verification failure" }
2420
+ "File the follow-up task.\n" +
2421
+ "Run in shell and return the stdout verbatim:\n" +
2422
+ "node " + CREW_API + " --crew-home " + crewHome + " create-task --json '" +
2423
+ JSON.stringify({ title: "QA content finding: " + cf.observation.slice(0, 120), description: "Content finding from QA on task " + taskId + " (subject: " + cf.subject + ", classified task-change: " + cfc.reason + "). Observation: " + cf.observation, project: LAUNCH_PROJECT_ID, workflow: "bugfix", filed_by: "hazel" }).split("'").join("'\\''") + "'\n" ,
2424
+ { key: attemptKey("qa-content-bugfix-" + taskId + "-" + cfi, totalReworkCount), label: "Filing attributable content bugfix" }
2425
+ );
2426
+ } else {
2427
+ // environment-attributable and unknown findings are recorded as task
2428
+ // note events with their attribution — never suppressed, never a
2429
+ // bugfix, never a park by themselves. The QA verdict stands on its
2430
+ // own: we do not override a FAIL based on finding attribution, because
2431
+ // the verdict may rest on functional grounds the findings do not capture.
2432
+ log("QA content finding " + cfc.attribution + " — recording, no bugfix: " + cf.subject + " :: " + cf.observation.slice(0, 120));
2433
+ await agent(
2434
+ "Record the " + cfc.attribution + " content finding.\n" +
2435
+ "Run in shell and return the stdout verbatim:\n" +
2436
+ crewCmd("log-event", { task_id: taskId, type: "note", message: "content-finding: " + cfc.attribution + " — subject: " + cf.subject + " — " + cf.observation + " (" + cfc.reason + ")" }) + "\n" ,
2437
+ { key: attemptKey("qa-content-record-" + taskId + "-" + cfi, totalReworkCount), label: "Recording " + cfc.attribution + " finding" }
2151
2438
  );
2152
- return {
2153
- __hatchWorkflowControl: "blocked",
2154
- result: {
2155
- blocked_reason: "Build already-merged declaration failed verification",
2156
- message: "The builder declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of the integration target. The work is not on the integration target; the phase is marked failed and the dispatcher will retry Build.",
2157
- task_id: taskId
2158
- }
2159
- };
2160
2439
  }
2161
- alreadyMergedSha = am.sha;
2162
- log("Build already-merged declaration verified: " + am.sha + " is an ancestor of the integration target");
2163
2440
  }
2441
+ // No automatic FAIL override. The QA prompt instructs Hazel that her
2442
+ // VERDICT judges this task's change and unattributable content is not a
2443
+ // failure of this task; if she still reports FAIL, it stands. The
2444
+ // recorded finding attributions (above) give the human the evidence to
2445
+ // distinguish environment residue from task failure. Overriding a FAIL
2446
+ // merely because all listed findings are environment-attributable would
2447
+ // be too broad — the prose may name functional failures the findings do
2448
+ // not capture.
2164
2449
  }
2165
2450
 
2166
2451
  // Deterministic closeout: no formatter agent. The verdict is mechanical
@@ -2172,6 +2457,9 @@ while (i < STEPS.length) {
2172
2457
  summary: workerText
2173
2458
  };
2174
2459
  let passed = stepResult.passed === true;
2460
+ // Blocker 38: content findings are recorded with their attribution in the
2461
+ // session notes above; the QA verdict stands on its own and is never
2462
+ // overridden by finding attribution.
2175
2463
 
2176
2464
  // See docs/decisions/publish-path.md#integrate-verify: the agent cannot verify integrate mechanically; the workflow checks the diff.
2177
2465
  if (step.name === "Integrate" && passed) {
@@ -2274,9 +2562,23 @@ while (i < STEPS.length) {
2274
2562
  } else {
2275
2563
  summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
2276
2564
  }
2277
- // See docs/decisions/publish-path.md#already-merged: when the Build gate verifies already-merged, attestation is recorded.
2278
- if (step.name === "Build" && alreadyMergedSha) {
2279
- summary += "\nalready_merged_verified: " + alreadyMergedSha;
2565
+ // Room #26 blocker 34 (redesign, 2026-09-21): machine-readable review
2566
+ // basis, prepended to Review session notes for observability (proof
2567
+ // rooms read the summary to see what Review examined and why a
2568
+ // reviewer was or was not dispatched).
2569
+ if (step.name === "Review" && branchState) {
2570
+ var reviewBasisNote;
2571
+ if (mechanicalReviewFail) {
2572
+ reviewBasisNote = "REVIEW_BASIS: none — mechanical FAIL (" + mechanicalFailReason + "), no reviewer dispatched";
2573
+ } else if (branchState.indexOf("already-merged:") === 0) {
2574
+ var rbSha = branchState.slice("already-merged:".length);
2575
+ reviewBasisNote = "REVIEW_BASIS: frozen merge " + rbSha + " — reviewed via git diff " + rbSha + "^1 " + rbSha;
2576
+ } else if (branchState === "has-work") {
2577
+ reviewBasisNote = "REVIEW_BASIS: branch diff — reviewed via inspect output (task branch vs integration target)";
2578
+ } else {
2579
+ reviewBasisNote = "REVIEW_BASIS: runtime-state-none — no repo change; plausibility of Build's repo_diff: none claim";
2580
+ }
2581
+ summary = reviewBasisNote + "\n" + summary;
2280
2582
  }
2281
2583
 
2282
2584
  // Visual verdict evidence (2026-09-15): no parent marker is appended. Hazel
@@ -2311,18 +2613,45 @@ while (i < STEPS.length) {
2311
2613
  releaseDecision = extractReleaseDecision(workerText);
2312
2614
  }
2313
2615
 
2616
+ // Room #26 blocker 34 (redesign, 2026-09-21): structured review basis
2617
+ // for the verdict row — what the review actually examined. A mechanical
2618
+ // FAIL has no reviewer and no basis beyond the mechanical fact. Only
2619
+ // computed for Review (branchState is null on other steps).
2620
+ var reviewBasis = null;
2621
+ if (step.name === "Review" && branchState) {
2622
+ reviewBasis =
2623
+ mechanicalReviewFail ? "mechanical-fail" :
2624
+ branchState.indexOf("already-merged:") === 0 ? "frozen-merge:" + branchState.slice("already-merged:".length) :
2625
+ branchState === "has-work" ? "branch-diff" : "runtime-state-none";
2626
+ }
2627
+
2314
2628
  await agent(
2315
2629
  "Update the session and log the event.\n" +
2316
2630
  "Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
2317
2631
  task_id: taskId,
2318
- // Room #16 blocker 11: the workflow-verified already-merged sha as
2319
- // structured control state. Only the Build gate sets alreadyMergedSha
2320
- // (after the mechanical ancestor check); the API validates the shape
2321
- // and a later write without the field never clears it (COALESCE).
2632
+ // Room #26 blocker 34 (2026-09-21): the already_merged_sha session
2633
+ // field is no longer written — the agent-authored declaration path
2634
+ // is deleted and nothing reads it. Removing the Crew API/database
2635
+ // column is a separate API-surface follow-up, not bundled here.
2322
2636
  session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name,
2323
- status: status, notes: summary, already_merged_sha: (step.name === "Build" ? alreadyMergedSha : null) },
2637
+ status: status, notes: summary },
2324
2638
  event: { task_id: taskId, type: status, identity: step.identity,
2325
- message: step.name + " " + status + " by " + step.identity }
2639
+ message: step.name + " " + status + " by " + step.identity },
2640
+ // Room #26 blocker 33: structured, non-lossy Review verdict record.
2641
+ // The session note above is truncated; the verdict row carries the
2642
+ // full worker report as grounds, written in the same transaction.
2643
+ // attempt is the rework round (0 = first Review).
2644
+ verdict: (step.name === "Review" && verdictPassed !== null ? {
2645
+ step: "Review",
2646
+ attempt: totalReworkCount,
2647
+ // Room #26 blocker 34 (redesign): a mechanical FAIL has no
2648
+ // reviewer — the workflow wrote the verdict. Identity rows still say
2649
+ // the Review step ran.
2650
+ reviewer: mechanicalReviewFail ? "workflow" : step.identity,
2651
+ verdict: verdictPassed ? "PASS" : "FAIL",
2652
+ review_basis: reviewBasis,
2653
+ grounds: workerText
2654
+ } : null)
2326
2655
  }),
2327
2656
  {
2328
2657
  key: "record-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "") + (mapGateBounceCount > 0 ? "-g" + mapGateBounceCount : ""),
@@ -2342,21 +2671,20 @@ while (i < STEPS.length) {
2342
2671
 
2343
2672
  // Review/QA rejection bounces to Build
2344
2673
  if (!passed && (step.name === "Review" || step.name === "QA")) {
2674
+ // Room #26 blocker 34 (redesign, 2026-09-21): the false-negative guard,
2675
+ // the Case B budget skip, and the already-merged corrective are all
2676
+ // gone. Emptiness is classified mechanically before Review, so a
2677
+ // rejected Review is never re-adjudicated here — the workflow owns the
2678
+ // git facts, the reviewer owns quality/spec compliance, and the budget
2679
+ // applies uniformly to every rejection. A rejected frozen merge
2680
+ // bounces with the reviewer's notes; a real fix commit becomes
2681
+ // has-work, and a do-nothing loop exhausts the normal budget.
2345
2682
  totalReworkCount++;
2346
2683
  if (totalReworkCount > MAX_TOTAL_REWORK) {
2347
2684
  log("Shared rework budget exhausted for task " + taskId + " — worktree preserved at " + WORKTREE_PRESERVED_HINT + " for manual inspection");
2348
2685
  return await parkTask("Exceeded shared rework budget (" + MAX_TOTAL_REWORK + " total rework attempts across Review and QA) after " + step.name + " rejection. Worktree preserved.");
2349
2686
  }
2350
2687
  rejectionNotes = summary;
2351
- // Already-merged corrective (room #16 blocker 11): when Review rejected
2352
- // an empty branch but the work is already on the integration target (the workflow verified
2353
- // the sha), Wren must declare it — not re-implement or re-commit
2354
- // already-landed work. Scoped to the empty-branch rejection; any other
2355
- // rejection already carries its own specific notes.
2356
- if (step.name === "Review" && alreadyMergedSha && /no commits ahead of (main|the integration target)/i.test(summary)) {
2357
- rejectionNotes += "\n\nCORRECTIVE (from the workflow, not the reviewer): the deliverable is already on the integration target — the workflow mechanically verified that " + alreadyMergedSha + " is an ancestor of the integration target. Do NOT re-implement the work and do NOT create a new commit for it. In your Build report, declare exactly: repo_diff: none (already-merged: " + alreadyMergedSha + ") — then end with VERDICT: PASS.";
2358
- log("Rework corrective appended for task " + taskId + ": already-merged " + alreadyMergedSha + " — Wren must declare, not rebuild");
2359
- }
2360
2688
  i = BUILD_INDEX;
2361
2689
  log(step.name + " rejected — bouncing to Build (rework #" + totalReworkCount + " of " + MAX_TOTAL_REWORK + ")");
2362
2690
  continue;