muse-crew 0.14.6 → 0.14.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +45 -5
- package/docs/decisions/AGENTS.md +2 -0
- package/docs/decisions/publish-path.md +56 -8
- package/docs/decisions/qa-reproduce.md +30 -0
- package/lib/AGENTS.md +6 -5
- package/lib/crew-api.js +403 -7
- package/lib/merge-lock.sh +153 -41
- package/lib/qa-db.js +132 -0
- package/lib/schema.sql +50 -0
- package/lib/serve-artifact.js +46 -2
- package/lib/test-detached-integrate.sh +122 -0
- package/lib/test-merge-lock.sh +30 -1
- package/lib/worktree-lifecycle.sh +284 -34
- package/package.json +1 -1
- package/seed/cron-body-template.md +6 -6
- package/workflows/AGENTS.md +1 -1
- package/workflows/bugfix.js +434 -106
- package/workflows/chore.js +251 -108
- package/workflows/crew-dispatch.js +57 -4
- package/workflows/docs.js +24 -2
- package/workflows/standard.js +439 -111
- package/workflows/upgrade.js +13 -1
package/workflows/standard.js
CHANGED
|
@@ -163,6 +163,62 @@ function extractVerdict(workerText) {
|
|
|
163
163
|
if (uniq.length !== 1) return { ok: false, count: matches.length };
|
|
164
164
|
return { ok: true, passed: last.value === "PASS" };
|
|
165
165
|
}
|
|
166
|
+
// Room #26 blocker 38 (2026-09-21): content-finding attribution. Hazel
|
|
167
|
+
// reports user-visible content observations in a machine-readable
|
|
168
|
+
// CONTENT-FINDINGS block; the workflow classifies each against the task's
|
|
169
|
+
// publish diff — mechanical set membership, never prose judgment. A finding
|
|
170
|
+
// is attributable to the task's change iff its subject file is in the
|
|
171
|
+
// publish diff. The special subject "live-data" (content in the artifact's
|
|
172
|
+
// runtime data stores — decks, cards, rows) is never in a git diff, so it
|
|
173
|
+
// always classifies environment-attributable: the crew has no live-DB write
|
|
174
|
+
// path (blocker 32: QA writes go to a fresh per-run temp DB), so live
|
|
175
|
+
// content is never the task's change. Findings are always recorded — never
|
|
176
|
+
// suppressed for looking audit-y — but environment-attributable findings
|
|
177
|
+
// never park the journey and never spawn an artifact bugfix. The
|
|
178
|
+
// platform-side half (audit sessions writing to production via the
|
|
179
|
+
// shared-concurrent path) is out of crew scope; see
|
|
180
|
+
// docs/decisions/qa-reproduce.md#content-finding-attribution.
|
|
181
|
+
function extractContentFindings(workerText) {
|
|
182
|
+
// The block is the LAST "CONTENT-FINDINGS:" line; the JSON array follows
|
|
183
|
+
// on the same line. Fail closed (ok:false) when missing or malformed —
|
|
184
|
+
// the caller blocks the phase for retry; unknown attribution must never
|
|
185
|
+
// degrade to "no findings" (the prose grounds stay preserved in the
|
|
186
|
+
// session notes).
|
|
187
|
+
var text = workerText || "";
|
|
188
|
+
var regex = /^CONTENT-FINDINGS:\s*(\[.*\])\s*$/gim;
|
|
189
|
+
var matches = [];
|
|
190
|
+
var m;
|
|
191
|
+
while ((m = regex.exec(text)) !== null) {
|
|
192
|
+
matches.push(m[1]);
|
|
193
|
+
}
|
|
194
|
+
if (matches.length === 0) return { ok: false, reason: "no CONTENT-FINDINGS block" };
|
|
195
|
+
var findings;
|
|
196
|
+
try {
|
|
197
|
+
findings = JSON.parse(matches[matches.length - 1]);
|
|
198
|
+
} catch (e) {
|
|
199
|
+
return { ok: false, reason: "CONTENT-FINDINGS block is not valid JSON" };
|
|
200
|
+
}
|
|
201
|
+
if (!Array.isArray(findings)) return { ok: false, reason: "CONTENT-FINDINGS is not an array" };
|
|
202
|
+
for (var i = 0; i < findings.length; i++) {
|
|
203
|
+
var f = findings[i];
|
|
204
|
+
if (!f || typeof f !== "object" || typeof f.subject !== "string" || typeof f.observation !== "string") {
|
|
205
|
+
return { ok: false, reason: "finding " + i + " needs string subject and observation" };
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
return { ok: true, findings: findings };
|
|
209
|
+
}
|
|
210
|
+
function classifyContentFinding(finding, diffFiles) {
|
|
211
|
+
// diffFiles: the task's publish-diff file set (repo-relative paths).
|
|
212
|
+
// Pure set membership — no prose, no judgment.
|
|
213
|
+
var subject = finding.subject;
|
|
214
|
+
if (subject === "live-data") {
|
|
215
|
+
return { attribution: "environment-attributable", reason: "live-data is never in a git publish diff — the crew has no live-DB write path (blocker 32)" };
|
|
216
|
+
}
|
|
217
|
+
if (diffFiles.indexOf(subject) !== -1) {
|
|
218
|
+
return { attribution: "task-change", reason: "subject is in the task's publish diff" };
|
|
219
|
+
}
|
|
220
|
+
return { attribution: "environment-attributable", reason: "subject is not in the task's publish diff" };
|
|
221
|
+
}
|
|
166
222
|
// See docs/decisions/workflow-core.md#verdict-reask: a report that fails extractVerdict gets bounded re-ask calls.
|
|
167
223
|
function verdictReaskKey(stepName, reworkSuffix, attempt) {
|
|
168
224
|
return "verdict-reask-" + stepName + reworkSuffix + "-a" + attempt;
|
|
@@ -466,13 +522,6 @@ function extractMarkerLines(workerText) {
|
|
|
466
522
|
return markers.join("\n");
|
|
467
523
|
}
|
|
468
524
|
|
|
469
|
-
// See docs/decisions/publish-path.md#already-merged-idem2: idempotency for already-merged tasks.
|
|
470
|
-
function extractAlreadyMerged(workerText) {
|
|
471
|
-
var m = /repo_diff:\s*none\s*\(already-merged:\s*([0-9a-f]{7,40})(?=[\s)]|$)/i.exec(workerText || "");
|
|
472
|
-
return m ? { sha: m[1].toLowerCase() } : { sha: null };
|
|
473
|
-
}
|
|
474
|
-
|
|
475
|
-
|
|
476
525
|
// See docs/decisions/qa-reproduce.md#worktree-confinement: the Build agent must declare its worktree.
|
|
477
526
|
function extractWorktree(workerText) {
|
|
478
527
|
var lines = (workerText || "").split("\n");
|
|
@@ -630,12 +679,24 @@ let mapGateBounceCount = 0;
|
|
|
630
679
|
// rationalized a skip against explicit instruction text — text alone did not
|
|
631
680
|
// hold, so the decision now lives in workflow code, not agent judgment.
|
|
632
681
|
let releaseDecision = null; // { release: "yes"|"no", version_bump: "patch"|"minor"|"major"|null }
|
|
633
|
-
// Already-merged
|
|
634
|
-
//
|
|
635
|
-
//
|
|
636
|
-
//
|
|
637
|
-
//
|
|
682
|
+
// Already-merged attribution (blocker 34, 2026-09-21): the sha of the
|
|
683
|
+
// task-attributed merge when classify-branch reports already-merged.
|
|
684
|
+
// Set ONLY from the classifier's mechanical output — never from agent
|
|
685
|
+
// prose. The agent-authored sha-bearing repo_diff declaration path is
|
|
686
|
+
// deleted; attribution is identity-first from git, then the task's own
|
|
687
|
+
// durable merge records.
|
|
638
688
|
let alreadyMergedSha = null;
|
|
689
|
+
// Room #26 blocker 34 (redesign, 2026-09-21): mechanical branch-state
|
|
690
|
+
// classification at Review entry. Git facts are never adjudicated by a
|
|
691
|
+
// reviewer — the workflow classifies the branch with `classify-branch`
|
|
692
|
+
// (pure git) before Cass is dispatched: "has-work",
|
|
693
|
+
// "already-merged:<sha>", or "empty-no-work". An empty branch with no
|
|
694
|
+
// verified sha and no `repo_diff: none` claim fails mechanically without
|
|
695
|
+
// dispatching Cass; reviewer prose is never parsed for git identity.
|
|
696
|
+
let branchState = null; // "has-work" | "already-merged:<sha>" | "empty-no-work"
|
|
697
|
+
let buildClaimedNoDiff = false; // Build declared plain `repo_diff: none` (no repo change — runtime-state deliverable, or believed already-merged). Same-process Build report only; there is no cross-process hydration — an empty branch on a fresh dispatch with no task-attributed merge fails mechanically.
|
|
698
|
+
let versionTouched = false; // npm: branch changed package.json's `version` (mechanical git fact, blocker-34 follow-up)
|
|
699
|
+
let branchStateErr = ""; // classifier failure detail (fail-closed grounds)
|
|
639
700
|
// Deterministic publish target — computed by the workflow (registry base +
|
|
640
701
|
// bumpVersion), never by the Publish agent.
|
|
641
702
|
let publishTarget = null; // { base, scope, target }
|
|
@@ -1104,55 +1165,126 @@ while (i < STEPS.length) {
|
|
|
1104
1165
|
"cd " + WORKTREE_HINT + "\n" +
|
|
1105
1166
|
"git add -A\n" +
|
|
1106
1167
|
"git commit -m \"" + safeTitle + "\"\n\n" +
|
|
1107
|
-
"If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of the integration target and declare `repo_diff: none` in your report, naming the runtime-state deliverable. If you verified the deliverable is already on the integration target (a prior merge
|
|
1168
|
+
"If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of the integration target and declare `repo_diff: none` in your report, naming the runtime-state deliverable. If you verified the deliverable is already on the integration target (a prior merge landed it — do NOT re-implement working code), make no commit and declare `repo_diff: none` in your report. The workflow verifies the claim mechanically from git — the task's own merge records, never your declaration, are the proof of delivery. Otherwise commit your changes normally.\n\n" +
|
|
1108
1169
|
(rejectionNotes ? "This is REWORK after rejection. Address these specific issues:\n" + rejectionNotes + "\n\n" : "") +
|
|
1109
1170
|
"Report back in plain prose: what you built and the outcome." +
|
|
1110
1171
|
(PUBLISH_TYPE === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
|
|
1111
1172
|
|
|
1112
1173
|
} else if (step.name === "Review") {
|
|
1113
|
-
//
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
1174
|
+
// Room #26 blocker 34 (redesign, 2026-09-21): classify the branch
|
|
1175
|
+
// mechanically BEFORE Cass is dispatched. The classifier is pure git
|
|
1176
|
+
// (classify-branch in the lifecycle script): has-work |
|
|
1177
|
+
// already-merged:<sha> | empty-no-work. Attribution is identity-first
|
|
1178
|
+
// from git, then the task's own durable merge records — there is no
|
|
1179
|
+
// agent-authored declared-sha input and no cross-process hydration
|
|
1180
|
+
// ferry (the re-entry case it served is covered by the durable record
|
|
1181
|
+
// → classifier directly). An unparseable result fails closed as
|
|
1182
|
+
// empty-no-work — an empty branch never passes Review silently, and
|
|
1183
|
+
// the bounce to Build gives the next round a chance to classify.
|
|
1184
|
+
try {
|
|
1185
|
+
var bsResult = await agent(
|
|
1186
|
+
"Classify the task branch state. This is a mechanical git check, not a judgment call.\n" +
|
|
1187
|
+
"Run in shell and return the stdout verbatim, then the single stderr line starting with `DIAG:` verbatim:\n" +
|
|
1188
|
+
LIFECYCLE_ENV + LIFECYCLE + " classify-branch " + taskId,
|
|
1189
|
+
{ key: "classify-branch" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Classifying branch state (mechanical)" }
|
|
1190
|
+
);
|
|
1191
|
+
var bsStr = (typeof bsResult === "string") ? bsResult : JSON.stringify(bsResult);
|
|
1192
|
+
var bsMatch = /^\s*BRANCH_STATE:\s*(has-work|already-merged:[0-9a-f]{40}|empty-no-work)\s*$/im.exec(bsStr);
|
|
1193
|
+
if (bsMatch) {
|
|
1194
|
+
branchState = bsMatch[1].toLowerCase();
|
|
1195
|
+
} else {
|
|
1196
|
+
branchStateErr = "unparseable classifier output: " + bsStr.slice(0, 120);
|
|
1127
1197
|
}
|
|
1128
|
-
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1198
|
+
// DIAG pin (blocker 34 design v2.1): the merge-record scan is
|
|
1199
|
+
// load-bearing with no fallback — the classifier's stderr counter is
|
|
1200
|
+
// logged verbatim so silent record loss surfaces in the evidence.
|
|
1201
|
+
var bsDiagMatch = /^[ \t]*DIAG:\s*classify-branch\s+records=\d+\s+resolvable=\d+[ \t]*$/im.exec(bsStr);
|
|
1202
|
+
if (bsDiagMatch) {
|
|
1203
|
+
log(bsDiagMatch[0].trim());
|
|
1133
1204
|
} else {
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1205
|
+
log("classify-branch: no DIAG: record-scan counter returned (scan unverifiable)");
|
|
1206
|
+
}
|
|
1207
|
+
} catch (bsErr) {
|
|
1208
|
+
branchStateErr = "classifier call failed: " + String(bsErr && bsErr.message || bsErr).slice(0, 120);
|
|
1209
|
+
}
|
|
1210
|
+
if (!branchState) {
|
|
1211
|
+
branchState = "empty-no-work";
|
|
1212
|
+
log("Branch-state classification failed (" + branchStateErr + ") — failing closed as empty-no-work");
|
|
1213
|
+
} else if (branchState.indexOf("already-merged:") === 0) {
|
|
1214
|
+
alreadyMergedSha = branchState.slice("already-merged:".length);
|
|
1215
|
+
log("Branch state already-merged: " + alreadyMergedSha + " (mechanically verified) — Cass reviews the frozen merge diff");
|
|
1216
|
+
} else {
|
|
1217
|
+
log("Branch state: " + branchState);
|
|
1218
|
+
}
|
|
1219
|
+
// buildClaimedNoDiff was set at the Build gate from the full Build
|
|
1220
|
+
// report (same process only). Only an empty branch WITH this claim
|
|
1221
|
+
// goes to Cass for a plausibility judgment; without it, empty-no-work
|
|
1222
|
+
// fails mechanically and Cass is never dispatched.
|
|
1223
|
+
// Blocker-34 follow-up: the package.json `version` check is a mechanical
|
|
1224
|
+
// git fact, not a reviewer judgment. For npm projects the workflow checks
|
|
1225
|
+
// it here, before Cass is dispatched — a touched `version` fails Review
|
|
1226
|
+
// mechanically (versions are assigned at publish time, never in
|
|
1227
|
+
// branches). Fail closed: anything but an explicit VERSION_CLEAN counts
|
|
1228
|
+
// as touched.
|
|
1229
|
+
if (PUBLISH_TYPE === "npm") {
|
|
1230
|
+
try {
|
|
1231
|
+
var vtResult = await agent(
|
|
1232
|
+
"Check whether the task branch changed package.json's `version` field. This is a mechanical git check, not a judgment call.\n" +
|
|
1233
|
+
"Run in shell and return the stdout verbatim:\n" +
|
|
1234
|
+
LIFECYCLE_ENV + LIFECYCLE + " version-check " + taskId,
|
|
1235
|
+
{ key: "version-check" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Checking package.json version (mechanical)" }
|
|
1236
|
+
);
|
|
1237
|
+
var vtStr = (typeof vtResult === "string") ? vtResult : JSON.stringify(vtResult);
|
|
1238
|
+
var vtMatch = /^\s*VERSION_(TOUCHED|CLEAN)\s*$/im.exec(vtStr);
|
|
1239
|
+
versionTouched = !(vtMatch && vtMatch[1].toUpperCase() === "CLEAN");
|
|
1240
|
+
} catch (vtErr) {
|
|
1241
|
+
versionTouched = true;
|
|
1242
|
+
log("Version check failed (" + String(vtErr && vtErr.message || vtErr).slice(0, 120) + ") — failing closed as touched");
|
|
1139
1243
|
}
|
|
1244
|
+
log("Review: package.json `version` " + (versionTouched ? "touched by the branch — mechanical FAIL, Cass not dispatched" : "untouched — mechanical check clean"));
|
|
1245
|
+
}
|
|
1246
|
+
// The empty-branch rule Cass used to adjudicate is gone. What she
|
|
1247
|
+
// examines and what rule she applies depend on the classified state —
|
|
1248
|
+
// never on her own judgment of git identity.
|
|
1249
|
+
var reviewChangeExam, reviewBranchRule;
|
|
1250
|
+
if (branchState === "has-work") {
|
|
1251
|
+
reviewChangeExam =
|
|
1252
|
+
"Examine the code changes by running:\n" +
|
|
1253
|
+
LIFECYCLE_ENV + LIFECYCLE + " inspect " + taskId + "\n\n" +
|
|
1254
|
+
"The inspect output is authoritative: it prints the task branch's actual tip commit (TIP) and every commit ahead of the integration target (inspect prints the target name). Base your review ONLY on this output — do NOT run git log yourself to pick commits, and do NOT discuss commit hashes from any other source (they may come from stale rework rounds or a different repo).\n\n";
|
|
1255
|
+
reviewBranchRule =
|
|
1256
|
+
"MECHANICAL FACT (computed by the workflow from git — never by a reviewer): branch state = has-work. The branch has commits ahead of the integration target. Review their content; branch emptiness is settled and is not yours to judge.\n";
|
|
1257
|
+
} else if (branchState.indexOf("already-merged:") === 0) {
|
|
1258
|
+
reviewChangeExam =
|
|
1259
|
+
"The change under review is the frozen merge " + alreadyMergedSha + " — it is already on the integration target (see MECHANICAL FACT below). Examine it by running:\n" +
|
|
1260
|
+
"cd " + REPO_PATH + " && git diff " + alreadyMergedSha + "^1 " + alreadyMergedSha + "\n\n" +
|
|
1261
|
+
"That first-parent diff is the frozen, attributable change. Base your review ONLY on this diff — do NOT run git log to pick commits, and do NOT discuss commit hashes from any other source (they may come from stale rework rounds or a different repo).\n\n";
|
|
1262
|
+
reviewBranchRule =
|
|
1263
|
+
"MECHANICAL FACT (computed by the workflow from git — never by a reviewer): branch state = already-merged:" + alreadyMergedSha + ". The deliverable landed on the integration target via this task's own prior merge " + alreadyMergedSha + " (verified ancestor of the integration target). The branch is empty by design — emptiness is settled fact, not a finding.\n";
|
|
1264
|
+
} else if (buildClaimedNoDiff) {
|
|
1265
|
+
reviewChangeExam =
|
|
1266
|
+
"The branch has no commits ahead of the integration target (see MECHANICAL FACT below) — there is no diff to examine. Judge the Build report's `repo_diff: none` claim on its plausibility.\n\n";
|
|
1267
|
+
reviewBranchRule =
|
|
1268
|
+
"MECHANICAL FACT (computed by the workflow from git — never by a reviewer): the branch has no commits ahead of the integration target and no task-attributed merge on the target was verified; the Build report declares `repo_diff: none` (no repo change). Approve ONLY if the task's deliverable is plausibly runtime state (e.g. a cron definition, scheduler change, or dashboard/config state created outside the repo). Otherwise report 'no commits ahead of the integration target and no plausible runtime-state deliverable — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n";
|
|
1269
|
+
} else {
|
|
1270
|
+
// Mechanical FAIL — Cass is never dispatched (see the dispatch gate
|
|
1271
|
+
// below). Instructions are unused; keep the shape.
|
|
1272
|
+
reviewChangeExam = "";
|
|
1273
|
+
reviewBranchRule = "";
|
|
1140
1274
|
}
|
|
1141
1275
|
instructions = "Review independently and cold. You have NOT seen any reasoning from the builder.\nDo NOT access the task dashboard, event log, or any comments. Your review is based solely on the spec and the code.\n\n" +
|
|
1142
1276
|
(mapperSpec ? "MAPPER'S SPEC (the builder was asked to implement exactly this):\n" + mapperSpec + "\n\n" : "Read the spec at exactly " + SPEC_PATH + " (fall back to the task description if the file is absent).\n\n") +
|
|
1143
|
-
|
|
1144
|
-
LIFECYCLE_ENV + LIFECYCLE + " inspect " + taskId + "\n\n" +
|
|
1145
|
-
"The inspect output is authoritative: it prints the task branch's actual tip commit (TIP) and every commit ahead of the integration target (inspect prints the target name). Base your review ONLY on this output — do NOT run git log yourself to pick commits, and do NOT discuss commit hashes from any other source (they may come from stale rework rounds or a different repo).\n\n" +
|
|
1277
|
+
reviewChangeExam +
|
|
1146
1278
|
"You can also read specific files in the worktree at:\n" +
|
|
1147
1279
|
WORKTREE_HINT + "/\n\n" +
|
|
1148
1280
|
"Check quality, correctness, and spec compliance.\n" +
|
|
1149
1281
|
(SURFACE_TERMINAL ? "TERMINAL UX REVIEW: judge the CLI surface against " + UX_DOCTRINE_PATH + " — help accuracy, error quality, exit codes, output clarity. Reject when the bar is not met.\n" : "") +
|
|
1150
1282
|
(SURFACE_ARTIFACT ? "ARTIFACT UX REVIEW: judge the rendered surface against " + UX_DOCTRINE_PATH + " — alignment, spacing, hierarchy, composition, balance, finish, correctness. Reject when the bar is not met.\n" : "") +
|
|
1151
1283
|
"Check that public-affecting changes have matching public doc updates (API.md or the published API contract). If the docs are missing or inaccurate, report what is stale, then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1152
|
-
|
|
1153
|
-
(PUBLISH_TYPE === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry, and versions are assigned at publish time — never in branches
|
|
1154
|
-
"
|
|
1155
|
-
"
|
|
1284
|
+
reviewBranchRule +
|
|
1285
|
+
(PUBLISH_TYPE === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry, and versions are assigned at publish time — never in branches.\n" +
|
|
1286
|
+
"MECHANICAL FACT (computed by the workflow from git — never by a reviewer): the task branch did not change package.json's `version` field. A branch that touches `version` fails Review mechanically before any reviewer is dispatched, so this is settled — do not re-check it.\n" +
|
|
1287
|
+
"The accepted Build report declares: " + releaseDecisionText() + ". " +
|
|
1156
1288
|
(releaseDecision
|
|
1157
1289
|
? "Validate this decision against the change: release must be 'yes' when the change is consumer-observable and 'no' when internal-only; the version_bump scope must fit the change (patch for fixes, minor for new behavior, major for breaking changes). If the decision is wrong or mis-scoped, report your notes, then end with exactly this line: VERDICT: FAIL."
|
|
1158
1290
|
: "The decision is missing or malformed — report 'Build report must end with release: yes|no and (when release is yes) version_bump: patch|minor|major lines', then end with exactly this line: VERDICT: FAIL.") + "\n" : "") +
|
|
@@ -1166,8 +1298,9 @@ while (i < STEPS.length) {
|
|
|
1166
1298
|
"Then run: "+ LIFECYCLE_ENV + "WORKFLOW_RUN_ID=" + lockHolder + " CREW_STAGED_BASE=<provenance.source_commit, empty when null> integrate " + taskId + " \"merge: " + safeTitle + "\\\"\\n" +
|
|
1167
1299
|
"The script merges, reconciles, records, and pushes under the merge lock. Informational only: RECONCILED, NO_REMOTE, NO_REMOTE_RECONCILE. Read the output:\n" +
|
|
1168
1300
|
"- STAGED_BASE_MISMATCH: the line left the staged base (stray checkout). VERDICT: FAIL.\n" +
|
|
1169
|
-
"- MERGED_EMPTY — check FIRST (it contains the word MERGED): either a runtime-state deliverable or
|
|
1301
|
+
"- MERGED_EMPTY — check FIRST (it contains the word MERGED): either a runtime-state deliverable or a task-attributed merge the workflow verified from the task's own merge records. No lock taken, no new commit. Report 'merged empty: no repo changes'. VERDICT: PASS.\n" +
|
|
1170
1302
|
"- MERGED: the merge landed; the script pushed inline. Read the push line: PUSHED — report the hash (detached prints PUSHED: origin/main (refspec HEAD:main)), VERDICT: PASS; NO_REMOTE_PUSH — no remote configured, VERDICT: PASS; ERROR after MERGED — a retry recovers the push; report it, VERDICT: FAIL.\n" +
|
|
1303
|
+
"- STALE_MERGE: the recorded merge no longer matches the task branch (rework moved it after the merge) — the script refused to push stale state. Report the STALE_MERGE line, VERDICT: FAIL.\n" +
|
|
1171
1304
|
"- LOCK_HELD: 10-minute backoff exhausted. VERDICT: FAIL.\n" +
|
|
1172
1305
|
"- CONFLICT: the plain merge failed — aborted, the target is clean, and your task still holds the merge lock. Do NOT fail yet. Resolve it:\n" +
|
|
1173
1306
|
"RESOLUTION:\n" +
|
|
@@ -1217,8 +1350,9 @@ while (i < STEPS.length) {
|
|
|
1217
1350
|
// See docs/decisions/publish-path.md#deterministic-artifact-publish: the work agent never publishes; the parent runs the deterministic publish script.
|
|
1218
1351
|
// Blocker 22 (one-party publish, room #24): verified re-entry guard.
|
|
1219
1352
|
// The workflow parks at publish intent; the tick worker issues
|
|
1220
|
-
// artifact_edit
|
|
1221
|
-
//
|
|
1353
|
+
// artifact_edit; the ack scan stamps provenance on exact per-attempt
|
|
1354
|
+
// version acknowledgement (sole positive completion criterion; no
|
|
1355
|
+
// read-back — docs/publish-verification.md) and re-queues. The dispatcher resumes this task at the Publish step
|
|
1222
1356
|
// (the parked session's own step — the failed→retry path). If the
|
|
1223
1357
|
// stamped provenance names THIS task and its source_commit is the
|
|
1224
1358
|
// repo's current HEAD, the merge commit was already published and
|
|
@@ -1274,8 +1408,10 @@ while (i < STEPS.length) {
|
|
|
1274
1408
|
}
|
|
1275
1409
|
// See docs/decisions/publish-path.md#step1-builder-source: the builder's source tree is NOT the crew's repo; verify report before stamping.
|
|
1276
1410
|
// (2026-09-20, one-party worker-owned publish) A verified re-entry
|
|
1277
|
-
// skips the whole issuance tail: the edit already went out,
|
|
1278
|
-
//
|
|
1411
|
+
// skips the whole issuance tail: the edit already went out, the
|
|
1412
|
+
// ack scan saw the exact per-attempt version acknowledged (sole
|
|
1413
|
+
// positive completion criterion; no read-back) and it was stamped.
|
|
1414
|
+
// The empty-diff
|
|
1279
1415
|
// park must not fire on this branch.
|
|
1280
1416
|
if (!publishSkippedNoLock && !publishAlreadyVerified) {
|
|
1281
1417
|
// The trigger key of the attempt that last ran, for the publish ledger.
|
|
@@ -1421,6 +1557,20 @@ while (i < STEPS.length) {
|
|
|
1421
1557
|
if (diffSummary.has_rename) {
|
|
1422
1558
|
return await parkTask("Publish diff contains a rename — the diff transport cannot carry renames. Human attention needed.");
|
|
1423
1559
|
}
|
|
1560
|
+
// Room #26 blocker 38 (2026-09-21): persist the publish-diff file
|
|
1561
|
+
// set for QA content-finding attribution. The workflow classifies
|
|
1562
|
+
// Hazel's content findings against this set mechanically at QA
|
|
1563
|
+
// closeout — the file list is the complete record of what the task
|
|
1564
|
+
// changed. The courier is pure hands: it writes the deterministic
|
|
1565
|
+
// script's own file list, never interprets it.
|
|
1566
|
+
var publishDiffFilesPath = crewHome + "/.publish-diffs/" + taskId + ".files.json";
|
|
1567
|
+
await agent(
|
|
1568
|
+
"Write the publish diff file list.\n" +
|
|
1569
|
+
"Run exactly this and return the stdout verbatim:\n" +
|
|
1570
|
+
"mkdir -p " + crewHome + "/.publish-diffs && cat > \"" + publishDiffFilesPath + "\" <<'DIFFFILES_EOF'\n" +
|
|
1571
|
+
JSON.stringify(diffSummary.files || []) + "\nDIFFFILES_EOF\n",
|
|
1572
|
+
{ key: attemptKey("publish-diff-files-" + taskId, totalReworkCount), label: "Persisting publish diff file list" }
|
|
1573
|
+
);
|
|
1424
1574
|
// Budget counts CHANGED lines (added + removed), not raw unified-diff
|
|
1425
1575
|
// output lines: context lines and file headers inflated the old
|
|
1426
1576
|
// split("\n").length count ~2x, parking a 95-line change against a
|
|
@@ -1530,11 +1680,13 @@ while (i < STEPS.length) {
|
|
|
1530
1680
|
// mechanical outcome above. It must not rebuild or re-stamp: a second
|
|
1531
1681
|
// artifact_edit would trigger a duplicate build.
|
|
1532
1682
|
// (2026-09-20, one-party worker-owned publish) On a verified re-entry
|
|
1533
|
-
// the publish was issued
|
|
1534
|
-
//
|
|
1683
|
+
// the publish was issued, the ack scan observed the artifact's exact
|
|
1684
|
+
// version acknowledgement, and provenance was stamped — already
|
|
1685
|
+
// verified, nothing to issue. The work agent reports that,
|
|
1686
|
+
// VERDICT: PASS.
|
|
1535
1687
|
if (publishAlreadyVerified) {
|
|
1536
1688
|
instructions = "Publish was already completed and verified for this task — do NOT call artifact_edit, artifact_status, setprovenance, or post-deploy yourself; doing so would disturb the finalized state.\n\n" +
|
|
1537
|
-
"
|
|
1689
|
+
"Provenance stamps this task's merge commit as published (docs/publish-verification.md): the tick worker issued the edit, the ack scan observed the artifact's exact version acknowledgement — the sole positive completion criterion — and the publish was stamped. Already verified, nothing to issue and nothing to re-verify.\n\n" +
|
|
1538
1690
|
"For the change summary, run: cd " + REPO_PATH + " && git log -1 --stat\n\n" +
|
|
1539
1691
|
"Write plain prose describing what was published, then on its own line: VERDICT: PASS\n" +
|
|
1540
1692
|
"The VERDICT line must be the last line of your report.";
|
|
@@ -1547,10 +1699,10 @@ while (i < STEPS.length) {
|
|
|
1547
1699
|
instructions = "Publish the merged code to the live artifact.\n\n" +
|
|
1548
1700
|
"The publish was performed deterministically by the workflow before your step — do NOT call artifact_edit, artifact_status, setprovenance, or post-deploy yourself; doing so would trigger a duplicate build or disturb the finalized state. You perform no publish actions.\n\n" +
|
|
1549
1701
|
"For the change summary, run: cd " + REPO_PATH + " && git log -1 --stat\n\n" +
|
|
1550
|
-
"Mechanical outcome (the workflow's mechanical steps;
|
|
1702
|
+
"Mechanical outcome (the workflow's mechanical steps; the ack scan's version-acknowledgement check is still pending):\n" +
|
|
1551
1703
|
"- merge lock refreshed: yes\n" +
|
|
1552
1704
|
"- artifact rebuild triggered and completed: yes\n" +
|
|
1553
|
-
"- provenance stamped: NO — not yet, and you must NOT stamp it. The
|
|
1705
|
+
"- provenance stamped: NO — not yet, and you must NOT stamp it. The ack scan stamps it when it observes the artifact's exact on-disk acknowledgement of this attempt's version — version acknowledgement is the sole positive completion criterion (docs/publish-verification.md). A stamped-but-hollow build is exactly how canary run 8 (2026-09-11) went green on stale content.\n" +
|
|
1554
1706
|
"- post-deploy finalized: yes (worktree removed, merge lock released)\n\n" +
|
|
1555
1707
|
"POLICY: The live artifact is rebuilt only in this phase, from the repo. Never use artifact_edit to change the artifact directly — fixes go through the repo and the loop. A source fix is not done until the artifact is rebuilt from it here.\n\n" +
|
|
1556
1708
|
"The repo push already happened in Integrate — do NOT push to git in this phase.\n\n" +
|
|
@@ -1564,6 +1716,17 @@ while (i < STEPS.length) {
|
|
|
1564
1716
|
"Report the situation in prose, then end your report with exactly this line: VERDICT: FAIL.";
|
|
1565
1717
|
}
|
|
1566
1718
|
} else if (step.name === "QA") {
|
|
1719
|
+
// Room #26 blocker 38 (2026-09-21): content-finding attribution
|
|
1720
|
+
// protocol. Hazel reports user-visible content/data observations in a
|
|
1721
|
+
// machine-readable CONTENT-FINDINGS block; the workflow classifies each
|
|
1722
|
+
// against the task's publish diff and files follow-ups itself — Hazel
|
|
1723
|
+
// never files bugfixes for content findings directly, and
|
|
1724
|
+
// environment-attributable content never fails her verdict alone.
|
|
1725
|
+
var contentFindingsProtocol =
|
|
1726
|
+
"CONTENT FINDINGS (machine-read — room #26 blocker 38): if you observe user-visible content or data that looks wrong, stale, or out of place (decks, cards, rows, text, files in the artifact — anything this task did not obviously produce), report it here — do NOT file a follow-up task for it yourself and do NOT fail your verdict for it alone. Emit exactly one line in this shape, immediately BEFORE your final VERDICT line (the verdict stays the last line of your report):\n" +
|
|
1727
|
+
"CONTENT-FINDINGS: [{\"subject\": \"<repo-relative file path, or the literal live-data for content in the artifact's runtime data stores>\", \"observation\": \"<what you saw, one line>\"}, ...]\n" +
|
|
1728
|
+
"Use subject \"live-data\" for anything you saw in the running artifact (UI content, database rows, uploaded files) — you are code-blind and cannot name its file. Use a repo-relative file path only when you know the content lives in a specific file (e.g. from the task description or public docs). When you saw no such content, emit the empty array: CONTENT-FINDINGS: [].\n" +
|
|
1729
|
+
"The workflow classifies each finding against the task's publish diff: a finding whose subject file is in the diff is attributable to this task's change and the workflow files a bugfix for it; anything else is recorded with its attribution (environment-attributable, or unknown when the diff is unavailable) — the finding itself never spawns an artifact bugfix. No finding is ever dropped for looking like test residue: it is classified by attribution and recorded with it. Your VERDICT judges this task's change; content you cannot attribute to it is not a failure of this task.\n";
|
|
1567
1730
|
// Backstop for merge-time versioning: when the accepted Build summary
|
|
1568
1731
|
// declared release: yes, QA verifies the registry actually moved. A silent
|
|
1569
1732
|
// publish skip becomes a loud QA failure with evidence, not a pass.
|
|
@@ -1606,7 +1769,7 @@ while (i < STEPS.length) {
|
|
|
1606
1769
|
"1. Run: cd " + REPO_PATH + " && git merge-base --is-ancestor <provenance.source_commit> LIVE_HEAD && echo ANCESTOR_OK (substitute the real stamped hash and LIVE_HEAD; do not run the literal placeholders). If this command fails, FAIL: { \"passed\": false, \"summary\": \"provenance mismatch: stamped source_commit is not an ancestor of live HEAD\" }.\n" +
|
|
1607
1770
|
"2. Run: cd " + REPO_PATH + " && git log --format=%s <provenance.source_commit>..LIVE_HEAD (substitute real values). Every subject line MUST start with \"rebuild: \". If any line does not, report 'provenance mismatch: live HEAD moved past the stamped commit with non-rebuild source commits: [paste the offending subject lines]', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1608
1771
|
"If both pass, the source check passes — the only drift since the stamp is builder staging output committed by post-deploy. Continue to STEP 3.\n\n" +
|
|
1609
|
-
"STEP 3: File follow-up tasks for any related issues you discover.\n" +
|
|
1772
|
+
"STEP 3: File follow-up tasks for any related issues you discover — EXCEPT content findings (user-visible content/data): those go in the CONTENT-FINDINGS block at the end of these instructions, never through create-task. The workflow files follow-ups for attributable content itself.\n" +
|
|
1610
1773
|
"For each issue, run in shell:\n" +
|
|
1611
1774
|
"node " + CREW_API + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + LAUNCH_PROJECT_ID + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
1612
1775
|
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
@@ -1634,7 +1797,7 @@ while (i < STEPS.length) {
|
|
|
1634
1797
|
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
|
|
1635
1798
|
"Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
|
|
1636
1799
|
"DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale for a public-affecting change, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n\n" +
|
|
1637
|
-
"STEP 3: File follow-up tasks for any related issues you discover.\n" +
|
|
1800
|
+
"STEP 3: File follow-up tasks for any related issues you discover — EXCEPT content findings (user-visible content/data): those go in the CONTENT-FINDINGS block at the end of these instructions, never through create-task. The workflow files follow-ups for attributable content itself.\n" +
|
|
1638
1801
|
"For each issue, run in shell:\n" +
|
|
1639
1802
|
"node " + CREW_API + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + LAUNCH_PROJECT_ID + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
1640
1803
|
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
@@ -1646,12 +1809,16 @@ while (i < STEPS.length) {
|
|
|
1646
1809
|
"Public docs (API.md, README) are NOT source code — read them freely, exactly as a user would.\n" +
|
|
1647
1810
|
"Verify the change is working as described in the task.\n" +
|
|
1648
1811
|
"DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n" +
|
|
1649
|
-
"File follow-up tasks for related issues found by running in shell:\n" +
|
|
1812
|
+
"File follow-up tasks for related issues found (EXCEPT content findings — user-visible content/data goes in the CONTENT-FINDINGS block at the end of these instructions, never through create-task) by running in shell:\n" +
|
|
1650
1813
|
"node " + CREW_API + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + LAUNCH_PROJECT_ID + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
1651
1814
|
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
1652
1815
|
npmPublishCheck +
|
|
1653
1816
|
"Report back in plain prose — what you tested and found. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
1654
1817
|
}
|
|
1818
|
+
// The content-findings block is machine-read at closeout (blocker 38).
|
|
1819
|
+
// Hazel emits it immediately before the VERDICT line, so the verdict
|
|
1820
|
+
// keeps its trailing-window contract with extractVerdict.
|
|
1821
|
+
instructions += "\n" + contentFindingsProtocol;
|
|
1655
1822
|
// qaExperiential: Hazel owns the experiential verdict through the QA
|
|
1656
1823
|
// prompt built in the SURFACE_ARTIFACT / SURFACE_TERMINAL branches above
|
|
1657
1824
|
// (see-act loop or terminal loop + OODA report). There is no parent
|
|
@@ -1692,7 +1859,31 @@ while (i < STEPS.length) {
|
|
|
1692
1859
|
var workKeyBase = "work-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "");
|
|
1693
1860
|
var workerResult = null;
|
|
1694
1861
|
var workAttempts = [];
|
|
1695
|
-
|
|
1862
|
+
// Room #26 blocker 34 (redesign): an empty branch with no verified sha and
|
|
1863
|
+
// no repo_diff: none claim fails Review mechanically — Cass is never
|
|
1864
|
+
// dispatched, so no reviewer adjudicates the emptiness. The synthetic
|
|
1865
|
+
// report below is written by the workflow and flows through the same
|
|
1866
|
+
// verdict extraction, structured verdict recording, and rejection/bounce
|
|
1867
|
+
// path as a real FAIL.
|
|
1868
|
+
// Blocker-34 follow-up: the npm `version` check joins the mechanical gate.
|
|
1869
|
+
// Either mechanical fact fails Review without dispatching Cass.
|
|
1870
|
+
var mechanicalFailReason = (versionTouched ? "version-touched" : (branchState === "empty-no-work" && !buildClaimedNoDiff ? "empty-no-work" : null));
|
|
1871
|
+
var mechanicalReviewFail = (step.name === "Review" && mechanicalFailReason !== null);
|
|
1872
|
+
if (mechanicalReviewFail) {
|
|
1873
|
+
log("Review: " + mechanicalFailReason + " — mechanical FAIL, Cass not dispatched");
|
|
1874
|
+
workerResult =
|
|
1875
|
+
"MECHANICAL REVIEW VERDICT (written by the workflow — no reviewer was dispatched).\n" +
|
|
1876
|
+
(mechanicalFailReason === "version-touched"
|
|
1877
|
+
? "Package.json `version` (checked by the workflow from git): the task branch changed the `version` field. Versions are assigned at publish time — never in branches. Remove the version change.\n"
|
|
1878
|
+
: "Branch state (classified by the workflow from git): empty-no-work — the task branch has no commits ahead of the integration target, " +
|
|
1879
|
+
"no task-attributed merge on the target was verified" + (branchStateErr ? " (branch classification itself failed: " + branchStateErr + ")" : "") + ", " +
|
|
1880
|
+
"and the accepted Build report declared no `repo_diff: none` deliverable. " +
|
|
1881
|
+
"No change was reviewed because there is no change to review. The builder likely forgot to commit.\n") +
|
|
1882
|
+
"Worktree: " + WORKTREE_HINT + "\n" +
|
|
1883
|
+
"Recovery: commit the deliverable on the task branch in the worktree above and re-run Review. If the deliverable is genuinely runtime state outside the repo, declare `repo_diff: none` in the Build report instead of leaving the branch empty without a declaration.\n" +
|
|
1884
|
+
"VERDICT: FAIL";
|
|
1885
|
+
}
|
|
1886
|
+
for (var workAttempt = 0; workAttempt <= 2 && !mechanicalReviewFail; workAttempt++) {
|
|
1696
1887
|
var workKey = workAttempt === 0 ? workKeyBase : workRetryKey(step.name, (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), workAttempt);
|
|
1697
1888
|
var prevAttempt = workAttempt === 0 ? null : workAttempts[workAttempt - 1];
|
|
1698
1889
|
var retryReason = workAttempt === 0 ? null : (prevAttempt.threw ? "discarded" : (prevAttempt.outcome === "missing-artifact-tools" ? "no-tools" : (prevAttempt.outcome === "unavailable-shell-transport" ? "no-transport" : "empty")));
|
|
@@ -1790,7 +1981,18 @@ while (i < STEPS.length) {
|
|
|
1790
1981
|
"Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
|
|
1791
1982
|
task_id: taskId,
|
|
1792
1983
|
session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: "failed", notes: "Worker report had no single unambiguous VERDICT: PASS/FAIL line (bounded re-ask exhausted)" },
|
|
1793
|
-
event: { task_id: taskId, type: "failed", message: step.name + " verdict line missing or ambiguous, re-ask exhausted — phase failed, dispatcher will retry" }
|
|
1984
|
+
event: { task_id: taskId, type: "failed", message: step.name + " verdict line missing or ambiguous, re-ask exhausted — phase failed, dispatcher will retry" },
|
|
1985
|
+
// Room #26 blocker 33: INDETERMINATE verdict record — the report
|
|
1986
|
+
// could not be read at all, but its full text is still preserved
|
|
1987
|
+
// as grounds. Review-scoped; other verdict steps keep the
|
|
1988
|
+
// existing failed-session behavior with no verdict row.
|
|
1989
|
+
verdict: (step.name === "Review" ? {
|
|
1990
|
+
step: "Review",
|
|
1991
|
+
attempt: totalReworkCount,
|
|
1992
|
+
reviewer: step.identity,
|
|
1993
|
+
verdict: "INDETERMINATE",
|
|
1994
|
+
grounds: workerText
|
|
1995
|
+
} : null)
|
|
1794
1996
|
}),
|
|
1795
1997
|
{ key: "record-block-" + step.name, label: "Recording verdict failure" }
|
|
1796
1998
|
);
|
|
@@ -1837,40 +2039,13 @@ while (i < STEPS.length) {
|
|
|
1837
2039
|
}
|
|
1838
2040
|
log("Build worktree confinement passed: " + wt.path);
|
|
1839
2041
|
|
|
1840
|
-
//
|
|
1841
|
-
|
|
1842
|
-
|
|
1843
|
-
|
|
1844
|
-
|
|
1845
|
-
|
|
1846
|
-
|
|
1847
|
-
{ key: "verify-already-merged" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Verifying already-merged declaration" }
|
|
1848
|
-
);
|
|
1849
|
-
var amOut = (typeof amCheck === "string") ? amCheck : JSON.stringify(amCheck);
|
|
1850
|
-
if (!/ALREADY_MERGED_YES/.test(amOut)) {
|
|
1851
|
-
log("Build already-merged declaration failed verification — " + am.sha + " is not an ancestor of the integration target — marking failed for retry");
|
|
1852
|
-
await agent(
|
|
1853
|
-
"Record already-merged verification failure.\n" +
|
|
1854
|
-
"Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
|
|
1855
|
-
task_id: taskId,
|
|
1856
|
-
session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: "failed",
|
|
1857
|
-
notes: "Build declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of the integration target in the configured repo. The declaration is fabricated or mistaken; the work is not on the integration target. Phase failed for retry" },
|
|
1858
|
-
event: { task_id: taskId, type: "failed", message: "Build already-merged declaration failed verification — " + am.sha + " not an ancestor of the integration target, phase failed, dispatcher will retry" }
|
|
1859
|
-
}),
|
|
1860
|
-
{ key: "record-already-merged-fail-" + step.name, label: "Recording already-merged verification failure" }
|
|
1861
|
-
);
|
|
1862
|
-
return {
|
|
1863
|
-
__hatchWorkflowControl: "blocked",
|
|
1864
|
-
result: {
|
|
1865
|
-
blocked_reason: "Build already-merged declaration failed verification",
|
|
1866
|
-
message: "The builder declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of the integration target. The work is not on the integration target; the phase is marked failed and the dispatcher will retry Build.",
|
|
1867
|
-
task_id: taskId
|
|
1868
|
-
}
|
|
1869
|
-
};
|
|
1870
|
-
}
|
|
1871
|
-
alreadyMergedSha = am.sha;
|
|
1872
|
-
log("Build already-merged declaration verified: " + am.sha + " is an ancestor of the integration target");
|
|
1873
|
-
}
|
|
2042
|
+
// Room #26 blocker 34 (redesign, 2026-09-21): capture the plain
|
|
2043
|
+
// `repo_diff: none` (no repo change) claim from the FULL Build report.
|
|
2044
|
+
// Each Build round sets this fresh from its own report; there is no
|
|
2045
|
+
// cross-process fallback. The agent-authored already-merged
|
|
2046
|
+
// declaration and its Build-closeout verification are deleted —
|
|
2047
|
+
// attribution is the classifier's job at Review entry.
|
|
2048
|
+
buildClaimedNoDiff = /repo_diff:\s*none/im.test(workerText || "");
|
|
1874
2049
|
}
|
|
1875
2050
|
|
|
1876
2051
|
// See docs/decisions/qa-reproduce.md#experiential-loop-guard: a PASS with missing experiential evidence parks fail-closed.
|
|
@@ -1915,6 +2090,116 @@ while (i < STEPS.length) {
|
|
|
1915
2090
|
log("QA " + qaLoopSurface + "-loop guard passed: experiential evidence present");
|
|
1916
2091
|
}
|
|
1917
2092
|
|
|
2093
|
+
// Room #26 blocker 38 (2026-09-21): content-finding attribution at QA
|
|
2094
|
+
// closeout. Hazel's CONTENT-FINDINGS block is extracted deterministically
|
|
2095
|
+
// and each finding is classified against the task's publish-diff file set
|
|
2096
|
+
// (persisted at Publish). Attributable findings become bugfix tasks filed
|
|
2097
|
+
// by the workflow; environment-attributable findings are recorded as task
|
|
2098
|
+
// note events with their attribution — never suppressed, never a bugfix,
|
|
2099
|
+
// never a park. No automatic FAIL override: if the QA agent still
|
|
2100
|
+
// reports FAIL, it stands — finding attribution informs follow-up
|
|
2101
|
+
// filing only.
|
|
2102
|
+
if (step.name === "QA") {
|
|
2103
|
+
var contentFindings = extractContentFindings(workerText);
|
|
2104
|
+
if (!contentFindings.ok) {
|
|
2105
|
+
// Unknown attribution fails closed: a missing or malformed
|
|
2106
|
+
// CONTENT-FINDINGS block blocks the phase for retry — the way an
|
|
2107
|
+
// unreadable VERDICT line does (record-phase failed + halted run).
|
|
2108
|
+
// It must never degrade to "no findings" (fail-open): unknown
|
|
2109
|
+
// attribution converted to an empty set silently drops the
|
|
2110
|
+
// observations the prose grounds carry. Do NOT re-ask an LLM to
|
|
2111
|
+
// repair a machine-readable contract — enforce the contract.
|
|
2112
|
+
log("QA CONTENT-FINDINGS unreadable (" + contentFindings.reason + ") — marking failed for retry");
|
|
2113
|
+
await agent(
|
|
2114
|
+
"Record content-findings failure.\n" +
|
|
2115
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
|
|
2116
|
+
task_id: taskId,
|
|
2117
|
+
session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: "failed",
|
|
2118
|
+
notes: "QA report had no machine-readable CONTENT-FINDINGS block (" + contentFindings.reason + "); phase failed for retry" },
|
|
2119
|
+
event: { task_id: taskId, type: "failed", message: "QA CONTENT-FINDINGS block unreadable (" + contentFindings.reason + ") — phase failed, dispatcher will retry" }
|
|
2120
|
+
}),
|
|
2121
|
+
{ key: "record-block-" + step.name, label: "Recording content-findings failure" }
|
|
2122
|
+
);
|
|
2123
|
+
return {
|
|
2124
|
+
__hatchWorkflowControl: "blocked",
|
|
2125
|
+
result: {
|
|
2126
|
+
blocked_reason: "QA worker report had no machine-readable CONTENT-FINDINGS block (" + contentFindings.reason + ")",
|
|
2127
|
+
message: "The " + step.identity + " agent's work may be valid — its report did not carry a CONTENT-FINDINGS block the workflow could read, and the attribution contract fails closed rather than degrading to \"no findings\". The report is preserved in the workflow log.",
|
|
2128
|
+
task_id: taskId
|
|
2129
|
+
}
|
|
2130
|
+
};
|
|
2131
|
+
}
|
|
2132
|
+
// The publish-diff file set, persisted at Publish. Unreadable →
|
|
2133
|
+
// attribution unknown: every finding records "unknown" with the
|
|
2134
|
+
// reason naming the missing diff (never environment-attributable,
|
|
2135
|
+
// never a bugfix). The QA verdict stands on its own.
|
|
2136
|
+
var qaDiffFiles = [];
|
|
2137
|
+
var qaDiffFilesOk = false;
|
|
2138
|
+
try {
|
|
2139
|
+
var qaDiffFilesOut = await agent(
|
|
2140
|
+
"Run exactly one command and return the stdout verbatim:\n" +
|
|
2141
|
+
"cat \"" + crewHome + "/.publish-diffs/" + taskId + ".files.json\" 2>/dev/null || echo DIFF_FILES_MISSING\n" +
|
|
2142
|
+
"Return JSON { \"output\": \"<the command's full stdout, trimmed>\" } and nothing else.",
|
|
2143
|
+
{ key: attemptKey("qa-diff-files-" + taskId, totalReworkCount), label: "Loading publish diff file list",
|
|
2144
|
+
schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
|
|
2145
|
+
);
|
|
2146
|
+
var qaDiffFilesRaw = ((qaDiffFilesOut && qaDiffFilesOut.output) || "").trim();
|
|
2147
|
+
if (qaDiffFilesRaw && qaDiffFilesRaw !== "DIFF_FILES_MISSING") {
|
|
2148
|
+
var qaDiffParsed = JSON.parse(qaDiffFilesRaw);
|
|
2149
|
+
if (Array.isArray(qaDiffParsed)) {
|
|
2150
|
+
qaDiffFiles = qaDiffParsed;
|
|
2151
|
+
qaDiffFilesOk = true;
|
|
2152
|
+
}
|
|
2153
|
+
}
|
|
2154
|
+
} catch (e) {
|
|
2155
|
+
log("QA diff file list unreadable: " + ((e && e.message ? e.message : String(e)) || "").slice(0, 200));
|
|
2156
|
+
}
|
|
2157
|
+
if (!qaDiffFilesOk) {
|
|
2158
|
+
log("QA publish diff file list unavailable — content findings record attribution unknown");
|
|
2159
|
+
}
|
|
2160
|
+
for (var cfi = 0; cfi < contentFindings.findings.length; cfi++) {
|
|
2161
|
+
var cf = contentFindings.findings[cfi];
|
|
2162
|
+
// Unknown attribution fails closed: when the diff file set is
|
|
2163
|
+
// unavailable we cannot attribute, so the finding is recorded as
|
|
2164
|
+
// "unknown" (never environment-attributable, never a bugfix) and the
|
|
2165
|
+
// QA verdict stands — the journey does not continue on unproven
|
|
2166
|
+
// attribution.
|
|
2167
|
+
var cfc = qaDiffFilesOk ? classifyContentFinding(cf, qaDiffFiles)
|
|
2168
|
+
: { attribution: "unknown", reason: "publish diff file set unavailable — cannot attribute" };
|
|
2169
|
+
if (cfc.attribution === "task-change") {
|
|
2170
|
+
log("QA content finding attributable to task change — filing bugfix: " + cf.subject + " :: " + cf.observation.slice(0, 120));
|
|
2171
|
+
await agent(
|
|
2172
|
+
"File the follow-up task.\n" +
|
|
2173
|
+
"Run in shell and return the stdout verbatim:\n" +
|
|
2174
|
+
"node " + CREW_API + " --crew-home " + crewHome + " create-task --json '" +
|
|
2175
|
+
JSON.stringify({ title: "QA content finding: " + cf.observation.slice(0, 120), description: "Content finding from QA on task " + taskId + " (subject: " + cf.subject + ", classified task-change: " + cfc.reason + "). Observation: " + cf.observation, project: LAUNCH_PROJECT_ID, workflow: "bugfix", filed_by: "hazel" }).split("'").join("'\\''") + "'\n",
|
|
2176
|
+
{ key: attemptKey("qa-content-bugfix-" + taskId + "-" + cfi, totalReworkCount), label: "Filing attributable content bugfix" }
|
|
2177
|
+
);
|
|
2178
|
+
} else {
|
|
2179
|
+
// environment-attributable and unknown findings are recorded as task
|
|
2180
|
+
// note events with their attribution — never suppressed, never a
|
|
2181
|
+
// bugfix, never a park by themselves. The QA verdict stands on its
|
|
2182
|
+
// own: we do not override a FAIL based on finding attribution, because
|
|
2183
|
+
// the verdict may rest on functional grounds the findings do not capture.
|
|
2184
|
+
log("QA content finding " + cfc.attribution + " — recording, no bugfix: " + cf.subject + " :: " + cf.observation.slice(0, 120));
|
|
2185
|
+
await agent(
|
|
2186
|
+
"Record the " + cfc.attribution + " content finding.\n" +
|
|
2187
|
+
"Run in shell and return the stdout verbatim:\n" +
|
|
2188
|
+
crewCmd("log-event", { task_id: taskId, type: "note", message: "content-finding: " + cfc.attribution + " — subject: " + cf.subject + " — " + cf.observation + " (" + cfc.reason + ")" }) + "\n",
|
|
2189
|
+
{ key: attemptKey("qa-content-record-" + taskId + "-" + cfi, totalReworkCount), label: "Recording " + cfc.attribution + " finding" }
|
|
2190
|
+
);
|
|
2191
|
+
}
|
|
2192
|
+
}
|
|
2193
|
+
// No automatic FAIL override. The QA prompt instructs Hazel that her
|
|
2194
|
+
// VERDICT judges this task's change and unattributable content is not a
|
|
2195
|
+
// failure of this task; if she still reports FAIL, it stands. The
|
|
2196
|
+
// recorded finding attributions (above) give the human the evidence to
|
|
2197
|
+
// distinguish environment residue from task failure. Overriding a FAIL
|
|
2198
|
+
// merely because all listed findings are environment-attributable would
|
|
2199
|
+
// be too broad — the prose may name functional failures the findings do
|
|
2200
|
+
// not capture.
|
|
2201
|
+
}
|
|
2202
|
+
|
|
1918
2203
|
// Deterministic closeout: no formatter agent. The verdict is mechanical
|
|
1919
2204
|
// (extractVerdict above); the summary is the worker's report truncated.
|
|
1920
2205
|
// For verdict steps passed comes from the verdict; for non-verdict steps
|
|
@@ -1924,6 +2209,9 @@ while (i < STEPS.length) {
|
|
|
1924
2209
|
summary: workerText
|
|
1925
2210
|
};
|
|
1926
2211
|
let passed = stepResult.passed === true;
|
|
2212
|
+
// Blocker 38: content findings are recorded with their attribution in the
|
|
2213
|
+
// session notes above; the QA verdict stands on its own and is never
|
|
2214
|
+
// overridden by finding attribution.
|
|
1927
2215
|
|
|
1928
2216
|
// See docs/decisions/publish-path.md#integrate-verify: the agent cannot verify integrate mechanically; the workflow checks the diff.
|
|
1929
2217
|
if (step.name === "Integrate" && passed) {
|
|
@@ -2064,9 +2352,23 @@ while (i < STEPS.length) {
|
|
|
2064
2352
|
} else {
|
|
2065
2353
|
summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
|
|
2066
2354
|
}
|
|
2067
|
-
//
|
|
2068
|
-
|
|
2069
|
-
|
|
2355
|
+
// Room #26 blocker 34 (redesign, 2026-09-21): machine-readable review
|
|
2356
|
+
// basis, prepended to Review session notes for observability (proof
|
|
2357
|
+
// rooms read the summary to see what Review examined and why a
|
|
2358
|
+
// reviewer was or was not dispatched).
|
|
2359
|
+
if (step.name === "Review" && branchState) {
|
|
2360
|
+
var reviewBasisNote;
|
|
2361
|
+
if (mechanicalReviewFail) {
|
|
2362
|
+
reviewBasisNote = "REVIEW_BASIS: none — mechanical FAIL (" + mechanicalFailReason + "), no reviewer dispatched";
|
|
2363
|
+
} else if (branchState.indexOf("already-merged:") === 0) {
|
|
2364
|
+
var rbSha = branchState.slice("already-merged:".length);
|
|
2365
|
+
reviewBasisNote = "REVIEW_BASIS: frozen merge " + rbSha + " — reviewed via git diff " + rbSha + "^1 " + rbSha;
|
|
2366
|
+
} else if (branchState === "has-work") {
|
|
2367
|
+
reviewBasisNote = "REVIEW_BASIS: branch diff — reviewed via inspect output (task branch vs integration target)";
|
|
2368
|
+
} else {
|
|
2369
|
+
reviewBasisNote = "REVIEW_BASIS: runtime-state-none — no repo change; plausibility of Build's repo_diff: none claim";
|
|
2370
|
+
}
|
|
2371
|
+
summary = reviewBasisNote + "\n" + summary;
|
|
2070
2372
|
}
|
|
2071
2373
|
|
|
2072
2374
|
// Visual verdict evidence: for experiential artifact tasks, append the
|
|
@@ -2096,17 +2398,44 @@ while (i < STEPS.length) {
|
|
|
2096
2398
|
releaseDecision = extractReleaseDecision(workerText);
|
|
2097
2399
|
}
|
|
2098
2400
|
|
|
2401
|
+
// Room #26 blocker 34 (redesign, 2026-09-21): structured review basis
|
|
2402
|
+
// for the verdict row — what the review actually examined. A mechanical
|
|
2403
|
+
// FAIL has no reviewer and no basis beyond the mechanical fact. Only
|
|
2404
|
+
// computed for Review (branchState is null on other steps).
|
|
2405
|
+
var reviewBasis = null;
|
|
2406
|
+
if (step.name === "Review" && branchState) {
|
|
2407
|
+
reviewBasis =
|
|
2408
|
+
mechanicalReviewFail ? "mechanical-fail" :
|
|
2409
|
+
branchState.indexOf("already-merged:") === 0 ? "frozen-merge:" + branchState.slice("already-merged:".length) :
|
|
2410
|
+
branchState === "has-work" ? "branch-diff" : "runtime-state-none";
|
|
2411
|
+
}
|
|
2412
|
+
|
|
2099
2413
|
// Record session result
|
|
2100
2414
|
await agent(
|
|
2101
2415
|
"Update the session and log the event.\n" +
|
|
2102
2416
|
"Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
|
|
2103
2417
|
task_id: taskId,
|
|
2104
|
-
// Room #
|
|
2105
|
-
//
|
|
2106
|
-
//
|
|
2107
|
-
//
|
|
2108
|
-
session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: status, notes: summary
|
|
2109
|
-
event: { task_id: taskId, type: status, identity: step.identity, message: step.name + " " + status + " by " + step.identity }
|
|
2418
|
+
// Room #26 blocker 34 (2026-09-21): the already_merged_sha session
|
|
2419
|
+
// field is no longer written — the agent-authored declaration path
|
|
2420
|
+
// is deleted and nothing reads it. Removing the Crew API/database
|
|
2421
|
+
// column is a separate API-surface follow-up, not bundled here.
|
|
2422
|
+
session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: status, notes: summary },
|
|
2423
|
+
event: { task_id: taskId, type: status, identity: step.identity, message: step.name + " " + status + " by " + step.identity },
|
|
2424
|
+
// Room #26 blocker 33: structured, non-lossy Review verdict record.
|
|
2425
|
+
// The session note above is truncated; the verdict row carries the
|
|
2426
|
+
// full worker report as grounds, written in the same transaction.
|
|
2427
|
+
// attempt is the rework round (0 = first Review).
|
|
2428
|
+
verdict: (step.name === "Review" && verdictPassed !== null ? {
|
|
2429
|
+
step: "Review",
|
|
2430
|
+
attempt: totalReworkCount,
|
|
2431
|
+
// Room #26 blocker 34 (redesign): a mechanical FAIL has no
|
|
2432
|
+
// reviewer — the workflow wrote the verdict. Identity rows still say
|
|
2433
|
+
// the Review step ran.
|
|
2434
|
+
reviewer: mechanicalReviewFail ? "workflow" : step.identity,
|
|
2435
|
+
verdict: verdictPassed ? "PASS" : "FAIL",
|
|
2436
|
+
review_basis: reviewBasis,
|
|
2437
|
+
grounds: workerText
|
|
2438
|
+
} : null)
|
|
2110
2439
|
}),
|
|
2111
2440
|
{
|
|
2112
2441
|
key: "record-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "") + (mapGateBounceCount > 0 ? "-g" + mapGateBounceCount : ""),
|
|
@@ -2116,21 +2445,20 @@ while (i < STEPS.length) {
|
|
|
2116
2445
|
|
|
2117
2446
|
// Handle rejection — bounce back to Build
|
|
2118
2447
|
if (!passed && (step.name === "Review" || step.name === "QA")) {
|
|
2448
|
+
// Room #26 blocker 34 (redesign, 2026-09-21): the false-negative guard,
|
|
2449
|
+
// the Case B budget skip, and the already-merged corrective are all
|
|
2450
|
+
// gone. Emptiness is classified mechanically before Review, so a
|
|
2451
|
+
// rejected Review is never re-adjudicated here — the workflow owns the
|
|
2452
|
+
// git facts, the reviewer owns quality/spec compliance, and the budget
|
|
2453
|
+
// applies uniformly to every rejection. A rejected frozen merge
|
|
2454
|
+
// bounces with the reviewer's notes; a real fix commit becomes
|
|
2455
|
+
// has-work, and a do-nothing loop exhausts the normal budget.
|
|
2119
2456
|
totalReworkCount++;
|
|
2120
2457
|
if (totalReworkCount > MAX_TOTAL_REWORK) {
|
|
2121
2458
|
log("Shared rework budget exhausted for task " + taskId + " — worktree preserved at " + WORKTREE_PRESERVED_HINT + " for manual inspection");
|
|
2122
2459
|
return await parkTask("Exceeded shared rework budget (" + MAX_TOTAL_REWORK + " total rework attempts across Review and QA) after " + step.name + " rejection. Worktree preserved.");
|
|
2123
2460
|
}
|
|
2124
2461
|
rejectionNotes = summary;
|
|
2125
|
-
// Already-merged corrective (room #16 blocker 11): when Review rejected
|
|
2126
|
-
// an empty branch but the work is already on the integration target (the workflow verified
|
|
2127
|
-
// the sha), Wren must declare it — not re-implement or re-commit
|
|
2128
|
-
// already-landed work. Scoped to the empty-branch rejection; any other
|
|
2129
|
-
// rejection already carries its own specific notes.
|
|
2130
|
-
if (step.name === "Review" && alreadyMergedSha && /no commits ahead of (main|the integration target)/i.test(summary)) {
|
|
2131
|
-
rejectionNotes += "\n\nCORRECTIVE (from the workflow, not the reviewer): the deliverable is already on the integration target — the workflow mechanically verified that " + alreadyMergedSha + " is an ancestor of the integration target. Do NOT re-implement the work and do NOT create a new commit for it. In your Build report, declare exactly: repo_diff: none (already-merged: " + alreadyMergedSha + ") — then end with VERDICT: PASS.";
|
|
2132
|
-
log("Rework corrective appended for task " + taskId + ": already-merged " + alreadyMergedSha + " — Wren must declare, not rebuild");
|
|
2133
|
-
}
|
|
2134
2462
|
i = BUILD_INDEX;
|
|
2135
2463
|
log(step.name + " rejected — bouncing to Build (rework #" + totalReworkCount + " of " + MAX_TOTAL_REWORK + ")");
|
|
2136
2464
|
continue;
|