muse-crew 0.7.9 → 0.7.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +19 -11
- package/docs/guide.md +2 -2
- package/docs/ooda-report.md +123 -0
- package/docs/publish-verification.md +253 -88
- package/docs/visual-verdict.md +81 -67
- package/lib/AGENTS.md +7 -0
- package/lib/append-ooda-step.js +167 -0
- package/lib/build-readback-request.js +130 -0
- package/lib/compose-evidence-caption.js +141 -0
- package/lib/crew-api.js +281 -2
- package/lib/edit-image.py +216 -0
- package/lib/render-html.js +142 -0
- package/lib/see-act.js +327 -0
- package/lib/serve-artifact.js +203 -0
- package/lib/verify-publish.js +265 -0
- package/lib/write-ooda-verdict.js +130 -0
- package/package.json +1 -1
- package/seed/cron-body-template.md +25 -3
- package/workflows/bugfix.js +220 -214
- package/workflows/chore.js +161 -113
- package/workflows/crew-dispatch.js +1 -1
- package/workflows/docs.js +1 -1
- package/workflows/standard.js +185 -229
package/workflows/standard.js
CHANGED
|
@@ -472,52 +472,23 @@ function verifyAppliedChanges(expected, applied) {
|
|
|
472
472
|
return { ok: true };
|
|
473
473
|
}
|
|
474
474
|
|
|
475
|
-
// Publish read-back request
|
|
476
|
-
//
|
|
477
|
-
//
|
|
478
|
-
//
|
|
479
|
-
//
|
|
480
|
-
//
|
|
481
|
-
//
|
|
482
|
-
//
|
|
483
|
-
//
|
|
484
|
-
//
|
|
485
|
-
//
|
|
475
|
+
// Publish read-back request (currently unavailable): the verbatim_request
|
|
476
|
+
// the parent protocol (docs/publish-verification.md) would hand to an
|
|
477
|
+
// independent read-back tool after the artifact build lands. artifact_inspect
|
|
478
|
+
// was removed by the platform (2026-09-14); artifact.inspect is malfunction
|
|
479
|
+
// diagnosis, not a substitute — so no agent-callable read-back tool exists
|
|
480
|
+
// and this request cannot currently be issued. Pure function — no I/O, no
|
|
481
|
+
// clock. The request carries the merged diff as the expected change and asks
|
|
482
|
+
// for an independent read of the artifact's actual source: for each file, the
|
|
483
|
+
// exact current text of the changed regions plus a per-line present/absent
|
|
484
|
+
// finding. Until a read-back path exists, the parent cannot independently
|
|
485
|
+
// confirm content and verification parks at "publish: verification-requested"
|
|
486
|
+
// (see docs/publish-verification.md). This preserves the circularity break
|
|
487
|
+
// that hollowed canary run 8 (2026-09-11): verifyAppliedChanges compares the
|
|
488
|
+
// builder's applied-report against the diff the report was derived from — a
|
|
489
|
+
// fabricated report passes by construction. Independent read-back cannot be
|
|
490
|
+
// fabricated from the diff;
|
|
486
491
|
// it must match the artifact's real content.
|
|
487
|
-
function buildPublishReadbackRequest(taskId, commit, diff, buildAgentId) {
|
|
488
|
-
// Build-ID correlation (2026-09-12): buildAgentId is the build.agent_id the
|
|
489
|
-
// workflow observed for the publish attempt (the artifact system's in-flight
|
|
490
|
-
// build correlation ID — not a durable post-completion identifier). The read-back request carries it so the parent can
|
|
491
|
-
// prove the read-back inspected the live build of THIS attempt — not a
|
|
492
|
-
// different build's output. Null/empty means the edit was accepted but
|
|
493
|
-
// never correlated to a builder run. Pure function of inputs — no I/O,
|
|
494
|
-
// no clock.
|
|
495
|
-
var buildIdLine = (typeof buildAgentId === "string" && buildAgentId.length > 0)
|
|
496
|
-
? "Expected builder build agent_id: " + buildAgentId + " (the artifact system's in-flight correlation ID for this publish attempt — not a durable post-completion identifier).\n"
|
|
497
|
-
: "No build agent_id was observed for this publish attempt (the edit was accepted but never correlated to a builder run) — say so explicitly in your report.\n";
|
|
498
|
-
return (
|
|
499
|
-
"Publish content read-back for task " + taskId + ", merge commit " + commit + ".\n" +
|
|
500
|
-
"The unified diff below was supposed to be applied to this artifact's source tree and deployed. Do NOT modify anything.\n" +
|
|
501
|
-
"Do NOT rely on the builder's applied-changes report — it is derived from this same diff, so it cannot confirm the content. Read the artifact's CURRENT source directly.\n" +
|
|
502
|
-
"\n" +
|
|
503
|
-
buildIdLine +
|
|
504
|
-
"Report the live build's agent_id as seen in artifact_status (or state explicitly that no build/agent_id is visible). If an expected agent_id is given above and the live one differs, say so exactly — the read-back may be inspecting a different build's output.\n" +
|
|
505
|
-
"\n" +
|
|
506
|
-
"UNIFIED DIFF (expected change):\n" +
|
|
507
|
-
"```diff\n" + diff + "\n```\n" +
|
|
508
|
-
"\n" +
|
|
509
|
-
"For each file in the diff:\n" +
|
|
510
|
-
"1. Read the file's CURRENT content in the artifact source tree.\n" +
|
|
511
|
-
"2. Quote the exact current text of the regions around the changed lines.\n" +
|
|
512
|
-
"3. For every added (+) line in the diff, state whether that exact line is PRESENT in the current source.\n" +
|
|
513
|
-
"4. For every removed (-) line in the diff, state whether that exact line is ABSENT from the current source.\n" +
|
|
514
|
-
"5. Report build/deploy health and the console error count.\n" +
|
|
515
|
-
"\n" +
|
|
516
|
-
"Return the per-file present/absent findings with the quoted observed lines. Do not modify anything.\n" +
|
|
517
|
-
"This read-back feeds the parent content-verification protocol (docs/publish-verification.md): the parent stamps provenance only when every added line is present and every removed line is absent."
|
|
518
|
-
);
|
|
519
|
-
}
|
|
520
|
-
|
|
521
492
|
// Pre-publish base observation (diagnostic, 2026-09-12): instruction fragment
|
|
522
493
|
// for the builder's edit request, asking it to report the sha256 of each
|
|
523
494
|
// touched file's CURRENT content BEFORE applying the diff. Pure function —
|
|
@@ -751,44 +722,6 @@ async function baselineStatus() {
|
|
|
751
722
|
return { baseline_found: false, baseline_kind: "", baseline_refs: "", requested_count: 0, evidence_count: 0 };
|
|
752
723
|
}
|
|
753
724
|
}
|
|
754
|
-
// Visual verdict status: the parent records the visual verdict as a note
|
|
755
|
-
// event after the rendered post-change inspection results arrive. Finds the
|
|
756
|
-
// latest step "QA" session's started_at, then note events NEWER than it
|
|
757
|
-
// whose message starts exactly "visual_verdict: PASS" / "visual_verdict:
|
|
758
|
-
// FAIL". Returns { found, verdict, detail }. A fresh key per call: the
|
|
759
|
-
// verdict lands while this run is parked.
|
|
760
|
-
let visualVerdictCallCount = 0;
|
|
761
|
-
async function visualVerdictStatus() {
|
|
762
|
-
visualVerdictCallCount++;
|
|
763
|
-
try {
|
|
764
|
-
var vv = await agent(
|
|
765
|
-
"Read this task's QA session and note events.\n" +
|
|
766
|
-
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
|
|
767
|
-
"Find the LATEST session with task_id \"" + taskId + "\" and step \"QA\" in the returned sessions array and note its started_at timestamp (call it QA_START; use empty string if there is no QA session).\n" +
|
|
768
|
-
"Then run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId }) + "\n" +
|
|
769
|
-
"Consider only events with type \"note\" whose timestamp is newer than QA_START. Among them, find messages starting exactly with \"visual_verdict: PASS\" or \"visual_verdict: FAIL\" (exact prefix, case-sensitive); use the latest such message.\n" +
|
|
770
|
-
"Return JSON { \"found\": <true if such a message exists>, \"verdict\": \"<\"PASS\" or \"FAIL\" from that message, or empty string>\", \"detail\": \"<the text after the prefix in that message, or empty string>\" } and nothing else.",
|
|
771
|
-
{
|
|
772
|
-
key: "visual-verdict-status-" + taskId + "-" + visualVerdictCallCount,
|
|
773
|
-
label: "Reading visual verdict status",
|
|
774
|
-
schema: {
|
|
775
|
-
type: "object",
|
|
776
|
-
properties: {
|
|
777
|
-
found: { type: "boolean" },
|
|
778
|
-
verdict: { type: "string" },
|
|
779
|
-
detail: { type: "string" }
|
|
780
|
-
},
|
|
781
|
-
required: ["found", "verdict", "detail"]
|
|
782
|
-
}
|
|
783
|
-
}
|
|
784
|
-
);
|
|
785
|
-
return { found: !!(vv && vv.found), verdict: (vv && vv.verdict) || "", detail: (vv && vv.detail) || "" };
|
|
786
|
-
} catch (e) {
|
|
787
|
-
log("visualVerdictStatus: agent call failed (" + (e && e.message ? e.message : e) + ") — treating as not found");
|
|
788
|
-
return { found: false, verdict: "", detail: "" };
|
|
789
|
-
}
|
|
790
|
-
}
|
|
791
|
-
|
|
792
725
|
// STEPS inline — export const meta is parsed as metadata, not a runtime binding
|
|
793
726
|
const STEPS = [
|
|
794
727
|
{ name: "Triage", identity: "sage" },
|
|
@@ -1227,12 +1160,10 @@ while (i < STEPS.length) {
|
|
|
1227
1160
|
}
|
|
1228
1161
|
}
|
|
1229
1162
|
|
|
1230
|
-
//
|
|
1231
|
-
//
|
|
1232
|
-
//
|
|
1233
|
-
//
|
|
1234
|
-
// parks for the parent protocol when no visual_verdict: note event exists
|
|
1235
|
-
// yet.
|
|
1163
|
+
// Experiential routing: experiential artifact tasks run the Hazel
|
|
1164
|
+
// experiential QA prompt (built in the PUBLISH_TYPE === "artifact" branch
|
|
1165
|
+
// of the QA step) — Hazel drives the artifact herself and owns the visual
|
|
1166
|
+
// verdict through her OODA report and write-ooda-verdict ledger.
|
|
1236
1167
|
var qaVisual = false;
|
|
1237
1168
|
if (step.name === "QA") {
|
|
1238
1169
|
qaVisual = (await resolveExperiential()) === "yes" && PUBLISH_TYPE === "artifact";
|
|
@@ -1243,7 +1174,7 @@ while (i < STEPS.length) {
|
|
|
1243
1174
|
var instructions = "";
|
|
1244
1175
|
|
|
1245
1176
|
if (step.name === "Triage") {
|
|
1246
|
-
instructions = "Validate the task,
|
|
1177
|
+
instructions = "Validate the task against the project's repo at " + REPO_PATH + " — that exact checkout, not any other copy of the project on disk. If you run git commands, cd " + REPO_PATH + " first.\nCheck clarity, note dependencies, confirm the standard workflow assignment.\nWrite a brief triage assessment as notes for the next step.\nReport back in plain prose — what you found.\nEXPERIENTIAL FLAG: does this task change anything rendered and visible in the project's user-facing artifact (pages, components, styles, layout, copy, visual states)? If yes it is experiential and gets baseline captures (plus a visual verdict where the workflow has a QA phase). End your report with exactly one line on its own, lowercase, unrephrased: experiential: yes — or experiential: no. This line is machine-read.";
|
|
1247
1178
|
|
|
1248
1179
|
} else if (step.name === "Map") {
|
|
1249
1180
|
var mapGatePara = "";
|
|
@@ -1371,10 +1302,11 @@ while (i < STEPS.length) {
|
|
|
1371
1302
|
// (fail-closed). There is deliberately NO workflow-side provenance
|
|
1372
1303
|
// stamp: the builder's applied-report is circular (canary run 8,
|
|
1373
1304
|
// 2026-09-11), so the stamp moved to the parent — after the build
|
|
1374
|
-
// lands, the workflow
|
|
1375
|
-
//
|
|
1376
|
-
//
|
|
1377
|
-
//
|
|
1305
|
+
// lands, the workflow records the session completed and parks with
|
|
1306
|
+
// "publish: verification-requested". The parent owns verification
|
|
1307
|
+
// (docs/publish-verification.md); the independent read-back step is
|
|
1308
|
+
// currently unavailable (no agent-callable read-back tool exists —
|
|
1309
|
+
// artifact_inspect was removed by the platform 2026-09-14).
|
|
1378
1310
|
// QA's provenance check enforces the stamp mechanically.
|
|
1379
1311
|
var artifactPublish = null;
|
|
1380
1312
|
var publishLockRefreshed = false;
|
|
@@ -1423,14 +1355,45 @@ while (i < STEPS.length) {
|
|
|
1423
1355
|
// changes — no prose claim to trust. If the artifact tool namespace
|
|
1424
1356
|
// is missing from this child it reports honestly and the workflow
|
|
1425
1357
|
// retries once with a fresh key (bounded); anything else parks.
|
|
1358
|
+
// Publish diff base (2026-09-14, task 0c53af4e): the carried diff is
|
|
1359
|
+
// BASE..HEAD where BASE is the previously-stamped provenance
|
|
1360
|
+
// source_commit — NOT HEAD^1. A push-time reconcile merge puts the
|
|
1361
|
+
// task's own changes behind an intermediate merge, so HEAD^1..HEAD
|
|
1362
|
+
// silently drops the task's fix while the artifact builds without
|
|
1363
|
+
// it. The stamped base is the artifact's actual content; BASE..HEAD
|
|
1364
|
+
// is the complete unpublished delta. Empty tree only for a genuine
|
|
1365
|
+
// first publish (no provenance stamped yet).
|
|
1366
|
+
var EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
|
|
1367
|
+
var provResult = await agent(
|
|
1368
|
+
crewCmd("get-provenance", {}) + "\n" +
|
|
1369
|
+
"Return JSON { \"provenance\": <the CLI's provenance object, or null when nothing is stamped> } and nothing else. Do not interpret it.",
|
|
1370
|
+
{ key: attemptKey("publish-provenance-base-" + taskId, totalReworkCount), label: "Reading stamped publish base",
|
|
1371
|
+
schema: { type: "object", properties: { provenance: { type: ["object", "null"] } }, required: ["provenance"] } }
|
|
1372
|
+
);
|
|
1373
|
+
var publishBase = (provResult.provenance && provResult.provenance.source_commit) || "";
|
|
1374
|
+
publishBase = String(publishBase).trim();
|
|
1375
|
+
if (!publishBase) {
|
|
1376
|
+
publishBase = EMPTY_TREE_SHA;
|
|
1377
|
+
log("Publish base for task " + taskId + ": no provenance stamped yet — using empty tree (first publish)");
|
|
1378
|
+
} else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
|
|
1379
|
+
return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
|
|
1380
|
+
}
|
|
1426
1381
|
var diffResult = await agent(
|
|
1427
|
-
"Run: cd " + REPO_PATH + " &&
|
|
1428
|
-
"
|
|
1429
|
-
|
|
1430
|
-
|
|
1382
|
+
"Run: cd " + REPO_PATH + " && BASE='" + publishBase + "' && HEAD=$(git rev-parse HEAD) && " +
|
|
1383
|
+
"if [ \"$BASE\" = '" + EMPTY_TREE_SHA + "' ]; then ANCESTOR=yes; else git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no; fi && " +
|
|
1384
|
+
"echo '---COMMIT---' && echo \"$HEAD\" && echo '---BASE---' && echo \"$BASE\" && echo '---ANCESTOR---' && echo \"$ANCESTOR\" && " +
|
|
1385
|
+
"if [ \"$ANCESTOR\" = yes ]; then echo '---DIFF---' && git diff \"$BASE\" \"$HEAD\" && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r \"$BASE\" \"$HEAD\"; fi\n" +
|
|
1386
|
+
"Return JSON { \"commit\": \"<HEAD trimmed>\", \"base\": \"<BASE trimmed>\", \"ancestor\": \"<yes|no>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
|
|
1387
|
+
{ key: attemptKey("publish-artifact-diff-" + taskId, totalReworkCount), label: "Computing publish diff from stamped base",
|
|
1388
|
+
schema: { type: "object", properties: { commit: { type: "string" }, base: { type: "string" }, ancestor: { type: "string" }, diff: { type: "string" }, files: { type: "string" } }, required: ["commit", "base", "ancestor", "diff"] } }
|
|
1431
1389
|
);
|
|
1390
|
+
if ((diffResult.ancestor || "").trim() !== "yes") {
|
|
1391
|
+
return await parkTask("Publish base " + publishBase.slice(0, 12) + " is not an ancestor of HEAD " + (diffResult.commit || "").trim().slice(0, 12) + " — the stamped provenance does not lead to the integrated commit. Human attention needed.");
|
|
1392
|
+
}
|
|
1393
|
+
if ((diffResult.base || "").trim() !== publishBase) {
|
|
1394
|
+
return await parkTask("Publish diff base mismatch: agent reported '" + (diffResult.base || "").trim().slice(0, 12) + "' but the stamped base is '" + publishBase.slice(0, 12) + "'. Human attention needed.");
|
|
1395
|
+
}
|
|
1432
1396
|
var mergeCommitForPublish = (diffResult.commit || "").trim();
|
|
1433
|
-
var mergeParentForPublish = (diffResult.parent || "").trim();
|
|
1434
1397
|
var mergeDiff = diffResult.diff || "";
|
|
1435
1398
|
if (!mergeDiff.trim()) {
|
|
1436
1399
|
return await parkTask("Publish diff is empty for commit " + (mergeCommitForPublish || "unknown") + " — a merge lock was held but there is no change to carry. Human attention needed.");
|
|
@@ -1441,11 +1404,15 @@ while (i < STEPS.length) {
|
|
|
1441
1404
|
if (/^rename from /m.test(mergeDiff)) {
|
|
1442
1405
|
return await parkTask("Publish diff contains a rename — the diff transport cannot carry renames. Human attention needed.");
|
|
1443
1406
|
}
|
|
1444
|
-
var mergeDiffLines = mergeDiff.split("\n").length;
|
|
1445
|
-
if (mergeDiffLines > 200) {
|
|
1446
|
-
return await parkTask("Publish diff is " + mergeDiffLines + " lines (budget 200) — too large for the diff transport. Human attention needed.");
|
|
1447
|
-
}
|
|
1448
1407
|
var expectedChanges = parseUnifiedDiff(mergeDiff);
|
|
1408
|
+
// Budget counts CHANGED lines (added + removed), not raw unified-diff
|
|
1409
|
+
// output lines: context lines and file headers inflated the old
|
|
1410
|
+
// split("\n").length count ~2x, parking a 95-line change against a
|
|
1411
|
+
// 200-line budget (Gate 1 Journey 3 attempt 3, 2026-09-13).
|
|
1412
|
+
var mergeDiffChangedLines = expectedChanges.reduce(function (n, f) { return n + f.added.length + f.removed.length; }, 0);
|
|
1413
|
+
if (mergeDiffChangedLines > 200) {
|
|
1414
|
+
return await parkTask("Publish diff changes " + mergeDiffChangedLines + " lines (budget 200) — too large for the diff transport. Human attention needed.");
|
|
1415
|
+
}
|
|
1449
1416
|
if (expectedChanges.length === 0) {
|
|
1450
1417
|
return await parkTask("Publish diff parsed to zero files for commit " + (mergeCommitForPublish || "unknown") + " — cannot verify application. Human attention needed.");
|
|
1451
1418
|
}
|
|
@@ -1458,12 +1425,16 @@ while (i < STEPS.length) {
|
|
|
1458
1425
|
// but never parks. The observation tells us what the publish actually
|
|
1459
1426
|
// reads, so the subsequent fix can require the right base.
|
|
1460
1427
|
var expectedBaseHashes = {};
|
|
1461
|
-
|
|
1428
|
+
if (publishBase === EMPTY_TREE_SHA) {
|
|
1429
|
+
// First publish: every file in the diff is new to the artifact.
|
|
1430
|
+
expectedChanges.forEach(function (f) { expectedBaseHashes[f.path] = "NEW-FILE"; });
|
|
1431
|
+
log("Publish expected base hashes for task " + taskId + ": empty tree (first publish) — all " + expectedChanges.length + " file(s) new");
|
|
1432
|
+
} else try {
|
|
1462
1433
|
// Shell-quote helper (no regex-with-quote: the test parser does not
|
|
1463
1434
|
// understand regex literals containing quotes).
|
|
1464
1435
|
var sq = function(s) { return "'" + String(s).split("'").join("'\\''") + "'"; };
|
|
1465
1436
|
var baseHashResult = await agent(
|
|
1466
|
-
"Run: cd " + REPO_PATH + " && parent=" + sq(
|
|
1437
|
+
"Run: cd " + REPO_PATH + " && parent=" + sq(publishBase) + " && for f in " + expectedChanges.map(function(f) { return sq(f.path); }).join(" ") + "; do printf '%s:' \"$f\"; git show \"$parent:$f\" 2>/dev/null | sha256sum | cut -d' ' -f1; done\n" +
|
|
1467
1438
|
"Return JSON { \"hashes\": \"<newline-separated <path>:<sha256> lines, empty hash means the file is new in this diff>\" } and nothing else.",
|
|
1468
1439
|
{ key: attemptKey("publish-base-hashes-" + taskId, totalReworkCount), label: "Computing expected base content hashes",
|
|
1469
1440
|
schema: { type: "object", properties: { hashes: { type: "string" } }, required: ["hashes"] } }
|
|
@@ -1472,7 +1443,7 @@ while (i < STEPS.length) {
|
|
|
1472
1443
|
var m = /^([^:]+):([0-9a-f]*)$/.exec(line.trim());
|
|
1473
1444
|
if (m) expectedBaseHashes[m[1]] = m[2] || "NEW-FILE";
|
|
1474
1445
|
});
|
|
1475
|
-
log("Publish expected base hashes for task " + taskId + " (
|
|
1446
|
+
log("Publish expected base hashes for task " + taskId + " (stamped base " + publishBase.slice(0, 12) + "): " + JSON.stringify(expectedBaseHashes));
|
|
1476
1447
|
} catch (e) {
|
|
1477
1448
|
log("Publish expected base hash computation failed for task " + taskId + " (non-fatal, observation degraded): " + (e && e.message ? e.message : e));
|
|
1478
1449
|
}
|
|
@@ -1530,6 +1501,7 @@ while (i < STEPS.length) {
|
|
|
1530
1501
|
required: ["edit_started", "build_agent_id", "applied"] };
|
|
1531
1502
|
var rebuildTrigger = null;
|
|
1532
1503
|
var rebuildReportMissing = false; // true if the edit went through but the agent returned no applied report (structured-output failure) — the smoke-check is skipped; the parent's independent read-back is the verification
|
|
1504
|
+
var rebuildEvidenceNote = null; // human-readable evidence line for the ledger when the edit is confirmed via fallback evidence (in-flight poll or durable audit dir) rather than the trigger's own report
|
|
1533
1505
|
// The trigger key of the attempt that last ran, for the publish ledger.
|
|
1534
1506
|
// Minted once here (not re-minted per use site) so the ledger always
|
|
1535
1507
|
// records the exact key that was issued — and so a re-minted duplicate
|
|
@@ -1548,6 +1520,32 @@ while (i < STEPS.length) {
|
|
|
1548
1520
|
// (2026-09-12, task 23ca8f3f): computed once the trigger outcome is
|
|
1549
1521
|
// known, logged loudly, never a park.
|
|
1550
1522
|
var publishAppliedObservation = null; // "match" | "mismatch: <reason>" | "missing-report" — observation only, never a park
|
|
1523
|
+
// Durable-evidence snapshot (2026-09-14): the structured-output
|
|
1524
|
+
// fallback below only observes IN-FLIGHT builds. A build that
|
|
1525
|
+
// finished before the poll leaves no in-flight trace — but the
|
|
1526
|
+
// platform's audit harness leaves a durable one:
|
|
1527
|
+
// ~/workspace/ts-spaces/<slug>/audits/<timestamp>-<id>/ per
|
|
1528
|
+
// completed build. Snapshot the listing BEFORE the trigger so the
|
|
1529
|
+
// fallback can diff before/after: a directory appearing during the
|
|
1530
|
+
// trigger window is positive evidence the edit went through and
|
|
1531
|
+
// the build completed. Best-effort and non-gating: if the snapshot
|
|
1532
|
+
// fails, the durable check is skipped and the fallback behaves as
|
|
1533
|
+
// before. No wall-clock in-script (deterministic replay) — the
|
|
1534
|
+
// comparison is a pure before/after set diff.
|
|
1535
|
+
var auditDirsBeforeTrigger = [];
|
|
1536
|
+
try {
|
|
1537
|
+
var auditBefore = await agent(
|
|
1538
|
+
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
|
|
1539
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1540
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
|
|
1541
|
+
{ key: attemptKey("publish-audit-before-" + taskId, totalReworkCount), label: "Snapshotting audit dirs before rebuild trigger",
|
|
1542
|
+
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1543
|
+
);
|
|
1544
|
+
auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1545
|
+
log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
|
|
1546
|
+
} catch (auditBeforeErr) {
|
|
1547
|
+
log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
|
|
1548
|
+
}
|
|
1551
1549
|
try {
|
|
1552
1550
|
rebuildTrigger = await agent(rebuildPrompt,
|
|
1553
1551
|
{ key: rebuildAttemptKey, label: "Triggering artifact rebuild", schema: rebuildSchema });
|
|
@@ -1606,7 +1604,46 @@ while (i < STEPS.length) {
|
|
|
1606
1604
|
rebuildTrigger = { edit_started: true, error: "", applied: null };
|
|
1607
1605
|
rebuildReportMissing = true;
|
|
1608
1606
|
rebuildAgentId = acceptedAgentId;
|
|
1607
|
+
rebuildEvidenceNote = "edit confirmed via build-state poll after structured-output failure (build " + acceptedAgentId + "); builder applied-report missing";
|
|
1609
1608
|
} else {
|
|
1609
|
+
// Durable completion check (2026-09-14): the in-flight poll
|
|
1610
|
+
// above only sees RUNNING builds. Attempt 7 (2026-09-14) proved
|
|
1611
|
+
// the gap: the trigger child applied the edit, the build ran
|
|
1612
|
+
// and completed — the platform's audit harness captured it
|
|
1613
|
+
// mid-window — then the child failed to return JSON. The
|
|
1614
|
+
// fallback poll saw no in-flight build, so a successful publish
|
|
1615
|
+
// parked as "unknown". Diff the audit-dir listing against the
|
|
1616
|
+
// pre-trigger snapshot: a timestamped directory that appeared
|
|
1617
|
+
// during the trigger window is positive evidence the edit went
|
|
1618
|
+
// through and the build completed. This never re-issues the
|
|
1619
|
+
// edit and never stamps provenance — it only routes to the
|
|
1620
|
+
// parent's independent content read-back, which remains the
|
|
1621
|
+
// real verification.
|
|
1622
|
+
var newAuditDirs = [];
|
|
1623
|
+
try {
|
|
1624
|
+
var auditAfter = await agent(
|
|
1625
|
+
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
|
|
1626
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1627
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
|
|
1628
|
+
{ key: attemptKey("publish-audit-after-" + taskId, totalReworkCount), label: "Re-listing audit dirs after trigger failure",
|
|
1629
|
+
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1630
|
+
);
|
|
1631
|
+
var auditDirsAfterTrigger = String((auditAfter && auditAfter.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1632
|
+
// Only timestamped build dirs count — the "latest" symlink
|
|
1633
|
+
// and anything else are not builds.
|
|
1634
|
+
newAuditDirs = auditDirsAfterTrigger.filter(function (d) {
|
|
1635
|
+
return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
|
|
1636
|
+
});
|
|
1637
|
+
} catch (auditAfterErr) {
|
|
1638
|
+
log("Publish audit-dir re-list after trigger failure failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
|
|
1639
|
+
}
|
|
1640
|
+
if (newAuditDirs.length > 0) {
|
|
1641
|
+
log("Publish rebuild trigger: new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and the build completed despite the structured-output failure. Skipping applied-report smoke-check; parent read-back is the verification.");
|
|
1642
|
+
rebuildTrigger = { edit_started: true, error: "", applied: null };
|
|
1643
|
+
rebuildReportMissing = true;
|
|
1644
|
+
rebuildAgentId = null;
|
|
1645
|
+
rebuildEvidenceNote = "edit confirmed via durable audit evidence after structured-output failure (new audit dir " + newAuditDirs[0] + "); builder applied-report missing";
|
|
1646
|
+
} else {
|
|
1610
1647
|
// No build observed — but that proves nothing (a fast-completing
|
|
1611
1648
|
// build can finish between polls, or the check itself failed). The
|
|
1612
1649
|
// outcome is UNKNOWN. No retry: re-issuing the edit here duplicated
|
|
@@ -1622,6 +1659,7 @@ while (i < STEPS.length) {
|
|
|
1622
1659
|
detail: "structured-output failure on rebuild trigger; build-state poll saw no build (or the check itself failed); edit may have been accepted as pending_init"
|
|
1623
1660
|
}, totalReworkCount);
|
|
1624
1661
|
return await parkTask("Publish outcome unknown: the rebuild trigger's child did not return JSON, and the follow-up build-state poll could not observe a build for slug " + PUBLISH_SLUG + ". The edit may have been accepted as pending_init, so no retry was issued — a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion before re-driving Publish. Fail-closed.");
|
|
1662
|
+
}
|
|
1625
1663
|
}
|
|
1626
1664
|
}
|
|
1627
1665
|
if (!rebuildReportMissing && !rebuildTrigger.edit_started && rebuildTrigger.error === "artifact_tools missing after load") {
|
|
@@ -1691,7 +1729,7 @@ while (i < STEPS.length) {
|
|
|
1691
1729
|
applied_report: publishAppliedObservation,
|
|
1692
1730
|
outcome: "submitted",
|
|
1693
1731
|
detail: rebuildReportMissing
|
|
1694
|
-
? "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing"
|
|
1732
|
+
? (rebuildEvidenceNote || "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing")
|
|
1695
1733
|
: "edit accepted; builder applied-report received"
|
|
1696
1734
|
}, totalReworkCount);
|
|
1697
1735
|
} else if (rebuildTrigger) {
|
|
@@ -1789,12 +1827,14 @@ while (i < STEPS.length) {
|
|
|
1789
1827
|
// builder's applied-report is derived from the carried diff, so
|
|
1790
1828
|
// verifyAppliedChanges above is circular — a fabricated report
|
|
1791
1829
|
// passes by construction, and every phase went green on a hollow
|
|
1792
|
-
// build. The stamp moves to the parent (docs/publish-verification.md)
|
|
1793
|
-
//
|
|
1794
|
-
//
|
|
1795
|
-
//
|
|
1796
|
-
//
|
|
1797
|
-
//
|
|
1830
|
+
// build. The stamp moves to the parent (docs/publish-verification.md);
|
|
1831
|
+
// the independent read-back step is currently unavailable (no
|
|
1832
|
+
// agent-callable read-back tool exists — artifact_inspect was
|
|
1833
|
+
// removed by the platform 2026-09-14), so the parent cannot
|
|
1834
|
+
// confirm content and the task parks for verification.
|
|
1835
|
+
// QA's provenance check enforces the stamp mechanically.
|
|
1836
|
+
// An unverified publish fails loudly in QA instead of passing
|
|
1837
|
+
// silently here.
|
|
1798
1838
|
publishBuildLanded = true;
|
|
1799
1839
|
artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
|
|
1800
1840
|
log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
|
|
@@ -1892,12 +1932,19 @@ while (i < STEPS.length) {
|
|
|
1892
1932
|
var safeDesc = taskDescription.replace(/"/g, "'").replace(/\\/g, "\\\\").slice(0, 500);
|
|
1893
1933
|
instructions = "You are code-blind QA. You NEVER read source files.\n" +
|
|
1894
1934
|
"Public docs (API.md, README, published action schemas) are NOT source code — read them freely, exactly as a user would.\n\n" +
|
|
1895
|
-
"STEP 1:
|
|
1896
|
-
"
|
|
1897
|
-
"
|
|
1898
|
-
"
|
|
1899
|
-
"
|
|
1900
|
-
"
|
|
1935
|
+
"STEP 1: Experiential visual inspection — drive the artifact as a user would, one browser step at a time.\n" +
|
|
1936
|
+
"You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
|
|
1937
|
+
"Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
|
|
1938
|
+
"a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and continue with the mechanical checks.\n" +
|
|
1939
|
+
"b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
|
|
1940
|
+
"c. Bounded see-act loop, at most 8 steps: drive the artifact as a user would, one browser step at a time. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
|
|
1941
|
+
"c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
|
|
1942
|
+
"c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
|
|
1943
|
+
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
1944
|
+
"d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the task's change needs such a sequence, report NOT POSSIBLE for that part and judge what you can.\n" +
|
|
1945
|
+
"e. Judge as a user against the task description: does the change render correctly? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
|
|
1946
|
+
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
1947
|
+
"Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
|
|
1901
1948
|
"STEP 2: Verify data integrity via the crew API.\n" +
|
|
1902
1949
|
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
|
|
1903
1950
|
"Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
|
|
@@ -1915,7 +1962,8 @@ while (i < STEPS.length) {
|
|
|
1915
1962
|
"For each issue, run in shell:\n" +
|
|
1916
1963
|
"node " + CREW_API + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + LAUNCH_PROJECT_ID + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
1917
1964
|
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
1918
|
-
"
|
|
1965
|
+
"BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
|
|
1966
|
+
"Report back in plain prose — what checks you ran and their results. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<what the task required>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten).";
|
|
1919
1967
|
} else {
|
|
1920
1968
|
instructions = "Test from a user's perspective. You are CODE-BLIND — do NOT read source code.\n" +
|
|
1921
1969
|
"Public docs (API.md, README) are NOT source code — read them freely, exactly as a user would.\n" +
|
|
@@ -1927,37 +1975,10 @@ while (i < STEPS.length) {
|
|
|
1927
1975
|
npmPublishCheck +
|
|
1928
1976
|
"Report back in plain prose — what you tested and found. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
1929
1977
|
}
|
|
1930
|
-
//
|
|
1931
|
-
//
|
|
1932
|
-
//
|
|
1933
|
-
//
|
|
1934
|
-
// Mechanical checks are kept verbatim from the artifact path above.
|
|
1935
|
-
if (qaVisual) {
|
|
1936
|
-
instructions = "You are code-blind QA. You NEVER read source files. Public docs are not source — read them as a user would.\n" +
|
|
1937
|
-
"VISUAL VERDICT OWNERSHIP: this task is experiential. The visual verdict is NOT yours to issue — it is produced after the rendered post-change inspection results arrive, by Hazel with the baseline and post-change evidence in hand.\n" +
|
|
1938
|
-
"Your VERDICT below covers the MECHANICAL CHECKS only. Do NOT call artifact_inspect (async; the parent triggers the post-change capture after your step).\n\n" +
|
|
1939
|
-
"MECHANICAL CHECKS:\n" +
|
|
1940
|
-
"STEP 2: Verify data integrity via the crew API.\n" +
|
|
1941
|
-
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
|
|
1942
|
-
"Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
|
|
1943
|
-
"DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale for a public-affecting change, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n\n" +
|
|
1944
|
-
"PROVENANCE CHECK: Run in shell and return the stdout verbatim:\n" + crewCmd("get-provenance", {}) + "\n" +
|
|
1945
|
-
"If provenance is null, report 'provenance missing — publish did not stamp source/crew release', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1946
|
-
"Run: cd " + REPO_PATH + " && git rev-parse HEAD — call this LIVE_HEAD.\n" +
|
|
1947
|
-
"Run: test -d " + crewHome + "/releases/<provenance.crew_release> (substitute the real stamped hash; do not run the literal placeholder). If the directory does not exist, FAIL: { \"passed\": false, \"summary\": \"provenance mismatch: crew_release [value from get-provenance] not found in release registry\" }.\n" +
|
|
1948
|
-
"If provenance.source_commit equals LIVE_HEAD, the source check passes — continue to STEP 3.\n" +
|
|
1949
|
-
"Otherwise the check is NOT failed yet: the parent stamps provenance AFTER post-deploy (docs/publish-verification.md), and post-deploy may commit an artifact-builder staging commit (\"rebuild: <task_id>\"), so LIVE_HEAD may sit ahead of the stamped commit ONLY IF every commit in between is such a rebuild marker. Verify exactly:\n" +
|
|
1950
|
-
"1. Run: cd " + REPO_PATH + " && git merge-base --is-ancestor <provenance.source_commit> LIVE_HEAD && echo ANCESTOR_OK (substitute the real stamped hash and LIVE_HEAD; do not run the literal placeholders). If this command fails, FAIL: { \"passed\": false, \"summary\": \"provenance mismatch: stamped source_commit is not an ancestor of live HEAD\" }.\n" +
|
|
1951
|
-
"2. Run: cd " + REPO_PATH + " && git log --format=%s <provenance.source_commit>..LIVE_HEAD (substitute real values). Every subject line MUST start with \"rebuild: \". If any line does not, report 'provenance mismatch: live HEAD moved past the stamped commit with non-rebuild source commits: [paste the offending subject lines]', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1952
|
-
"If both pass, the source check passes — the only drift since the stamp is builder staging output committed by post-deploy. Continue to STEP 3.\n\n" +
|
|
1953
|
-
"STEP 3: File follow-up tasks for any related issues you discover.\n" +
|
|
1954
|
-
"For each issue, run in shell:\n" +
|
|
1955
|
-
"node " + CREW_API + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + LAUNCH_PROJECT_ID + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
1956
|
-
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
1957
|
-
"BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
|
|
1958
|
-
"File follow-up tasks as today.\n" +
|
|
1959
|
-
"Report back in plain prose — what checks you ran and their results. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL on the mechanical checks.";
|
|
1960
|
-
}
|
|
1978
|
+
// qaVisual (experiential artifact tasks): Hazel owns the visual verdict
|
|
1979
|
+
// through the experiential QA prompt built in the PUBLISH_TYPE ===
|
|
1980
|
+
// "artifact" branch above (STEP 1 see-act loop + OODA report). There is
|
|
1981
|
+
// no parent visual-verdict protocol anymore — no override here.
|
|
1961
1982
|
}
|
|
1962
1983
|
|
|
1963
1984
|
// Task event history — Review and Publish are excluded. Review is cold by
|
|
@@ -2266,9 +2287,10 @@ while (i < STEPS.length) {
|
|
|
2266
2287
|
// it to HEAD: that verifies the stamp, not the content. Canary run 8
|
|
2267
2288
|
// (2026-09-11) passed it with a hollow build — the stamp was honest, the
|
|
2268
2289
|
// artifact was stale, all eight phases green. The stamp now moves to the
|
|
2269
|
-
// parent
|
|
2270
|
-
//
|
|
2271
|
-
//
|
|
2290
|
+
// parent (docs/publish-verification.md); the independent read-back step
|
|
2291
|
+
// is currently unavailable (no agent-callable read-back tool exists —
|
|
2292
|
+
// artifact_inspect was removed by the platform 2026-09-14), so the parent
|
|
2293
|
+
// cannot confirm content and the task parks for verification. QA's
|
|
2272
2294
|
// provenance check enforces the stamp — an unverified publish fails loudly
|
|
2273
2295
|
// there instead of passing silently here.
|
|
2274
2296
|
// Skip-aware (park 2026-09-11): an empty-diff Integrate takes no merge
|
|
@@ -2276,37 +2298,13 @@ while (i < STEPS.length) {
|
|
|
2276
2298
|
// there is no new content to verify, so verification is vacuous.
|
|
2277
2299
|
// publishSkippedNoLock is workflow-computed state from the explicit
|
|
2278
2300
|
// lock-status read in STEP 0, not agent prose.
|
|
2279
|
-
|
|
2280
|
-
|
|
2281
|
-
|
|
2282
|
-
|
|
2283
|
-
|
|
2284
|
-
|
|
2285
|
-
|
|
2286
|
-
try {
|
|
2287
|
-
var inspectResult = await agent(
|
|
2288
|
-
ARTIFACT_LOAD_PREAMBLE +
|
|
2289
|
-
"Call artifact_inspect with slug \"" + PUBLISH_SLUG + "\", repair_authorized false, and verbatim_request exactly as follows:\n" +
|
|
2290
|
-
"<<<READBACK_REQUEST\n" + buildPublishReadbackRequest(taskId, mergeCommitForPublish, mergeDiff, rebuildAgentId) + "\nREADBACK_REQUEST\n" +
|
|
2291
|
-
"If artifact_inspect is still not available after the load, do NOT improvise — return { \"triggered\": false, \"inspection_id\": \"\", \"error\": \"artifact_tools missing after load\" } and nothing else.\n" +
|
|
2292
|
-
"Return JSON { \"triggered\": <true if the inspection started, false otherwise>, \"inspection_id\": \"<the inspection id, or empty string>\", \"error\": \"<details or empty string>\" } and nothing else.",
|
|
2293
|
-
{ key: attemptKey("publish-verify-inspect-" + taskId, totalReworkCount), label: "Triggering publish content read-back",
|
|
2294
|
-
schema: { type: "object", properties: { triggered: { type: "boolean" }, inspection_id: { type: "string" }, error: { type: "string" } }, required: ["triggered"] } }
|
|
2295
|
-
);
|
|
2296
|
-
publishVerifyInspect.triggered = !!(inspectResult && inspectResult.triggered);
|
|
2297
|
-
publishVerifyInspect.inspection_id = (inspectResult && inspectResult.inspection_id) || "";
|
|
2298
|
-
publishVerifyInspect.error = (inspectResult && inspectResult.error) || "";
|
|
2299
|
-
if (publishVerifyInspect.triggered) {
|
|
2300
|
-
log("Publish content read-back inspection triggered for task " + taskId + ": " + publishVerifyInspect.inspection_id);
|
|
2301
|
-
} else {
|
|
2302
|
-
log("Publish content read-back inspect trigger failed for task " + taskId + ": " + (publishVerifyInspect.error || "not started") + " — the park below asks the parent to trigger it manually");
|
|
2303
|
-
}
|
|
2304
|
-
} catch (e) {
|
|
2305
|
-
publishVerifyInspect.error = (e && e.message ? e.message : String(e)).slice(0, 200);
|
|
2306
|
-
log("Publish content read-back inspect trigger threw for task " + taskId + ": " + publishVerifyInspect.error + " — the park below asks the parent to trigger it manually");
|
|
2307
|
-
}
|
|
2308
|
-
} // end: !publishSkippedNoLock && publishBuildLanded — a skipped or failed publish has nothing to verify
|
|
2309
|
-
}
|
|
2301
|
+
// The parent (tick worker) triggers the ONE read-back inspection it can
|
|
2302
|
+
// actually receive (async results go to the root agent, never into a
|
|
2303
|
+
// workflow run — a workflow-side trigger would be an orphan). The workflow
|
|
2304
|
+
// only parks; the parent's scan builds the request deterministically via
|
|
2305
|
+
// lib/build-readback-request.js and ferries the inspection.
|
|
2306
|
+
// publishBuildLanded and publishSkippedNoLock are workflow-computed state;
|
|
2307
|
+
// a skipped or failed publish has nothing to verify.
|
|
2310
2308
|
|
|
2311
2309
|
// Session notes. Machine-readable marker lines are extracted from the full
|
|
2312
2310
|
// worker report and appended AFTER the slice so a long report can never
|
|
@@ -2329,11 +2327,7 @@ while (i < STEPS.length) {
|
|
|
2329
2327
|
// Visual verdict evidence: for experiential artifact tasks, append the
|
|
2330
2328
|
// deterministic post-change capture plan to the QA session notes. The
|
|
2331
2329
|
// parent protocol (docs/visual-verdict.md) triggers the inspection with
|
|
2332
|
-
|
|
2333
|
-
if (step.name === "QA" && passed && qaVisual) {
|
|
2334
|
-
summary = (summary + "\nvisual: pending\ncapture_plan: " +
|
|
2335
|
-
buildVisualCapturePlan(taskTitle, taskDescription, "postchange", captureTargets).replace(/\s+/g, " ")).slice(0, 2900);
|
|
2336
|
-
}
|
|
2330
|
+
|
|
2337
2331
|
|
|
2338
2332
|
// Capture mapper's spec for Build and Review
|
|
2339
2333
|
if (step.name === "Map" && passed) {
|
|
@@ -2368,41 +2362,6 @@ while (i < STEPS.length) {
|
|
|
2368
2362
|
}
|
|
2369
2363
|
);
|
|
2370
2364
|
|
|
2371
|
-
// Visual-verdict gate: for experiential artifact tasks the task is done
|
|
2372
|
-
// only when the parent has recorded a visual_verdict: note event. The QA
|
|
2373
|
-
// work agent above covered the mechanical checks only. No recorded
|
|
2374
|
-
// verdict → park for the parent protocol (never mark done on a pending
|
|
2375
|
-
// visual verdict). A recorded FAIL with budget left → rework at Build.
|
|
2376
|
-
if (step.name === "QA" && passed && qaVisual) {
|
|
2377
|
-
var vvStatus = await visualVerdictStatus();
|
|
2378
|
-
if (vvStatus.found && vvStatus.verdict === "PASS") {
|
|
2379
|
-
log("Visual verdict PASS recorded for task " + taskId + (vvStatus.detail ? " — " + vvStatus.detail : ""));
|
|
2380
|
-
} else if (vvStatus.found && vvStatus.verdict === "FAIL") {
|
|
2381
|
-
// A FAIL whose reason begins exactly "rendering impossible:" is not
|
|
2382
|
-
// reworkable — there is no rendered evidence to fix against. Park for
|
|
2383
|
-
// human attention instead of bouncing to Build.
|
|
2384
|
-
if (vvStatus.detail.indexOf("rendering impossible:") === 0) {
|
|
2385
|
-
log("Visual verdict FAIL (rendering impossible) for task " + taskId + " — parking for human attention");
|
|
2386
|
-
return await parkTask("Visual verdict FAIL — rendering impossible, human attention required: " + vvStatus.detail);
|
|
2387
|
-
}
|
|
2388
|
-
totalReworkCount++;
|
|
2389
|
-
if (totalReworkCount > MAX_TOTAL_REWORK) {
|
|
2390
|
-
log("Shared rework budget exhausted for task " + taskId + " — parking after visual verdict FAIL");
|
|
2391
|
-
return await parkTask("Exceeded shared rework budget (" + MAX_TOTAL_REWORK + " total rework attempts across Review and QA) after visual verdict FAIL: " + vvStatus.detail);
|
|
2392
|
-
}
|
|
2393
|
-
rejectionNotes = "Visual verdict FAIL: " + vvStatus.detail;
|
|
2394
|
-
i = BUILD_INDEX;
|
|
2395
|
-
log("Visual verdict FAIL — bouncing to Build (rework #" + totalReworkCount + " of " + MAX_TOTAL_REWORK + ")");
|
|
2396
|
-
continue;
|
|
2397
|
-
} else {
|
|
2398
|
-
if (!VISUAL_PROTOCOL_AVAILABLE) {
|
|
2399
|
-
log("Visual verdict protocol not available (VISUAL_PROTOCOL_AVAILABLE=false) for task " + taskId + " — skipping visual gate, QA mechanical checks already passed");
|
|
2400
|
-
} else {
|
|
2401
|
-
return await parkTask("Visual verdict pending — parent: run the post-change capture + visual verdict protocol in docs/visual-verdict.md (capture plan is in the QA session notes)");
|
|
2402
|
-
}
|
|
2403
|
-
}
|
|
2404
|
-
}
|
|
2405
|
-
|
|
2406
2365
|
// Handle rejection — bounce back to Build
|
|
2407
2366
|
if (!passed && (step.name === "Review" || step.name === "QA")) {
|
|
2408
2367
|
totalReworkCount++;
|
|
@@ -2442,10 +2401,7 @@ while (i < STEPS.length) {
|
|
|
2442
2401
|
if (passed && step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG && !publishSkippedNoLock && publishBuildLanded) {
|
|
2443
2402
|
return await parkTask("publish: verification-requested " + mergeCommitForPublish +
|
|
2444
2403
|
" (build " + (rebuildAgentId || "agent_id unobserved") + ")" +
|
|
2445
|
-
" — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md"
|
|
2446
|
-
(publishVerifyInspect.triggered
|
|
2447
|
-
? " (content read-back inspection " + publishVerifyInspect.inspection_id + " already triggered)."
|
|
2448
|
-
: " (read-back inspect trigger failed: " + (publishVerifyInspect.error || "not started") + " — parent: trigger artifact_inspect manually)."));
|
|
2404
|
+
" — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md.");
|
|
2449
2405
|
}
|
|
2450
2406
|
|
|
2451
2407
|
i++;
|