muse-crew 0.7.10 → 0.7.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +36 -14
- package/docs/guide.md +5 -5
- package/docs/ooda-report.md +150 -0
- package/docs/publish-verification.md +277 -88
- package/docs/visual-verdict.md +81 -67
- package/lib/AGENTS.md +8 -0
- package/lib/append-ooda-step.js +167 -0
- package/lib/build-readback-request.js +130 -0
- package/lib/compose-evidence-caption.js +141 -0
- package/lib/crew-api.js +467 -43
- package/lib/edit-image.py +216 -0
- package/lib/read-ooda-verdict.js +94 -0
- package/lib/readback-disk.js +186 -0
- package/lib/render-html.js +142 -0
- package/lib/see-act.js +327 -0
- package/lib/serve-artifact.js +203 -0
- package/lib/verify-publish.js +323 -0
- package/lib/write-ooda-verdict.js +147 -0
- package/package.json +1 -1
- package/seed/cron-body-template.md +30 -4
- package/seed/crons.json +1 -1
- package/workflows/bugfix.js +512 -220
- package/workflows/chore.js +389 -119
- package/workflows/crew-dispatch.js +72 -13
- package/workflows/docs.js +11 -2
- package/workflows/standard.js +413 -235
package/workflows/bugfix.js
CHANGED
|
@@ -28,6 +28,15 @@ const startStepIndex = inputs.start_step_index || 0;
|
|
|
28
28
|
// resolution back via updatetask in the self-claim below.
|
|
29
29
|
const RESOLVED_WORKFLOW = inputs.resolved_workflow || null;
|
|
30
30
|
const WORKFLOW_WAS_NULL = inputs.workflow_was_null === true;
|
|
31
|
+
// One-shot recovery routing: the dispatcher sets inputs.next_phase when it
|
|
32
|
+
// routes this run via an explicit recover-task redirect. The value is
|
|
33
|
+
// consumed (cleared) atomically by the successful self-claim below:
|
|
34
|
+
// claim-task takes expected_next_phase and clears the matching next_phase in
|
|
35
|
+
// the same transaction as the winning session insert, so no platform death
|
|
36
|
+
// can slip between claim and consumption and replay the routing. A stale or
|
|
37
|
+
// superseded routing survives — only an exact match clears.
|
|
38
|
+
// what the dispatcher routed on.
|
|
39
|
+
const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_phase.length > 0) ? inputs.next_phase : null;
|
|
31
40
|
const CLAIM_WORKFLOW_PERSIST = (WORKFLOW_WAS_NULL && RESOLVED_WORKFLOW) ? ", \"workflow\": \"" + RESOLVED_WORKFLOW + "\"" : "";
|
|
32
41
|
|
|
33
42
|
// Visual verdict protocol availability — the workflow parks for parent-run
|
|
@@ -472,52 +481,22 @@ function verifyAppliedChanges(expected, applied) {
|
|
|
472
481
|
return { ok: true };
|
|
473
482
|
}
|
|
474
483
|
|
|
475
|
-
// Publish read-back request
|
|
476
|
-
//
|
|
477
|
-
//
|
|
478
|
-
//
|
|
479
|
-
//
|
|
480
|
-
//
|
|
481
|
-
//
|
|
482
|
-
//
|
|
483
|
-
//
|
|
484
|
-
//
|
|
485
|
-
//
|
|
486
|
-
//
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
// prove the read-back inspected the live build of THIS attempt — not a
|
|
492
|
-
// different build's output. Null/empty means the edit was accepted but
|
|
493
|
-
// never correlated to a builder run. Pure function of inputs — no I/O,
|
|
494
|
-
// no clock.
|
|
495
|
-
var buildIdLine = (typeof buildAgentId === "string" && buildAgentId.length > 0)
|
|
496
|
-
? "Expected builder build agent_id: " + buildAgentId + " (the artifact system's in-flight correlation ID for this publish attempt — not a durable post-completion identifier).\n"
|
|
497
|
-
: "No build agent_id was observed for this publish attempt (the edit was accepted but never correlated to a builder run) — say so explicitly in your report.\n";
|
|
498
|
-
return (
|
|
499
|
-
"Publish content read-back for task " + taskId + ", merge commit " + commit + ".\n" +
|
|
500
|
-
"The unified diff below was supposed to be applied to this artifact's source tree and deployed. Do NOT modify anything.\n" +
|
|
501
|
-
"Do NOT rely on the builder's applied-changes report — it is derived from this same diff, so it cannot confirm the content. Read the artifact's CURRENT source directly.\n" +
|
|
502
|
-
"\n" +
|
|
503
|
-
buildIdLine +
|
|
504
|
-
"Report the live build's agent_id as seen in artifact_status (or state explicitly that no build/agent_id is visible). If an expected agent_id is given above and the live one differs, say so exactly — the read-back may be inspecting a different build's output.\n" +
|
|
505
|
-
"\n" +
|
|
506
|
-
"UNIFIED DIFF (expected change):\n" +
|
|
507
|
-
"```diff\n" + diff + "\n```\n" +
|
|
508
|
-
"\n" +
|
|
509
|
-
"For each file in the diff:\n" +
|
|
510
|
-
"1. Read the file's CURRENT content in the artifact source tree.\n" +
|
|
511
|
-
"2. Quote the exact current text of the regions around the changed lines.\n" +
|
|
512
|
-
"3. For every added (+) line in the diff, state whether that exact line is PRESENT in the current source.\n" +
|
|
513
|
-
"4. For every removed (-) line in the diff, state whether that exact line is ABSENT from the current source.\n" +
|
|
514
|
-
"5. Report build/deploy health and the console error count.\n" +
|
|
515
|
-
"\n" +
|
|
516
|
-
"Return the per-file present/absent findings with the quoted observed lines. Do not modify anything.\n" +
|
|
517
|
-
"This read-back feeds the parent content-verification protocol (docs/publish-verification.md): the parent stamps provenance only when every added line is present and every removed line is absent."
|
|
518
|
-
);
|
|
519
|
-
}
|
|
520
|
-
|
|
484
|
+
// Publish read-back request (currently unavailable): the verbatim_request
|
|
485
|
+
// the parent protocol (docs/publish-verification.md) would hand to an
|
|
486
|
+
// independent read-back tool after the artifact build lands. artifact_inspect
|
|
487
|
+
// was removed by the platform (2026-09-14); artifact.inspect is malfunction
|
|
488
|
+
// diagnosis, not a substitute — so no agent-callable read-back tool exists
|
|
489
|
+
// and this request cannot currently be issued. Pure function — no I/O, no
|
|
490
|
+
// clock. The request carries the merged diff as the expected change and asks
|
|
491
|
+
// for an independent read of the artifact's actual source: for each file, the
|
|
492
|
+
// exact current text of the changed regions plus a per-line present/absent
|
|
493
|
+
// finding. Until a read-back path exists, the parent cannot independently
|
|
494
|
+
// confirm content and verification parks at "publish: verification-requested"
|
|
495
|
+
// (see docs/publish-verification.md). This preserves the circularity break
|
|
496
|
+
// that hollowed canary run 8 (2026-09-11): verifyAppliedChanges compares the
|
|
497
|
+
// builder's applied-report against the diff the report was derived from — a
|
|
498
|
+
// fabricated report passes by construction. Independent read-back cannot be
|
|
499
|
+
// fabricated from the diff; it must match the artifact's real content.
|
|
521
500
|
// Pre-publish base observation (diagnostic, 2026-09-12): instruction fragment
|
|
522
501
|
// for the builder's edit request, asking it to report the sha256 of each
|
|
523
502
|
// touched file's CURRENT content BEFORE applying the diff. Pure function —
|
|
@@ -611,6 +590,18 @@ function extractMarkerLines(workerText) {
|
|
|
611
590
|
return markers.join("\n");
|
|
612
591
|
}
|
|
613
592
|
|
|
593
|
+
// Already-merged idempotency (canary 2026-09-15, task 1d692d91): when the
|
|
594
|
+
// builder correctly makes no commit because the deliverable is already on
|
|
595
|
+
// main (a prior merge or hand-repair landed it), it declares
|
|
596
|
+
// `repo_diff: none (already-merged: <sha>)` naming the main commit that
|
|
597
|
+
// carries the work. The sha is hex-only (7-40 chars) so the workflow can
|
|
598
|
+
// interpolate it into the mechanical ancestor check without injection
|
|
599
|
+
// risk. Pure — pinned byte-identical across standard/bugfix/chore.
|
|
600
|
+
function extractAlreadyMerged(workerText) {
|
|
601
|
+
var m = /^repo_diff:\s*none\s*\(already-merged:\s*([0-9a-f]{7,40})\)/im.exec(workerText || "");
|
|
602
|
+
return m ? { sha: m[1].toLowerCase() } : { sha: null };
|
|
603
|
+
}
|
|
604
|
+
|
|
614
605
|
// Worktree confinement: the Build agent must declare the exact worktree
|
|
615
606
|
// path it built in on a `worktree:` marker line. The workflow compares it
|
|
616
607
|
// against WORKTREE_HINT mechanically (exact string match) — never by
|
|
@@ -751,43 +742,14 @@ async function baselineStatus() {
|
|
|
751
742
|
return { baseline_found: false, baseline_kind: "", baseline_refs: "", requested_count: 0, evidence_count: 0 };
|
|
752
743
|
}
|
|
753
744
|
}
|
|
754
|
-
//
|
|
755
|
-
//
|
|
756
|
-
//
|
|
757
|
-
//
|
|
758
|
-
//
|
|
759
|
-
//
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
visualVerdictCallCount++;
|
|
763
|
-
try {
|
|
764
|
-
var vv = await agent(
|
|
765
|
-
"Read this task's QA session and note events from the crew API.\n" +
|
|
766
|
-
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
|
|
767
|
-
"Find the LATEST session with task_id \"" + taskId + "\" and step \"QA\" in the returned sessions array and note its started_at timestamp (call it QA_START; use empty string if there is no QA session).\n" +
|
|
768
|
-
"Then run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId }) + "\n" +
|
|
769
|
-
"Consider only events with type \"note\" whose timestamp is newer than QA_START. Among them, find messages starting exactly with \"visual_verdict: PASS\" or \"visual_verdict: FAIL\" (exact prefix, case-sensitive); use the latest such message.\n" +
|
|
770
|
-
"Return JSON { \"found\": <true if such a message exists>, \"verdict\": \"<\"PASS\" or \"FAIL\" from that message, or empty string>\", \"detail\": \"<the text after the prefix in that message, or empty string>\" } and nothing else.",
|
|
771
|
-
{
|
|
772
|
-
key: "visual-verdict-status-" + taskId + "-" + visualVerdictCallCount,
|
|
773
|
-
label: "Reading visual verdict status",
|
|
774
|
-
schema: {
|
|
775
|
-
type: "object",
|
|
776
|
-
properties: {
|
|
777
|
-
found: { type: "boolean" },
|
|
778
|
-
verdict: { type: "string" },
|
|
779
|
-
detail: { type: "string" }
|
|
780
|
-
},
|
|
781
|
-
required: ["found", "verdict", "detail"]
|
|
782
|
-
}
|
|
783
|
-
}
|
|
784
|
-
);
|
|
785
|
-
return { found: !!(vv && vv.found), verdict: (vv && vv.verdict) || "", detail: (vv && vv.detail) || "" };
|
|
786
|
-
} catch (e) {
|
|
787
|
-
log("visualVerdictStatus: agent call failed (" + (e && e.message ? e.message : e) + ") — treating as not found");
|
|
788
|
-
return { found: false, verdict: "", detail: "" };
|
|
789
|
-
}
|
|
790
|
-
}
|
|
745
|
+
// The parent visual-verdict reader was removed 2026-09-15: Hazel (the QA
|
|
746
|
+
// work agent) now owns the visual verdict experientially — she drives the
|
|
747
|
+
// now owns the visual verdict experientially — she drives the see-act loop
|
|
748
|
+
// herself and records verdict.json + the append-only verdicts.jsonl. The old
|
|
749
|
+
// see-act loop herself and records verdict.json + the append-only
|
|
750
|
+
// verdicts.jsonl. The old parent note gate is obsolete and has been
|
|
751
|
+
// deleted from the QA closeout below. (See docs/visual-verdict.md.)
|
|
752
|
+
// from the QA closeout below. (See docs/visual-verdict.md.)
|
|
791
753
|
|
|
792
754
|
// STEPS inline — export const meta is parsed as metadata, not a runtime binding
|
|
793
755
|
const STEPS = [
|
|
@@ -828,6 +790,12 @@ let mapGateBounceCount = 0;
|
|
|
828
790
|
// rationalized a skip against explicit instruction text — text alone did not
|
|
829
791
|
// hold, so the decision now lives in workflow code, not agent judgment.
|
|
830
792
|
let releaseDecision = null; // { release: "yes"|"no", version_bump: "patch"|"minor"|"major"|null }
|
|
793
|
+
// Already-merged idempotency: the verified sha from the builder's
|
|
794
|
+
// `repo_diff: none (already-merged: <sha>)` declaration (null when the
|
|
795
|
+
// builder made commits or declared a runtime-state deliverable). The
|
|
796
|
+
// workflow verifies the sha is an ancestor of main at Build closeout;
|
|
797
|
+
// Review's no-diff branch reads this, never the builder's prose.
|
|
798
|
+
let alreadyMergedSha = null;
|
|
831
799
|
// Deterministic publish target — computed by the workflow (registry base +
|
|
832
800
|
// bumpVersion), never by the Publish agent.
|
|
833
801
|
let publishTarget = null; // { base, scope, target }
|
|
@@ -1059,7 +1027,7 @@ while (i < STEPS.length) {
|
|
|
1059
1027
|
const claimResult = await agent(
|
|
1060
1028
|
"Claim this task for the " + step.name + " step.\n" +
|
|
1061
1029
|
"Run in shell and return the stdout verbatim:\n" + crewCmd("update-task", firstClaimUpdateArgs) + "\n" +
|
|
1062
|
-
"Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started" }) + "\n" +
|
|
1030
|
+
"Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started", ...(NEXT_PHASE_ROUTED ? { expected_next_phase: NEXT_PHASE_ROUTED } : {}) }) + "\n" +
|
|
1063
1031
|
"If the claim response has claimed=true, then run in shell and return the stdout verbatim:\n" + crewCmd("clear-reservation", { task_id: taskId }) + "\n" +
|
|
1064
1032
|
"Do not interpret the claim response. It already contains an explicit \"claimed\" field — copy it verbatim.\n" +
|
|
1065
1033
|
"Return { claimed: <verbatim>, session_id: \"<...>\" }. If claimed is false there is no session_id; return { claimed: false, session_id: \"\" }.",
|
|
@@ -1245,12 +1213,10 @@ while (i < STEPS.length) {
|
|
|
1245
1213
|
}
|
|
1246
1214
|
}
|
|
1247
1215
|
|
|
1248
|
-
// Visual verdict routing: experiential artifact tasks
|
|
1249
|
-
//
|
|
1250
|
-
//
|
|
1251
|
-
//
|
|
1252
|
-
// parks for the parent protocol when no visual_verdict: note event exists
|
|
1253
|
-
// yet.
|
|
1216
|
+
// Visual verdict routing (2026-09-15): experiential artifact tasks route
|
|
1217
|
+
// to Hazel's own experiential QA instructions below — she drives the
|
|
1218
|
+
// see-act loop herself and owns the visual verdict (verdict.json +
|
|
1219
|
+
// verdicts.jsonl). No parent verdict gate remains.
|
|
1254
1220
|
var qaVisual = false;
|
|
1255
1221
|
if (step.name === "QA") {
|
|
1256
1222
|
qaVisual = (await resolveExperiential()) === "yes" && PUBLISH_TYPE === "artifact";
|
|
@@ -1260,18 +1226,25 @@ while (i < STEPS.length) {
|
|
|
1260
1226
|
var instructions = "";
|
|
1261
1227
|
|
|
1262
1228
|
if (step.name === "Triage") {
|
|
1263
|
-
instructions = "Validate the task,
|
|
1229
|
+
instructions = "Validate the task against the project's repo at " + REPO_PATH + " — that exact checkout, not any other copy of the project on disk. If you run git commands, cd " + REPO_PATH + " first.\nCheck clarity, note dependencies, confirm the bugfix workflow assignment.\nIf the task needs decomposition, note that in your assessment.\nReport back in plain prose — what you found.\nEXPERIENTIAL FLAG: does this task change anything rendered and visible in the project's user-facing artifact (pages, components, styles, layout, copy, visual states)? If yes it is experiential and gets baseline captures (plus a visual verdict where the workflow has a QA phase). End your report with exactly one line on its own, lowercase, unrephrased: experiential: yes — or experiential: no. This line is machine-read.";
|
|
1264
1230
|
|
|
1265
1231
|
} else if (step.name === "Reproduce") {
|
|
1266
|
-
instructions = "Reproduce the bug from a user's perspective. You are CODE-BLIND — do NOT read source code.\n" +
|
|
1267
|
-
"
|
|
1268
|
-
"
|
|
1269
|
-
"
|
|
1270
|
-
"
|
|
1271
|
-
"
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1232
|
+
instructions = "Reproduce the bug from a user's perspective — by USING the artifact, not by reading data. You are CODE-BLIND — do NOT read source code.\n" +
|
|
1233
|
+
"You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
|
|
1234
|
+
"Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and fall back to the data-level investigation at the end.\n" +
|
|
1235
|
+
"a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and fall back to the data-level investigation.\n" +
|
|
1236
|
+
"b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-repro > /tmp/repro-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/repro-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/repro-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/repro-server-" + taskId + ".log) and fall back to the data-level investigation.\n" +
|
|
1237
|
+
"c. Bounded see-act loop, at most 8 steps: drive the artifact to TRIGGER the reported bug. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/repro/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
|
|
1238
|
+
"c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/repro/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you. If the server stops responding mid-loop (curl --max-time 3 http://localhost:<N>/ fails), restart it per (b).\n" +
|
|
1239
|
+
"c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
|
|
1240
|
+
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/repro/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/repro/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
1241
|
+
"d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the bug needs such a sequence, report NOT POSSIBLE for that part and reproduce what you can.\n" +
|
|
1242
|
+
"e. Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/repro/ and indexed in ooda-log.jsonl — that log plus the frames is your OODA report for this reproduction. A frame you did not read is not evidence. Loading, error, or blank frames never reproduce anything.\n" +
|
|
1243
|
+
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-repro' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
1244
|
+
"Data-level fallback (only if the browser loop above is NOT POSSIBLE): run in shell and read the stdout JSON:\n" + crewCmd("get-state", {}) + "\n" +
|
|
1245
|
+
"This returns current sessions, events, and tasks — capture concrete evidence from the data you retrieve.\n" +
|
|
1246
|
+
"Report your reproduction steps and evidence as plain prose.\n" +
|
|
1247
|
+
"End your report with exactly one line: VERDICT: PASS if you reproduced the reported bug (your frames show the reported misbehavior), VERDICT: FAIL if you could not. --expected names the bug as reported (its visible manifestation); --actual names what your frames actually showed. Checks you could not run are evidence gaps, not silent drops: name every one in --missing. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/repro/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/repro/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the bug as reported — its visible manifestation>\" --actual \"<what your frames actually showed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL or NOT_POSSIBLE verdict without a machine-readable --reason cannot be written — state the reason.";
|
|
1275
1248
|
|
|
1276
1249
|
} else if (step.name === "Map") {
|
|
1277
1250
|
var mapGatePara = "";
|
|
@@ -1307,12 +1280,33 @@ while (i < STEPS.length) {
|
|
|
1307
1280
|
"cd " + WORKTREE_HINT + "\n" +
|
|
1308
1281
|
"git add -A\n" +
|
|
1309
1282
|
"git commit -m \"fix: " + safeTitle + "\"\n\n" +
|
|
1310
|
-
"If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of main and declare `repo_diff: none` in your report, naming the runtime-state deliverable. Otherwise commit your changes normally.\n\n" +
|
|
1283
|
+
"If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of main and declare `repo_diff: none` in your report, naming the runtime-state deliverable. If you verified the deliverable is already on main (a prior merge or hand-repair landed it — do NOT re-implement working code), make no commit and declare `repo_diff: none (already-merged: <sha>)` naming the main commit that carries the work; the workflow verifies the sha is an ancestor of main, and a false declaration fails the phase. Otherwise commit your changes normally.\n\n" +
|
|
1311
1284
|
(rejectionNotes ? "This is REWORK after rejection. Address these specific issues:\n" + rejectionNotes + "\n\n" : "") +
|
|
1312
1285
|
"Report back in plain prose: what you built and the outcome." +
|
|
1313
1286
|
(PUBLISH_TYPE === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
|
|
1314
1287
|
|
|
1315
1288
|
} else if (step.name === "Review") {
|
|
1289
|
+
// Already-merged hydration: when this run did not execute Build itself
|
|
1290
|
+
// (dispatcher resume at Review after a platform death between phases),
|
|
1291
|
+
// recover the workflow-attested verification from the latest completed
|
|
1292
|
+
// Build session notes. The `already_merged_verified:` line was written
|
|
1293
|
+
// by the workflow after a mechanical ancestor check — it is trusted;
|
|
1294
|
+
// the builder's bare declaration never is. Absent the line, the
|
|
1295
|
+
// mechanical fact below reads "none declared" and Cass fails closed.
|
|
1296
|
+
if (!alreadyMergedSha) {
|
|
1297
|
+
var hydNotes = await agent(
|
|
1298
|
+
"Read the latest completed Build session notes for task " + taskId + ".\n" +
|
|
1299
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
|
|
1300
|
+
"In the returned sessions array, find the most recent session (by started_at) with task_id \"" + taskId + "\", step \"Build\", and status \"completed\". Return ONLY its notes field, verbatim, with no commentary.",
|
|
1301
|
+
{ key: "hydrate-already-merged" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Hydrating already-merged verification" }
|
|
1302
|
+
);
|
|
1303
|
+
var hydStr = (typeof hydNotes === "string") ? hydNotes : JSON.stringify(hydNotes);
|
|
1304
|
+
var hvm = /already_merged_verified:\s*([0-9a-f]{7,40})/i.exec(hydStr);
|
|
1305
|
+
if (hvm) {
|
|
1306
|
+
alreadyMergedSha = hvm[1].toLowerCase();
|
|
1307
|
+
log("Hydrated already-merged verification from Build session notes: " + alreadyMergedSha);
|
|
1308
|
+
}
|
|
1309
|
+
}
|
|
1316
1310
|
instructions = "Review independently and cold. You have NOT seen any reasoning from the builder.\nDo NOT access the task dashboard, event log, or any comments. Your review is based solely on the spec and the code.\n\n" +
|
|
1317
1311
|
(mapperSpec ? "MAPPER'S SPEC (the builder was asked to implement exactly this):\n" + mapperSpec + "\n\n" : "Read the spec (from the task description or spec files under " + crewHome + "/).\n\n") +
|
|
1318
1312
|
"Examine the code changes by running:\n" +
|
|
@@ -1322,7 +1316,7 @@ while (i < STEPS.length) {
|
|
|
1322
1316
|
WORKTREE_HINT + "/\n\n" +
|
|
1323
1317
|
"Check quality, correctness, and spec compliance.\n" +
|
|
1324
1318
|
"Check that public-affecting changes have matching public doc updates (API.md or the published API contract). If the docs are missing or inaccurate, report what is stale, then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1325
|
-
"If the branch has no commits ahead of main (inspect shows an empty commit log), approve ONLY if the Build summary declares `repo_diff: none` with a plausible runtime-state deliverable (e.g. a cron created via the cron tool). Otherwise report 'no commits ahead of main and no repo_diff: none declaration — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1319
|
+
"If the branch has no commits ahead of main (inspect shows an empty commit log), approve ONLY if the Build summary declares `repo_diff: none` with (a) a plausible runtime-state deliverable (e.g. a cron created via the cron tool), or (b) an already-merged declaration `repo_diff: none (already-merged: <sha>)` AND the mechanical fact below confirms the sha verified. MECHANICAL FACT (computed by the workflow, never by the builder): already_merged sha = " + (alreadyMergedSha ? alreadyMergedSha + " (verified ancestor of main: YES)" : "none declared") + ". Otherwise report 'no commits ahead of main and no valid repo_diff: none declaration — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1326
1320
|
(PUBLISH_TYPE === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry, and versions are assigned at publish time — never in branches. Two checks:\n" +
|
|
1327
1321
|
"(a) The task branch must NOT have changed package.json's `version` field. Check: cd " + REPO_PATH + " && git diff main..." + TASK_BRANCH + " -- package.json. If the branch touched `version` in any way, report 'versions are assigned at publish time, never in branches — remove the version change' in your notes, then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1328
1322
|
"(b) The accepted Build report declares: " + releaseDecisionText() + ". " +
|
|
@@ -1336,7 +1330,7 @@ while (i < STEPS.length) {
|
|
|
1336
1330
|
"Run: "+ LIFECYCLE_ENV + "WORKFLOW_RUN_ID=" + lockHolder + " integrate " + taskId + " \"merge: fix: " + safeTitle + "\"\n\n" +
|
|
1337
1331
|
"Read the output:\n" +
|
|
1338
1332
|
"- If it contains MERGED, integration succeeded. Report the merged commit hash.\n" +
|
|
1339
|
-
"- If it contains MERGED_EMPTY, the branch had no commits ahead of main (a runtime-state deliverable
|
|
1333
|
+
"- If it contains MERGED_EMPTY, the branch had no commits ahead of main (declared by Build as repo_diff: none — either a runtime-state deliverable or an already-merged sha the workflow verified). Integration succeeded vacuously: the merge lock was NOT taken and there is no new commit. Report 'merged empty: no repo changes — deliverable was runtime state or already on main', then end your report with exactly this line: VERDICT: PASS. SKIP STEP 2 (push): there is no new commit to push.\n" +
|
|
1340
1334
|
"- If it contains LOCK_HELD, another task holds the merge lock (mid Integrate/Publish) and the 10-minute bounded backoff is exhausted. Report 'merge lock held after bounded backoff', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1341
1335
|
"- If it contains CONFLICT, the plain merge failed — the merge was aborted, main is clean, and your task still holds the merge lock. Do NOT fail yet. Resolve it:\n" +
|
|
1342
1336
|
"RESOLUTION:\n" +
|
|
@@ -1399,10 +1393,11 @@ while (i < STEPS.length) {
|
|
|
1399
1393
|
// (fail-closed). There is deliberately NO workflow-side provenance
|
|
1400
1394
|
// stamp: the builder's applied-report is circular (canary run 8,
|
|
1401
1395
|
// 2026-09-11), so the stamp moved to the parent — after the build
|
|
1402
|
-
// lands, the workflow
|
|
1403
|
-
//
|
|
1404
|
-
//
|
|
1405
|
-
//
|
|
1396
|
+
// lands, the workflow records the session completed and parks with
|
|
1397
|
+
// "publish: verification-requested". The parent owns verification
|
|
1398
|
+
// (docs/publish-verification.md); the independent read-back step is
|
|
1399
|
+
// currently unavailable (no agent-callable read-back tool exists —
|
|
1400
|
+
// artifact_inspect was removed by the platform 2026-09-14).
|
|
1406
1401
|
// QA's provenance check enforces the stamp mechanically.
|
|
1407
1402
|
var artifactPublish = null;
|
|
1408
1403
|
var publishLockRefreshed = false;
|
|
@@ -1451,14 +1446,45 @@ while (i < STEPS.length) {
|
|
|
1451
1446
|
// changes — no prose claim to trust. If the artifact tool namespace
|
|
1452
1447
|
// is missing from this child it reports honestly and the workflow
|
|
1453
1448
|
// retries once with a fresh key (bounded); anything else parks.
|
|
1449
|
+
// Publish diff base (2026-09-14, task 0c53af4e): the carried diff is
|
|
1450
|
+
// BASE..HEAD where BASE is the previously-stamped provenance
|
|
1451
|
+
// source_commit — NOT HEAD^1. A push-time reconcile merge puts the
|
|
1452
|
+
// task's own changes behind an intermediate merge, so HEAD^1..HEAD
|
|
1453
|
+
// silently drops the task's fix while the artifact builds without
|
|
1454
|
+
// it. The stamped base is the artifact's actual content; BASE..HEAD
|
|
1455
|
+
// is the complete unpublished delta. Empty tree only for a genuine
|
|
1456
|
+
// first publish (no provenance stamped yet).
|
|
1457
|
+
var EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
|
|
1458
|
+
var provResult = await agent(
|
|
1459
|
+
crewCmd("get-provenance", {}) + "\n" +
|
|
1460
|
+
"Return JSON { \"provenance\": <the CLI's provenance object, or null when nothing is stamped> } and nothing else. Do not interpret it.",
|
|
1461
|
+
{ key: attemptKey("publish-provenance-base-" + taskId, totalReworkCount), label: "Reading stamped publish base",
|
|
1462
|
+
schema: { type: "object", properties: { provenance: { type: ["object", "null"] } }, required: ["provenance"] } }
|
|
1463
|
+
);
|
|
1464
|
+
var publishBase = (provResult.provenance && provResult.provenance.source_commit) || "";
|
|
1465
|
+
publishBase = String(publishBase).trim();
|
|
1466
|
+
if (!publishBase) {
|
|
1467
|
+
publishBase = EMPTY_TREE_SHA;
|
|
1468
|
+
log("Publish base for task " + taskId + ": no provenance stamped yet — using empty tree (first publish)");
|
|
1469
|
+
} else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
|
|
1470
|
+
return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
|
|
1471
|
+
}
|
|
1454
1472
|
var diffResult = await agent(
|
|
1455
|
-
"Run: cd " + REPO_PATH + " &&
|
|
1456
|
-
"
|
|
1457
|
-
|
|
1458
|
-
|
|
1473
|
+
"Run: cd " + REPO_PATH + " && BASE='" + publishBase + "' && HEAD=$(git rev-parse HEAD) && " +
|
|
1474
|
+
"if [ \"$BASE\" = '" + EMPTY_TREE_SHA + "' ]; then ANCESTOR=yes; else git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no; fi && " +
|
|
1475
|
+
"echo '---COMMIT---' && echo \"$HEAD\" && echo '---BASE---' && echo \"$BASE\" && echo '---ANCESTOR---' && echo \"$ANCESTOR\" && " +
|
|
1476
|
+
"if [ \"$ANCESTOR\" = yes ]; then echo '---DIFF---' && git diff \"$BASE\" \"$HEAD\" && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r \"$BASE\" \"$HEAD\"; fi\n" +
|
|
1477
|
+
"Return JSON { \"commit\": \"<HEAD trimmed>\", \"base\": \"<BASE trimmed>\", \"ancestor\": \"<yes|no>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
|
|
1478
|
+
{ key: attemptKey("publish-artifact-diff-" + taskId, totalReworkCount), label: "Computing publish diff from stamped base",
|
|
1479
|
+
schema: { type: "object", properties: { commit: { type: "string" }, base: { type: "string" }, ancestor: { type: "string" }, diff: { type: "string" }, files: { type: "string" } }, required: ["commit", "base", "ancestor", "diff"] } }
|
|
1459
1480
|
);
|
|
1481
|
+
if ((diffResult.ancestor || "").trim() !== "yes") {
|
|
1482
|
+
return await parkTask("Publish base " + publishBase.slice(0, 12) + " is not an ancestor of HEAD " + (diffResult.commit || "").trim().slice(0, 12) + " — the stamped provenance does not lead to the integrated commit. Human attention needed.");
|
|
1483
|
+
}
|
|
1484
|
+
if ((diffResult.base || "").trim() !== publishBase) {
|
|
1485
|
+
return await parkTask("Publish diff base mismatch: agent reported '" + (diffResult.base || "").trim().slice(0, 12) + "' but the stamped base is '" + publishBase.slice(0, 12) + "'. Human attention needed.");
|
|
1486
|
+
}
|
|
1460
1487
|
var mergeCommitForPublish = (diffResult.commit || "").trim();
|
|
1461
|
-
var mergeParentForPublish = (diffResult.parent || "").trim();
|
|
1462
1488
|
var mergeDiff = diffResult.diff || "";
|
|
1463
1489
|
if (!mergeDiff.trim()) {
|
|
1464
1490
|
return await parkTask("Publish diff is empty for commit " + (mergeCommitForPublish || "unknown") + " — a merge lock was held but there is no change to carry. Human attention needed.");
|
|
@@ -1490,12 +1516,16 @@ while (i < STEPS.length) {
|
|
|
1490
1516
|
// but never parks. The observation tells us what the publish actually
|
|
1491
1517
|
// reads, so the subsequent fix can require the right base.
|
|
1492
1518
|
var expectedBaseHashes = {};
|
|
1493
|
-
|
|
1519
|
+
if (publishBase === EMPTY_TREE_SHA) {
|
|
1520
|
+
// First publish: every file in the diff is new to the artifact.
|
|
1521
|
+
expectedChanges.forEach(function (f) { expectedBaseHashes[f.path] = "NEW-FILE"; });
|
|
1522
|
+
log("Publish expected base hashes for task " + taskId + ": empty tree (first publish) — all " + expectedChanges.length + " file(s) new");
|
|
1523
|
+
} else try {
|
|
1494
1524
|
// Shell-quote helper (no regex-with-quote: the test parser does not
|
|
1495
1525
|
// understand regex literals containing quotes).
|
|
1496
1526
|
var sq = function(s) { return "'" + String(s).split("'").join("'\\''") + "'"; };
|
|
1497
1527
|
var baseHashResult = await agent(
|
|
1498
|
-
"Run: cd " + REPO_PATH + " && parent=" + sq(
|
|
1528
|
+
"Run: cd " + REPO_PATH + " && parent=" + sq(publishBase) + " && for f in " + expectedChanges.map(function(f) { return sq(f.path); }).join(" ") + "; do printf '%s:' \"$f\"; git show \"$parent:$f\" 2>/dev/null | sha256sum | cut -d' ' -f1; done\n" +
|
|
1499
1529
|
"Return JSON { \"hashes\": \"<newline-separated <path>:<sha256> lines, empty hash means the file is new in this diff>\" } and nothing else.",
|
|
1500
1530
|
{ key: attemptKey("publish-base-hashes-" + taskId, totalReworkCount), label: "Computing expected base content hashes",
|
|
1501
1531
|
schema: { type: "object", properties: { hashes: { type: "string" } }, required: ["hashes"] } }
|
|
@@ -1504,7 +1534,7 @@ while (i < STEPS.length) {
|
|
|
1504
1534
|
var m = /^([^:]+):([0-9a-f]*)$/.exec(line.trim());
|
|
1505
1535
|
if (m) expectedBaseHashes[m[1]] = m[2] || "NEW-FILE";
|
|
1506
1536
|
});
|
|
1507
|
-
log("Publish expected base hashes for task " + taskId + " (
|
|
1537
|
+
log("Publish expected base hashes for task " + taskId + " (stamped base " + publishBase.slice(0, 12) + "): " + JSON.stringify(expectedBaseHashes));
|
|
1508
1538
|
} catch (e) {
|
|
1509
1539
|
log("Publish expected base hash computation failed for task " + taskId + " (non-fatal, observation degraded): " + (e && e.message ? e.message : e));
|
|
1510
1540
|
}
|
|
@@ -1562,6 +1592,7 @@ while (i < STEPS.length) {
|
|
|
1562
1592
|
required: ["edit_started", "build_agent_id", "applied"] };
|
|
1563
1593
|
var rebuildTrigger = null;
|
|
1564
1594
|
var rebuildReportMissing = false; // true if the edit went through but the agent returned no applied report (structured-output failure) — the smoke-check is skipped; the parent's independent read-back is the verification
|
|
1595
|
+
var rebuildEvidenceNote = null; // human-readable evidence line for the ledger when the edit is confirmed via fallback evidence (in-flight poll or durable audit dir) rather than the trigger's own report
|
|
1565
1596
|
// The trigger key of the attempt that last ran, for the publish ledger.
|
|
1566
1597
|
// Minted once here (not re-minted per use site) so the ledger always
|
|
1567
1598
|
// records the exact key that was issued — and so a re-minted duplicate
|
|
@@ -1580,6 +1611,32 @@ while (i < STEPS.length) {
|
|
|
1580
1611
|
// (2026-09-12, task 23ca8f3f): computed once the trigger outcome is
|
|
1581
1612
|
// known, logged loudly, never a park.
|
|
1582
1613
|
var publishAppliedObservation = null; // "match" | "mismatch: <reason>" | "missing-report" — observation only, never a park
|
|
1614
|
+
// Durable-evidence snapshot (2026-09-14): the structured-output
|
|
1615
|
+
// fallback below only observes IN-FLIGHT builds. A build that
|
|
1616
|
+
// finished before the poll leaves no in-flight trace — but the
|
|
1617
|
+
// platform's audit harness leaves a durable one:
|
|
1618
|
+
// ~/workspace/ts-spaces/<slug>/audits/<timestamp>-<id>/ per
|
|
1619
|
+
// completed build. Snapshot the listing BEFORE the trigger so the
|
|
1620
|
+
// fallback can diff before/after: a directory appearing during the
|
|
1621
|
+
// trigger window is positive evidence the edit went through and
|
|
1622
|
+
// the build completed. Best-effort and non-gating: if the snapshot
|
|
1623
|
+
// fails, the durable check is skipped and the fallback behaves as
|
|
1624
|
+
// before. No wall-clock in-script (deterministic replay) — the
|
|
1625
|
+
// comparison is a pure before/after set diff.
|
|
1626
|
+
var auditDirsBeforeTrigger = [];
|
|
1627
|
+
try {
|
|
1628
|
+
var auditBefore = await agent(
|
|
1629
|
+
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
|
|
1630
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1631
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
|
|
1632
|
+
{ key: attemptKey("publish-audit-before-" + taskId, totalReworkCount), label: "Snapshotting audit dirs before rebuild trigger",
|
|
1633
|
+
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1634
|
+
);
|
|
1635
|
+
auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1636
|
+
log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
|
|
1637
|
+
} catch (auditBeforeErr) {
|
|
1638
|
+
log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
|
|
1639
|
+
}
|
|
1583
1640
|
try {
|
|
1584
1641
|
rebuildTrigger = await agent(rebuildPrompt,
|
|
1585
1642
|
{ key: rebuildAttemptKey, label: "Triggering artifact rebuild", schema: rebuildSchema });
|
|
@@ -1638,7 +1695,46 @@ while (i < STEPS.length) {
|
|
|
1638
1695
|
rebuildTrigger = { edit_started: true, error: "", applied: null };
|
|
1639
1696
|
rebuildReportMissing = true;
|
|
1640
1697
|
rebuildAgentId = acceptedAgentId;
|
|
1698
|
+
rebuildEvidenceNote = "edit confirmed via build-state poll after structured-output failure (build " + acceptedAgentId + "); builder applied-report missing";
|
|
1641
1699
|
} else {
|
|
1700
|
+
// Durable completion check (2026-09-14): the in-flight poll
|
|
1701
|
+
// above only sees RUNNING builds. Attempt 7 (2026-09-14) proved
|
|
1702
|
+
// the gap: the trigger child applied the edit, the build ran
|
|
1703
|
+
// and completed — the platform's audit harness captured it
|
|
1704
|
+
// mid-window — then the child failed to return JSON. The
|
|
1705
|
+
// fallback poll saw no in-flight build, so a successful publish
|
|
1706
|
+
// parked as "unknown". Diff the audit-dir listing against the
|
|
1707
|
+
// pre-trigger snapshot: a timestamped directory that appeared
|
|
1708
|
+
// during the trigger window is positive evidence the edit went
|
|
1709
|
+
// through and the build completed. This never re-issues the
|
|
1710
|
+
// edit and never stamps provenance — it only routes to the
|
|
1711
|
+
// parent's independent content read-back, which remains the
|
|
1712
|
+
// real verification.
|
|
1713
|
+
var newAuditDirs = [];
|
|
1714
|
+
try {
|
|
1715
|
+
var auditAfter = await agent(
|
|
1716
|
+
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
|
|
1717
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1718
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
|
|
1719
|
+
{ key: attemptKey("publish-audit-after-" + taskId, totalReworkCount), label: "Re-listing audit dirs after trigger failure",
|
|
1720
|
+
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1721
|
+
);
|
|
1722
|
+
var auditDirsAfterTrigger = String((auditAfter && auditAfter.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1723
|
+
// Only timestamped build dirs count — the "latest" symlink
|
|
1724
|
+
// and anything else are not builds.
|
|
1725
|
+
newAuditDirs = auditDirsAfterTrigger.filter(function (d) {
|
|
1726
|
+
return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
|
|
1727
|
+
});
|
|
1728
|
+
} catch (auditAfterErr) {
|
|
1729
|
+
log("Publish audit-dir re-list after trigger failure failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
|
|
1730
|
+
}
|
|
1731
|
+
if (newAuditDirs.length > 0) {
|
|
1732
|
+
log("Publish rebuild trigger: new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and the build completed despite the structured-output failure. Skipping applied-report smoke-check; parent read-back is the verification.");
|
|
1733
|
+
rebuildTrigger = { edit_started: true, error: "", applied: null };
|
|
1734
|
+
rebuildReportMissing = true;
|
|
1735
|
+
rebuildAgentId = null;
|
|
1736
|
+
rebuildEvidenceNote = "edit confirmed via durable audit evidence after structured-output failure (new audit dir " + newAuditDirs[0] + "); builder applied-report missing";
|
|
1737
|
+
} else {
|
|
1642
1738
|
// No build observed — but that proves nothing (a fast-completing
|
|
1643
1739
|
// build can finish between polls, or the check itself failed). The
|
|
1644
1740
|
// outcome is UNKNOWN. No retry: re-issuing the edit here duplicated
|
|
@@ -1654,6 +1750,7 @@ while (i < STEPS.length) {
|
|
|
1654
1750
|
detail: "structured-output failure on rebuild trigger; build-state poll saw no build (or the check itself failed); edit may have been accepted as pending_init"
|
|
1655
1751
|
}, totalReworkCount);
|
|
1656
1752
|
return await parkTask("Publish outcome unknown: the rebuild trigger's child did not return JSON, and the follow-up build-state poll could not observe a build for slug " + PUBLISH_SLUG + ". The edit may have been accepted as pending_init, so no retry was issued — a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion before re-driving Publish. Fail-closed.");
|
|
1753
|
+
}
|
|
1657
1754
|
}
|
|
1658
1755
|
}
|
|
1659
1756
|
if (!rebuildReportMissing && !rebuildTrigger.edit_started && rebuildTrigger.error === "artifact_tools missing after load") {
|
|
@@ -1723,7 +1820,7 @@ while (i < STEPS.length) {
|
|
|
1723
1820
|
applied_report: publishAppliedObservation,
|
|
1724
1821
|
outcome: "submitted",
|
|
1725
1822
|
detail: rebuildReportMissing
|
|
1726
|
-
? "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing"
|
|
1823
|
+
? (rebuildEvidenceNote || "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing")
|
|
1727
1824
|
: "edit accepted; builder applied-report received"
|
|
1728
1825
|
}, totalReworkCount);
|
|
1729
1826
|
} else if (rebuildTrigger) {
|
|
@@ -1780,6 +1877,15 @@ while (i < STEPS.length) {
|
|
|
1780
1877
|
// lock was lost: stop the run and park the task — never continue to
|
|
1781
1878
|
// a provenance stamp or version assignment without holding the lock.
|
|
1782
1879
|
var buildPoll = null;
|
|
1880
|
+
// STEP 1b poll-signal accumulators (2026-09-15, task aadeccc3):
|
|
1881
|
+
// the durable audit-dir fallback below needs the poll's own
|
|
1882
|
+
// observations, not just its final verdict — whether our build was
|
|
1883
|
+
// ever seen, whether a stranger's build was ever in flight, and
|
|
1884
|
+
// what the last check observed. OR-ed across all three chunks so
|
|
1885
|
+
// a signal seen in any chunk survives the chunk boundary.
|
|
1886
|
+
var pollSawOurBuild = false;
|
|
1887
|
+
var pollSawStranger = false;
|
|
1888
|
+
var lastObservedAgentId = null;
|
|
1783
1889
|
for (var chunk = 1; chunk <= 3; chunk++) {
|
|
1784
1890
|
if (chunk > 1) {
|
|
1785
1891
|
var refreshPoll = await agent(
|
|
@@ -1802,36 +1908,152 @@ while (i < STEPS.length) {
|
|
|
1802
1908
|
: attemptKey("publish-artifact-poll-" + taskId + "-c" + chunk, totalReworkCount);
|
|
1803
1909
|
buildPoll = await agent(
|
|
1804
1910
|
"First call tool_search.load_tool_namespace with paths [\"artifact\"]. Then poll artifact_status for slug \"" + PUBLISH_SLUG + "\" \u2014 for OUR build only, the one whose agent_id is \"" + rebuildAgentId + "\" (the receipt captured when the edit was accepted; the agent_id is the artifact system's in-flight build correlation ID, stable across polls while the build runs). Check every 30 seconds, up to 7 checks (3.5 minutes max). On each check, read the raw build object:\n" +
|
|
1805
|
-
"
|
|
1911
|
+
"On every check, record whether you have positively OBSERVED our build: a running build whose agent_id equals \"" + rebuildAgentId + "\", or a completed-build record whose agent_id equals \"" + rebuildAgentId + "\" (if the tool surfaces one \u2014 match it mechanically, never assume).\n" +
|
|
1912
|
+
"- If no build is running (build is null) and you have NOT observed our build: our build's completion is UNPROVEN. Absence of a running build is not evidence our build ran. Do NOT report done.\n" +
|
|
1913
|
+
"- If no build is running (build is null) and you previously observed our build running: our build finished. Stop and report done.\n" +
|
|
1806
1914
|
"- If the running build's agent_id equals \"" + rebuildAgentId + "\": still ours \u2014 keep waiting.\n" +
|
|
1807
|
-
"- If the running build's agent_id is present but DIFFERENT:
|
|
1808
|
-
"Return JSON { \"build_done\": <true
|
|
1915
|
+
"- If the running build's agent_id is present but DIFFERENT: that is a stranger's build. Do NOT attribute its completion to our attempt and do NOT wait on it \u2014 keep checking within budget; if the budget expires without observing our build, report done=false. Record it in saw_stranger regardless of what else you observe.\n" +
|
|
1916
|
+
"Return JSON { \"build_done\": <true ONLY when you positively observed our build and it is no longer running, false otherwise>, \"saw_our_build\": <true if you observed our build at any check, false if never>, \"saw_stranger\": true if at ANY check a running build had an agent_id different from ours (\"" + rebuildAgentId + "\"), false otherwise, \"status\": \"<final status or timeout note>\", \"observed_agent_id\": \"<the agent_id seen on the last check, or null when no build was running>\" } and nothing else.",
|
|
1809
1917
|
{ key: pollKey, label: "Waiting for artifact build to complete (chunk " + chunk + " of 3)",
|
|
1810
|
-
schema: { type: "object", properties: { build_done: { type: "boolean" }, status: { type: "string" }, observed_agent_id: { type: ["string", "null"] } }, required: ["build_done"] },
|
|
1918
|
+
schema: { type: "object", properties: { build_done: { type: "boolean" }, saw_our_build: { type: "boolean" }, saw_stranger: { type: "boolean" }, status: { type: "string" }, observed_agent_id: { type: ["string", "null"] } }, required: ["build_done"] },
|
|
1811
1919
|
timeoutMs: 270000 }
|
|
1812
1920
|
);
|
|
1921
|
+
pollSawOurBuild = pollSawOurBuild || (buildPoll && buildPoll.saw_our_build === true);
|
|
1922
|
+
pollSawStranger = pollSawStranger || (buildPoll && buildPoll.saw_stranger === true);
|
|
1923
|
+
lastObservedAgentId = (buildPoll && buildPoll.observed_agent_id) || null;
|
|
1813
1924
|
if (buildPoll && buildPoll.build_done) { break; }
|
|
1814
1925
|
}
|
|
1815
1926
|
if (!buildPoll || !buildPoll.build_done) {
|
|
1816
1927
|
buildPoll = { build_done: false, status: (buildPoll && buildPoll.status) || "build still running after the 10.5-minute bounded poll" };
|
|
1817
1928
|
}
|
|
1818
|
-
if (buildPoll.build_done) {
|
|
1929
|
+
if (buildPoll.build_done && pollSawOurBuild) {
|
|
1819
1930
|
// STEP 1c (mechanical): NO provenance stamp here. Canary run 8
|
|
1820
1931
|
// (2026-09-11) proved the stamp cannot certify content: the
|
|
1821
1932
|
// builder's applied-report is derived from the carried diff, so
|
|
1822
1933
|
// verifyAppliedChanges above is circular — a fabricated report
|
|
1823
1934
|
// passes by construction, and every phase went green on a hollow
|
|
1824
|
-
// build. The stamp moves to the parent (docs/publish-verification.md)
|
|
1825
|
-
//
|
|
1826
|
-
//
|
|
1827
|
-
//
|
|
1828
|
-
//
|
|
1829
|
-
//
|
|
1935
|
+
// build. The stamp moves to the parent (docs/publish-verification.md);
|
|
1936
|
+
// the independent read-back step is currently unavailable (no
|
|
1937
|
+
// agent-callable read-back tool exists — artifact_inspect was
|
|
1938
|
+
// removed by the platform 2026-09-14), so the parent cannot
|
|
1939
|
+
// confirm content and the task parks for verification.
|
|
1940
|
+
// QA's provenance check enforces the stamp mechanically.
|
|
1941
|
+
// An unverified publish fails loudly in QA instead of passing
|
|
1942
|
+
// silently here.
|
|
1830
1943
|
publishBuildLanded = true;
|
|
1831
1944
|
artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
|
|
1832
1945
|
log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
|
|
1833
1946
|
} else {
|
|
1834
|
-
|
|
1947
|
+
// STEP 1b durable audit-dir fallback (2026-09-15, task aadeccc3):
|
|
1948
|
+
// the poll above only observes IN-FLIGHT builds. A build that
|
|
1949
|
+
// finished between the receipt capture and the poll's first check
|
|
1950
|
+
// leaves no in-flight trace — but the platform's audit harness
|
|
1951
|
+
// leaves a durable one (~/workspace/ts-spaces/<slug>/audits/
|
|
1952
|
+
// <timestamp>-<id>/ per completed build). Diff the audit-dir
|
|
1953
|
+
// listing against the pre-trigger snapshot: a timestamped dir
|
|
1954
|
+
// that appeared during the attempt window is evidence a build
|
|
1955
|
+
// completed. Attribution is by window, not by build identity:
|
|
1956
|
+
// the poll's saw_stranger signal only catches stranger builds in
|
|
1957
|
+
// flight AT a check — a stranger that finished entirely inside
|
|
1958
|
+
// the window is indistinguishable, so any observed stranger
|
|
1959
|
+
// blocks attribution and the outcome stays unknown. This never
|
|
1960
|
+
// re-issues the edit and never stamps provenance — ok=true only
|
|
1961
|
+
// routes to the parent's independent content read-back, which
|
|
1962
|
+
// remains the real verification.
|
|
1963
|
+
//
|
|
1964
|
+
// The poll end-state is read from the poll's own observations,
|
|
1965
|
+
// not from build_done alone: a build in flight at the last check
|
|
1966
|
+
// means the budget was shorter than the latency (or the build is
|
|
1967
|
+
// stuck) — NOT that no build ever started; nothing observed at
|
|
1968
|
+
// any check is the never-started signal.
|
|
1969
|
+
var pollEndState = lastObservedAgentId ? "build-still-running-at-poll-end"
|
|
1970
|
+
: (pollSawOurBuild ? "our-build-observed-then-unconfirmed" : "no-build-observed-in-window");
|
|
1971
|
+
var strangerObserved = pollSawStranger;
|
|
1972
|
+
var newAuditDirsAfterPoll = [];
|
|
1973
|
+
try {
|
|
1974
|
+
var auditAfterPoll = await agent(
|
|
1975
|
+
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
|
|
1976
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1977
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
|
|
1978
|
+
{ key: attemptKey("publish-audit-after-poll-" + taskId, totalReworkCount), label: "Re-listing audit dirs after build poll",
|
|
1979
|
+
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1980
|
+
);
|
|
1981
|
+
var auditDirsAfterPollList = String((auditAfterPoll && auditAfterPoll.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1982
|
+
newAuditDirsAfterPoll = auditDirsAfterPollList.filter(function (d) {
|
|
1983
|
+
return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
|
|
1984
|
+
});
|
|
1985
|
+
log("Publish audit-dir re-list after build poll for task " + taskId + ": " + newAuditDirsAfterPoll.length + " new timestamped dir(s)");
|
|
1986
|
+
} catch (auditAfterPollErr) {
|
|
1987
|
+
log("Publish audit-dir re-list after build poll failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterPollErr && auditAfterPollErr.message ? auditAfterPollErr.message : auditAfterPollErr));
|
|
1988
|
+
}
|
|
1989
|
+
// auditReportOk: pure tri-state read of a report.json body —
|
|
1990
|
+
// true (build ok), false (build failed), null (missing or
|
|
1991
|
+
// unreadable — not evidence either way). The child returns the
|
|
1992
|
+
// raw body verbatim; interpretation lives here, never in prose.
|
|
1993
|
+
var auditReportOk = function (raw) {
|
|
1994
|
+
if (typeof raw !== "string") return null;
|
|
1995
|
+
var trimmed = raw.trim();
|
|
1996
|
+
if (trimmed === "" || trimmed === "MISSING") return null;
|
|
1997
|
+
var parsed;
|
|
1998
|
+
try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
|
|
1999
|
+
if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
|
|
2000
|
+
return null;
|
|
2001
|
+
};
|
|
2002
|
+
var auditOkAfterPoll = null;
|
|
2003
|
+
var newestAuditDirAfterPoll = null;
|
|
2004
|
+
if (newAuditDirsAfterPoll.length > 0 && !strangerObserved) {
|
|
2005
|
+
newAuditDirsAfterPoll.sort();
|
|
2006
|
+
newestAuditDirAfterPoll = newAuditDirsAfterPoll[newAuditDirsAfterPoll.length - 1];
|
|
2007
|
+
try {
|
|
2008
|
+
var auditOkRead = await agent(
|
|
2009
|
+
"Read the build report for artifact slug \"" + PUBLISH_SLUG + "\", audit dir \"" + newestAuditDirAfterPoll + "\" (verbatim read, never interpreted, never a gate).\n" +
|
|
2010
|
+
"Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestAuditDirAfterPoll + "/report.json 2>/dev/null || echo MISSING\n" +
|
|
2011
|
+
"Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
|
|
2012
|
+
{ key: attemptKey("publish-audit-ok-after-poll-" + taskId, totalReworkCount), label: "Reading build report after build poll",
|
|
2013
|
+
schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
|
|
2014
|
+
);
|
|
2015
|
+
auditOkAfterPoll = auditReportOk(auditOkRead && auditOkRead.raw);
|
|
2016
|
+
} catch (auditOkReadErr) {
|
|
2017
|
+
log("Publish build-report read after build poll failed for task " + taskId + " (non-fatal, treated as unknown): " + (auditOkReadErr && auditOkReadErr.message ? auditOkReadErr.message : auditOkReadErr));
|
|
2018
|
+
auditOkAfterPoll = null;
|
|
2019
|
+
}
|
|
2020
|
+
}
|
|
2021
|
+
if (auditOkAfterPoll === true) {
|
|
2022
|
+
publishBuildLanded = true;
|
|
2023
|
+
artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
|
|
2024
|
+
log("Publish build landed for task " + taskId + " via durable audit evidence — provenance stamp deferred to parent content verification");
|
|
2025
|
+
await recordPublishLedger({
|
|
2026
|
+
commit: mergeCommitForPublish,
|
|
2027
|
+
attempt: rebuildAttemptKey,
|
|
2028
|
+
agent_id: rebuildAgentId,
|
|
2029
|
+
applied_report: publishAppliedObservation,
|
|
2030
|
+
outcome: "submitted",
|
|
2031
|
+
detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
|
|
2032
|
+
}, totalReworkCount);
|
|
2033
|
+
} else if (auditOkAfterPoll === false) {
|
|
2034
|
+
publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestAuditDirAfterPoll + ", report ok=false). Explicit negative evidence: a build ran and failed (attribution by window, not by build identity — no stranger build was observed in flight during the poll). The publish did not land — provenance was not stamped. Fail-closed.";
|
|
2035
|
+
await recordPublishLedger({
|
|
2036
|
+
commit: mergeCommitForPublish,
|
|
2037
|
+
attempt: rebuildAttemptKey,
|
|
2038
|
+
agent_id: rebuildAgentId,
|
|
2039
|
+
applied_report: publishAppliedObservation,
|
|
2040
|
+
outcome: "failed",
|
|
2041
|
+
detail: "a build ran and failed (attribution by window, not by build identity): audit dir " + newestAuditDirAfterPoll + " report ok=false; no stranger build observed in flight during the poll"
|
|
2042
|
+
}, totalReworkCount);
|
|
2043
|
+
} else {
|
|
2044
|
+
var unattributableReason = strangerObserved ? "stranger-build-observed-during-poll"
|
|
2045
|
+
: (pollEndState === "build-still-running-at-poll-end" ? "build-still-running-at-poll-end"
|
|
2046
|
+
: (newAuditDirsAfterPoll.length === 0 ? "no-new-audit-dir-in-window" : "audit-report-unreadable-or-missing"));
|
|
2047
|
+
publishFailure = "Artifact build completion unproven (fail-closed, no provenance stamped): unattributable_reason=" + unattributableReason + "; poll_end_state=" + pollEndState + "; " + "saw_our_build=" + pollSawOurBuild + "; new_audit_dirs=" + newAuditDirsAfterPoll.length + ". Attribution is by window, not by build identity. The publish may or may not have landed. Fail-closed.";
|
|
2048
|
+
await recordPublishLedger({
|
|
2049
|
+
commit: mergeCommitForPublish,
|
|
2050
|
+
attempt: rebuildAttemptKey,
|
|
2051
|
+
agent_id: rebuildAgentId,
|
|
2052
|
+
applied_report: publishAppliedObservation,
|
|
2053
|
+
outcome: "unknown",
|
|
2054
|
+
detail: "durable audit-dir fallback could not attribute a completed build to this attempt (unattributable_reason=" + unattributableReason + ", poll_end_state=" + pollEndState + ")"
|
|
2055
|
+
}, totalReworkCount);
|
|
2056
|
+
}
|
|
1835
2057
|
}
|
|
1836
2058
|
} else {
|
|
1837
2059
|
publishFailure = "Artifact rebuild trigger failed: " + (rebuildTrigger.error || "artifact_edit not accepted") + ". The publish did not land.";
|
|
@@ -1924,32 +2146,40 @@ while (i < STEPS.length) {
|
|
|
1924
2146
|
"Public docs (API.md, README, published action schemas) are NOT source code — read them freely, exactly as a user would.\n" +
|
|
1925
2147
|
"DOCS GATE: If the fix is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n" +
|
|
1926
2148
|
"To test, run in shell and read the stdout JSON:\n" + crewCmd("get-state", {}) + "\nThis returns current sessions, events, and tasks.\n" +
|
|
1927
|
-
"Verify the fix
|
|
2149
|
+
"Verify the fix against the task description in the returned data: the state the bug corrupted should now read correctly, and the task's own expected behavior should hold.\n" +
|
|
1928
2150
|
"You can also check specific data with: node " + CREW_API + " --crew-home " + crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
|
|
1929
|
-
"Do NOT use artifact_inspect — it
|
|
2151
|
+
"Do NOT use artifact_inspect — it was removed by the platform (2026-09-14) and does not exist; do not substitute artifact.inspect (malfunction diagnosis, not an inspection tool).\n" +
|
|
1930
2152
|
"File follow-up tasks by running in shell:\n" + crewCmd("create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n" +
|
|
1931
2153
|
npmPublishCheck +
|
|
1932
2154
|
"Report your test results as plain prose.\n" +
|
|
1933
2155
|
"End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails.";
|
|
1934
|
-
//
|
|
1935
|
-
//
|
|
1936
|
-
// artifact_inspect (async; the parent triggers the post-change capture
|
|
1937
|
-
// after this step). The visual verdict is produced by the parent after
|
|
1938
|
-
// the rendered post-change inspection results arrive.
|
|
2156
|
+
// qaVisual: Hazel runs the experiential see-act loop herself (STEP 1
|
|
2157
|
+
// below) and owns the visual verdict — no parent capture protocol.
|
|
1939
2158
|
if (qaVisual) {
|
|
1940
2159
|
instructions = "You are code-blind QA. You NEVER read source files. Public docs are not source — read them as a user would.\n" +
|
|
1941
|
-
"
|
|
1942
|
-
"
|
|
2160
|
+
"STEP 1: Experiential visual inspection — drive the fixed artifact as a user would, one browser step at a time, and verify the reported bug is actually fixed.\n" +
|
|
2161
|
+
"You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
|
|
2162
|
+
"Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
|
|
2163
|
+
"a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and continue with the mechanical checks.\n" +
|
|
2164
|
+
"b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
|
|
2165
|
+
"c. Bounded see-act loop, at most 8 steps: re-run the reproduction steps for the reported bug — does it still occur? Then check the surrounding views for regressions. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
|
|
2166
|
+
"c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
|
|
2167
|
+
"c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
|
|
2168
|
+
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
2169
|
+
"d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the fix needs such a sequence to verify, report NOT POSSIBLE for that part and judge what you can.\n" +
|
|
2170
|
+
"e. Judge as a user against the task description: is the reported bug fixed AND is nothing else visibly broken? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
|
|
2171
|
+
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
2172
|
+
"Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
|
|
1943
2173
|
"MECHANICAL CHECKS:\n" +
|
|
1944
2174
|
"DOCS GATE: If the fix is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n" +
|
|
1945
2175
|
"To test, run in shell and read the stdout JSON:\n" + crewCmd("get-state", {}) + "\nThis returns current sessions, events, and tasks.\n" +
|
|
1946
|
-
"Verify the fix
|
|
2176
|
+
"Verify the fix against the task description in the returned data: the state the bug corrupted should now read correctly, and the task's own expected behavior should hold.\n" +
|
|
1947
2177
|
"You can also check specific data with: node " + CREW_API + " --crew-home " + crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
|
|
1948
2178
|
npmPublishCheck +
|
|
1949
2179
|
"File follow-up tasks by running in shell:\n" + crewCmd("create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n\n" +
|
|
1950
2180
|
"BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
|
|
1951
2181
|
"Report your test results as plain prose.\n" +
|
|
1952
|
-
"End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails on the mechanical checks.";
|
|
2182
|
+
"End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails on the visual or the mechanical checks. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the reported bug, fixed>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL verdict must carry a machine-readable reason: the workflow closeout cross-checks verdict.json against your prose VERDICT line, and an unreasoned or contradictory verdict fails the phase (never routes to rework).";
|
|
1953
2183
|
}
|
|
1954
2184
|
if (PUBLISH_TYPE === "artifact") {
|
|
1955
2185
|
instructions = "PROVENANCE CHECK (this project publishes to a dashboard artifact).\n" +
|
|
@@ -2117,6 +2347,70 @@ while (i < STEPS.length) {
|
|
|
2117
2347
|
verdictPassed = verdict.passed;
|
|
2118
2348
|
}
|
|
2119
2349
|
|
|
2350
|
+
// QA verdict.json closeout gate (task 30dceb78): the prose VERDICT: line is
|
|
2351
|
+
// the routing signal, but verdict.json is the reason-carrying record the
|
|
2352
|
+
// workflow actually reads. A FAIL verdict must carry a machine-readable
|
|
2353
|
+
// reason; the workflow refuses to route to rework on an unreasoned or
|
|
2354
|
+
// contradictory verdict. The deterministic cross-checker
|
|
2355
|
+
// (lib/read-ooda-verdict.js) runs against the prose verdict: a missing,
|
|
2356
|
+
// corrupt, contradictory, or reason-less record fails the phase for retry
|
|
2357
|
+
// — the dispatcher re-runs QA at the same step under its
|
|
2358
|
+
// consecutive-failure cap — instead of routing to rework. Without this
|
|
2359
|
+
// gate, a bare VERDICT: FAIL with an all-positive report (canary
|
|
2360
|
+
// 2026-09-15, task 1d692d91) rebuilt nothing and parked at Publish on an
|
|
2361
|
+
// unobserved artifact build. The gate applies only to the experiential QA
|
|
2362
|
+
// path (qaVisual), the only path that writes verdict.json.
|
|
2363
|
+
if (step.name === "QA" && qaVisual && verdictPassed !== null) {
|
|
2364
|
+
var qaVerdictDir = crewHome + "/task-evidence/" + taskId + "/postchange";
|
|
2365
|
+
var qaVerdictExpect = verdictPassed ? "PASS" : "FAIL";
|
|
2366
|
+
var qaVerdictOut = "";
|
|
2367
|
+
try {
|
|
2368
|
+
var qaVerdictCheck = await agent(
|
|
2369
|
+
"Run: node " + crewHome + "/current/lib/read-ooda-verdict.js --dir " + qaVerdictDir + " --expect " + qaVerdictExpect + "\n" +
|
|
2370
|
+
"Return JSON { \"output\": \"<the command's full stdout, trimmed>\" } and nothing else.",
|
|
2371
|
+
{ key: attemptKey("qa-verdict-check-" + taskId, totalReworkCount), label: "Cross-checking QA verdict.json",
|
|
2372
|
+
schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
|
|
2373
|
+
);
|
|
2374
|
+
qaVerdictOut = (qaVerdictCheck && qaVerdictCheck.output ? qaVerdictCheck.output : "").trim();
|
|
2375
|
+
} catch (e) {
|
|
2376
|
+
qaVerdictOut = "";
|
|
2377
|
+
}
|
|
2378
|
+
// The script prints exactly one JSON line; exit info does not survive
|
|
2379
|
+
// the schema'd return, so ok:false on that line is the gate signal.
|
|
2380
|
+
var qaVerdictGate = null;
|
|
2381
|
+
try {
|
|
2382
|
+
var qaVerdictLines = qaVerdictOut.split("\n");
|
|
2383
|
+
qaVerdictGate = JSON.parse(qaVerdictLines[qaVerdictLines.length - 1]);
|
|
2384
|
+
} catch (e) {
|
|
2385
|
+
qaVerdictGate = null;
|
|
2386
|
+
}
|
|
2387
|
+
if (!qaVerdictGate || qaVerdictGate.ok !== true) {
|
|
2388
|
+
var qaGateCode = (qaVerdictGate && qaVerdictGate.code) ? qaVerdictGate.code : "unreadable";
|
|
2389
|
+
var qaGateDetail = (qaVerdictGate && qaVerdictGate.error) ? qaVerdictGate.error : (qaVerdictOut ? qaVerdictOut.slice(0, 200) : "cross-checker produced no usable output");
|
|
2390
|
+
log("QA verdict.json closeout gate failed (" + qaGateCode + "): " + qaGateDetail + " — marking QA failed for retry, NOT routing to rework");
|
|
2391
|
+
await agent(
|
|
2392
|
+
"Record QA verdict gate failure.\n" +
|
|
2393
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
|
|
2394
|
+
task_id: taskId,
|
|
2395
|
+
session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name,
|
|
2396
|
+
status: "failed", notes: "QA verdict.json closeout gate failed (" + qaGateCode + "): " + qaGateDetail + ". The prose VERDICT line said " + qaVerdictExpect + " but verdict.json is missing, corrupt, contradictory, or (for FAIL) carries no machine-readable reason. Unreasoned or contradictory verdicts never route to rework; phase failed for retry" },
|
|
2397
|
+
event: { task_id: taskId, type: "failed",
|
|
2398
|
+
message: "QA verdict.json closeout gate failed (" + qaGateCode + ") — verdict unreasoned or contradictory, phase failed, dispatcher will retry QA" }
|
|
2399
|
+
}),
|
|
2400
|
+
{ key: "record-qa-verdict-gate-fail-" + step.name, label: "Recording QA verdict gate failure" }
|
|
2401
|
+
);
|
|
2402
|
+
return {
|
|
2403
|
+
__hatchWorkflowControl: "blocked",
|
|
2404
|
+
result: {
|
|
2405
|
+
blocked_reason: "QA verdict.json closeout gate failed (" + qaGateCode + ")",
|
|
2406
|
+
message: "QA's prose VERDICT line said " + qaVerdictExpect + " but the machine-readable verdict.json is " + qaGateCode + " (" + qaGateDetail + "). A FAIL verdict must carry a machine-readable reason and the record must agree with the prose line. The phase is marked failed and the dispatcher will retry QA; it is not routed to rework.",
|
|
2407
|
+
task_id: taskId
|
|
2408
|
+
}
|
|
2409
|
+
};
|
|
2410
|
+
}
|
|
2411
|
+
log("QA verdict.json closeout gate passed: verdict.json agrees with prose VERDICT: " + qaVerdictExpect);
|
|
2412
|
+
}
|
|
2413
|
+
|
|
2120
2414
|
|
|
2121
2415
|
// Worktree confinement (Build only): the declared worktree path must
|
|
2122
2416
|
// match WORKTREE_HINT exactly. A builder that worked in any other
|
|
@@ -2148,6 +2442,52 @@ while (i < STEPS.length) {
|
|
|
2148
2442
|
};
|
|
2149
2443
|
}
|
|
2150
2444
|
log("Build worktree confinement passed: " + wt.path);
|
|
2445
|
+
|
|
2446
|
+
// Already-merged idempotency: a `repo_diff: none (already-merged:
|
|
2447
|
+
// <sha>)` declaration is verified mechanically — <sha> must resolve
|
|
2448
|
+
// and be an ancestor of main in the configured repo. A fabricated or
|
|
2449
|
+
// mistaken declaration fails the phase here (the dispatcher retries
|
|
2450
|
+
// Build under its consecutive-failure cap); a verified declaration is
|
|
2451
|
+
// recorded in alreadyMergedSha for Review's no-diff branch. Without
|
|
2452
|
+
// this guard, Build correctly doing nothing left Review with no
|
|
2453
|
+
// mechanical way to accept an empty diff, and Cass rejected for "no
|
|
2454
|
+
// commits ahead of main — the builder likely forgot to commit" while
|
|
2455
|
+
// the deliverable sat on main (canary 2026-09-15, task 1d692d91).
|
|
2456
|
+
// The sha is hex-only by construction (extractAlreadyMerged), so
|
|
2457
|
+
// interpolating it into the shell command cannot inject.
|
|
2458
|
+
var am = extractAlreadyMerged(workerText);
|
|
2459
|
+
if (am.sha) {
|
|
2460
|
+
var amCheck = await agent(
|
|
2461
|
+
"Verify the builder's already-merged declaration.\n" +
|
|
2462
|
+
"Run in shell and return the stdout verbatim:\n" +
|
|
2463
|
+
"cd " + REPO_PATH + " && git rev-parse --verify --quiet " + am.sha + " >/dev/null && git merge-base --is-ancestor " + am.sha + " main && echo ALREADY_MERGED_YES || echo ALREADY_MERGED_NO",
|
|
2464
|
+
{ key: "verify-already-merged" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Verifying already-merged declaration" }
|
|
2465
|
+
);
|
|
2466
|
+
var amOut = (typeof amCheck === "string") ? amCheck : JSON.stringify(amCheck);
|
|
2467
|
+
if (!/ALREADY_MERGED_YES/.test(amOut)) {
|
|
2468
|
+
log("Build already-merged declaration failed verification — " + am.sha + " is not an ancestor of main — marking failed for retry");
|
|
2469
|
+
await agent(
|
|
2470
|
+
"Record already-merged verification failure.\n" +
|
|
2471
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
|
|
2472
|
+
task_id: taskId,
|
|
2473
|
+
session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: "failed",
|
|
2474
|
+
notes: "Build declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of main in the configured repo. The declaration is fabricated or mistaken; the work is not on main. Phase failed for retry" },
|
|
2475
|
+
event: { task_id: taskId, type: "failed", message: "Build already-merged declaration failed verification — " + am.sha + " not an ancestor of main, phase failed, dispatcher will retry" }
|
|
2476
|
+
}),
|
|
2477
|
+
{ key: "record-already-merged-fail-" + step.name, label: "Recording already-merged verification failure" }
|
|
2478
|
+
);
|
|
2479
|
+
return {
|
|
2480
|
+
__hatchWorkflowControl: "blocked",
|
|
2481
|
+
result: {
|
|
2482
|
+
blocked_reason: "Build already-merged declaration failed verification",
|
|
2483
|
+
message: "The builder declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of main. The work is not on main; the phase is marked failed and the dispatcher will retry Build.",
|
|
2484
|
+
task_id: taskId
|
|
2485
|
+
}
|
|
2486
|
+
};
|
|
2487
|
+
}
|
|
2488
|
+
alreadyMergedSha = am.sha;
|
|
2489
|
+
log("Build already-merged declaration verified: " + am.sha + " is an ancestor of main");
|
|
2490
|
+
}
|
|
2151
2491
|
}
|
|
2152
2492
|
|
|
2153
2493
|
// Deterministic closeout: no formatter agent. The verdict is mechanical
|
|
@@ -2276,9 +2616,10 @@ while (i < STEPS.length) {
|
|
|
2276
2616
|
// it to HEAD: that verifies the stamp, not the content. Canary run 8
|
|
2277
2617
|
// (2026-09-11) passed it with a hollow build — the stamp was honest, the
|
|
2278
2618
|
// artifact was stale, all eight phases green. The stamp now moves to the
|
|
2279
|
-
// parent
|
|
2280
|
-
//
|
|
2281
|
-
//
|
|
2619
|
+
// parent (docs/publish-verification.md); the independent read-back step
|
|
2620
|
+
// is currently unavailable (no agent-callable read-back tool exists —
|
|
2621
|
+
// artifact_inspect was removed by the platform 2026-09-14), so the parent
|
|
2622
|
+
// cannot confirm content and the task parks for verification. QA's
|
|
2282
2623
|
// provenance check enforces the stamp — an unverified publish fails loudly
|
|
2283
2624
|
// there instead of passing silently here.
|
|
2284
2625
|
// Skip-aware (park 2026-09-11): an empty-diff Integrate takes no merge
|
|
@@ -2286,38 +2627,13 @@ while (i < STEPS.length) {
|
|
|
2286
2627
|
// there is no new content to verify, so verification is vacuous.
|
|
2287
2628
|
// publishSkippedNoLock is workflow-computed state from the explicit
|
|
2288
2629
|
// lock-status read in STEP 0, not agent prose.
|
|
2289
|
-
|
|
2290
|
-
|
|
2291
|
-
|
|
2292
|
-
|
|
2293
|
-
|
|
2294
|
-
|
|
2295
|
-
|
|
2296
|
-
try {
|
|
2297
|
-
var inspectResult = await agent(
|
|
2298
|
-
ARTIFACT_LOAD_PREAMBLE +
|
|
2299
|
-
"Call artifact_inspect with slug \"" + PUBLISH_SLUG + "\", repair_authorized false, and verbatim_request exactly as follows:\n" +
|
|
2300
|
-
"<<<READBACK_REQUEST\n" + buildPublishReadbackRequest(taskId, mergeCommitForPublish, mergeDiff, rebuildAgentId) + "\nREADBACK_REQUEST\n" +
|
|
2301
|
-
"If artifact_inspect is still not available after the load, do NOT improvise — return { \"triggered\": false, \"inspection_id\": \"\", \"error\": \"artifact_tools missing after load\" } and nothing else.\n" +
|
|
2302
|
-
"Return JSON { \"triggered\": <true if the inspection started, false otherwise>, \"inspection_id\": \"<the inspection id, or empty string>\", \"error\": \"<details or empty string>\" } and nothing else.",
|
|
2303
|
-
{ key: attemptKey("publish-verify-inspect-" + taskId, totalReworkCount), label: "Triggering publish content read-back",
|
|
2304
|
-
schema: { type: "object", properties: { triggered: { type: "boolean" }, inspection_id: { type: "string" }, error: { type: "string" } }, required: ["triggered"] } }
|
|
2305
|
-
);
|
|
2306
|
-
publishVerifyInspect.triggered = !!(inspectResult && inspectResult.triggered);
|
|
2307
|
-
publishVerifyInspect.inspection_id = (inspectResult && inspectResult.inspection_id) || "";
|
|
2308
|
-
publishVerifyInspect.error = (inspectResult && inspectResult.error) || "";
|
|
2309
|
-
if (publishVerifyInspect.triggered) {
|
|
2310
|
-
log("Publish content read-back inspection triggered for task " + taskId + ": " + publishVerifyInspect.inspection_id);
|
|
2311
|
-
} else {
|
|
2312
|
-
log("Publish content read-back inspect trigger failed for task " + taskId + ": " + (publishVerifyInspect.error || "not started") + " — the park below asks the parent to trigger it manually");
|
|
2313
|
-
}
|
|
2314
|
-
} catch (e) {
|
|
2315
|
-
publishVerifyInspect.error = (e && e.message ? e.message : String(e)).slice(0, 200);
|
|
2316
|
-
log("Publish content read-back inspect trigger threw for task " + taskId + ": " + publishVerifyInspect.error + " — the park below asks the parent to trigger it manually");
|
|
2317
|
-
}
|
|
2318
|
-
} // end: !publishSkippedNoLock && publishBuildLanded — a skipped or failed publish has nothing to verify
|
|
2319
|
-
}
|
|
2320
|
-
|
|
2630
|
+
// The parent (tick worker) triggers the ONE read-back inspection it can
|
|
2631
|
+
// actually receive (async results go to the root agent, never into a
|
|
2632
|
+
// workflow run — a workflow-side trigger would be an orphan). The workflow
|
|
2633
|
+
// only parks; the parent's scan builds the request deterministically via
|
|
2634
|
+
// lib/build-readback-request.js and ferries the inspection.
|
|
2635
|
+
// publishBuildLanded and publishSkippedNoLock are workflow-computed state;
|
|
2636
|
+
// a skipped or failed publish has nothing to verify.
|
|
2321
2637
|
// Session notes. Machine-readable marker lines are extracted from the full
|
|
2322
2638
|
// worker report and appended AFTER the slice so a long report can never
|
|
2323
2639
|
// amputate them; later phases (Review reading repo_diff:, QA backstop
|
|
@@ -2335,16 +2651,23 @@ while (i < STEPS.length) {
|
|
|
2335
2651
|
} else {
|
|
2336
2652
|
summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
|
|
2337
2653
|
}
|
|
2338
|
-
|
|
2339
|
-
//
|
|
2340
|
-
//
|
|
2341
|
-
//
|
|
2342
|
-
//
|
|
2343
|
-
|
|
2344
|
-
|
|
2345
|
-
|
|
2654
|
+
// Already-merged attestation: when the Build gate verified the builder's
|
|
2655
|
+
// already-merged declaration, the workflow records its own marker line in
|
|
2656
|
+
// the session notes (like the builder markers above, it is appended after
|
|
2657
|
+
// the slice so it can never be amputated). A later run resumed at Review
|
|
2658
|
+
// hydrates alreadyMergedSha from this workflow-attested line — never from
|
|
2659
|
+
// the builder's declaration alone.
|
|
2660
|
+
if (step.name === "Build" && alreadyMergedSha) {
|
|
2661
|
+
summary += "\nalready_merged_verified: " + alreadyMergedSha;
|
|
2346
2662
|
}
|
|
2347
2663
|
|
|
2664
|
+
// Visual verdict evidence (2026-09-15): no parent marker is appended. Hazel
|
|
2665
|
+
// records her own verdict.json + the append-only verdicts.jsonl in the
|
|
2666
|
+
// task-evidence dir during the QA loop itself — the QA session notes carry
|
|
2667
|
+
// her prose report, and the OODA report carries the machine-readable state.
|
|
2668
|
+
// The old parent-protocol marker for the QA closeout is deleted.
|
|
2669
|
+
// is deleted.
|
|
2670
|
+
|
|
2348
2671
|
// Capture mapper's spec for Build and Review
|
|
2349
2672
|
if (step.name === "Map" && passed) {
|
|
2350
2673
|
mapperSpec = summary;
|
|
@@ -2387,40 +2710,12 @@ while (i < STEPS.length) {
|
|
|
2387
2710
|
return await parkTask("Reproduction failed — needs PM attention");
|
|
2388
2711
|
}
|
|
2389
2712
|
|
|
2390
|
-
// Visual
|
|
2391
|
-
//
|
|
2392
|
-
//
|
|
2393
|
-
//
|
|
2394
|
-
//
|
|
2395
|
-
|
|
2396
|
-
var vvStatus = await visualVerdictStatus();
|
|
2397
|
-
if (vvStatus.found && vvStatus.verdict === "PASS") {
|
|
2398
|
-
log("Visual verdict PASS recorded for task " + taskId + (vvStatus.detail ? " — " + vvStatus.detail : ""));
|
|
2399
|
-
} else if (vvStatus.found && vvStatus.verdict === "FAIL") {
|
|
2400
|
-
// A FAIL whose reason begins exactly "rendering impossible:" is not
|
|
2401
|
-
// reworkable — there is no rendered evidence to fix against. Park for
|
|
2402
|
-
// human attention instead of bouncing to Build.
|
|
2403
|
-
if (vvStatus.detail.indexOf("rendering impossible:") === 0) {
|
|
2404
|
-
log("Visual verdict FAIL (rendering impossible) for task " + taskId + " — parking for human attention");
|
|
2405
|
-
return await parkTask("Visual verdict FAIL — rendering impossible, human attention required: " + vvStatus.detail);
|
|
2406
|
-
}
|
|
2407
|
-
totalReworkCount++;
|
|
2408
|
-
if (totalReworkCount > MAX_TOTAL_REWORK) {
|
|
2409
|
-
log("Shared rework budget exhausted for task " + taskId + " — parking after visual verdict FAIL");
|
|
2410
|
-
return await parkTask("Exceeded shared rework budget (" + MAX_TOTAL_REWORK + " total rework attempts across Review and QA) after visual verdict FAIL: " + vvStatus.detail);
|
|
2411
|
-
}
|
|
2412
|
-
rejectionNotes = "Visual verdict FAIL: " + vvStatus.detail;
|
|
2413
|
-
i = BUILD_INDEX;
|
|
2414
|
-
log("Visual verdict FAIL — bouncing to Build (rework #" + totalReworkCount + " of " + MAX_TOTAL_REWORK + ")");
|
|
2415
|
-
continue;
|
|
2416
|
-
} else {
|
|
2417
|
-
if (!VISUAL_PROTOCOL_AVAILABLE) {
|
|
2418
|
-
log("Visual verdict protocol not available (VISUAL_PROTOCOL_AVAILABLE=false) for task " + taskId + " — skipping visual gate, QA mechanical checks already passed");
|
|
2419
|
-
} else {
|
|
2420
|
-
return await parkTask("Visual verdict pending — parent: run the post-change capture + visual verdict protocol in docs/visual-verdict.md (capture plan is in the QA session notes)");
|
|
2421
|
-
}
|
|
2422
|
-
}
|
|
2423
|
-
}
|
|
2713
|
+
// Visual verdict ownership (2026-09-15): Hazel owns the visual verdict
|
|
2714
|
+
// experientially — the QA instructions above have her drive the see-act
|
|
2715
|
+
// loop herself and record verdict.json + the append-only verdicts.jsonl.
|
|
2716
|
+
// Her prose VERDICT: line drives `passed` via extractVerdict; a FAIL
|
|
2717
|
+
// bounces to Build through the standard Review/QA rejection path below.
|
|
2718
|
+
// The old parent-recorded note gate is deleted.
|
|
2424
2719
|
|
|
2425
2720
|
// Review/QA rejection bounces to Build
|
|
2426
2721
|
if (!passed && (step.name === "Review" || step.name === "QA")) {
|
|
@@ -2461,10 +2756,7 @@ while (i < STEPS.length) {
|
|
|
2461
2756
|
if (passed && step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG && !publishSkippedNoLock && publishBuildLanded) {
|
|
2462
2757
|
return await parkTask("publish: verification-requested " + mergeCommitForPublish +
|
|
2463
2758
|
" (build " + (rebuildAgentId || "agent_id unobserved") + ")" +
|
|
2464
|
-
" — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md"
|
|
2465
|
-
(publishVerifyInspect.triggered
|
|
2466
|
-
? " (content read-back inspection " + publishVerifyInspect.inspection_id + " already triggered)."
|
|
2467
|
-
: " (read-back inspect trigger failed: " + (publishVerifyInspect.error || "not started") + " — parent: trigger artifact_inspect manually)."));
|
|
2759
|
+
" — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md.");
|
|
2468
2760
|
}
|
|
2469
2761
|
|
|
2470
2762
|
i++;
|