muse-crew 0.7.10 → 0.7.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +36 -14
- package/docs/guide.md +5 -5
- package/docs/ooda-report.md +150 -0
- package/docs/publish-verification.md +277 -88
- package/docs/visual-verdict.md +81 -67
- package/lib/AGENTS.md +8 -0
- package/lib/append-ooda-step.js +167 -0
- package/lib/build-readback-request.js +130 -0
- package/lib/compose-evidence-caption.js +141 -0
- package/lib/crew-api.js +467 -43
- package/lib/edit-image.py +216 -0
- package/lib/read-ooda-verdict.js +94 -0
- package/lib/readback-disk.js +186 -0
- package/lib/render-html.js +142 -0
- package/lib/see-act.js +327 -0
- package/lib/serve-artifact.js +203 -0
- package/lib/verify-publish.js +323 -0
- package/lib/write-ooda-verdict.js +147 -0
- package/package.json +1 -1
- package/seed/cron-body-template.md +30 -4
- package/seed/crons.json +1 -1
- package/workflows/bugfix.js +512 -220
- package/workflows/chore.js +389 -119
- package/workflows/crew-dispatch.js +72 -13
- package/workflows/docs.js +11 -2
- package/workflows/standard.js +413 -235
package/workflows/chore.js
CHANGED
|
@@ -26,6 +26,15 @@ const startStepIndex = inputs.start_step_index || 0;
|
|
|
26
26
|
// resolution back via updatetask in the self-claim below.
|
|
27
27
|
const RESOLVED_WORKFLOW = inputs.resolved_workflow || null;
|
|
28
28
|
const WORKFLOW_WAS_NULL = inputs.workflow_was_null === true;
|
|
29
|
+
// One-shot recovery routing: the dispatcher sets inputs.next_phase when it
|
|
30
|
+
// routes this run via an explicit recover-task redirect. The value is
|
|
31
|
+
// consumed (cleared) atomically by the successful self-claim below:
|
|
32
|
+
// claim-task takes expected_next_phase and clears the matching next_phase in
|
|
33
|
+
// the same transaction as the winning session insert, so no platform death
|
|
34
|
+
// can slip between claim and consumption and replay the routing. A stale or
|
|
35
|
+
// superseded routing survives — only an exact match clears.
|
|
36
|
+
// what the dispatcher routed on.
|
|
37
|
+
const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_phase.length > 0) ? inputs.next_phase : null;
|
|
29
38
|
|
|
30
39
|
// Visual verdict protocol availability — the workflow parks for parent-run
|
|
31
40
|
// baseline capture ONLY when the protocol is fully shipped. The protocol
|
|
@@ -529,52 +538,22 @@ function verifyAppliedChanges(expected, applied) {
|
|
|
529
538
|
return { ok: true };
|
|
530
539
|
}
|
|
531
540
|
|
|
532
|
-
// Publish read-back request
|
|
533
|
-
//
|
|
534
|
-
//
|
|
535
|
-
//
|
|
536
|
-
//
|
|
537
|
-
//
|
|
538
|
-
//
|
|
539
|
-
//
|
|
540
|
-
//
|
|
541
|
-
//
|
|
542
|
-
//
|
|
543
|
-
//
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
// prove the read-back inspected the live build of THIS attempt — not a
|
|
549
|
-
// different build's output. Null/empty means the edit was accepted but
|
|
550
|
-
// never correlated to a builder run. Pure function of inputs — no I/O,
|
|
551
|
-
// no clock.
|
|
552
|
-
var buildIdLine = (typeof buildAgentId === "string" && buildAgentId.length > 0)
|
|
553
|
-
? "Expected builder build agent_id: " + buildAgentId + " (the artifact system's in-flight correlation ID for this publish attempt — not a durable post-completion identifier).\n"
|
|
554
|
-
: "No build agent_id was observed for this publish attempt (the edit was accepted but never correlated to a builder run) — say so explicitly in your report.\n";
|
|
555
|
-
return (
|
|
556
|
-
"Publish content read-back for task " + taskId + ", merge commit " + commit + ".\n" +
|
|
557
|
-
"The unified diff below was supposed to be applied to this artifact's source tree and deployed. Do NOT modify anything.\n" +
|
|
558
|
-
"Do NOT rely on the builder's applied-changes report — it is derived from this same diff, so it cannot confirm the content. Read the artifact's CURRENT source directly.\n" +
|
|
559
|
-
"\n" +
|
|
560
|
-
buildIdLine +
|
|
561
|
-
"Report the live build's agent_id as seen in artifact_status (or state explicitly that no build/agent_id is visible). If an expected agent_id is given above and the live one differs, say so exactly — the read-back may be inspecting a different build's output.\n" +
|
|
562
|
-
"\n" +
|
|
563
|
-
"UNIFIED DIFF (expected change):\n" +
|
|
564
|
-
"```diff\n" + diff + "\n```\n" +
|
|
565
|
-
"\n" +
|
|
566
|
-
"For each file in the diff:\n" +
|
|
567
|
-
"1. Read the file's CURRENT content in the artifact source tree.\n" +
|
|
568
|
-
"2. Quote the exact current text of the regions around the changed lines.\n" +
|
|
569
|
-
"3. For every added (+) line in the diff, state whether that exact line is PRESENT in the current source.\n" +
|
|
570
|
-
"4. For every removed (-) line in the diff, state whether that exact line is ABSENT from the current source.\n" +
|
|
571
|
-
"5. Report build/deploy health and the console error count.\n" +
|
|
572
|
-
"\n" +
|
|
573
|
-
"Return the per-file present/absent findings with the quoted observed lines. Do not modify anything.\n" +
|
|
574
|
-
"This read-back feeds the parent content-verification protocol (docs/publish-verification.md): the parent stamps provenance only when every added line is present and every removed line is absent."
|
|
575
|
-
);
|
|
576
|
-
}
|
|
577
|
-
|
|
541
|
+
// Publish read-back request (currently unavailable): the verbatim_request
|
|
542
|
+
// the parent protocol (docs/publish-verification.md) would hand to an
|
|
543
|
+
// independent read-back tool after the artifact build lands. artifact_inspect
|
|
544
|
+
// was removed by the platform (2026-09-14); artifact.inspect is malfunction
|
|
545
|
+
// diagnosis, not a substitute — so no agent-callable read-back tool exists
|
|
546
|
+
// and this request cannot currently be issued. Pure function — no I/O, no
|
|
547
|
+
// clock. The request carries the merged diff as the expected change and asks
|
|
548
|
+
// for an independent read of the artifact's actual source: for each file, the
|
|
549
|
+
// exact current text of the changed regions plus a per-line present/absent
|
|
550
|
+
// finding. Until a read-back path exists, the parent cannot independently
|
|
551
|
+
// confirm content and verification parks at "publish: verification-requested"
|
|
552
|
+
// (see docs/publish-verification.md). This preserves the circularity break
|
|
553
|
+
// that hollowed canary run 8 (2026-09-11): verifyAppliedChanges compares the
|
|
554
|
+
// builder's applied-report against the diff the report was derived from — a
|
|
555
|
+
// fabricated report passes by construction. Independent read-back cannot be
|
|
556
|
+
// fabricated from the diff; it must match the artifact's real content.
|
|
578
557
|
// Pre-publish base observation (diagnostic, 2026-09-12): instruction fragment
|
|
579
558
|
// for the builder's edit request, asking it to report the sha256 of each
|
|
580
559
|
// touched file's CURRENT content BEFORE applying the diff. Pure function —
|
|
@@ -668,6 +647,18 @@ function extractMarkerLines(workerText) {
|
|
|
668
647
|
return markers.join("\n");
|
|
669
648
|
}
|
|
670
649
|
|
|
650
|
+
// Already-merged idempotency (canary 2026-09-15, task 1d692d91): when the
|
|
651
|
+
// builder correctly makes no commit because the deliverable is already on
|
|
652
|
+
// main (a prior merge or hand-repair landed it), it declares
|
|
653
|
+
// `repo_diff: none (already-merged: <sha>)` naming the main commit that
|
|
654
|
+
// carries the work. The sha is hex-only (7-40 chars) so the workflow can
|
|
655
|
+
// interpolate it into the mechanical ancestor check without injection
|
|
656
|
+
// risk. Pure — pinned byte-identical across standard/bugfix/chore.
|
|
657
|
+
function extractAlreadyMerged(workerText) {
|
|
658
|
+
var m = /^repo_diff:\s*none\s*\(already-merged:\s*([0-9a-f]{7,40})\)/im.exec(workerText || "");
|
|
659
|
+
return m ? { sha: m[1].toLowerCase() } : { sha: null };
|
|
660
|
+
}
|
|
661
|
+
|
|
671
662
|
// Worktree confinement: the Build agent must declare the exact worktree
|
|
672
663
|
// path it built in on a `worktree:` marker line. The workflow compares it
|
|
673
664
|
// against WORKTREE_HINT mechanically (exact string match) — never by
|
|
@@ -845,6 +836,12 @@ let mapGateBounceCount = 0;
|
|
|
845
836
|
// rationalized a skip against explicit instruction text — text alone did not
|
|
846
837
|
// hold, so the decision now lives in workflow code, not agent judgment.
|
|
847
838
|
let releaseDecision = null; // { release: "yes"|"no", version_bump: "patch"|"minor"|"major"|null }
|
|
839
|
+
// Already-merged idempotency: the verified sha from the builder's
|
|
840
|
+
// `repo_diff: none (already-merged: <sha>)` declaration (null when the
|
|
841
|
+
// builder made commits or declared a runtime-state deliverable). The
|
|
842
|
+
// workflow verifies the sha is an ancestor of main at Build closeout;
|
|
843
|
+
// Review's no-diff branch reads this, never the builder's prose.
|
|
844
|
+
let alreadyMergedSha = null;
|
|
848
845
|
// Deterministic publish target — computed by the workflow (registry base +
|
|
849
846
|
// bumpVersion), never by the Publish agent.
|
|
850
847
|
let publishTarget = null; // { base, scope, target }
|
|
@@ -1084,7 +1081,7 @@ while (i < STEPS.length) {
|
|
|
1084
1081
|
const claimResult = await agent(
|
|
1085
1082
|
"Claim this task for the " + step.name + " step.\n" +
|
|
1086
1083
|
"Run in shell and return the stdout verbatim:\n" + crewCmd("update-task", firstClaimUpdateArgs) + "\n" +
|
|
1087
|
-
"Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started" }) + "\n" +
|
|
1084
|
+
"Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started", ...(NEXT_PHASE_ROUTED ? { expected_next_phase: NEXT_PHASE_ROUTED } : {}) }) + "\n" +
|
|
1088
1085
|
"If the claim response has claimed=true, then run in shell and return the stdout verbatim:\n" + crewCmd("clear-reservation", { task_id: taskId }) + "\n" +
|
|
1089
1086
|
"Do not interpret the claim response. It already contains an explicit \"claimed\" field — copy it verbatim.\n" +
|
|
1090
1087
|
"Return { claimed: <verbatim>, session_id: \"<...>\" }. If claimed is false there is no session_id; return { claimed: false, session_id: \"\" }.",
|
|
@@ -1260,7 +1257,7 @@ while (i < STEPS.length) {
|
|
|
1260
1257
|
var instructions = "";
|
|
1261
1258
|
|
|
1262
1259
|
if (step.name === "Triage") {
|
|
1263
|
-
instructions = "Validate the task,
|
|
1260
|
+
instructions = "Validate the task against the project's repo at " + REPO_PATH + " — that exact checkout, not any other copy of the project on disk. If you run git commands, cd " + REPO_PATH + " first.\nCheck clarity, note dependencies, confirm the chore workflow assignment.\nReport back in plain prose — what you found.\nEXPERIENTIAL FLAG: does this task change anything rendered and visible in the project's user-facing artifact (pages, components, styles, layout, copy, visual states)? If yes it is experiential and gets baseline captures. End your report with exactly one line on its own, lowercase, unrephrased: experiential: yes — or experiential: no. This line is machine-read.";
|
|
1264
1261
|
|
|
1265
1262
|
} else if (step.name === "Map") {
|
|
1266
1263
|
var mapGatePara = "";
|
|
@@ -1272,7 +1269,7 @@ while (i < STEPS.length) {
|
|
|
1272
1269
|
" If the baseline evidence is missing with no baseline:none recorded, do not write the spec — report 'baseline evidence missing — Map gate bounce required' and stop.\n" +
|
|
1273
1270
|
"Declare capture targets for the post-change visual capture: end your report with a line `capture_targets: <comma-separated views/controls this change affects>` (optional; falls back to the task description).";
|
|
1274
1271
|
}
|
|
1275
|
-
instructions = "Research options, pick the path, write a clear spec for the builder.\nThe builder will edit source files in a git worktree of the project at " + REPO_PATH + ".\nProject: " + PROJECT_DESC + "\nTo understand the current code, read source files directly using the read tool. Do NOT use artifact_inspect — it
|
|
1272
|
+
instructions = "Research options, pick the path, write a clear spec for the builder.\nThe builder will edit source files in a git worktree of the project at " + REPO_PATH + ".\nProject: " + PROJECT_DESC + "\nTo understand the current code, read source files directly using the read tool. Do NOT use artifact_inspect — it was removed by the platform (2026-09-14) and does not exist; do not substitute artifact.inspect (malfunction diagnosis, not an inspection tool).\nIdentify the exact files and changes needed. Be specific: file paths, what to add or change.\nReport back in plain prose — what you specified." + mapGatePara;
|
|
1276
1273
|
|
|
1277
1274
|
} else if (step.name === "Build") {
|
|
1278
1275
|
instructions = "STEP 1: Prepare your worktree.\n" +
|
|
@@ -1296,12 +1293,33 @@ while (i < STEPS.length) {
|
|
|
1296
1293
|
"cd " + WORKTREE_HINT + "\n" +
|
|
1297
1294
|
"git add -A\n" +
|
|
1298
1295
|
"git commit -m \"chore: " + safeTitle + "\"\n\n" +
|
|
1299
|
-
"If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of main and declare `repo_diff: none` in your report, naming the runtime-state deliverable. Otherwise commit your changes normally.\n\n" +
|
|
1296
|
+
"If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of main and declare `repo_diff: none` in your report, naming the runtime-state deliverable. If you verified the deliverable is already on main (a prior merge or hand-repair landed it — do NOT re-implement working code), make no commit and declare `repo_diff: none (already-merged: <sha>)` naming the main commit that carries the work; the workflow verifies the sha is an ancestor of main, and a false declaration fails the phase. Otherwise commit your changes normally.\n\n" +
|
|
1300
1297
|
(rejectionNotes ? "REWORK after rejection. Address:\n" + rejectionNotes + "\n\n" : "") +
|
|
1301
1298
|
"Report back in plain prose: what you built and the outcome." +
|
|
1302
1299
|
(PUBLISH_TYPE === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
|
|
1303
1300
|
|
|
1304
1301
|
} else if (step.name === "Review") {
|
|
1302
|
+
// Already-merged hydration: when this run did not execute Build itself
|
|
1303
|
+
// (dispatcher resume at Review after a platform death between phases),
|
|
1304
|
+
// recover the workflow-attested verification from the latest completed
|
|
1305
|
+
// Build session notes. The `already_merged_verified:` line was written
|
|
1306
|
+
// by the workflow after a mechanical ancestor check — it is trusted;
|
|
1307
|
+
// the builder's bare declaration never is. Absent the line, the
|
|
1308
|
+
// mechanical fact below reads "none declared" and Cass fails closed.
|
|
1309
|
+
if (!alreadyMergedSha) {
|
|
1310
|
+
var hydNotes = await agent(
|
|
1311
|
+
"Read the latest completed Build session notes for task " + taskId + ".\n" +
|
|
1312
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
|
|
1313
|
+
"In the returned sessions array, find the most recent session (by started_at) with task_id \"" + taskId + "\", step \"Build\", and status \"completed\". Return ONLY its notes field, verbatim, with no commentary.",
|
|
1314
|
+
{ key: "hydrate-already-merged" + (reworkCount > 0 ? "-r" + reworkCount : ""), label: "Hydrating already-merged verification" }
|
|
1315
|
+
);
|
|
1316
|
+
var hydStr = (typeof hydNotes === "string") ? hydNotes : JSON.stringify(hydNotes);
|
|
1317
|
+
var hvm = /already_merged_verified:\s*([0-9a-f]{7,40})/i.exec(hydStr);
|
|
1318
|
+
if (hvm) {
|
|
1319
|
+
alreadyMergedSha = hvm[1].toLowerCase();
|
|
1320
|
+
log("Hydrated already-merged verification from Build session notes: " + alreadyMergedSha);
|
|
1321
|
+
}
|
|
1322
|
+
}
|
|
1305
1323
|
instructions = "Review independently and cold. No prior context from the builder.\nDo NOT access the task dashboard, event log, or any comments. Your review is based solely on the spec and the code.\n\n" +
|
|
1306
1324
|
(mapperSpec ? "MAPPER'S SPEC (the builder was asked to implement exactly this):\n" + mapperSpec + "\n\n" : "Read the spec from the task description.\n\n") +
|
|
1307
1325
|
"Examine the code changes by running:\n" +
|
|
@@ -1311,7 +1329,7 @@ while (i < STEPS.length) {
|
|
|
1311
1329
|
WORKTREE_HINT + "/\n\n" +
|
|
1312
1330
|
"Check quality, correctness, spec compliance.\n" +
|
|
1313
1331
|
"Check that public-affecting changes have matching public doc updates (API.md or the published API contract). If the docs are missing or inaccurate, report what is stale, then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1314
|
-
"If the branch has no commits ahead of main (inspect shows an empty commit log), approve ONLY if the Build summary declares `repo_diff: none` with a plausible runtime-state deliverable (e.g. a cron created via the cron tool). Otherwise report 'no commits ahead of main and no repo_diff: none declaration — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1332
|
+
"If the branch has no commits ahead of main (inspect shows an empty commit log), approve ONLY if the Build summary declares `repo_diff: none` with (a) a plausible runtime-state deliverable (e.g. a cron created via the cron tool), or (b) an already-merged declaration `repo_diff: none (already-merged: <sha>)` AND the mechanical fact below confirms the sha verified. MECHANICAL FACT (computed by the workflow, never by the builder): already_merged sha = " + (alreadyMergedSha ? alreadyMergedSha + " (verified ancestor of main: YES)" : "none declared") + ". Otherwise report 'no commits ahead of main and no valid repo_diff: none declaration — the builder likely forgot to commit', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1315
1333
|
(PUBLISH_TYPE === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry, and versions are assigned at publish time — never in branches. Two checks:\n" +
|
|
1316
1334
|
"(a) The task branch must NOT have changed package.json's `version` field. Check: cd " + REPO_PATH + " && git diff main..." + TASK_BRANCH + " -- package.json. If the branch touched `version` in any way, report 'versions are assigned at publish time, never in branches — remove the version change' in your notes, then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1317
1335
|
"(b) The accepted Build report declares: " + releaseDecisionText() + ". " +
|
|
@@ -1325,7 +1343,7 @@ while (i < STEPS.length) {
|
|
|
1325
1343
|
"Run: "+ LIFECYCLE_ENV + "WORKFLOW_RUN_ID=" + lockHolder + " integrate " + taskId + " \"merge: chore: " + safeTitle + "\"\n\n" +
|
|
1326
1344
|
"Read the output:\n" +
|
|
1327
1345
|
"- If it contains MERGED, integration succeeded. Report the merged commit hash.\n" +
|
|
1328
|
-
"- If it contains MERGED_EMPTY, the branch had no commits ahead of main (a runtime-state deliverable
|
|
1346
|
+
"- If it contains MERGED_EMPTY, the branch had no commits ahead of main (declared by Build as repo_diff: none — either a runtime-state deliverable or an already-merged sha the workflow verified). Integration succeeded vacuously: the merge lock was NOT taken and there is no new commit. Report 'merged empty: no repo changes — deliverable was runtime state or already on main', then end your report with exactly this line: VERDICT: PASS. SKIP STEP 2 (push): there is no new commit to push.\n" +
|
|
1329
1347
|
"- If it contains LOCK_HELD, another task holds the merge lock (mid Integrate/Publish) and the 10-minute bounded backoff is exhausted. Report 'merge lock held after bounded backoff', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
1330
1348
|
"- If it contains CONFLICT, the plain merge failed — the merge was aborted, main is clean, and your task still holds the merge lock. Do NOT fail yet. Resolve it:\n" +
|
|
1331
1349
|
"RESOLUTION:\n" +
|
|
@@ -1388,11 +1406,12 @@ while (i < STEPS.length) {
|
|
|
1388
1406
|
// (fail-closed). There is deliberately NO workflow-side provenance
|
|
1389
1407
|
// stamp: the builder's applied-report is circular (canary run 8,
|
|
1390
1408
|
// 2026-09-11), so the stamp moved to the parent — after the build
|
|
1391
|
-
// lands, the workflow
|
|
1392
|
-
//
|
|
1393
|
-
//
|
|
1394
|
-
//
|
|
1395
|
-
//
|
|
1409
|
+
// lands, the workflow records the session completed and parks with
|
|
1410
|
+
// "publish: verification-requested". The parent owns verification
|
|
1411
|
+
// (docs/publish-verification.md); the independent read-back step is
|
|
1412
|
+
// currently unavailable (no agent-callable read-back tool exists —
|
|
1413
|
+
// artifact_inspect was removed by the platform 2026-09-14).
|
|
1414
|
+
// Chore has no QA: the parent's verification is the final gate.
|
|
1396
1415
|
var artifactPublish = null;
|
|
1397
1416
|
var publishLockRefreshed = false;
|
|
1398
1417
|
var publishSkippedNoLock = false;
|
|
@@ -1440,14 +1459,45 @@ while (i < STEPS.length) {
|
|
|
1440
1459
|
// changes — no prose claim to trust. If the artifact tool namespace
|
|
1441
1460
|
// is missing from this child it reports honestly and the workflow
|
|
1442
1461
|
// retries once with a fresh key (bounded); anything else parks.
|
|
1462
|
+
// Publish diff base (2026-09-14, task 0c53af4e): the carried diff is
|
|
1463
|
+
// BASE..HEAD where BASE is the previously-stamped provenance
|
|
1464
|
+
// source_commit — NOT HEAD^1. A push-time reconcile merge puts the
|
|
1465
|
+
// task's own changes behind an intermediate merge, so HEAD^1..HEAD
|
|
1466
|
+
// silently drops the task's fix while the artifact builds without
|
|
1467
|
+
// it. The stamped base is the artifact's actual content; BASE..HEAD
|
|
1468
|
+
// is the complete unpublished delta. Empty tree only for a genuine
|
|
1469
|
+
// first publish (no provenance stamped yet).
|
|
1470
|
+
var EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
|
|
1471
|
+
var provResult = await agent(
|
|
1472
|
+
crewCmd("get-provenance", {}) + "\n" +
|
|
1473
|
+
"Return JSON { \"provenance\": <the CLI's provenance object, or null when nothing is stamped> } and nothing else. Do not interpret it.",
|
|
1474
|
+
{ key: attemptKey("publish-provenance-base-" + taskId, reworkCount), label: "Reading stamped publish base",
|
|
1475
|
+
schema: { type: "object", properties: { provenance: { type: ["object", "null"] } }, required: ["provenance"] } }
|
|
1476
|
+
);
|
|
1477
|
+
var publishBase = (provResult.provenance && provResult.provenance.source_commit) || "";
|
|
1478
|
+
publishBase = String(publishBase).trim();
|
|
1479
|
+
if (!publishBase) {
|
|
1480
|
+
publishBase = EMPTY_TREE_SHA;
|
|
1481
|
+
log("Publish base for task " + taskId + ": no provenance stamped yet — using empty tree (first publish)");
|
|
1482
|
+
} else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
|
|
1483
|
+
return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
|
|
1484
|
+
}
|
|
1443
1485
|
var diffResult = await agent(
|
|
1444
|
-
"Run: cd " + REPO_PATH + " &&
|
|
1445
|
-
"
|
|
1446
|
-
|
|
1447
|
-
|
|
1486
|
+
"Run: cd " + REPO_PATH + " && BASE='" + publishBase + "' && HEAD=$(git rev-parse HEAD) && " +
|
|
1487
|
+
"if [ \"$BASE\" = '" + EMPTY_TREE_SHA + "' ]; then ANCESTOR=yes; else git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no; fi && " +
|
|
1488
|
+
"echo '---COMMIT---' && echo \"$HEAD\" && echo '---BASE---' && echo \"$BASE\" && echo '---ANCESTOR---' && echo \"$ANCESTOR\" && " +
|
|
1489
|
+
"if [ \"$ANCESTOR\" = yes ]; then echo '---DIFF---' && git diff \"$BASE\" \"$HEAD\" && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r \"$BASE\" \"$HEAD\"; fi\n" +
|
|
1490
|
+
"Return JSON { \"commit\": \"<HEAD trimmed>\", \"base\": \"<BASE trimmed>\", \"ancestor\": \"<yes|no>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
|
|
1491
|
+
{ key: attemptKey("publish-artifact-diff-" + taskId, reworkCount), label: "Computing publish diff from stamped base",
|
|
1492
|
+
schema: { type: "object", properties: { commit: { type: "string" }, base: { type: "string" }, ancestor: { type: "string" }, diff: { type: "string" }, files: { type: "string" } }, required: ["commit", "base", "ancestor", "diff"] } }
|
|
1448
1493
|
);
|
|
1494
|
+
if ((diffResult.ancestor || "").trim() !== "yes") {
|
|
1495
|
+
return await parkTask("Publish base " + publishBase.slice(0, 12) + " is not an ancestor of HEAD " + (diffResult.commit || "").trim().slice(0, 12) + " — the stamped provenance does not lead to the integrated commit. Human attention needed.");
|
|
1496
|
+
}
|
|
1497
|
+
if ((diffResult.base || "").trim() !== publishBase) {
|
|
1498
|
+
return await parkTask("Publish diff base mismatch: agent reported '" + (diffResult.base || "").trim().slice(0, 12) + "' but the stamped base is '" + publishBase.slice(0, 12) + "'. Human attention needed.");
|
|
1499
|
+
}
|
|
1449
1500
|
var mergeCommitForPublish = (diffResult.commit || "").trim();
|
|
1450
|
-
var mergeParentForPublish = (diffResult.parent || "").trim();
|
|
1451
1501
|
var mergeDiff = diffResult.diff || "";
|
|
1452
1502
|
if (!mergeDiff.trim()) {
|
|
1453
1503
|
return await parkTask("Publish diff is empty for commit " + (mergeCommitForPublish || "unknown") + " — a merge lock was held but there is no change to carry. Human attention needed.");
|
|
@@ -1479,12 +1529,16 @@ while (i < STEPS.length) {
|
|
|
1479
1529
|
// but never parks. The observation tells us what the publish actually
|
|
1480
1530
|
// reads, so the subsequent fix can require the right base.
|
|
1481
1531
|
var expectedBaseHashes = {};
|
|
1482
|
-
|
|
1532
|
+
if (publishBase === EMPTY_TREE_SHA) {
|
|
1533
|
+
// First publish: every file in the diff is new to the artifact.
|
|
1534
|
+
expectedChanges.forEach(function (f) { expectedBaseHashes[f.path] = "NEW-FILE"; });
|
|
1535
|
+
log("Publish expected base hashes for task " + taskId + ": empty tree (first publish) — all " + expectedChanges.length + " file(s) new");
|
|
1536
|
+
} else try {
|
|
1483
1537
|
// Shell-quote helper (no regex-with-quote: the test parser does not
|
|
1484
1538
|
// understand regex literals containing quotes).
|
|
1485
1539
|
var sq = function(s) { return "'" + String(s).split("'").join("'\\''") + "'"; };
|
|
1486
1540
|
var baseHashResult = await agent(
|
|
1487
|
-
"Run: cd " + REPO_PATH + " && parent=" + sq(
|
|
1541
|
+
"Run: cd " + REPO_PATH + " && parent=" + sq(publishBase) + " && for f in " + expectedChanges.map(function(f) { return sq(f.path); }).join(" ") + "; do printf '%s:' \"$f\"; git show \"$parent:$f\" 2>/dev/null | sha256sum | cut -d' ' -f1; done\n" +
|
|
1488
1542
|
"Return JSON { \"hashes\": \"<newline-separated <path>:<sha256> lines, empty hash means the file is new in this diff>\" } and nothing else.",
|
|
1489
1543
|
{ key: attemptKey("publish-base-hashes-" + taskId, reworkCount), label: "Computing expected base content hashes",
|
|
1490
1544
|
schema: { type: "object", properties: { hashes: { type: "string" } }, required: ["hashes"] } }
|
|
@@ -1493,7 +1547,7 @@ while (i < STEPS.length) {
|
|
|
1493
1547
|
var m = /^([^:]+):([0-9a-f]*)$/.exec(line.trim());
|
|
1494
1548
|
if (m) expectedBaseHashes[m[1]] = m[2] || "NEW-FILE";
|
|
1495
1549
|
});
|
|
1496
|
-
log("Publish expected base hashes for task " + taskId + " (
|
|
1550
|
+
log("Publish expected base hashes for task " + taskId + " (stamped base " + publishBase.slice(0, 12) + "): " + JSON.stringify(expectedBaseHashes));
|
|
1497
1551
|
} catch (e) {
|
|
1498
1552
|
log("Publish expected base hash computation failed for task " + taskId + " (non-fatal, observation degraded): " + (e && e.message ? e.message : e));
|
|
1499
1553
|
}
|
|
@@ -1551,6 +1605,7 @@ while (i < STEPS.length) {
|
|
|
1551
1605
|
required: ["edit_started", "build_agent_id", "applied"] };
|
|
1552
1606
|
var rebuildTrigger = null;
|
|
1553
1607
|
var rebuildReportMissing = false; // true if the edit went through but the agent returned no applied report (structured-output failure) — the smoke-check is skipped; the parent's independent read-back is the verification
|
|
1608
|
+
var rebuildEvidenceNote = null; // human-readable evidence line for the ledger when the edit is confirmed via fallback evidence (in-flight poll or durable audit dir) rather than the trigger's own report
|
|
1554
1609
|
// The trigger key of the attempt that last ran, for the publish ledger.
|
|
1555
1610
|
// Minted once here (not re-minted per use site) so the ledger always
|
|
1556
1611
|
// records the exact key that was issued — and so a re-minted duplicate
|
|
@@ -1569,6 +1624,32 @@ while (i < STEPS.length) {
|
|
|
1569
1624
|
// (2026-09-12, task 23ca8f3f): computed once the trigger outcome is
|
|
1570
1625
|
// known, logged loudly, never a park.
|
|
1571
1626
|
var publishAppliedObservation = null; // "match" | "mismatch: <reason>" | "missing-report" — observation only, never a park
|
|
1627
|
+
// Durable-evidence snapshot (2026-09-14): the structured-output
|
|
1628
|
+
// fallback below only observes IN-FLIGHT builds. A build that
|
|
1629
|
+
// finished before the poll leaves no in-flight trace — but the
|
|
1630
|
+
// platform's audit harness leaves a durable one:
|
|
1631
|
+
// ~/workspace/ts-spaces/<slug>/audits/<timestamp>-<id>/ per
|
|
1632
|
+
// completed build. Snapshot the listing BEFORE the trigger so the
|
|
1633
|
+
// fallback can diff before/after: a directory appearing during the
|
|
1634
|
+
// trigger window is positive evidence the edit went through and
|
|
1635
|
+
// the build completed. Best-effort and non-gating: if the snapshot
|
|
1636
|
+
// fails, the durable check is skipped and the fallback behaves as
|
|
1637
|
+
// before. No wall-clock in-script (deterministic replay) — the
|
|
1638
|
+
// comparison is a pure before/after set diff.
|
|
1639
|
+
var auditDirsBeforeTrigger = [];
|
|
1640
|
+
try {
|
|
1641
|
+
var auditBefore = await agent(
|
|
1642
|
+
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
|
|
1643
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1644
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
|
|
1645
|
+
{ key: attemptKey("publish-audit-before-" + taskId, reworkCount), label: "Snapshotting audit dirs before rebuild trigger",
|
|
1646
|
+
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1647
|
+
);
|
|
1648
|
+
auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1649
|
+
log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
|
|
1650
|
+
} catch (auditBeforeErr) {
|
|
1651
|
+
log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
|
|
1652
|
+
}
|
|
1572
1653
|
try {
|
|
1573
1654
|
rebuildTrigger = await agent(rebuildPrompt,
|
|
1574
1655
|
{ key: rebuildAttemptKey, label: "Triggering artifact rebuild", schema: rebuildSchema });
|
|
@@ -1627,7 +1708,46 @@ while (i < STEPS.length) {
|
|
|
1627
1708
|
rebuildTrigger = { edit_started: true, error: "", applied: null };
|
|
1628
1709
|
rebuildReportMissing = true;
|
|
1629
1710
|
rebuildAgentId = acceptedAgentId;
|
|
1711
|
+
rebuildEvidenceNote = "edit confirmed via build-state poll after structured-output failure (build " + acceptedAgentId + "); builder applied-report missing";
|
|
1630
1712
|
} else {
|
|
1713
|
+
// Durable completion check (2026-09-14): the in-flight poll
|
|
1714
|
+
// above only sees RUNNING builds. Attempt 7 (2026-09-14) proved
|
|
1715
|
+
// the gap: the trigger child applied the edit, the build ran
|
|
1716
|
+
// and completed — the platform's audit harness captured it
|
|
1717
|
+
// mid-window — then the child failed to return JSON. The
|
|
1718
|
+
// fallback poll saw no in-flight build, so a successful publish
|
|
1719
|
+
// parked as "unknown". Diff the audit-dir listing against the
|
|
1720
|
+
// pre-trigger snapshot: a timestamped directory that appeared
|
|
1721
|
+
// during the trigger window is positive evidence the edit went
|
|
1722
|
+
// through and the build completed. This never re-issues the
|
|
1723
|
+
// edit and never stamps provenance — it only routes to the
|
|
1724
|
+
// parent's independent content read-back, which remains the
|
|
1725
|
+
// real verification.
|
|
1726
|
+
var newAuditDirs = [];
|
|
1727
|
+
try {
|
|
1728
|
+
var auditAfter = await agent(
|
|
1729
|
+
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
|
|
1730
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1731
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
|
|
1732
|
+
{ key: attemptKey("publish-audit-after-" + taskId, reworkCount), label: "Re-listing audit dirs after trigger failure",
|
|
1733
|
+
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1734
|
+
);
|
|
1735
|
+
var auditDirsAfterTrigger = String((auditAfter && auditAfter.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1736
|
+
// Only timestamped build dirs count — the "latest" symlink
|
|
1737
|
+
// and anything else are not builds.
|
|
1738
|
+
newAuditDirs = auditDirsAfterTrigger.filter(function (d) {
|
|
1739
|
+
return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
|
|
1740
|
+
});
|
|
1741
|
+
} catch (auditAfterErr) {
|
|
1742
|
+
log("Publish audit-dir re-list after trigger failure failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
|
|
1743
|
+
}
|
|
1744
|
+
if (newAuditDirs.length > 0) {
|
|
1745
|
+
log("Publish rebuild trigger: new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and the build completed despite the structured-output failure. Skipping applied-report smoke-check; parent read-back is the verification.");
|
|
1746
|
+
rebuildTrigger = { edit_started: true, error: "", applied: null };
|
|
1747
|
+
rebuildReportMissing = true;
|
|
1748
|
+
rebuildAgentId = null;
|
|
1749
|
+
rebuildEvidenceNote = "edit confirmed via durable audit evidence after structured-output failure (new audit dir " + newAuditDirs[0] + "); builder applied-report missing";
|
|
1750
|
+
} else {
|
|
1631
1751
|
// No build observed — but that proves nothing (a fast-completing
|
|
1632
1752
|
// build can finish between polls, or the check itself failed). The
|
|
1633
1753
|
// outcome is UNKNOWN. No retry: re-issuing the edit here duplicated
|
|
@@ -1643,6 +1763,7 @@ while (i < STEPS.length) {
|
|
|
1643
1763
|
detail: "structured-output failure on rebuild trigger; build-state poll saw no build (or the check itself failed); edit may have been accepted as pending_init"
|
|
1644
1764
|
}, reworkCount);
|
|
1645
1765
|
return await parkTask("Publish outcome unknown: the rebuild trigger's child did not return JSON, and the follow-up build-state poll could not observe a build for slug " + PUBLISH_SLUG + ". The edit may have been accepted as pending_init, so no retry was issued — a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion before re-driving Publish. Fail-closed.");
|
|
1766
|
+
}
|
|
1646
1767
|
}
|
|
1647
1768
|
}
|
|
1648
1769
|
if (!rebuildReportMissing && !rebuildTrigger.edit_started && rebuildTrigger.error === "artifact_tools missing after load") {
|
|
@@ -1712,7 +1833,7 @@ while (i < STEPS.length) {
|
|
|
1712
1833
|
applied_report: publishAppliedObservation,
|
|
1713
1834
|
outcome: "submitted",
|
|
1714
1835
|
detail: rebuildReportMissing
|
|
1715
|
-
? "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing"
|
|
1836
|
+
? (rebuildEvidenceNote || "edit confirmed via build-state poll after structured-output failure (build " + (rebuildAgentId || "agent_id unknown") + "); builder applied-report missing")
|
|
1716
1837
|
: "edit accepted; builder applied-report received"
|
|
1717
1838
|
}, reworkCount);
|
|
1718
1839
|
} else if (rebuildTrigger) {
|
|
@@ -1769,6 +1890,15 @@ while (i < STEPS.length) {
|
|
|
1769
1890
|
// lock was lost: stop the run and park the task — never continue to
|
|
1770
1891
|
// a provenance stamp or version assignment without holding the lock.
|
|
1771
1892
|
var buildPoll = null;
|
|
1893
|
+
// STEP 1b poll-signal accumulators (2026-09-15, task aadeccc3):
|
|
1894
|
+
// the durable audit-dir fallback below needs the poll's own
|
|
1895
|
+
// observations, not just its final verdict — whether our build was
|
|
1896
|
+
// ever seen, whether a stranger's build was ever in flight, and
|
|
1897
|
+
// what the last check observed. OR-ed across all three chunks so
|
|
1898
|
+
// a signal seen in any chunk survives the chunk boundary.
|
|
1899
|
+
var pollSawOurBuild = false;
|
|
1900
|
+
var pollSawStranger = false;
|
|
1901
|
+
var lastObservedAgentId = null;
|
|
1772
1902
|
for (var chunk = 1; chunk <= 3; chunk++) {
|
|
1773
1903
|
if (chunk > 1) {
|
|
1774
1904
|
var refreshPoll = await agent(
|
|
@@ -1791,36 +1921,150 @@ while (i < STEPS.length) {
|
|
|
1791
1921
|
: attemptKey("publish-artifact-poll-" + taskId + "-c" + chunk, reworkCount);
|
|
1792
1922
|
buildPoll = await agent(
|
|
1793
1923
|
"First call tool_search.load_tool_namespace with paths [\"artifact\"]. Then poll artifact_status for slug \"" + PUBLISH_SLUG + "\" \u2014 for OUR build only, the one whose agent_id is \"" + rebuildAgentId + "\" (the receipt captured when the edit was accepted; the agent_id is the artifact system's in-flight build correlation ID, stable across polls while the build runs). Check every 30 seconds, up to 7 checks (3.5 minutes max). On each check, read the raw build object:\n" +
|
|
1794
|
-
"
|
|
1924
|
+
"On every check, record whether you have positively OBSERVED our build: a running build whose agent_id equals \"" + rebuildAgentId + "\", or a completed-build record whose agent_id equals \"" + rebuildAgentId + "\" (if the tool surfaces one \u2014 match it mechanically, never assume).\n" +
|
|
1925
|
+
"- If no build is running (build is null) and you have NOT observed our build: our build's completion is UNPROVEN. Absence of a running build is not evidence our build ran. Do NOT report done.\n" +
|
|
1926
|
+
"- If no build is running (build is null) and you previously observed our build running: our build finished. Stop and report done.\n" +
|
|
1795
1927
|
"- If the running build's agent_id equals \"" + rebuildAgentId + "\": still ours \u2014 keep waiting.\n" +
|
|
1796
|
-
"- If the running build's agent_id is present but DIFFERENT:
|
|
1797
|
-
"Return JSON { \"build_done\": <true
|
|
1928
|
+
"- If the running build's agent_id is present but DIFFERENT: that is a stranger's build. Do NOT attribute its completion to our attempt and do NOT wait on it \u2014 keep checking within budget; if the budget expires without observing our build, report done=false. Record it in saw_stranger regardless of what else you observe.\n" +
|
|
1929
|
+
"Return JSON { \"build_done\": <true ONLY when you positively observed our build and it is no longer running, false otherwise>, \"saw_our_build\": <true if you observed our build at any check, false if never>, \"saw_stranger\": true if at ANY check a running build had an agent_id different from ours (\"" + rebuildAgentId + "\"), false otherwise, \"status\": \"<final status or timeout note>\", \"observed_agent_id\": \"<the agent_id seen on the last check, or null when no build was running>\" } and nothing else.",
|
|
1798
1930
|
{ key: pollKey, label: "Waiting for artifact build to complete (chunk " + chunk + " of 3)",
|
|
1799
|
-
schema: { type: "object", properties: { build_done: { type: "boolean" }, status: { type: "string" }, observed_agent_id: { type: ["string", "null"] } }, required: ["build_done"] },
|
|
1931
|
+
schema: { type: "object", properties: { build_done: { type: "boolean" }, saw_our_build: { type: "boolean" }, saw_stranger: { type: "boolean" }, status: { type: "string" }, observed_agent_id: { type: ["string", "null"] } }, required: ["build_done"] },
|
|
1800
1932
|
timeoutMs: 270000 }
|
|
1801
1933
|
);
|
|
1934
|
+
pollSawOurBuild = pollSawOurBuild || (buildPoll && buildPoll.saw_our_build === true);
|
|
1935
|
+
pollSawStranger = pollSawStranger || (buildPoll && buildPoll.saw_stranger === true);
|
|
1936
|
+
lastObservedAgentId = (buildPoll && buildPoll.observed_agent_id) || null;
|
|
1802
1937
|
if (buildPoll && buildPoll.build_done) { break; }
|
|
1803
1938
|
}
|
|
1804
1939
|
if (!buildPoll || !buildPoll.build_done) {
|
|
1805
1940
|
buildPoll = { build_done: false, status: (buildPoll && buildPoll.status) || "build still running after the 10.5-minute bounded poll" };
|
|
1806
1941
|
}
|
|
1807
|
-
if (buildPoll.build_done) {
|
|
1942
|
+
if (buildPoll.build_done && pollSawOurBuild) {
|
|
1808
1943
|
// STEP 1c (mechanical): NO provenance stamp here. Canary run 8
|
|
1809
1944
|
// (2026-09-11) proved the stamp cannot certify content: the
|
|
1810
1945
|
// builder's applied-report is derived from the carried diff, so
|
|
1811
1946
|
// verifyAppliedChanges above is circular — a fabricated report
|
|
1812
1947
|
// passes by construction, and every phase went green on a hollow
|
|
1813
|
-
// build. The stamp moves to the parent (docs/publish-verification.md)
|
|
1814
|
-
//
|
|
1815
|
-
//
|
|
1816
|
-
//
|
|
1817
|
-
//
|
|
1818
|
-
//
|
|
1948
|
+
// build. The stamp moves to the parent (docs/publish-verification.md);
|
|
1949
|
+
// the independent read-back step is currently unavailable (no
|
|
1950
|
+
// agent-callable read-back tool exists — artifact_inspect was
|
|
1951
|
+
// removed by the platform 2026-09-14), so the parent cannot
|
|
1952
|
+
// confirm content and the task parks for verification.
|
|
1953
|
+
// Chore has no QA: the parent's verification is the final gate.
|
|
1819
1954
|
publishBuildLanded = true;
|
|
1820
1955
|
artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
|
|
1821
1956
|
log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
|
|
1822
1957
|
} else {
|
|
1823
|
-
|
|
1958
|
+
// STEP 1b durable audit-dir fallback (2026-09-15, task aadeccc3):
|
|
1959
|
+
// the poll above only observes IN-FLIGHT builds. A build that
|
|
1960
|
+
// finished between the receipt capture and the poll's first check
|
|
1961
|
+
// leaves no in-flight trace — but the platform's audit harness
|
|
1962
|
+
// leaves a durable one (~/workspace/ts-spaces/<slug>/audits/
|
|
1963
|
+
// <timestamp>-<id>/ per completed build). Diff the audit-dir
|
|
1964
|
+
// listing against the pre-trigger snapshot: a timestamped dir
|
|
1965
|
+
// that appeared during the attempt window is evidence a build
|
|
1966
|
+
// completed. Attribution is by window, not by build identity:
|
|
1967
|
+
// the poll's saw_stranger signal only catches stranger builds in
|
|
1968
|
+
// flight AT a check — a stranger that finished entirely inside
|
|
1969
|
+
// the window is indistinguishable, so any observed stranger
|
|
1970
|
+
// blocks attribution and the outcome stays unknown. This never
|
|
1971
|
+
// re-issues the edit and never stamps provenance — ok=true only
|
|
1972
|
+
// routes to the parent's independent content read-back, which
|
|
1973
|
+
// remains the real verification.
|
|
1974
|
+
//
|
|
1975
|
+
// The poll end-state is read from the poll's own observations,
|
|
1976
|
+
// not from build_done alone: a build in flight at the last check
|
|
1977
|
+
// means the budget was shorter than the latency (or the build is
|
|
1978
|
+
// stuck) — NOT that no build ever started; nothing observed at
|
|
1979
|
+
// any check is the never-started signal.
|
|
1980
|
+
var pollEndState = lastObservedAgentId ? "build-still-running-at-poll-end"
|
|
1981
|
+
: (pollSawOurBuild ? "our-build-observed-then-unconfirmed" : "no-build-observed-in-window");
|
|
1982
|
+
var strangerObserved = pollSawStranger;
|
|
1983
|
+
var newAuditDirsAfterPoll = [];
|
|
1984
|
+
try {
|
|
1985
|
+
var auditAfterPoll = await agent(
|
|
1986
|
+
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
|
|
1987
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1988
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
|
|
1989
|
+
{ key: attemptKey("publish-audit-after-poll-" + taskId, reworkCount), label: "Re-listing audit dirs after build poll",
|
|
1990
|
+
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1991
|
+
);
|
|
1992
|
+
var auditDirsAfterPollList = String((auditAfterPoll && auditAfterPoll.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1993
|
+
newAuditDirsAfterPoll = auditDirsAfterPollList.filter(function (d) {
|
|
1994
|
+
return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
|
|
1995
|
+
});
|
|
1996
|
+
log("Publish audit-dir re-list after build poll for task " + taskId + ": " + newAuditDirsAfterPoll.length + " new timestamped dir(s)");
|
|
1997
|
+
} catch (auditAfterPollErr) {
|
|
1998
|
+
log("Publish audit-dir re-list after build poll failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterPollErr && auditAfterPollErr.message ? auditAfterPollErr.message : auditAfterPollErr));
|
|
1999
|
+
}
|
|
2000
|
+
// auditReportOk: pure tri-state read of a report.json body —
|
|
2001
|
+
// true (build ok), false (build failed), null (missing or
|
|
2002
|
+
// unreadable — not evidence either way). The child returns the
|
|
2003
|
+
// raw body verbatim; interpretation lives here, never in prose.
|
|
2004
|
+
var auditReportOk = function (raw) {
|
|
2005
|
+
if (typeof raw !== "string") return null;
|
|
2006
|
+
var trimmed = raw.trim();
|
|
2007
|
+
if (trimmed === "" || trimmed === "MISSING") return null;
|
|
2008
|
+
var parsed;
|
|
2009
|
+
try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
|
|
2010
|
+
if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
|
|
2011
|
+
return null;
|
|
2012
|
+
};
|
|
2013
|
+
var auditOkAfterPoll = null;
|
|
2014
|
+
var newestAuditDirAfterPoll = null;
|
|
2015
|
+
if (newAuditDirsAfterPoll.length > 0 && !strangerObserved) {
|
|
2016
|
+
newAuditDirsAfterPoll.sort();
|
|
2017
|
+
newestAuditDirAfterPoll = newAuditDirsAfterPoll[newAuditDirsAfterPoll.length - 1];
|
|
2018
|
+
try {
|
|
2019
|
+
var auditOkRead = await agent(
|
|
2020
|
+
"Read the build report for artifact slug \"" + PUBLISH_SLUG + "\", audit dir \"" + newestAuditDirAfterPoll + "\" (verbatim read, never interpreted, never a gate).\n" +
|
|
2021
|
+
"Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestAuditDirAfterPoll + "/report.json 2>/dev/null || echo MISSING\n" +
|
|
2022
|
+
"Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
|
|
2023
|
+
{ key: attemptKey("publish-audit-ok-after-poll-" + taskId, reworkCount), label: "Reading build report after build poll",
|
|
2024
|
+
schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
|
|
2025
|
+
);
|
|
2026
|
+
auditOkAfterPoll = auditReportOk(auditOkRead && auditOkRead.raw);
|
|
2027
|
+
} catch (auditOkReadErr) {
|
|
2028
|
+
log("Publish build-report read after build poll failed for task " + taskId + " (non-fatal, treated as unknown): " + (auditOkReadErr && auditOkReadErr.message ? auditOkReadErr.message : auditOkReadErr));
|
|
2029
|
+
auditOkAfterPoll = null;
|
|
2030
|
+
}
|
|
2031
|
+
}
|
|
2032
|
+
if (auditOkAfterPoll === true) {
|
|
2033
|
+
publishBuildLanded = true;
|
|
2034
|
+
artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
|
|
2035
|
+
log("Publish build landed for task " + taskId + " via durable audit evidence — provenance stamp deferred to parent content verification");
|
|
2036
|
+
await recordPublishLedger({
|
|
2037
|
+
commit: mergeCommitForPublish,
|
|
2038
|
+
attempt: rebuildAttemptKey,
|
|
2039
|
+
agent_id: rebuildAgentId,
|
|
2040
|
+
applied_report: publishAppliedObservation,
|
|
2041
|
+
outcome: "submitted",
|
|
2042
|
+
detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
|
|
2043
|
+
}, reworkCount);
|
|
2044
|
+
} else if (auditOkAfterPoll === false) {
|
|
2045
|
+
publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestAuditDirAfterPoll + ", report ok=false). Explicit negative evidence: a build ran and failed (attribution by window, not by build identity — no stranger build was observed in flight during the poll). The publish did not land — provenance was not stamped. Fail-closed.";
|
|
2046
|
+
await recordPublishLedger({
|
|
2047
|
+
commit: mergeCommitForPublish,
|
|
2048
|
+
attempt: rebuildAttemptKey,
|
|
2049
|
+
agent_id: rebuildAgentId,
|
|
2050
|
+
applied_report: publishAppliedObservation,
|
|
2051
|
+
outcome: "failed",
|
|
2052
|
+
detail: "a build ran and failed (attribution by window, not by build identity): audit dir " + newestAuditDirAfterPoll + " report ok=false; no stranger build observed in flight during the poll"
|
|
2053
|
+
}, reworkCount);
|
|
2054
|
+
} else {
|
|
2055
|
+
var unattributableReason = strangerObserved ? "stranger-build-observed-during-poll"
|
|
2056
|
+
: (pollEndState === "build-still-running-at-poll-end" ? "build-still-running-at-poll-end"
|
|
2057
|
+
: (newAuditDirsAfterPoll.length === 0 ? "no-new-audit-dir-in-window" : "audit-report-unreadable-or-missing"));
|
|
2058
|
+
publishFailure = "Artifact build completion unproven (fail-closed, no provenance stamped): unattributable_reason=" + unattributableReason + "; poll_end_state=" + pollEndState + "; " + "saw_our_build=" + pollSawOurBuild + "; new_audit_dirs=" + newAuditDirsAfterPoll.length + ". Attribution is by window, not by build identity. The publish may or may not have landed. Fail-closed.";
|
|
2059
|
+
await recordPublishLedger({
|
|
2060
|
+
commit: mergeCommitForPublish,
|
|
2061
|
+
attempt: rebuildAttemptKey,
|
|
2062
|
+
agent_id: rebuildAgentId,
|
|
2063
|
+
applied_report: publishAppliedObservation,
|
|
2064
|
+
outcome: "unknown",
|
|
2065
|
+
detail: "durable audit-dir fallback could not attribute a completed build to this attempt (unattributable_reason=" + unattributableReason + ", poll_end_state=" + pollEndState + ")"
|
|
2066
|
+
}, reworkCount);
|
|
2067
|
+
}
|
|
1824
2068
|
}
|
|
1825
2069
|
} else {
|
|
1826
2070
|
publishFailure = "Artifact rebuild trigger failed: " + (rebuildTrigger.error || "artifact_edit not accepted") + ". The publish did not land.";
|
|
@@ -2071,6 +2315,52 @@ while (i < STEPS.length) {
|
|
|
2071
2315
|
};
|
|
2072
2316
|
}
|
|
2073
2317
|
log("Build worktree confinement passed: " + wt.path);
|
|
2318
|
+
|
|
2319
|
+
// Already-merged idempotency: a `repo_diff: none (already-merged:
|
|
2320
|
+
// <sha>)` declaration is verified mechanically — <sha> must resolve
|
|
2321
|
+
// and be an ancestor of main in the configured repo. A fabricated or
|
|
2322
|
+
// mistaken declaration fails the phase here (the dispatcher retries
|
|
2323
|
+
// Build under its consecutive-failure cap); a verified declaration is
|
|
2324
|
+
// recorded in alreadyMergedSha for Review's no-diff branch. Without
|
|
2325
|
+
// this guard, Build correctly doing nothing left Review with no
|
|
2326
|
+
// mechanical way to accept an empty diff, and Cass rejected for "no
|
|
2327
|
+
// commits ahead of main — the builder likely forgot to commit" while
|
|
2328
|
+
// the deliverable sat on main (canary 2026-09-15, task 1d692d91).
|
|
2329
|
+
// The sha is hex-only by construction (extractAlreadyMerged), so
|
|
2330
|
+
// interpolating it into the shell command cannot inject.
|
|
2331
|
+
var am = extractAlreadyMerged(workerText);
|
|
2332
|
+
if (am.sha) {
|
|
2333
|
+
var amCheck = await agent(
|
|
2334
|
+
"Verify the builder's already-merged declaration.\n" +
|
|
2335
|
+
"Run in shell and return the stdout verbatim:\n" +
|
|
2336
|
+
"cd " + REPO_PATH + " && git rev-parse --verify --quiet " + am.sha + " >/dev/null && git merge-base --is-ancestor " + am.sha + " main && echo ALREADY_MERGED_YES || echo ALREADY_MERGED_NO",
|
|
2337
|
+
{ key: "verify-already-merged" + (reworkCount > 0 ? "-r" + reworkCount : ""), label: "Verifying already-merged declaration" }
|
|
2338
|
+
);
|
|
2339
|
+
var amOut = (typeof amCheck === "string") ? amCheck : JSON.stringify(amCheck);
|
|
2340
|
+
if (!/ALREADY_MERGED_YES/.test(amOut)) {
|
|
2341
|
+
log("Build already-merged declaration failed verification — " + am.sha + " is not an ancestor of main — marking failed for retry");
|
|
2342
|
+
await agent(
|
|
2343
|
+
"Record already-merged verification failure.\n" +
|
|
2344
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
|
|
2345
|
+
task_id: taskId,
|
|
2346
|
+
session: { id: activeSessionId, task_id: taskId, identity: step.identity, step: step.name, status: "failed",
|
|
2347
|
+
notes: "Build declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of main in the configured repo. The declaration is fabricated or mistaken; the work is not on main. Phase failed for retry" },
|
|
2348
|
+
event: { task_id: taskId, type: "failed", message: "Build already-merged declaration failed verification — " + am.sha + " not an ancestor of main, phase failed, dispatcher will retry" }
|
|
2349
|
+
}),
|
|
2350
|
+
{ key: "record-already-merged-fail-" + step.name, label: "Recording already-merged verification failure" }
|
|
2351
|
+
);
|
|
2352
|
+
return {
|
|
2353
|
+
__hatchWorkflowControl: "blocked",
|
|
2354
|
+
result: {
|
|
2355
|
+
blocked_reason: "Build already-merged declaration failed verification",
|
|
2356
|
+
message: "The builder declared repo_diff: none (already-merged: " + am.sha + ") but " + am.sha + " is not an ancestor of main. The work is not on main; the phase is marked failed and the dispatcher will retry Build.",
|
|
2357
|
+
task_id: taskId
|
|
2358
|
+
}
|
|
2359
|
+
};
|
|
2360
|
+
}
|
|
2361
|
+
alreadyMergedSha = am.sha;
|
|
2362
|
+
log("Build already-merged declaration verified: " + am.sha + " is an ancestor of main");
|
|
2363
|
+
}
|
|
2074
2364
|
}
|
|
2075
2365
|
|
|
2076
2366
|
// Deterministic closeout: no formatter agent. The verdict is mechanical
|
|
@@ -2198,48 +2488,22 @@ while (i < STEPS.length) {
|
|
|
2198
2488
|
// it to HEAD: that verifies the stamp, not the content. Canary run 8
|
|
2199
2489
|
// (2026-09-11) passed it with a hollow build — the stamp was honest, the
|
|
2200
2490
|
// artifact was stale, all eight phases green. The stamp now moves to the
|
|
2201
|
-
// parent
|
|
2202
|
-
//
|
|
2203
|
-
//
|
|
2204
|
-
//
|
|
2205
|
-
// there instead of passing silently here.
|
|
2491
|
+
// parent (docs/publish-verification.md); the independent read-back step
|
|
2492
|
+
// is currently unavailable (no agent-callable read-back tool exists —
|
|
2493
|
+
// artifact_inspect was removed by the platform 2026-09-14), so the parent
|
|
2494
|
+
// cannot confirm content and the task parks for verification.
|
|
2206
2495
|
// Skip-aware (park 2026-09-11): an empty-diff Integrate takes no merge
|
|
2207
2496
|
// lock, and the deterministic publish path skips rebuild/stamp entirely —
|
|
2208
2497
|
// there is no new content to verify, so verification is vacuous.
|
|
2209
2498
|
// publishSkippedNoLock is workflow-computed state from the explicit
|
|
2210
2499
|
// lock-status read in STEP 0, not agent prose.
|
|
2211
|
-
|
|
2212
|
-
|
|
2213
|
-
|
|
2214
|
-
|
|
2215
|
-
|
|
2216
|
-
|
|
2217
|
-
|
|
2218
|
-
try {
|
|
2219
|
-
var inspectResult = await agent(
|
|
2220
|
-
ARTIFACT_LOAD_PREAMBLE +
|
|
2221
|
-
"Call artifact_inspect with slug \"" + PUBLISH_SLUG + "\", repair_authorized false, and verbatim_request exactly as follows:\n" +
|
|
2222
|
-
"<<<READBACK_REQUEST\n" + buildPublishReadbackRequest(taskId, mergeCommitForPublish, mergeDiff, rebuildAgentId) + "\nREADBACK_REQUEST\n" +
|
|
2223
|
-
"If artifact_inspect is still not available after the load, do NOT improvise — return { \"triggered\": false, \"inspection_id\": \"\", \"error\": \"artifact_tools missing after load\" } and nothing else.\n" +
|
|
2224
|
-
"Return JSON { \"triggered\": <true if the inspection started, false otherwise>, \"inspection_id\": \"<the inspection id, or empty string>\", \"error\": \"<details or empty string>\" } and nothing else.",
|
|
2225
|
-
{ key: attemptKey("publish-verify-inspect-" + taskId, reworkCount), label: "Triggering publish content read-back",
|
|
2226
|
-
schema: { type: "object", properties: { triggered: { type: "boolean" }, inspection_id: { type: "string" }, error: { type: "string" } }, required: ["triggered"] } }
|
|
2227
|
-
);
|
|
2228
|
-
publishVerifyInspect.triggered = !!(inspectResult && inspectResult.triggered);
|
|
2229
|
-
publishVerifyInspect.inspection_id = (inspectResult && inspectResult.inspection_id) || "";
|
|
2230
|
-
publishVerifyInspect.error = (inspectResult && inspectResult.error) || "";
|
|
2231
|
-
if (publishVerifyInspect.triggered) {
|
|
2232
|
-
log("Publish content read-back inspection triggered for task " + taskId + ": " + publishVerifyInspect.inspection_id);
|
|
2233
|
-
} else {
|
|
2234
|
-
log("Publish content read-back inspect trigger failed for task " + taskId + ": " + (publishVerifyInspect.error || "not started") + " — the park below asks the parent to trigger it manually");
|
|
2235
|
-
}
|
|
2236
|
-
} catch (e) {
|
|
2237
|
-
publishVerifyInspect.error = (e && e.message ? e.message : String(e)).slice(0, 200);
|
|
2238
|
-
log("Publish content read-back inspect trigger threw for task " + taskId + ": " + publishVerifyInspect.error + " — the park below asks the parent to trigger it manually");
|
|
2239
|
-
}
|
|
2240
|
-
} // end: !publishSkippedNoLock && publishBuildLanded — a skipped or failed publish has nothing to verify
|
|
2241
|
-
}
|
|
2242
|
-
|
|
2500
|
+
// The parent (tick worker) triggers the ONE read-back inspection it can
|
|
2501
|
+
// actually receive (async results go to the root agent, never into a
|
|
2502
|
+
// workflow run — a workflow-side trigger would be an orphan). The workflow
|
|
2503
|
+
// only parks; the parent's scan builds the request deterministically via
|
|
2504
|
+
// lib/build-readback-request.js and ferries the inspection.
|
|
2505
|
+
// publishBuildLanded and publishSkippedNoLock are workflow-computed state;
|
|
2506
|
+
// a skipped or failed publish has nothing to verify.
|
|
2243
2507
|
// Session notes. Machine-readable marker lines are extracted from the full
|
|
2244
2508
|
// worker report and appended AFTER the slice so a long report can never
|
|
2245
2509
|
// amputate them; later phases (Review reading repo_diff:, QA backstop
|
|
@@ -2257,6 +2521,15 @@ while (i < STEPS.length) {
|
|
|
2257
2521
|
} else {
|
|
2258
2522
|
summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
|
|
2259
2523
|
}
|
|
2524
|
+
// Already-merged attestation: when the Build gate verified the builder's
|
|
2525
|
+
// already-merged declaration, the workflow records its own marker line in
|
|
2526
|
+
// the session notes (like the builder markers above, it is appended after
|
|
2527
|
+
// the slice so it can never be amputated). A later run resumed at Review
|
|
2528
|
+
// hydrates alreadyMergedSha from this workflow-attested line — never from
|
|
2529
|
+
// the builder's declaration alone.
|
|
2530
|
+
if (step.name === "Build" && alreadyMergedSha) {
|
|
2531
|
+
summary += "\nalready_merged_verified: " + alreadyMergedSha;
|
|
2532
|
+
}
|
|
2260
2533
|
|
|
2261
2534
|
// Capture mapper's spec for Build and Review
|
|
2262
2535
|
if (step.name === "Map" && passed) {
|
|
@@ -2323,10 +2596,7 @@ while (i < STEPS.length) {
|
|
|
2323
2596
|
if (passed && step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG && !publishSkippedNoLock && publishBuildLanded) {
|
|
2324
2597
|
return await parkTask("publish: verification-requested " + mergeCommitForPublish +
|
|
2325
2598
|
" (build " + (rebuildAgentId || "agent_id unobserved") + ")" +
|
|
2326
|
-
" — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md"
|
|
2327
|
-
(publishVerifyInspect.triggered
|
|
2328
|
-
? " (content read-back inspection " + publishVerifyInspect.inspection_id + " already triggered)."
|
|
2329
|
-
: " (read-back inspect trigger failed: " + (publishVerifyInspect.error || "not started") + " — parent: trigger artifact_inspect manually)."));
|
|
2599
|
+
" — artifact build landed, post-deploy finalized, provenance NOT stamped. Parent: run docs/publish-verification.md.");
|
|
2330
2600
|
}
|
|
2331
2601
|
|
|
2332
2602
|
i++;
|