muse-crew 0.13.2 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/decisions/AGENTS.md +94 -0
- package/docs/decisions/publish-path.md +1280 -0
- package/docs/decisions/qa-reproduce.md +486 -0
- package/docs/decisions/workflow-core.md +234 -0
- package/docs/publish-unknown-recovery.md +113 -0
- package/docs/visual-verdict.md +6 -7
- package/lib/AGENTS.md +3 -1
- package/lib/classify-publish-absence.js +451 -0
- package/lib/compose-evidence-caption.js +36 -5
- package/lib/crew-api.js +584 -22
- package/lib/crew-release.sh +51 -9
- package/lib/publish-content.js +154 -0
- package/lib/retry-publish.js +370 -0
- package/lib/verify-publish.js +101 -84
- package/package.json +2 -2
- package/seed/cron-body-template.md +57 -2
- package/workflows/AGENTS.md +3 -1
- package/workflows/bugfix.js +301 -592
- package/workflows/chore.js +292 -526
- package/workflows/standard.js +294 -550
- package/workflows/upgrade.js +1 -1
package/workflows/standard.js
CHANGED
|
@@ -27,26 +27,11 @@ const startStepIndex = inputs.start_step_index || 0;
|
|
|
27
27
|
// resolution back via updatetask in the self-claim below.
|
|
28
28
|
const RESOLVED_WORKFLOW = inputs.resolved_workflow || null;
|
|
29
29
|
const WORKFLOW_WAS_NULL = inputs.workflow_was_null === true;
|
|
30
|
-
//
|
|
31
|
-
// routes this run via an explicit recover-task redirect. The value is
|
|
32
|
-
// consumed (cleared) atomically by the successful self-claim below:
|
|
33
|
-
// claim-task takes expected_next_phase and clears the matching next_phase in
|
|
34
|
-
// the same transaction as the winning session insert, so no platform death
|
|
35
|
-
// can slip between claim and consumption and replay the routing. A stale or
|
|
36
|
-
// superseded routing survives — only an exact match clears.
|
|
37
|
-
// what the dispatcher routed on.
|
|
30
|
+
// See docs/decisions/workflow-core.md#oneshot-recovery: the dispatcher sets input for one-shot recovery routing.
|
|
38
31
|
const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_phase.length > 0) ? inputs.next_phase : null;
|
|
39
32
|
const CLAIM_WORKFLOW_PERSIST = (WORKFLOW_WAS_NULL && RESOLVED_WORKFLOW) ? ", \"workflow\": \"" + RESOLVED_WORKFLOW + "\"" : "";
|
|
40
33
|
|
|
41
|
-
//
|
|
42
|
-
// baseline capture and visual verdict ONLY when the protocol is fully
|
|
43
|
-
// shipped. The protocol requires docs/visual-verdict.md in the release AND
|
|
44
|
-
// the parent-side capture tooling (task b309a97d, "QA owns the visual
|
|
45
|
-
// verdict"). Until both exist, the parks would deadlock waiting for a
|
|
46
|
-
// parent who cannot fulfill them.
|
|
47
|
-
// Effective value for this run, resolved by the dispatcher from the
|
|
48
|
-
// project's visual_protocol setting (null=inherits crew default=off).
|
|
49
|
-
// Manual launches without the arg default to off (previous behavior).
|
|
34
|
+
// See docs/decisions/qa-reproduce.md#visual-protocol-avail: the workflow parks if the visual protocol is unavailable.
|
|
50
35
|
var VISUAL_PROTOCOL_AVAILABLE = inputs.visual_protocol === true;
|
|
51
36
|
|
|
52
37
|
// crewHome is required — the dispatcher always passes it (crew-dispatch.js
|
|
@@ -93,15 +78,7 @@ const COMPUTE_DIFF_SRC = crewHome + "/current/lib/compute-publish-diff.js";
|
|
|
93
78
|
const COMPUTE_DIFF = RUN_LIB + "/compute-publish-diff.js";
|
|
94
79
|
const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
|
|
95
80
|
const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
|
|
96
|
-
//
|
|
97
|
-
// by workflow code from the verbatim listing, never from agent prose.
|
|
98
|
-
// COMPUTE_DIFF is the deterministic publish-diff computer (room #14,
|
|
99
|
-
// 2026-09-17): the diff is computed by this script, never ferried as an
|
|
100
|
-
// agent JSON string. Pinned like the other publish-critical modules so a
|
|
101
|
-
// mid-run release swap cannot change it under the workflow.
|
|
102
|
-
// CLASSIFY_SURFACE is the surface classifier (room #15, 2026-09-18):
|
|
103
|
-
// crew-api.js statically imports it, so the pin must carry it — a pin
|
|
104
|
-
// without it kills every claim with ERR_MODULE_NOT_FOUND.
|
|
81
|
+
// See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
|
|
105
82
|
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE].map(function (p) { return p.split("/").pop(); });
|
|
106
83
|
|
|
107
84
|
// Project config — passed by dispatcher, falls back to dashboard defaults
|
|
@@ -134,14 +111,7 @@ const WORKTREE_HINT = REPO_PATH + "/.worktrees/" + taskId;
|
|
|
134
111
|
const WORKTREE_PRESERVED_HINT = ".worktrees/" + taskId;
|
|
135
112
|
|
|
136
113
|
const PUBLISH_TYPE = projectConfig.deploy_type || "";
|
|
137
|
-
//
|
|
138
|
-
// web UI Hazel drives with the see-act browser loop) | 'terminal' (a CLI
|
|
139
|
-
// Hazel drives herself, keeping attempt-scoped transcripts) | null
|
|
140
|
-
// (unclassified — no experiential QA). environment_type is the canonical
|
|
141
|
-
// UX-surface axis; deploy_type names the deployment target, but
|
|
142
|
-
// deploy_type === "artifact" remains a legacy artifact-surface signal so
|
|
143
|
-
// pre-field projects keep today's experiential QA (the migration does not
|
|
144
|
-
// backfill the column).
|
|
114
|
+
// See docs/decisions/qa-reproduce.md#surface-routing: the user-facing surface determines experiential QA routing.
|
|
145
115
|
const ENV_TYPE = projectConfig.environment_type || null;
|
|
146
116
|
// Surface resolution: artifact wins on contradictory config (the deployed
|
|
147
117
|
// artifact is what users see). Unclassified surface => Capture skips, QA
|
|
@@ -156,12 +126,7 @@ const SURFACE_TRIAGE_DESC = SURFACE_ARTIFACT
|
|
|
156
126
|
: SURFACE_TERMINAL
|
|
157
127
|
? "This project's user-facing surface is terminal: a command-line interface."
|
|
158
128
|
: "This project's user-facing surface is unclassified (environment_type not set): judge by what a user would directly observe.";
|
|
159
|
-
//
|
|
160
|
-
// mechanically — every phase prompt reads UX_DOCTRINE_PATH, never a
|
|
161
|
-
// hardcoded filename. Canonical map: lib/ux-doctrine.js (mirrored here as a
|
|
162
|
-
// one-liner because the workflow runtime's relative-import support is
|
|
163
|
-
// unverified; tests pin the mirror). Null on unclassified surfaces: no
|
|
164
|
-
// shared page, and prompts say so instead of naming the wrong one.
|
|
129
|
+
// See docs/decisions/qa-reproduce.md#ux-doctrine-page: the shared UX bar for this run's surface.
|
|
165
130
|
const UX_DOCTRINE_PAGE = SURFACE_TERMINAL ? "terminal-ux.md" : (SURFACE_ARTIFACT ? "artifact-ux.md" : null);
|
|
166
131
|
const UX_DOCTRINE_PATH = UX_DOCTRINE_PAGE ? crewHome + "/current/docs/" + UX_DOCTRINE_PAGE : null;
|
|
167
132
|
const PUBLISH_SLUG = projectConfig.deploy_slug || "";
|
|
@@ -172,21 +137,7 @@ if (!taskId) {
|
|
|
172
137
|
throw new Error("task_id is required in args");
|
|
173
138
|
}
|
|
174
139
|
|
|
175
|
-
//
|
|
176
|
-
// transport envelope {"status": "ok", "result": "<prose report>"} with no
|
|
177
|
-
// schema, so the workflow receives the report as a plain string. There is no
|
|
178
|
-
// {"report"} wrapper: that invented shape invited agents to improvise sibling
|
|
179
|
-
// keys (notably "status"), which the runtime duck-types as its own envelope
|
|
180
|
-
// and fatally misparses. The envelope is the runtime's own documented shape
|
|
181
|
-
// — not a demand for machine-structured reasoning.
|
|
182
|
-
// The verdict is extracted mechanically by extractVerdict below — never by an
|
|
183
|
-
// agent. The summary is the worker's report truncated. The release decision
|
|
184
|
-
// comes from extractReleaseDecision. No formatter agent: it added a failure
|
|
185
|
-
// mode while contributing nothing the workflow doesn't compute itself.
|
|
186
|
-
// Steps whose passed=false drives a control-flow branch (rework bounce,
|
|
187
|
-
// block) declare their verdict explicitly on a VERDICT: line. The verdict is
|
|
188
|
-
// extracted DETERMINISTICALLY by workflow code (extractVerdict) — never by
|
|
189
|
-
// an agent. Missing, malformed, or contradictory lines fail the phase (never silently pass).
|
|
140
|
+
// See docs/decisions/workflow-core.md#closeout-envelope: the work agent returns the runtime's native envelope; verdict extracted mechanically.
|
|
190
141
|
const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
|
|
191
142
|
function extractVerdict(workerText) {
|
|
192
143
|
// The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
|
|
@@ -210,16 +161,7 @@ function extractVerdict(workerText) {
|
|
|
210
161
|
if (uniq.length !== 1) return { ok: false, count: matches.length };
|
|
211
162
|
return { ok: true, passed: last.value === "PASS" };
|
|
212
163
|
}
|
|
213
|
-
//
|
|
214
|
-
// extractVerdict is not failed immediately. Stochastic verdict-line
|
|
215
|
-
// non-compliance (the agent did the work but omitted or garbled the VERDICT
|
|
216
|
-
// line) gets up to two bounded follow-up agent() calls whose only job is to
|
|
217
|
-
// read the preserved report and emit exactly one VERDICT line. The verdict
|
|
218
|
-
// is still extracted mechanically by extractVerdict — the re-ask agent
|
|
219
|
-
// transcribes, never decides the phase outcome. Each attempt uses a fresh
|
|
220
|
-
// stable-key suffix so a cached failure can never replay deterministically.
|
|
221
|
-
// Exhaustion keeps the existing fail-closed behavior. This is structure, not
|
|
222
|
-
// prompt hardening: no instruction text was stern-ified to get here.
|
|
164
|
+
// See docs/decisions/workflow-core.md#verdict-reask: a report that fails extractVerdict gets bounded re-ask calls.
|
|
223
165
|
function verdictReaskKey(stepName, reworkSuffix, attempt) {
|
|
224
166
|
return "verdict-reask-" + stepName + reworkSuffix + "-a" + attempt;
|
|
225
167
|
}
|
|
@@ -277,12 +219,7 @@ function workRetryKey(stepName, reworkSuffix, attempt) {
|
|
|
277
219
|
function attemptKey(base, reworkCount) {
|
|
278
220
|
return base + (reworkCount > 0 ? "-r" + reworkCount : "");
|
|
279
221
|
}
|
|
280
|
-
//
|
|
281
|
-
// the verbatim `ls -1` listing so WORKFLOW CODE asserts the six pinned
|
|
282
|
-
// basenames; the agent cannot self-certify. (The pin step was the one place
|
|
283
|
-
// the workflows trusted agent prose: task 24be1cd6 walked to Publish on an
|
|
284
|
-
// empty pin dir.) Byte-identical across standard/bugfix/chore — pinned by
|
|
285
|
-
// tests/pin-location.test.js.
|
|
222
|
+
// See docs/decisions/qa-reproduce.md#pin-lifecycle: snapshot the lifecycle scripts in the pin.
|
|
286
223
|
function pinLifecycle(key) {
|
|
287
224
|
return agent(
|
|
288
225
|
"Snapshot lifecycle scripts for version pinning.\n" +
|
|
@@ -490,24 +427,7 @@ function hydrateReleaseDecision(rec) {
|
|
|
490
427
|
// fabricated from the diff;
|
|
491
428
|
// it must match the artifact's real content.
|
|
492
429
|
|
|
493
|
-
//
|
|
494
|
-
// attempt is recorded append-only at $CREW_HOME/.publish-ledger/<slug>.jsonl
|
|
495
|
-
// on persistent disk (NOT /tmp). The ledger is the correlation record for
|
|
496
|
-
// publish attempts whose outcome is UNKNOWN. When the rebuild trigger's
|
|
497
|
-
// child returns prose instead of JSON (structured-output failure), the edit
|
|
498
|
-
// may already have been accepted as pending_init — and artifact_status
|
|
499
|
-
// cannot see pending_init (diagnostic canary 2026-09-12: an edit accepted
|
|
500
|
-
// as pending_init was immediately followed by an all-false status check,
|
|
501
|
-
// and the old retry issued a DUPLICATE edit). "No build visible" is NOT
|
|
502
|
-
// evidence the edit did not go through, so the workflow never blind-retries
|
|
503
|
-
// on an unknown outcome: it records the attempt and parks fail-closed. A
|
|
504
|
-
// human or a later run correlates the accepted edit via the ledger (commit
|
|
505
|
-
// hash + attempt key + the artifact build's agent_id when one was observed)
|
|
506
|
-
// instead of guessing from a blind status poll.
|
|
507
|
-
// Best-effort observability: a failed write is logged loudly but never
|
|
508
|
-
// throws — the caller's park/proceed decision never depends on the ledger.
|
|
509
|
-
// Byte-identical across standard/bugfix/chore — pinned by
|
|
510
|
-
// tests/publish-ledger.test.js.
|
|
430
|
+
// See docs/decisions/publish-path.md#publish-attempt-ledger: every trigger outcome is recorded in the durable ledger.
|
|
511
431
|
async function recordPublishLedger(entry, rework) {
|
|
512
432
|
try {
|
|
513
433
|
var ledgerDir = crewHome + "/.publish-ledger";
|
|
@@ -520,6 +440,7 @@ async function recordPublishLedger(entry, rework) {
|
|
|
520
440
|
attempt: entry.attempt || null,
|
|
521
441
|
agent_id: entry.agent_id || null,
|
|
522
442
|
applied_report: entry.applied_report || null,
|
|
443
|
+
manifest_before: entry.manifest_before || null,
|
|
523
444
|
outcome: entry.outcome,
|
|
524
445
|
detail: entry.detail || ""
|
|
525
446
|
});
|
|
@@ -557,45 +478,19 @@ function extractMarkerLines(workerText) {
|
|
|
557
478
|
return markers.join("\n");
|
|
558
479
|
}
|
|
559
480
|
|
|
560
|
-
//
|
|
561
|
-
// builder correctly makes no commit because the deliverable is already on
|
|
562
|
-
// main (a prior merge or hand-repair landed it), it declares
|
|
563
|
-
// `repo_diff: none (already-merged: <sha>)` naming the main commit that
|
|
564
|
-
// carries the work. Room #16 blocker 11 (2026-09-18): the line anchor
|
|
565
|
-
// missed Wren's mid-paragraph declaration, and the persisted notes truncated
|
|
566
|
-
// the tail — so the anchor is gone and a sha followed by `)`, whitespace, or
|
|
567
|
-
// end-of-string (truncation) is accepted. The sha is hex-only (7-40 chars)
|
|
568
|
-
// so the workflow can interpolate it into the mechanical ancestor check
|
|
569
|
-
// without injection risk; a over-long hex run never matches (the lookahead
|
|
570
|
-
// fails on the extra hex char). Pure — pinned byte-identical across
|
|
571
|
-
// standard/bugfix/chore.
|
|
481
|
+
// See docs/decisions/publish-path.md#already-merged-idem2: idempotency for already-merged tasks.
|
|
572
482
|
function extractAlreadyMerged(workerText) {
|
|
573
483
|
var m = /repo_diff:\s*none\s*\(already-merged:\s*([0-9a-f]{7,40})(?=[\s)]|$)/i.exec(workerText || "");
|
|
574
484
|
return m ? { sha: m[1].toLowerCase() } : { sha: null };
|
|
575
485
|
}
|
|
576
486
|
|
|
577
|
-
//
|
|
578
|
-
// trigger child ends its turn with `ARTIFACT_EDIT_REFUSED: <text>` when
|
|
579
|
-
// artifact_edit explicitly refuses the edit (e.g. the artifact does not
|
|
580
|
-
// exist). A refusal is conclusive negative evidence — the edit provably did
|
|
581
|
-
// NOT go through — distinct from an unconsumed trigger return (unknown).
|
|
582
|
-
// Pure — pinned byte-identical across standard/bugfix/chore.
|
|
583
|
-
// The signal must be the ENTIRE trimmed turn output (not a line within prose):
|
|
584
|
-
// the trigger child is instructed to end its turn with exactly this line and
|
|
585
|
-
// nothing else. A confused child quoting the instructions back in prose must
|
|
586
|
-
// NOT produce a conclusive negative — that degrades to unknown (fail-closed).
|
|
487
|
+
// See docs/decisions/publish-path.md#refusal-signal: the refusal signal must be the ENTIRE trimmed turn output.
|
|
587
488
|
function extractRefusal(workerText) {
|
|
588
489
|
var m = /^ARTIFACT_EDIT_REFUSED:\s*(.+?)\s*$/.exec(String(workerText || "").trim());
|
|
589
490
|
return m ? m[1].slice(0, 300) : null;
|
|
590
491
|
}
|
|
591
492
|
|
|
592
|
-
//
|
|
593
|
-
// path it built in on a `worktree:` marker line. The workflow compares it
|
|
594
|
-
// against WORKTREE_HINT mechanically (exact string match) — never by
|
|
595
|
-
// reading agent prose. This closes the hole where a builder whose prepare
|
|
596
|
-
// failed freelanced into a different checkout (canary, 2026-09-11): the
|
|
597
|
-
// honest-but-confused case fails here, and a fabricated path is caught one
|
|
598
|
-
// phase later when Review's inspect finds no commits in the configured repo.
|
|
493
|
+
// See docs/decisions/qa-reproduce.md#worktree-confinement: the Build agent must declare its worktree.
|
|
599
494
|
function extractWorktree(workerText) {
|
|
600
495
|
var lines = (workerText || "").split("\n");
|
|
601
496
|
var found = null;
|
|
@@ -625,26 +520,7 @@ function extractExperiential(workerText) {
|
|
|
625
520
|
if (!r) return null;
|
|
626
521
|
return r[1].toLowerCase() === "yes";
|
|
627
522
|
}
|
|
628
|
-
|
|
629
|
-
// Deterministic visual-capture frame. kind: "baseline" | "postchange".
|
|
630
|
-
// This string IS the capture script: fixed viewport matrix, scroll
|
|
631
|
-
// positions, and interaction states — the inspection agent executes it
|
|
632
|
-
// verbatim, nothing is improvised. Task-specific targets fill the slots.
|
|
633
|
-
var title = String(taskTitle || "").replace(/"/g, "'").slice(0, 120);
|
|
634
|
-
var targets = String(captureTargets || "").trim() ||
|
|
635
|
-
String(taskDescription || "").replace(/"/g, "'").slice(0, 300);
|
|
636
|
-
return "VISUAL CAPTURE — " + kind.toUpperCase() + " — task: " + title + ". " +
|
|
637
|
-
"Target views/controls: " + targets + ". " +
|
|
638
|
-
"For EACH target, capture exactly: " +
|
|
639
|
-
"(1) desktop 1440x900, full view, scrolled to top; " +
|
|
640
|
-
"(2) desktop 1440x900, scrolled so the target is vertically centered; " +
|
|
641
|
-
"(3) mobile 390x844, scrolled so the target is vertically centered; " +
|
|
642
|
-
"(4) desktop 1440x900, hover state on the target control; " +
|
|
643
|
-
"(5) desktop 1440x900, keyboard-focus state on the target control; " +
|
|
644
|
-
"(6) desktop 1440x900, active/pressed state if the target is a button or control. " +
|
|
645
|
-
"Also record: console error count, the ARIA tree of the target region, any horizontal overflow. " +
|
|
646
|
-
"Name captures " + kind + "-<n>-<viewport>-<state>. Return the captures, not a summary.";
|
|
647
|
-
}
|
|
523
|
+
|
|
648
524
|
// Experiential flag resolution: the task is experiential when Sage's Triage
|
|
649
525
|
// report ends with the machine-read marker "experiential: yes". The flag is
|
|
650
526
|
// opt-in — a missing or garbled line degrades to "unknown", which callers
|
|
@@ -790,23 +666,9 @@ function releaseDecisionText() {
|
|
|
790
666
|
if (!releaseDecision) return "no machine-readable release decision from the Build report";
|
|
791
667
|
return "release: " + releaseDecision.release + (releaseDecision.version_bump ? ", version_bump: " + releaseDecision.version_bump : " (no version_bump line)");
|
|
792
668
|
}
|
|
793
|
-
//
|
|
794
|
-
//
|
|
795
|
-
//
|
|
796
|
-
// instead. Parking is one atomic dashboard action (parktask): the parked
|
|
797
|
-
// state and the explanatory note land in one transaction, never half.
|
|
798
|
-
// The dispatcher skips parked tasks; a human moving parked→todo
|
|
799
|
-
// mechanically resets the retry counters. Returns the workflow result
|
|
800
|
-
// envelope the launcher sees. If the park call itself fails, the run
|
|
801
|
-
// reports "failed" (retryable) so the next tick re-attempts the park —
|
|
802
|
-
// a lost park is never reported as parked.
|
|
803
|
-
// Terminal cleanup: the run's last act at every park/fail boundary. A run
|
|
804
|
-
// that parks or fails must not leak its worktree, branch, or merge lock.
|
|
805
|
-
// The lifecycle's terminal-cleanup releases the lock unconditionally and
|
|
806
|
-
// reclaims the worktree+branch ONLY when the task branch is fully merged
|
|
807
|
-
// into main (then it is redundant); unmerged work is preserved for the
|
|
808
|
-
// human by design. Fire-and-forget with one bounded retry — the merge-lock
|
|
809
|
-
// lease expiry and the orphan sweep are the backstop for a dead transport.
|
|
669
|
+
// Terminal cleanup: the run's last act — release the lock unconditionally;
|
|
670
|
+
// reclaim worktree+branch only when fully merged (unmerged work is preserved
|
|
671
|
+
// for the human by design). See `docs/decisions/workflow-core.md#park-contract`.
|
|
810
672
|
async function terminalCleanup() {
|
|
811
673
|
for (var attempt = 1; attempt <= 2; attempt++) {
|
|
812
674
|
try {
|
|
@@ -1057,14 +919,7 @@ while (i < STEPS.length) {
|
|
|
1057
919
|
lockHolder = taskId + "/" + activeSessionId;
|
|
1058
920
|
}
|
|
1059
921
|
|
|
1060
|
-
//
|
|
1061
|
-
// Hazel's QA capture pass runs right after Triage, before Map, for tasks
|
|
1062
|
-
// Sage flagged experiential. The capture itself is parent-driven (the
|
|
1063
|
-
// inspection handoff arrives at the root agent, outside this script), so
|
|
1064
|
-
// when no baseline evidence is recorded yet the script logs a note event
|
|
1065
|
-
// and parks with the exact parent protocol + resume path. Never fails the
|
|
1066
|
-
// task over missing evidence: after two requests, baseline:none is
|
|
1067
|
-
// recorded and final QA judges on the rubric alone.
|
|
922
|
+
// See docs/decisions/qa-reproduce.md#capture-baseline: baseline evidence for experiential tasks is captured before work begins.
|
|
1068
923
|
if (step.name === "Capture") {
|
|
1069
924
|
var capExp = await resolveExperiential();
|
|
1070
925
|
var bounceSuffix = (mapGateBounceCount > 0 ? "-g" + mapGateBounceCount : "");
|
|
@@ -1111,12 +966,7 @@ while (i < STEPS.length) {
|
|
|
1111
966
|
continue;
|
|
1112
967
|
}
|
|
1113
968
|
var capStatus = await baselineStatus();
|
|
1114
|
-
//
|
|
1115
|
-
// note is only durable while the protocol is unavailable. When
|
|
1116
|
-
// VISUAL_PROTOCOL_AVAILABLE is true, that old decision no longer
|
|
1117
|
-
// stands — fall through to the request path for a fresh capture
|
|
1118
|
-
// attempt. Exact-string trim comparison against the workflow's own
|
|
1119
|
-
// written message (explicit state, never English matching).
|
|
969
|
+
// See docs/decisions/qa-reproduce.md#stale-decision-guard: a stale baseline decision parks fail-closed.
|
|
1120
970
|
var baselineLatestMessage = ("baseline: " + capStatus.baseline_kind + capStatus.baseline_refs).trim();
|
|
1121
971
|
var baselineStale = VISUAL_PROTOCOL_AVAILABLE && baselineLatestMessage === "baseline: none (visual protocol unavailable)";
|
|
1122
972
|
if (capStatus.baseline_found && !baselineStale) {
|
|
@@ -1277,17 +1127,7 @@ while (i < STEPS.length) {
|
|
|
1277
1127
|
(PUBLISH_TYPE === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
|
|
1278
1128
|
|
|
1279
1129
|
} else if (step.name === "Review") {
|
|
1280
|
-
//
|
|
1281
|
-
// (dispatcher resume at Review after a platform death between phases),
|
|
1282
|
-
// recover the workflow-verified sha. Room #16 blocker 11: the structured
|
|
1283
|
-
// session field is read FIRST — the `already_merged_verified:` notes line
|
|
1284
|
-
// is only a fallback, because session notes are hard-capped at 3000
|
|
1285
|
-
// chars and a truthful declaration at the report's tail was silently
|
|
1286
|
-
// truncated. The structured value was written by the workflow after a
|
|
1287
|
-
// mechanical ancestor check — it is trusted; the builder's bare
|
|
1288
|
-
// declaration never is. Absent both, the mechanical fact below reads
|
|
1289
|
-
// "none declared" and Cass fails closed. The hydration read is best-effort:
|
|
1290
|
-
// a transport throw degrades to "none declared" rather than crashing Review.
|
|
1130
|
+
// See docs/decisions/publish-path.md#already-merged-hydra: when this run did not execute, hydration uses the existing merge.
|
|
1291
1131
|
if (!alreadyMergedSha) {
|
|
1292
1132
|
var hydResult = null;
|
|
1293
1133
|
try {
|
|
@@ -1395,25 +1235,7 @@ while (i < STEPS.length) {
|
|
|
1395
1235
|
"published: muse-crew@" + publishTarget.target + "\n" +
|
|
1396
1236
|
"VERDICT: PASS\n\n";
|
|
1397
1237
|
} else if (PUBLISH_TYPE === "artifact") {
|
|
1398
|
-
//
|
|
1399
|
-
// agent claimed "Rebuilt and deployed" while no build ran and no
|
|
1400
|
-
// provenance was stamped — prose-trusted side effects, the same failure
|
|
1401
|
-
// class as the npm double-skip (bb739316). The npm path already runs one
|
|
1402
|
-
// deterministic script; the artifact path now has the same shape. Lock
|
|
1403
|
-
// refresh, rebuild trigger, build-completion poll, and post-deploy are
|
|
1404
|
-
// narrow schema'd bookkeeping calls owned by the workflow — the work
|
|
1405
|
-
// agent reports on the mechanical outcome and cannot skip what it never
|
|
1406
|
-
// owned. Any step failing parks with an honest, step-specific reason
|
|
1407
|
-
// (fail-closed). There is deliberately NO workflow-side provenance
|
|
1408
|
-
// stamp: the builder's applied-report is circular (canary run 8,
|
|
1409
|
-
// 2026-09-11), so the stamp moved to the parent — after the build
|
|
1410
|
-
// lands, the workflow records the session completed and parks with
|
|
1411
|
-
// "publish: verification-requested". The parent owns verification
|
|
1412
|
-
// (docs/publish-verification.md); the primary sensor is the
|
|
1413
|
-
// deterministic lib/readback-disk.js (the agent-callable read-back
|
|
1414
|
-
// tool is unavailable — artifact_inspect was removed by the platform
|
|
1415
|
-
// 2026-09-14 — so the LLM-inspector path is manual-fallback only).
|
|
1416
|
-
// QA's provenance check enforces the stamp mechanically.
|
|
1238
|
+
// See docs/decisions/publish-path.md#deterministic-artifact-publish: the work agent never publishes; the parent runs the deterministic publish script.
|
|
1417
1239
|
var artifactPublish = null;
|
|
1418
1240
|
var publishLockRefreshed = false;
|
|
1419
1241
|
var publishSkippedNoLock = false;
|
|
@@ -1440,20 +1262,7 @@ while (i < STEPS.length) {
|
|
|
1440
1262
|
}
|
|
1441
1263
|
publishLockRefreshed = true;
|
|
1442
1264
|
}
|
|
1443
|
-
//
|
|
1444
|
-
// builder. The builder's source tree is NOT the crew's repo —
|
|
1445
|
-
// canary run 4 (2026-09-11) proved it: Publish asked for "rebuild
|
|
1446
|
-
// from current source. Do not modify any source files" and the
|
|
1447
|
-
// builder rebuilt a stale copy predating the canary's changes, then
|
|
1448
|
-
// the workflow stamped the new commit hash on the stale build.
|
|
1449
|
-
// Provenance fiction; all eight phases passed. The merge diff is
|
|
1450
|
-
// embedded in the edit request; the builder applies it to its own
|
|
1451
|
-
// tree and reports the applied changes; the workflow verifies the
|
|
1452
|
-
// report matches the diff BEFORE stamping provenance. A mismatch
|
|
1453
|
-
// parks without stamping — the stamp must never certify a build
|
|
1454
|
-
// whose content was not verified.
|
|
1455
|
-
// Skipped entirely when no lock was held — nothing merged, nothing
|
|
1456
|
-
// to ship.
|
|
1265
|
+
// See docs/decisions/publish-path.md#step1-builder-source: the builder's source tree is NOT the crew's repo; verify report before stamping.
|
|
1457
1266
|
if (!publishSkippedNoLock) {
|
|
1458
1267
|
// The trigger key of the attempt that last ran, for the publish ledger.
|
|
1459
1268
|
// Minted once here (not re-minted per use site) so the ledger always
|
|
@@ -1534,20 +1343,7 @@ while (i < STEPS.length) {
|
|
|
1534
1343
|
return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact target preflight was inconclusive (no parsable signal). Target existence is unproven, so no edit was issued and nothing was retried blindly. Human attention needed.");
|
|
1535
1344
|
}
|
|
1536
1345
|
log("Publish artifact preflight for task " + taskId + ": target " + artifactTargetDir + " present");
|
|
1537
|
-
//
|
|
1538
|
-
// verification, bounded poll, and provenance stamp. The builder
|
|
1539
|
-
// only makes the artifact_edit call and reports the applied
|
|
1540
|
-
// changes — no prose claim to trust. If the artifact tool namespace
|
|
1541
|
-
// is missing from this child it reports honestly and the workflow
|
|
1542
|
-
// retries once with a fresh key (bounded); anything else parks.
|
|
1543
|
-
// Publish diff base (2026-09-14, task 0c53af4e): the carried diff is
|
|
1544
|
-
// BASE..HEAD where BASE is the previously-stamped provenance
|
|
1545
|
-
// source_commit — NOT HEAD^1. A push-time reconcile merge puts the
|
|
1546
|
-
// task's own changes behind an intermediate merge, so HEAD^1..HEAD
|
|
1547
|
-
// silently drops the task's fix while the artifact builds without
|
|
1548
|
-
// it. The stamped base is the artifact's actual content; BASE..HEAD
|
|
1549
|
-
// is the complete unpublished delta. Empty tree only for a genuine
|
|
1550
|
-
// first publish (no provenance stamped yet).
|
|
1346
|
+
// See docs/decisions/publish-path.md#diff-computation: the diff is computed, the rebuild is triggered, and the report is verified.
|
|
1551
1347
|
var EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
|
|
1552
1348
|
var provResult = await agent(
|
|
1553
1349
|
crewCmd("get-provenance", { project_id: LAUNCH_PROJECT_ID }) + "\n" +
|
|
@@ -1563,16 +1359,7 @@ while (i < STEPS.length) {
|
|
|
1563
1359
|
} else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
|
|
1564
1360
|
return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
|
|
1565
1361
|
}
|
|
1566
|
-
//
|
|
1567
|
-
// ferried as a JSON string field in the agent's response — the agent
|
|
1568
|
-
// produced a valid 700-line diff on disk but the JSON ferry dropped
|
|
1569
|
-
// it, and the parse saw zero files ("Publish diff parsed to zero
|
|
1570
|
-
// files"). The diff now travels git -> file -> deterministic script
|
|
1571
|
-
// summary; the LLM never carries diff bytes. The agent is pure hands:
|
|
1572
|
-
// it runs exactly one command (the PINNED compute-publish-diff.js —
|
|
1573
|
-
// a mid-run release swap cannot change it under the workflow) and
|
|
1574
|
-
// returns the small JSON summary verbatim. All fail-closed parks
|
|
1575
|
-
// below are unchanged in meaning.
|
|
1362
|
+
// See docs/decisions/publish-path.md#diff-transport: the diff travels via file, not the LLM.
|
|
1576
1363
|
var publishDiffFile = crewHome + "/.publish-diffs/" + taskId + (totalReworkCount > 0 ? "-r" + totalReworkCount : "") + ".diff";
|
|
1577
1364
|
var diffResult = await agent(
|
|
1578
1365
|
"Run exactly one command and nothing else:\n" +
|
|
@@ -1597,6 +1384,7 @@ while (i < STEPS.length) {
|
|
|
1597
1384
|
return await parkTask("Publish diff base mismatch: script reported '" + String(diffSummary.base || "").slice(0, 12) + "' but the stamped base is '" + publishBase.slice(0, 12) + "'. Human attention needed.");
|
|
1598
1385
|
}
|
|
1599
1386
|
var mergeCommitForPublish = String(diffSummary.commit || "").trim();
|
|
1387
|
+
var mergeCommitShortForPublish = String(mergeCommitForPublish).substring(0, 7) || "unknown";
|
|
1600
1388
|
var publishDiffSha256 = String(diffSummary.sha256 || "");
|
|
1601
1389
|
if (!diffSummary.bytes) {
|
|
1602
1390
|
return await parkTask("Publish diff is empty for commit " + (mergeCommitForPublish || "unknown") + " — a merge lock was held but there is no change to carry. Human attention needed.");
|
|
@@ -1648,83 +1436,44 @@ while (i < STEPS.length) {
|
|
|
1648
1436
|
"- If artifact_edit is not available after the load, do NOT improvise — end your turn.\n" +
|
|
1649
1437
|
"- You do NOT call setprovenance, artifact_inspect, or post-deploy yourself.\n" +
|
|
1650
1438
|
"No report is needed: do not return JSON, do not summarize what you did, do not echo the diff. End your turn after the artifact_edit call.\n";
|
|
1651
|
-
//
|
|
1652
|
-
// workflow-owned observation below. The agent_id is the artifact
|
|
1653
|
-
// system's in-flight correlation ID (research 2026-09-12):
|
|
1654
|
-
// artifact.edit returns pending_init with NO agent_id, but
|
|
1655
|
-
// artifact_status exposes build.agent_id immediately after
|
|
1656
|
-
// acceptance, stable across polls. Recorded in the ledger so an
|
|
1657
|
-
// attempt correlates to the exact builder run; null when no build
|
|
1658
|
-
// was ever observed.
|
|
1439
|
+
// See docs/decisions/publish-path.md#agent-id-attribution: the artifact build's agent_id is attributed to the edit call.
|
|
1659
1440
|
var rebuildAgentId = null;
|
|
1660
|
-
//
|
|
1661
|
-
// trigger's JSON closeout contract, which is removed below. The
|
|
1662
|
-
// parent's independent read-back (docs/publish-verification.md) is
|
|
1663
|
-
// the verification — this field stays "missing-report" on ledger
|
|
1664
|
-
// lines for issued triggers; pre-trigger parks (toolcheck
|
|
1665
|
-
// rejected/inconclusive) and unattributed-unknown parks write null
|
|
1666
|
-
// (no trigger was observed, so there is nothing to report).
|
|
1441
|
+
// See docs/decisions/publish-path.md#applied-report-gone: the builder's applied report is gone; the workflow verifies differently.
|
|
1667
1442
|
var publishAppliedObservation = "missing-report";
|
|
1668
|
-
//
|
|
1669
|
-
// detects IN-FLIGHT builds. A build that finished before the
|
|
1670
|
-
// observation leaves no in-flight trace — but the platform's audit
|
|
1671
|
-
// harness leaves a durable one:
|
|
1672
|
-
// ~/workspace/ts-spaces/<slug>/audits/<timestamp>-<id>/ per
|
|
1673
|
-
// completed build. Snapshot the listing BEFORE the trigger so the
|
|
1674
|
-
// fallback can diff before/after: a directory appearing during the
|
|
1675
|
-
// trigger window is positive evidence the edit went through and
|
|
1676
|
-
// the build completed. Best-effort and non-gating: if the snapshot
|
|
1677
|
-
// fails, auditBeforeOk stays false and BOTH fallback comparisons
|
|
1678
|
-
// are disabled (2026-09-16, critic finding 4) — without a baseline,
|
|
1679
|
-
// an empty before-list would make every historical audit dir look
|
|
1680
|
-
// "new". No wall-clock in-script (deterministic replay) — the
|
|
1681
|
-
// comparison is a pure before/after set diff.
|
|
1443
|
+
// See docs/decisions/publish-path.md#durable-evidence-snapshot: snapshot the audit-dir listing BEFORE the trigger; fallback diffs before/after.
|
|
1682
1444
|
var auditDirsBeforeTrigger = [];
|
|
1683
1445
|
var auditBeforeOk = false;
|
|
1446
|
+
// Pre-trigger baselines (design §1.9): the audit-dir listing and the
|
|
1447
|
+
// manifest snapshot. The verified-path freshness check compares the
|
|
1448
|
+
// post-trigger manifest against the baseline (built_at advance +
|
|
1449
|
+
// content_sha256 change). Best-effort, never gates — a missing
|
|
1450
|
+
// manifest baseline fails the verified path closed.
|
|
1451
|
+
var preTriggerManifest = null;
|
|
1684
1452
|
try {
|
|
1685
1453
|
var auditBefore = await agent(
|
|
1686
|
-
"
|
|
1687
|
-
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1688
|
-
"Return JSON { \"dirs\": \"<newline-separated names, empty string when
|
|
1689
|
-
{ key: attemptKey("publish-
|
|
1690
|
-
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1454
|
+
"Capture pre-trigger baselines for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshots, never gates).\n" +
|
|
1455
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null; echo ---MANIFEST---; cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/.space-build/manifest.json 2>/dev/null\n" +
|
|
1456
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when missing>\", \"manifest\": \"<the manifest's full text, or empty string when missing/unreadable>\" } and nothing else.",
|
|
1457
|
+
{ key: attemptKey("publish-baseline-before-" + taskId, totalReworkCount), label: "Snapshotting baselines before rebuild trigger",
|
|
1458
|
+
schema: { type: "object", properties: { dirs: { type: "string" }, manifest: { type: "string" } }, required: ["dirs", "manifest"] } }
|
|
1691
1459
|
);
|
|
1692
1460
|
auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1461
|
+
var manifestText = String((auditBefore && auditBefore.manifest) || "").trim();
|
|
1462
|
+
if (manifestText) {
|
|
1463
|
+
var manifestJson = JSON.parse(manifestText);
|
|
1464
|
+
preTriggerManifest = {
|
|
1465
|
+
built_at: typeof manifestJson.built_at === "string" ? manifestJson.built_at : null,
|
|
1466
|
+
content_sha256: typeof manifestJson.content_sha256 === "string" ? manifestJson.content_sha256 : null,
|
|
1467
|
+
};
|
|
1468
|
+
}
|
|
1469
|
+
// Arm only after both baselines parse.
|
|
1693
1470
|
auditBeforeOk = true;
|
|
1694
|
-
log("Publish
|
|
1471
|
+
log("Publish pre-trigger baselines for task " + taskId + ": " + auditDirsBeforeTrigger.length + " audit dirs, manifest " + (preTriggerManifest ? "built_at=" + preTriggerManifest.built_at : "none"));
|
|
1695
1472
|
} catch (auditBeforeErr) {
|
|
1696
|
-
log("Publish
|
|
1473
|
+
log("Publish pre-trigger baselines failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
|
|
1697
1474
|
}
|
|
1698
|
-
|
|
1699
|
-
//
|
|
1700
|
-
// traveled over the stochastic text channel, and the runtime's
|
|
1701
|
-
// JSON-candidate heuristic misfired on it ("workflow agent output
|
|
1702
|
-
// was not JSON: no JSON object or array found in final response"),
|
|
1703
|
-
// parking a task whose edit may have gone through. The contract's
|
|
1704
|
-
// content was already observation-only (the applied report never
|
|
1705
|
-
// gated; the pre_hashes were diagnostic-only), so the contract is
|
|
1706
|
-
// removed: the trigger carries NO schema and its return value is
|
|
1707
|
-
// never consumed, which takes the extraction heuristic out of this
|
|
1708
|
-
// call entirely. The workflow attributes the edit itself through
|
|
1709
|
-
// the tiny schema'd reads below — no prose is parsed for the
|
|
1710
|
-
// trigger outcome.
|
|
1711
|
-
// (Probe, 2026-09-16: the workflow scope exposes only agent() —
|
|
1712
|
-
// tool_search, artifact_edit and artifact_status are undefined
|
|
1713
|
-
// there — so the workflow cannot call the artifact tools directly;
|
|
1714
|
-
// observation still goes through minimal child calls with tiny
|
|
1715
|
-
// schemas, never a broad JSON contract.)
|
|
1716
|
-
//
|
|
1717
|
-
// Pre-trigger toolcheck (tiny, schema'd): the artifact namespace is
|
|
1718
|
-
// deferred for workflow children — the child self-loads it and emits
|
|
1719
|
-
// one exact signal line, read mechanically (never English prose).
|
|
1720
|
-
// Only a parsed ARTIFACT_TOOLS: missing signal is explicit negative
|
|
1721
|
-
// evidence: it gets one bounded retry with a fresh key, then parks
|
|
1722
|
-
// rejected — without the tools the edit provably did NOT go through,
|
|
1723
|
-
// so this is the one safe retry on the publish path. A throw (or an
|
|
1724
|
-
// unparseable signal) is INCONCLUSIVE transport noise, never
|
|
1725
|
-
// evidence of missing tools (2026-09-16, critic finding 3): it is
|
|
1726
|
-
// recorded, it retries once in case the flake clears, but it can
|
|
1727
|
-
// never take the rejected path.
|
|
1475
|
+
|
|
1476
|
+
// See docs/decisions/publish-path.md#fire-and-forget-trigger: the trigger child returns immediately; the workflow owns observation and verdict.
|
|
1728
1477
|
var publishToolsOk = false;
|
|
1729
1478
|
var publishToolsMissing = false;
|
|
1730
1479
|
for (var toolcheckAttempt = 1; toolcheckAttempt <= 2 && !publishToolsOk; toolcheckAttempt++) {
|
|
@@ -1772,14 +1521,7 @@ while (i < STEPS.length) {
|
|
|
1772
1521
|
}, totalReworkCount);
|
|
1773
1522
|
return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was explicitly missing (parsed signal — the edit provably did not go through, so no trigger was issued and nothing was retried blindly). Human attention needed.");
|
|
1774
1523
|
}
|
|
1775
|
-
//
|
|
1776
|
-
// artifact_status. The post-trigger observation diffs against this
|
|
1777
|
-
// baseline — a build whose agent_id was absent from (or differs
|
|
1778
|
-
// from) the baseline is attributed to our edit; a build already in
|
|
1779
|
-
// flight at baseline predates the trigger and is never attributed
|
|
1780
|
-
// to it. If the baseline read itself fails, receipt attribution is
|
|
1781
|
-
// skipped and the durable audit-dir evidence below is the only
|
|
1782
|
-
// positive signal.
|
|
1524
|
+
// See docs/decisions/publish-path.md#pretrigger-baseline: a tiny schema'd baseline is captured before the trigger.
|
|
1783
1525
|
var baselineAgentId = null;
|
|
1784
1526
|
var baselineFailed = false;
|
|
1785
1527
|
try {
|
|
@@ -1796,32 +1538,13 @@ while (i < STEPS.length) {
|
|
|
1796
1538
|
baselineFailed = true;
|
|
1797
1539
|
log("Publish pre-trigger baseline read failed for task " + taskId + " (" + (baselineErr && baselineErr.message ? baselineErr.message : baselineErr) + ") — receipt attribution skipped; durable audit-dir evidence is the only positive signal");
|
|
1798
1540
|
}
|
|
1799
|
-
//
|
|
1800
|
-
// waits for it to complete) but its return value is intentionally
|
|
1801
|
-
// UNCONSUMED — NO schema, so no schema validation can fail this
|
|
1802
|
-
// call: a schema-less call resolves to the child's raw response as
|
|
1803
|
-
// a plain string (probed live 2026-09-16 — never parsed, never
|
|
1804
|
-
// throws on content). One caveat, also probed: the runtime still
|
|
1805
|
-
// scans the response for a JSON candidate, and an unparseable
|
|
1806
|
-
// {...}-looking substring in the child's prose throws ("response
|
|
1807
|
-
// JSON candidate", probe P6). The prompt tells the child to end its
|
|
1808
|
-
// turn with no prose at all, which keeps the common case clean —
|
|
1809
|
-
// but the channel is stochastic, so any throw is possible and
|
|
1810
|
-
// inconclusive: the edit may still have gone through, so the
|
|
1811
|
-
// outcome stays unknown until the observation below confirms it —
|
|
1812
|
-
// never inferred from the throw, and never blind-retried (a blind
|
|
1813
|
-
// re-trigger duplicated the edit on 2026-09-12).
|
|
1541
|
+
// See docs/decisions/publish-path.md#trigger-await: the artifact_edit call is awaited.
|
|
1814
1542
|
var rebuildTrigger = null;
|
|
1815
1543
|
try {
|
|
1816
1544
|
var triggerText = String(await agent(rebuildPrompt,
|
|
1817
1545
|
{ key: rebuildAttemptKey, label: "Triggering artifact rebuild" }) || "");
|
|
1818
1546
|
log("Publish rebuild trigger for task " + taskId + " returned (" + triggerText.length + " chars; awaited; scanned only for the explicit refusal signal)");
|
|
1819
|
-
//
|
|
1820
|
-
// with ARTIFACT_EDIT_REFUSED when artifact_edit explicitly refused.
|
|
1821
|
-
// Conclusive negative evidence — the edit provably did NOT go
|
|
1822
|
-
// through — so this parks rejected and skips observation polling.
|
|
1823
|
-
// A missing/unparseable signal is NOT a refusal: it stays unknown
|
|
1824
|
-
// and fail-closed below.
|
|
1547
|
+
// See docs/decisions/publish-path.md#explicit-refusal-16: the child must explicitly refuse artifact work.
|
|
1825
1548
|
var refusalText = extractRefusal(triggerText);
|
|
1826
1549
|
if (refusalText) {
|
|
1827
1550
|
await recordPublishLedger({
|
|
@@ -1868,17 +1591,79 @@ while (i < STEPS.length) {
|
|
|
1868
1591
|
log("Publish post-trigger build-state check failed for task " + taskId + " (" + (buildCheckErr && buildCheckErr.message ? buildCheckErr.message : buildCheckErr) + ") — this signal is unknown, not negative");
|
|
1869
1592
|
}
|
|
1870
1593
|
var observedAgentId = (buildState && buildState.build && typeof buildState.build.agent_id === "string" && buildState.build.agent_id) || null;
|
|
1871
|
-
//
|
|
1872
|
-
// is timing-based — any agent_id new relative to the baseline is
|
|
1873
|
-
// treated as this edit's receipt. A stranger's build starting inside
|
|
1874
|
-
// the trigger window is indistinguishable by timing and would be
|
|
1875
|
-
// misattributed here. The consequence is bounded: the completion
|
|
1876
|
-
// poll below tracks the recorded id, and the parent's mechanical
|
|
1877
|
-
// content read-back (docs/publish-verification.md) certifies the
|
|
1878
|
-
// exact commit's content — a wrong build's content fails closed as
|
|
1879
|
-
// verification-failed, never stamped. Timing narrows the candidate;
|
|
1880
|
-
// content decides.
|
|
1594
|
+
// See docs/decisions/publish-path.md#attribution-limitation: attribution is timing-based; content verification bounds the risk.
|
|
1881
1595
|
var receiptAgentId = (!buildStateFailed && !baselineFailed && observedAgentId && observedAgentId !== baselineAgentId) ? observedAgentId : null;
|
|
1596
|
+
// auditReportOk: pure tri-state read of a report.json body —
|
|
1597
|
+
// true (build ok), false (build failed), null (missing or
|
|
1598
|
+
// unreadable — not evidence either way). The child returns the
|
|
1599
|
+
// raw body verbatim; interpretation lives here, never in prose.
|
|
1600
|
+
// Hoisted to Publish-step scope (before the receipt branch) so both
|
|
1601
|
+
// the immediate and post-poll audit fallbacks share it on every path —
|
|
1602
|
+
// the receipt path skips the else below, which must not leave these
|
|
1603
|
+
// undefined.
|
|
1604
|
+
var auditReportOk = function (raw) {
|
|
1605
|
+
if (typeof raw !== "string") return null;
|
|
1606
|
+
var trimmed = raw.trim();
|
|
1607
|
+
if (trimmed === "" || trimmed === "MISSING") return null;
|
|
1608
|
+
var parsed;
|
|
1609
|
+
try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
|
|
1610
|
+
if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
|
|
1611
|
+
return null;
|
|
1612
|
+
};
|
|
1613
|
+
// (2026-09-18, H2 verdict-first) Publish-verdict vocabulary. The
|
|
1614
|
+
// verdict is one of "landed" | "unknown". Decided once,
|
|
1615
|
+
// before the receipt poll, and dispatched on — never re-derived.
|
|
1616
|
+
// Pure and self-contained: unit-tested by
|
|
1617
|
+
// tests/publish-verdict-first.test.js.
|
|
1618
|
+
// See docs/decisions/publish-path.md#h2-verdict-dispatch.
|
|
1619
|
+
var decidePublishVerdict = function (opts) {
|
|
1620
|
+
var immediateReport = (opts && "immediateReport" in opts) ? opts.immediateReport : null;
|
|
1621
|
+
var newDirCount = (opts && typeof opts.newDirCount === "number") ? opts.newDirCount : 0;
|
|
1622
|
+
if (newDirCount === 1 && immediateReport === true) return { verdict: "landed", unattributableReason: null };
|
|
1623
|
+
if (newDirCount === 1 && immediateReport === false) return { verdict: "unknown", unattributableReason: "audit-report-ok-false" };
|
|
1624
|
+
if (newDirCount === 1) return { verdict: "unknown", unattributableReason: "audit-report-unreadable-or-missing" };
|
|
1625
|
+
if (newDirCount === 0) return { verdict: "unknown", unattributableReason: "no-new-audit-dir-in-window" };
|
|
1626
|
+
return { verdict: "unknown", unattributableReason: "audit-dir-ambiguity", ambiguousDirCount: newDirCount };
|
|
1627
|
+
};
|
|
1628
|
+
// (2026-09-18, H2 message contract, design §1.5) The human line is
|
|
1629
|
+
// the output of a mechanical field→template mapping: the trigger
|
|
1630
|
+
// commit short-sha, one plain clause per unattributable reason, the
|
|
1631
|
+
// no-republish warning, the ledger path with (outcome: unknown), and
|
|
1632
|
+
// the unknown-recovery clause. The appendix carries every machine
|
|
1633
|
+
// field. Pure and self-contained: unit-tested by
|
|
1634
|
+
// tests/publish-verdict-first.test.js.
|
|
1635
|
+
var composeUnattributedParkReason = function (fields) {
|
|
1636
|
+
var f = fields || {};
|
|
1637
|
+
var task = f.taskId || taskId;
|
|
1638
|
+
var reason = f.unattributableReason || "unknown-outcome";
|
|
1639
|
+
var shortSha = String(f.commitShortSha || "").substring(0, 7) || "unknown";
|
|
1640
|
+
var reasonClauses = {
|
|
1641
|
+
"no-new-audit-dir-in-window": "no build could be tied to this attempt",
|
|
1642
|
+
"audit-report-unreadable-or-missing": "the build's report is unreadable",
|
|
1643
|
+
"audit-dir-ambiguity": "more than one build appeared in the check window",
|
|
1644
|
+
"stranger-build-observed-during-poll": "a different build was running during the check",
|
|
1645
|
+
"status-read-errors-during-poll": "status reads kept failing"
|
|
1646
|
+
};
|
|
1647
|
+
var clause = reasonClauses[reason] || "no build could be tied to this attempt";
|
|
1648
|
+
return "Publish outcome unknown for task " + task +
|
|
1649
|
+
": the trigger was sent for commit " + shortSha + " but the outcome could not be confirmed — " + clause + ". " +
|
|
1650
|
+
"Do NOT republish: if the trigger was accepted, a retry duplicates the build (2026-09-12). " +
|
|
1651
|
+
"The attempt is in the ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (outcome: unknown) " +
|
|
1652
|
+
"and the crew's unknown-recovery will re-examine it. " +
|
|
1653
|
+
"Appendix: unattributable_reason=" + reason +
|
|
1654
|
+
"; poll_end_state=" + (f.pollEndState || "not-polled") +
|
|
1655
|
+
"; saw_our_build=" + (f.sawOurBuild ? "true" : "false") +
|
|
1656
|
+
"; new_audit_dirs=" + (f.newAuditDirCount == null ? 0 : f.newAuditDirCount) +
|
|
1657
|
+
"; poll_chunks_failed=" + (f.chunkFailures == null ? 0 : f.chunkFailures) +
|
|
1658
|
+
"; poll_status_errors=" + (f.pollStatusErrors == null ? 0 : f.pollStatusErrors) + ".";
|
|
1659
|
+
};
|
|
1660
|
+
// publishFailure is declared here (per-Publish-pass scope) so the
|
|
1661
|
+
// post-poll explicit build failure survives to the final routing
|
|
1662
|
+
// below; publishUnknownFields carries the structured unknown fields
|
|
1663
|
+
// for the single composed fail-closed park. Both re-initialize on
|
|
1664
|
+
// every pass — a rework re-entry never leaks a stale verdict.
|
|
1665
|
+
var publishFailure = null;
|
|
1666
|
+
var publishUnknownFields = null;
|
|
1882
1667
|
if (receiptAgentId) {
|
|
1883
1668
|
// The edit went through — a build with a new agent_id appeared
|
|
1884
1669
|
// after the trigger. The parent's independent read-back
|
|
@@ -1892,36 +1677,12 @@ while (i < STEPS.length) {
|
|
|
1892
1677
|
attempt: rebuildAttemptKey,
|
|
1893
1678
|
agent_id: rebuildAgentId,
|
|
1894
1679
|
applied_report: publishAppliedObservation,
|
|
1680
|
+
manifest_before: preTriggerManifest,
|
|
1895
1681
|
outcome: "submitted",
|
|
1896
1682
|
detail: "fire-and-forget trigger; build receipt captured by workflow-owned build-state observation (pre/post-trigger diff)"
|
|
1897
1683
|
}, totalReworkCount);
|
|
1898
1684
|
} else {
|
|
1899
1685
|
var newAuditDirs = [];
|
|
1900
|
-
// auditReportOk: pure tri-state read of a report.json body —
|
|
1901
|
-
// true (build ok), false (build failed), null (missing or
|
|
1902
|
-
// unreadable — not evidence either way). The child returns the
|
|
1903
|
-
// raw body verbatim; interpretation lives here, never in prose.
|
|
1904
|
-
// Defined here so both the immediate and post-poll audit
|
|
1905
|
-
// fallbacks share it.
|
|
1906
|
-
var auditReportOk = function (raw) {
|
|
1907
|
-
if (typeof raw !== "string") return null;
|
|
1908
|
-
var trimmed = raw.trim();
|
|
1909
|
-
if (trimmed === "" || trimmed === "MISSING") return null;
|
|
1910
|
-
var parsed;
|
|
1911
|
-
try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
|
|
1912
|
-
if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
|
|
1913
|
-
return null;
|
|
1914
|
-
};
|
|
1915
|
-
// (2026-09-16, critic finding 2) When durable audit evidence
|
|
1916
|
-
// confirms (or refutes) the build, there is no receipt agent_id
|
|
1917
|
-
// to chain the completion poll to — skipReceiptPoll bypasses the
|
|
1918
|
-
// poll below, which with a null receipt could only observe
|
|
1919
|
-
// strangers or nothing.
|
|
1920
|
-
var skipReceiptPoll = false;
|
|
1921
|
-
// publishFailure is declared here (moved up from below) so the
|
|
1922
|
-
// immediate audit fallback can record an explicit build failure
|
|
1923
|
-
// without the later declaration resetting it.
|
|
1924
|
-
var publishFailure = null;
|
|
1925
1686
|
try {
|
|
1926
1687
|
var auditAfter = await agent(
|
|
1927
1688
|
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
|
|
@@ -1941,88 +1702,110 @@ while (i < STEPS.length) {
|
|
|
1941
1702
|
log("Publish audit-dir re-list after trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
|
|
1942
1703
|
}
|
|
1943
1704
|
if (newAuditDirs.length > 0) {
|
|
1944
|
-
rebuildTrigger = { edit_started: true };
|
|
1945
1705
|
rebuildAgentId = null;
|
|
1946
1706
|
newAuditDirs.sort();
|
|
1947
1707
|
var newestImmediateDir = newAuditDirs[newAuditDirs.length - 1];
|
|
1948
1708
|
log("Publish rebuild trigger for task " + taskId + ": new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and a build completed; no in-flight receipt was observed.");
|
|
1949
|
-
// (2026-09-
|
|
1709
|
+
// (2026-09-18, H2 verdict-first) Durable audit evidence exists,
|
|
1950
1710
|
// but there is no receipt agent_id to chain the completion poll
|
|
1951
1711
|
// to — polling with a null receipt can only observe strangers
|
|
1952
1712
|
// (any running build differs from "null") or nothing, burning
|
|
1953
|
-
// 10.5 minutes to park unknown. Read the build report now
|
|
1954
|
-
//
|
|
1955
|
-
//
|
|
1956
|
-
//
|
|
1713
|
+
// 10.5 minutes to park unknown. Read the build report now instead
|
|
1714
|
+
// of polling, then decide the verdict ONCE via
|
|
1715
|
+
// decidePublishVerdict: exactly one new dir with ok=true lands
|
|
1716
|
+
// (attribution by window, not identity — never poll blind);
|
|
1717
|
+
// ok=false is UNKNOWN with the failure evidence preserved in the
|
|
1718
|
+
// ledger detail (the evidence is explicit, the attribution is
|
|
1719
|
+
// not); unreadable / zero / ambiguous dirs are UNKNOWN.
|
|
1720
|
+
// Verdict-first: landed bypasses the poll below, unknown falls
|
|
1721
|
+
// through to post-deploy. STEP 2 runs on every path.
|
|
1722
|
+
// See docs/decisions/publish-path.md#h2-verdict-dispatch.
|
|
1957
1723
|
var immediateReportOk = null;
|
|
1958
|
-
|
|
1959
|
-
|
|
1960
|
-
|
|
1961
|
-
|
|
1962
|
-
|
|
1963
|
-
|
|
1964
|
-
|
|
1965
|
-
|
|
1966
|
-
|
|
1967
|
-
|
|
1968
|
-
|
|
1969
|
-
|
|
1724
|
+
// (N1) Read the report only when exactly one new dir exists:
|
|
1725
|
+
// ambiguity (>1) forces UNKNOWN regardless — don't shell out to
|
|
1726
|
+
// read a report that will be discarded.
|
|
1727
|
+
if (newAuditDirs.length === 1) {
|
|
1728
|
+
try {
|
|
1729
|
+
var immediateOkRead = await agent(
|
|
1730
|
+
"Read the artifact build report for slug \"" + PUBLISH_SLUG + "\".\n" +
|
|
1731
|
+
"Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestImmediateDir + "/report.json 2>/dev/null || echo MISSING\n" +
|
|
1732
|
+
"Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
|
|
1733
|
+
{ key: attemptKey("publish-audit-ok-immediate-" + taskId, totalReworkCount), label: "Reading build report for audit-confirmed build",
|
|
1734
|
+
schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
|
|
1735
|
+
);
|
|
1736
|
+
immediateReportOk = auditReportOk(immediateOkRead && immediateOkRead.raw);
|
|
1737
|
+
} catch (immediateOkErr) {
|
|
1738
|
+
log("Publish build-report read for audit-confirmed dir failed for task " + taskId + " (treated as unknown): " + (immediateOkErr && immediateOkErr.message ? immediateOkErr.message : immediateOkErr));
|
|
1739
|
+
immediateReportOk = null;
|
|
1740
|
+
}
|
|
1970
1741
|
}
|
|
1971
|
-
|
|
1742
|
+
var immediateVerdict = decidePublishVerdict({ editStarted: true, immediateReport: immediateReportOk, newDirCount: newAuditDirs.length });
|
|
1743
|
+
if (immediateVerdict.verdict === "landed") {
|
|
1972
1744
|
publishBuildLanded = true;
|
|
1973
1745
|
artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
|
|
1974
|
-
skipReceiptPoll = true;
|
|
1975
|
-
log("Publish build landed for task " + taskId + " via immediate durable audit evidence (audit dir " + newestImmediateDir + ", report ok=true) — receipt poll skipped (no receipt to chain to), routing directly to parent verification");
|
|
1976
1746
|
await recordPublishLedger({
|
|
1977
1747
|
commit: mergeCommitForPublish,
|
|
1978
1748
|
attempt: rebuildAttemptKey,
|
|
1979
1749
|
agent_id: null,
|
|
1980
1750
|
applied_report: publishAppliedObservation,
|
|
1981
|
-
|
|
1982
|
-
|
|
1983
|
-
|
|
1984
|
-
} else if (immediateReportOk === false) {
|
|
1985
|
-
skipReceiptPoll = true;
|
|
1986
|
-
publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestImmediateDir + ", report ok=false — immediate audit evidence, no receipt observed). Explicit negative evidence: a build ran and failed. The publish did not land — provenance was not stamped. Fail-closed.";
|
|
1987
|
-
await recordPublishLedger({
|
|
1988
|
-
commit: mergeCommitForPublish,
|
|
1989
|
-
attempt: rebuildAttemptKey,
|
|
1990
|
-
agent_id: null,
|
|
1991
|
-
applied_report: publishAppliedObservation,
|
|
1992
|
-
outcome: "failed",
|
|
1993
|
-
detail: "a build ran and failed: audit dir " + newestImmediateDir + " report ok=false (immediate audit evidence, no receipt)"
|
|
1751
|
+
manifest_before: preTriggerManifest,
|
|
1752
|
+
outcome: "build-observed",
|
|
1753
|
+
detail: "durable audit evidence shows a build completed during the attempt window (no receipt agent_id — attribution by window, not identity; receipt poll bypassed (verdict decided), routed to parent verification)"
|
|
1994
1754
|
}, totalReworkCount);
|
|
1755
|
+
log("Publish verdict LANDED for task " + taskId + ": a completed build was observed during the attempt window — receipt poll bypassed (verdict decided, no receipt to chain to), routing directly to parent verification.");
|
|
1995
1756
|
} else {
|
|
1757
|
+
// Verdict UNKNOWN on the immediate path. ok=false is explicit
|
|
1758
|
+
// failure evidence but not an attributable failure — the ledger
|
|
1759
|
+
// records unknown with the evidence preserved in the detail,
|
|
1760
|
+
// and the flow continues to post-deploy (never parks early).
|
|
1761
|
+
publishUnknownFields = {
|
|
1762
|
+
taskId: taskId,
|
|
1763
|
+
unattributableReason: immediateVerdict.unattributableReason,
|
|
1764
|
+
pollEndState: "not-polled",
|
|
1765
|
+
sawOurBuild: false,
|
|
1766
|
+
newAuditDirCount: newAuditDirs.length,
|
|
1767
|
+
chunkFailures: 0,
|
|
1768
|
+
pollStatusErrors: 0
|
|
1769
|
+
};
|
|
1770
|
+
var immediateDetail = "durable audit-dir fallback could not prove a completed build for this attempt (unattributable_reason=" + immediateVerdict.unattributableReason + ", no receipt agent_id)";
|
|
1771
|
+
if (immediateReportOk === false) {
|
|
1772
|
+
immediateDetail += "; explicit failure evidence preserved: report ok=false for audit dir " + newestImmediateDir;
|
|
1773
|
+
}
|
|
1774
|
+
if (immediateVerdict.unattributableReason === "audit-dir-ambiguity") {
|
|
1775
|
+
immediateDetail += "; audit-dir ambiguity: " + newAuditDirs.length + " new dirs in window";
|
|
1776
|
+
}
|
|
1996
1777
|
await recordPublishLedger({
|
|
1997
1778
|
commit: mergeCommitForPublish,
|
|
1998
1779
|
attempt: rebuildAttemptKey,
|
|
1999
1780
|
agent_id: null,
|
|
2000
|
-
applied_report:
|
|
1781
|
+
applied_report: publishAppliedObservation,
|
|
2001
1782
|
outcome: "unknown",
|
|
2002
|
-
detail:
|
|
1783
|
+
detail: immediateDetail
|
|
2003
1784
|
}, totalReworkCount);
|
|
2004
|
-
|
|
1785
|
+
log("Publish verdict UNKNOWN for task " + taskId + ": " + immediateDetail + " — continuing to post-deploy; never polling blind and never parking early.");
|
|
2005
1786
|
}
|
|
2006
1787
|
} else {
|
|
2007
|
-
//
|
|
2008
|
-
|
|
2009
|
-
|
|
2010
|
-
|
|
2011
|
-
|
|
2012
|
-
|
|
2013
|
-
|
|
1788
|
+
// See docs/decisions/publish-path.md#no-attributable-build: no attributable build and no durable evidence means no publish.
|
|
1789
|
+
publishUnknownFields = {
|
|
1790
|
+
taskId: taskId,
|
|
1791
|
+
unattributableReason: decidePublishVerdict({ editStarted: true, immediateReport: null, newDirCount: 0 }).unattributableReason,
|
|
1792
|
+
pollEndState: "not-polled",
|
|
1793
|
+
sawOurBuild: false,
|
|
1794
|
+
newAuditDirCount: 0,
|
|
1795
|
+
chunkFailures: 0,
|
|
1796
|
+
pollStatusErrors: 0
|
|
1797
|
+
};
|
|
2014
1798
|
await recordPublishLedger({
|
|
2015
1799
|
commit: mergeCommitForPublish,
|
|
2016
1800
|
attempt: rebuildAttemptKey,
|
|
2017
1801
|
agent_id: null,
|
|
2018
|
-
applied_report:
|
|
1802
|
+
applied_report: publishAppliedObservation,
|
|
2019
1803
|
outcome: "unknown",
|
|
2020
|
-
detail: "
|
|
1804
|
+
detail: "no new audit dir appeared in the trigger window and no receipt agent_id was observed — the edit was issued fire-and-forget, so completion is unproven; never poll blind on a null receipt"
|
|
2021
1805
|
}, totalReworkCount);
|
|
2022
|
-
|
|
1806
|
+
log("Publish verdict UNKNOWN for task " + taskId + ": no new audit dir in window and no receipt — continuing to post-deploy; never polling blind and never parking early.");
|
|
2023
1807
|
}
|
|
2024
1808
|
}
|
|
2025
|
-
|
|
2026
1809
|
// Durable publish-attempt ledger: record the trigger outcome while the
|
|
2027
1810
|
// attempt key and commit are in scope. Every attempt lands here with
|
|
2028
1811
|
// its outcome — submitted, rejected, or unknown (unknown is recorded
|
|
@@ -2032,9 +1815,45 @@ while (i < STEPS.length) {
|
|
|
2032
1815
|
// already recorded the ledger's submitted line on both positive paths
|
|
2033
1816
|
// and parked on unknown — there is no applied report to observe and
|
|
2034
1817
|
// no rejection signal to record.
|
|
2035
|
-
// (
|
|
2036
|
-
// above
|
|
2037
|
-
|
|
1818
|
+
// (2026-09-18, H2 verdict-first) The verdict was decided exactly
|
|
1819
|
+
// once above; dispatch on it. landed bypasses the receipt poll
|
|
1820
|
+
// (the audit evidence already proved completion); an explicit
|
|
1821
|
+
// publishFailure is preserved verbatim through post-deploy.
|
|
1822
|
+
// Otherwise the verdict is open — but the poll below is only
|
|
1823
|
+
// legitimate against a real receipt: the null-safe assertion records
|
|
1824
|
+
// UNKNOWN and continues to STEP 2 instead of polling blind.
|
|
1825
|
+
// See docs/decisions/publish-path.md#h2-verdict-dispatch.
|
|
1826
|
+
if (publishBuildLanded) {
|
|
1827
|
+
log("Publish verdict already LANDED for task " + taskId + " — bypassing receipt poll.");
|
|
1828
|
+
} else if (publishFailure) {
|
|
1829
|
+
// design §1.2 pre-poll branch — currently unassigned; kept for the converged dispatch shape.
|
|
1830
|
+
log("Publish verdict already FAILED for task " + taskId + " — preserved verbatim through post-deploy.");
|
|
1831
|
+
} else if (!rebuildTrigger || !rebuildTrigger.edit_started) {
|
|
1832
|
+
// Loud defensive assertion (replaces the old lying "Unreachable"
|
|
1833
|
+
// else): with no receipt state the poll would observe strangers or
|
|
1834
|
+
// nothing — record UNKNOWN and continue to STEP 2. Never park
|
|
1835
|
+
// early here; never poll blind.
|
|
1836
|
+
if (!publishUnknownFields) {
|
|
1837
|
+
publishUnknownFields = {
|
|
1838
|
+
taskId: taskId,
|
|
1839
|
+
unattributableReason: "no-receipt-state",
|
|
1840
|
+
pollEndState: "not-polled",
|
|
1841
|
+
sawOurBuild: false,
|
|
1842
|
+
newAuditDirCount: 0,
|
|
1843
|
+
chunkFailures: 0,
|
|
1844
|
+
pollStatusErrors: 0
|
|
1845
|
+
};
|
|
1846
|
+
await recordPublishLedger({
|
|
1847
|
+
commit: mergeCommitForPublish,
|
|
1848
|
+
attempt: rebuildAttemptKey,
|
|
1849
|
+
agent_id: null,
|
|
1850
|
+
applied_report: publishAppliedObservation,
|
|
1851
|
+
outcome: "unknown",
|
|
1852
|
+
detail: "no receipt state was recorded for this attempt — never polling blind; continuing to post-deploy"
|
|
1853
|
+
}, totalReworkCount);
|
|
1854
|
+
}
|
|
1855
|
+
log("Publish verdict UNKNOWN for task " + taskId + ": no receipt state recorded — never polling blind, continuing to post-deploy.");
|
|
1856
|
+
} else {
|
|
2038
1857
|
// (2026-09-16) There is no builder report: the fire-and-forget
|
|
2039
1858
|
// trigger carries no JSON contract, so there is nothing to
|
|
2040
1859
|
// compare and no pre-hash diagnostic. The builder's old
|
|
@@ -2107,13 +1926,7 @@ while (i < STEPS.length) {
|
|
|
2107
1926
|
timeoutMs: 270000 }
|
|
2108
1927
|
);
|
|
2109
1928
|
} catch (chunkErr) {
|
|
2110
|
-
//
|
|
2111
|
-
// record it and continue to the next chunk. (2026-09-16,
|
|
2112
|
-
// clean-room task 1febe8eb: the platform's 270s agent
|
|
2113
|
-
// timeout killed chunk 2, which threw out of this loop —
|
|
2114
|
-
// skipping chunk 3 AND the STEP 1b audit-dir fallback and
|
|
2115
|
-
// parking on the exception path.) Fail-closed still applies
|
|
2116
|
-
// after chunk 3 and the fallback are exhausted.
|
|
1929
|
+
// See docs/decisions/publish-path.md#hung-chunk: a hung or failed chunk is inconclusive, never terminal.
|
|
2117
1930
|
chunkFailures.push("chunk " + chunk + ": " + (chunkErr && chunkErr.message ? chunkErr.message : chunkErr));
|
|
2118
1931
|
log("Artifact build poll chunk " + chunk + " of 3 failed (" + (chunkErr && chunkErr.message ? chunkErr.message : chunkErr) + ") \u2014 continuing to the next chunk; build completion still unproven.");
|
|
2119
1932
|
}
|
|
@@ -2129,20 +1942,7 @@ while (i < STEPS.length) {
|
|
|
2129
1942
|
buildPoll = { build_done: false, status: (buildPoll && buildPoll.status) || "build still running after the 10.5-minute bounded poll" };
|
|
2130
1943
|
}
|
|
2131
1944
|
if (buildPoll.build_done && pollSawOurBuild) {
|
|
2132
|
-
//
|
|
2133
|
-
// (2026-09-11) proved the stamp cannot certify content: the
|
|
2134
|
-
// builder's applied-report was derived from the carried diff, so
|
|
2135
|
-
// the old report check was circular — a fabricated report
|
|
2136
|
-
// passed by construction, and every phase went green on a hollow
|
|
2137
|
-
// build. The stamp moves to the parent (docs/publish-verification.md);
|
|
2138
|
-
// the deterministic lib/readback-disk.js is the primary sensor
|
|
2139
|
-
// (the agent-callable read-back tool is unavailable —
|
|
2140
|
-
// artifact_inspect was removed by the platform 2026-09-14 — so
|
|
2141
|
-
// the LLM-inspector path is manual-fallback only), and the task
|
|
2142
|
-
// parks for parent verification.
|
|
2143
|
-
// QA's provenance check enforces the stamp mechanically.
|
|
2144
|
-
// An unverified publish fails loudly in QA instead of passing
|
|
2145
|
-
// silently here.
|
|
1945
|
+
// See docs/decisions/publish-path.md#step-1c-no-stamp: no provenance stamp in STEP 1c.
|
|
2146
1946
|
publishBuildLanded = true;
|
|
2147
1947
|
artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
|
|
2148
1948
|
log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
|
|
@@ -2221,11 +2021,16 @@ while (i < STEPS.length) {
|
|
|
2221
2021
|
attempt: rebuildAttemptKey,
|
|
2222
2022
|
agent_id: rebuildAgentId,
|
|
2223
2023
|
applied_report: publishAppliedObservation,
|
|
2224
|
-
|
|
2024
|
+
manifest_before: preTriggerManifest,
|
|
2025
|
+
outcome: "build-observed",
|
|
2225
2026
|
detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
|
|
2226
2027
|
}, totalReworkCount);
|
|
2227
2028
|
} else if (auditOkAfterPoll === false) {
|
|
2228
|
-
|
|
2029
|
+
// Explicit negative evidence under the stranger guard: a build
|
|
2030
|
+
// ran and failed during the poll window with no stranger in
|
|
2031
|
+
// flight. This is an attributable failure — preserved verbatim
|
|
2032
|
+
// through post-deploy. Message contract unchanged.
|
|
2033
|
+
publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestAuditDirAfterPoll + ", report ok=false). Explicit negative evidence: a build ran and failed (attribution by window, not by build identity — no stranger build was observed in flight during the poll). The publish did not land — provenance was not stamped — your change was NOT published. Build log: ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestAuditDirAfterPoll + "/. This explicit failure is not auto-retried. Appendix: attribution=window; report=ok=false; provenance=unstamped.";
|
|
2229
2034
|
await recordPublishLedger({
|
|
2230
2035
|
commit: mergeCommitForPublish,
|
|
2231
2036
|
attempt: rebuildAttemptKey,
|
|
@@ -2239,7 +2044,19 @@ while (i < STEPS.length) {
|
|
|
2239
2044
|
: ((pollStatusErrors > 0 && !pollSawOurBuild) ? "status-read-errors-during-poll"
|
|
2240
2045
|
: (pollEndState === "build-still-running-at-poll-end" ? "build-still-running-at-poll-end"
|
|
2241
2046
|
: (newAuditDirsAfterPoll.length === 0 ? "no-new-audit-dir-in-window" : "audit-report-unreadable-or-missing")));
|
|
2242
|
-
|
|
2047
|
+
// Verdict UNKNOWN (poll path): the build cannot be attributed
|
|
2048
|
+
// to this attempt. The fields feed the single composed
|
|
2049
|
+
// fail-closed park reason after post-deploy — never a
|
|
2050
|
+
// fabricated verdict, never a silent pass.
|
|
2051
|
+
publishUnknownFields = {
|
|
2052
|
+
taskId: taskId,
|
|
2053
|
+
unattributableReason: unattributableReason,
|
|
2054
|
+
pollEndState: pollEndState,
|
|
2055
|
+
sawOurBuild: pollSawOurBuild,
|
|
2056
|
+
newAuditDirCount: newAuditDirsAfterPoll.length,
|
|
2057
|
+
chunkFailures: chunkFailures.length,
|
|
2058
|
+
pollStatusErrors: pollStatusErrors
|
|
2059
|
+
};
|
|
2243
2060
|
await recordPublishLedger({
|
|
2244
2061
|
commit: mergeCommitForPublish,
|
|
2245
2062
|
attempt: rebuildAttemptKey,
|
|
@@ -2250,11 +2067,8 @@ while (i < STEPS.length) {
|
|
|
2250
2067
|
}, totalReworkCount);
|
|
2251
2068
|
}
|
|
2252
2069
|
}
|
|
2253
|
-
} else {
|
|
2254
|
-
// Unreachable: the observation above either attributes the edit
|
|
2255
|
-
// (edit_started) or parks. Defensive only — never a silent pass.
|
|
2256
|
-
publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish is unattributed (not proven landed, not proven failed) — provenance was not stamped. Fail-closed.";
|
|
2257
2070
|
}
|
|
2071
|
+
|
|
2258
2072
|
} // end: publishSkippedNoLock — no rebuild, no stamp, nothing to ship
|
|
2259
2073
|
// STEP 2 (mechanical, always — skip path included): post-deploy
|
|
2260
2074
|
// commits builder leftovers if any, removes the worktree, and releases
|
|
@@ -2278,6 +2092,19 @@ while (i < STEPS.length) {
|
|
|
2278
2092
|
? " Post-deploy finalized cleanup."
|
|
2279
2093
|
: " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
|
|
2280
2094
|
}
|
|
2095
|
+
if (!publishBuildLanded) {
|
|
2096
|
+
// Verdict UNKNOWN (single fail-closed park — post-deploy always
|
|
2097
|
+
// runs first; no early parks anywhere above). Machine contract:
|
|
2098
|
+
// "unattributable_reason=" and "poll_end_state=" are always
|
|
2099
|
+
// present; attribution is by window, not identity.
|
|
2100
|
+
// (2026-09-18, H2 message contract) Thread the commit short-sha
|
|
2101
|
+
// through the unknown fields so the human line names the trigger
|
|
2102
|
+
// commit (design §1.5: "the trigger was sent for commit <short-sha>").
|
|
2103
|
+
if (publishUnknownFields) publishUnknownFields.commitShortSha = mergeCommitShortForPublish;
|
|
2104
|
+
return await parkTask(composeUnattributedParkReason(publishUnknownFields) + (postDeploy.deployed
|
|
2105
|
+
? " Post-deploy finalized cleanup."
|
|
2106
|
+
: " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
|
|
2107
|
+
}
|
|
2281
2108
|
if (!postDeploy.deployed) {
|
|
2282
2109
|
return await parkTask("Post-deploy failed after the artifact build landed: " + (postDeploy.output || "no output") + ". The build may have landed but worktree cleanup and lock release are unknown — human attention needed.");
|
|
2283
2110
|
}
|
|
@@ -2377,12 +2204,7 @@ while (i < STEPS.length) {
|
|
|
2377
2204
|
"BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
|
|
2378
2205
|
"Report back in plain prose — what checks you ran and their results. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<what the task required>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten).";
|
|
2379
2206
|
} else if (SURFACE_TERMINAL) {
|
|
2380
|
-
//
|
|
2381
|
-
// the terminal counterpart to the see-act loop above. Same OODA
|
|
2382
|
-
// discipline (append-ooda-step with --action terminal and a transcript
|
|
2383
|
-
// per step; verdict via write-ooda-verdict), judged against the
|
|
2384
|
-
// shared bar resolved via UX_DOCTRINE_PATH (the terminal doctrine page here). Transcripts are delivered;
|
|
2385
|
-
// screenshots are never invented for terminal work.
|
|
2207
|
+
// See docs/decisions/qa-reproduce.md#terminal-qa: Hazel drives CLI transcripts for terminal surfaces.
|
|
2386
2208
|
var termEvidence = crewHome + "/task-evidence/" + taskId + "/postchange";
|
|
2387
2209
|
var termTargetsLine = terminalTargets || "not declared — derive from --help and the task description";
|
|
2388
2210
|
instructions = "You are code-blind QA. You NEVER read source files.\n" +
|
|
@@ -2606,18 +2428,7 @@ while (i < STEPS.length) {
|
|
|
2606
2428
|
}
|
|
2607
2429
|
log("Build worktree confinement passed: " + wt.path);
|
|
2608
2430
|
|
|
2609
|
-
//
|
|
2610
|
-
// <sha>)` declaration is verified mechanically — <sha> must resolve
|
|
2611
|
-
// and be an ancestor of main in the configured repo. A fabricated or
|
|
2612
|
-
// mistaken declaration fails the phase here (the dispatcher retries
|
|
2613
|
-
// Build under its consecutive-failure cap); a verified declaration is
|
|
2614
|
-
// recorded in alreadyMergedSha for Review's no-diff branch. Without
|
|
2615
|
-
// this guard, Build correctly doing nothing left Review with no
|
|
2616
|
-
// mechanical way to accept an empty diff, and Cass rejected for "no
|
|
2617
|
-
// commits ahead of main — the builder likely forgot to commit" while
|
|
2618
|
-
// the deliverable sat on main (canary 2026-09-15, task 1d692d91).
|
|
2619
|
-
// The sha is hex-only by construction (extractAlreadyMerged), so
|
|
2620
|
-
// interpolating it into the shell command cannot inject.
|
|
2431
|
+
// See docs/decisions/publish-path.md#already-merged-idem: an already-merged repo_diff is idempotent; no rebuild.
|
|
2621
2432
|
var am = extractAlreadyMerged(workerText);
|
|
2622
2433
|
if (am.sha) {
|
|
2623
2434
|
var amCheck = await agent(
|
|
@@ -2653,19 +2464,7 @@ while (i < STEPS.length) {
|
|
|
2653
2464
|
}
|
|
2654
2465
|
}
|
|
2655
2466
|
|
|
2656
|
-
//
|
|
2657
|
-
// is honest about missing experiential evidence, but the closeout treated a PASS
|
|
2658
|
-
// as terminal done even when the experiential loop never ran (playwright-core
|
|
2659
|
-
// was unresolvable from the release layout — the dependency lived in the
|
|
2660
|
-
// npm install dir, severed from the crew home). A PASS verdict with missing
|
|
2661
|
-
// experiential evidence must never be terminal: the task parks fail-closed
|
|
2662
|
-
// with unattributable_reason=qa-visual-loop-unavailable (artifact surface)
|
|
2663
|
-
// or qa-terminal-loop-unavailable (terminal surface) instead of
|
|
2664
|
-
// transitioning to done. Code, not prompt text: the check reads the
|
|
2665
|
-
// machine-readable verdict via lib/read-ooda-verdict.js, which reports
|
|
2666
|
-
// visual_loop_unavailable from the OODA log's NOT POSSIBLE browser steps
|
|
2667
|
-
// and terminal_loop_unavailable from NOT POSSIBLE terminal steps, plus the
|
|
2668
|
-
// verdict's missing_evidence tool-unavailability notes.
|
|
2467
|
+
// See docs/decisions/qa-reproduce.md#experiential-loop-guard: a PASS with missing experiential evidence parks fail-closed.
|
|
2669
2468
|
if (step.name === "QA" && qaExperiential && verdictPassed === true) {
|
|
2670
2469
|
var qaLoopDir = crewHome + "/task-evidence/" + taskId + "/postchange";
|
|
2671
2470
|
var qaLoopSurface = SURFACE_TERMINAL ? "terminal" : "visual";
|
|
@@ -2717,20 +2516,7 @@ while (i < STEPS.length) {
|
|
|
2717
2516
|
};
|
|
2718
2517
|
let passed = stepResult.passed === true;
|
|
2719
2518
|
|
|
2720
|
-
//
|
|
2721
|
-
// merge. After the Integrate agent claims success, the workflow confirms
|
|
2722
|
-
// mechanically that the task branch tip is an ancestor of main via the
|
|
2723
|
-
// lifecycle script's verify-merge command (which resolves the branch
|
|
2724
|
-
// through the crew registry — never by reconstructing "task/"+taskId — so
|
|
2725
|
-
// the check cannot verify the wrong branch). The VERIFIED marker is matched
|
|
2726
|
-
// by regex on the script's own stdout; agent prose is never read. This
|
|
2727
|
-
// closes the hole where an agent reported "merged empty" while approved
|
|
2728
|
-
// commits were still stranded on the task branch (bug b1b1f919). A genuine
|
|
2729
|
-
// empty-diff Integrate (MERGED_EMPTY: no commits ahead of main) verifies
|
|
2730
|
-
// vacuously — the tip is then an ancestor of main. Verification failure is
|
|
2731
|
-
// an operational step failure, not a park: the dispatcher retries Integrate
|
|
2732
|
-
// under its consecutive-failure cap, and the retry finds the commits still
|
|
2733
|
-
// on the branch and performs the real merge — self-healing.
|
|
2519
|
+
// See docs/decisions/publish-path.md#integrate-verify: the agent cannot verify integrate mechanically; the workflow checks the diff.
|
|
2734
2520
|
if (step.name === "Integrate" && passed) {
|
|
2735
2521
|
var integrateVerifyOut = "";
|
|
2736
2522
|
try {
|
|
@@ -2791,17 +2577,7 @@ while (i < STEPS.length) {
|
|
|
2791
2577
|
// dispatcher's consecutive-failure cap.
|
|
2792
2578
|
const status = passed ? "completed" : (step.name === "Integrate" || step.name === "Publish" || step.name === "Map" ? "failed" : "rejected");
|
|
2793
2579
|
|
|
2794
|
-
//
|
|
2795
|
-
// Skip-aware (park 2026-09-11): when the deterministic publish script found
|
|
2796
|
-
// no merge lock held (empty-diff Integrate), it skips the publish path
|
|
2797
|
-
// gracefully and emits the machine-readable PUBLISH_SKIPPED=no-lock-held
|
|
2798
|
-
// marker. The preflight (bugfix 2026-09-17) emits
|
|
2799
|
-
// PUBLISH_SKIPPED=no-npm-publish when npm publish is not configured on
|
|
2800
|
-
// this machine (helper or credential absent) — also before any mutation.
|
|
2801
|
-
// Verification is then vacuous — nothing was shipped, and the
|
|
2802
|
-
// registry must NOT have moved. The marker is script-emitted explicit state
|
|
2803
|
-
// (pasted verbatim per the Publish agent instructions), not agent prose; a
|
|
2804
|
-
// report without the marker still runs the full verification fail-closed.
|
|
2580
|
+
// See docs/decisions/publish-path.md#publish-verify: the agent cannot verify publish; the parent does.
|
|
2805
2581
|
var publishVerified = false;
|
|
2806
2582
|
var npmPublishSkipped = false;
|
|
2807
2583
|
if (step.name === "Publish" && PUBLISH_TYPE === "npm" && publishTarget) {
|
|
@@ -2827,15 +2603,7 @@ while (i < STEPS.length) {
|
|
|
2827
2603
|
publishTarget.target + " (" + publishTarget.base + " + " + publishTarget.scope + "). The publish did not land.");
|
|
2828
2604
|
}
|
|
2829
2605
|
publishVerified = true;
|
|
2830
|
-
//
|
|
2831
|
-
// installs and activates a new immutable release (crew-release.sh deploy
|
|
2832
|
-
// swaps the `current` symlink inside publish-npm.sh) but never stamped
|
|
2833
|
-
// the dashboard's provenance record — every crew release left
|
|
2834
|
-
// crew_release pointing at a pruned release. After a verified landed
|
|
2835
|
-
// publish, refresh the record's crew_release to the now-live release
|
|
2836
|
-
// identity, preserving the existing source_commit (the dashboard
|
|
2837
|
-
// artifact's build source — a crew-repo commit here would fail the
|
|
2838
|
-
// dashboard QA source check).
|
|
2606
|
+
// See docs/decisions/publish-path.md#provenance-refresh: refresh crew_release after landed publish, preserving source_commit.
|
|
2839
2607
|
try {
|
|
2840
2608
|
var provRefresh = await agent(
|
|
2841
2609
|
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-provenance", { project_id: LAUNCH_PROJECT_ID }) + "\n" +
|
|
@@ -2865,29 +2633,10 @@ while (i < STEPS.length) {
|
|
|
2865
2633
|
} // end: !npmPublishSkipped — a skipped publish has nothing to verify
|
|
2866
2634
|
}
|
|
2867
2635
|
|
|
2868
|
-
//
|
|
2869
|
-
//
|
|
2870
|
-
//
|
|
2871
|
-
//
|
|
2872
|
-
// artifact was stale, all eight phases green. The stamp now moves to the
|
|
2873
|
-
// parent (docs/publish-verification.md); the independent read-back step
|
|
2874
|
-
// is currently unavailable (no agent-callable read-back tool exists —
|
|
2875
|
-
// artifact_inspect was removed by the platform 2026-09-14), so the parent
|
|
2876
|
-
// cannot confirm content and the task parks for verification. QA's
|
|
2877
|
-
// provenance check enforces the stamp — an unverified publish fails loudly
|
|
2878
|
-
// there instead of passing silently here.
|
|
2879
|
-
// Skip-aware (park 2026-09-11): an empty-diff Integrate takes no merge
|
|
2880
|
-
// lock, and the deterministic publish path skips rebuild/stamp entirely —
|
|
2881
|
-
// there is no new content to verify, so verification is vacuous.
|
|
2882
|
-
// publishSkippedNoLock is workflow-computed state from the explicit
|
|
2883
|
-
// lock-status read in STEP 0, not agent prose.
|
|
2884
|
-
// The parent (tick worker) triggers the ONE read-back inspection it can
|
|
2885
|
-
// actually receive (async results go to the root agent, never into a
|
|
2886
|
-
// workflow run — a workflow-side trigger would be an orphan). The workflow
|
|
2887
|
-
// only parks; the parent's scan builds the request deterministically via
|
|
2888
|
-
// lib/build-readback-request.js and ferries the inspection.
|
|
2889
|
-
// publishBuildLanded and publishSkippedNoLock are workflow-computed state;
|
|
2890
|
-
// a skipped or failed publish has nothing to verify.
|
|
2636
|
+
// Parent-owned verification (docs/publish-verification.md): the parent's
|
|
2637
|
+
// scan builds the request via lib/build-readback-request.js and ferries
|
|
2638
|
+
// the inspection; a skipped or failed publish has nothing to verify.
|
|
2639
|
+
// See docs/decisions/publish-path.md#parent-owned-verification for history.
|
|
2891
2640
|
|
|
2892
2641
|
// Session notes. Machine-readable marker lines are extracted from the full
|
|
2893
2642
|
// worker report and appended AFTER the slice so a long report can never
|
|
@@ -2906,12 +2655,7 @@ while (i < STEPS.length) {
|
|
|
2906
2655
|
} else {
|
|
2907
2656
|
summary = (stepResult.summary || "Step completed").slice(0, 2000 - workerMarkers.length - 1) + (workerMarkers ? "\n" + workerMarkers : "");
|
|
2908
2657
|
}
|
|
2909
|
-
//
|
|
2910
|
-
// already-merged declaration, the workflow records its own marker line in
|
|
2911
|
-
// the session notes (like the builder markers above, it is appended after
|
|
2912
|
-
// the slice so it can never be amputated). A later run resumed at Review
|
|
2913
|
-
// hydrates alreadyMergedSha from this workflow-attested line — never from
|
|
2914
|
-
// the builder's declaration alone.
|
|
2658
|
+
// See docs/decisions/publish-path.md#already-merged: when the Build gate verifies already-merged, attestation is recorded.
|
|
2915
2659
|
if (step.name === "Build" && alreadyMergedSha) {
|
|
2916
2660
|
summary += "\nalready_merged_verified: " + alreadyMergedSha;
|
|
2917
2661
|
}
|
|
@@ -3008,18 +2752,18 @@ while (i < STEPS.length) {
|
|
|
3008
2752
|
return { status: "failed", task_id: taskId, reason: "Map spec verification failed: no spec file at " + SPEC_PATH };
|
|
3009
2753
|
}
|
|
3010
2754
|
|
|
3011
|
-
//
|
|
3012
|
-
// but provenance is UNSTAMPED until the parent's independent read-back
|
|
3013
|
-
// (docs/publish-verification.md) confirms the artifact's actual content
|
|
3014
|
-
// matches the merged diff. The parent stamps provenance, then re-queues;
|
|
3015
|
-
// the dispatcher resumes at QA, whose provenance check enforces the stamp
|
|
3016
|
-
// mechanically. A failed Publish never reaches this park — it returned
|
|
3017
|
-
// failed above and retries under the dispatcher's cap. The merge lock is
|
|
3018
|
-
// already released (post-deploy), so the parked task holds no resources.
|
|
2755
|
+
// See docs/decisions/publish-path.md#verification-park: the build landed but provenance is unstamped until parent verification.
|
|
3019
2756
|
if (passed && step.name === "Publish" && PUBLISH_TYPE === "artifact" && PUBLISH_SLUG && !publishSkippedNoLock && publishBuildLanded) {
|
|
2757
|
+
// Success contract (2026-09-18, H2): the parent asked for exactly
|
|
2758
|
+
// one build for this publish. The build landed — do NOT republish: a
|
|
2759
|
+
// duplicate build would re-publish the same change. content_check=pending
|
|
2760
|
+
// means the parent's independent read-back has not happened yet; this
|
|
2761
|
+
// park is NOT proof the content is correct.
|
|
3020
2762
|
return await parkTask("publish: verification-requested " + mergeCommitForPublish +
|
|
3021
|
-
"
|
|
3022
|
-
"
|
|
2763
|
+
" attempt=" + rebuildAttemptKey +
|
|
2764
|
+
" Do NOT republish: a duplicate build would re-publish the same change. " +
|
|
2765
|
+
"Artifact build landed, post-deploy finalized, provenance not stamped — the crew has not yet independently confirmed the live artifact contains exactly the change; waiting on the manual read-back in docs/publish-verification.md. " +
|
|
2766
|
+
"Appendix: build=" + (rebuildAgentId || "agent_id unobserved") + "; provenance=unstamped; content_check=pending.");
|
|
3023
2767
|
}
|
|
3024
2768
|
|
|
3025
2769
|
i++;
|