muse-crew 0.13.3 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -39,9 +39,18 @@
39
39
 
40
40
  import { execFileSync } from "node:child_process";
41
41
  import { readFileSync } from "node:fs";
42
+ import { homedir } from "node:os";
42
43
  import { join } from "node:path";
43
-
44
- const EMPTY_TREE = "4b825dc642cb6eb9a060e54bf8d69288fbee4904";
44
+ // Shared content-judgment primitives (2026-09-18, blocker 15): the same
45
+ // diff parser, findings parser, and collision-exemption rules the
46
+ // unknown-recovery classifier uses — one judgment, never two copies.
47
+ import {
48
+ EMPTY_TREE,
49
+ parseDiff,
50
+ parseFindings,
51
+ makeOldCounter,
52
+ discriminatingLines,
53
+ } from "./publish-content.js";
45
54
 
46
55
  function arg(name, required = true) {
47
56
  const i = process.argv.indexOf(name);
@@ -67,6 +76,8 @@ const resultFile = arg("--result-file");
67
76
  const crewRelease = arg("--crew-release");
68
77
  const buildAgentId = arg("--build-agent-id", false);
69
78
  const projectId = arg("--project-id");
79
+ const attempt = arg("--attempt", false); // 2026-09-18, blocker 4: exact attempt binding
80
+ const spacesRoot = arg("--spaces-root", false) || join(homedir(), "workspace", "ts-spaces");
70
81
 
71
82
  if (!/^[0-9a-f]{40}$/.test(commit)) fail("usage", "commit must be a 40-char hex sha.", 2);
72
83
  if (!/^[0-9a-f]{40}$/.test(base)) fail("usage", "base must be a 40-char hex sha.", 2);
@@ -135,33 +146,7 @@ try {
135
146
  // ADDED: <line> :: PRESENT|ABSENT
136
147
  // REMOVED: <line> :: PRESENT|ABSENT
137
148
  // END_FILE
138
- function parseFindings(text) {
139
- const findings = new Map(); // path -> { added: Map(line->verdict), removed: Map(line->verdict) }
140
- let cur = null;
141
- let malformed = null;
142
- for (const rawLine of text.split("\n")) {
143
- const line = rawLine.trimEnd();
144
- if (line.startsWith("FILE: ")) {
145
- cur = { added: new Map(), removed: new Map() };
146
- findings.set(line.slice(6).trim(), cur);
147
- } else if (line === "END_FILE") {
148
- cur = null;
149
- } else if (cur && (line.startsWith("ADDED: ") || line.startsWith("REMOVED: "))) {
150
- const kind = line.startsWith("ADDED: ") ? "added" : "removed";
151
- const rest = line.slice(kind === "added" ? 7 : 9);
152
- const sep = rest.lastIndexOf(" :: ");
153
- if (sep < 0) { malformed = `malformed finding line: ${line.slice(0, 80)}`; break; }
154
- const content = rest.slice(0, sep);
155
- const verdict = rest.slice(sep + 4).trim();
156
- if (verdict !== "PRESENT" && verdict !== "ABSENT") {
157
- malformed = `bad verdict: ${verdict}`;
158
- break;
159
- }
160
- cur[kind].set(content, verdict);
161
- }
162
- }
163
- return { findings, malformed };
164
- }
149
+ // (parsed by the shared parseFindings in lib/publish-content.js)
165
150
 
166
151
  // --- 2. Expected diff from git (never from the builder's report) -----------
167
152
  // The expected change is the publish delta base..commit, using the SAME
@@ -183,21 +168,7 @@ try {
183
168
  } catch (e) {
184
169
  terminal("read-back-unavailable", `git diff failed: ${e.message}`);
185
170
  }
186
- const expected = new Map(); // path -> { added: [], removed: [], isBinary: bool }
187
- let curFile = null;
188
- for (const line of diff.split("\n")) {
189
- if (line.startsWith("diff --git")) {
190
- const m = line.match(/^diff --git a\/(.+) b\/(.+)$/);
191
- curFile = m ? m[2] : "unknown";
192
- expected.set(curFile, { added: [], removed: [], isBinary: false });
193
- } else if (curFile && line.startsWith("Binary files ")) {
194
- expected.get(curFile).isBinary = true;
195
- } else if (curFile && line.startsWith("+") && !line.startsWith("+++")) {
196
- expected.get(curFile).added.push(line.slice(1));
197
- } else if (curFile && line.startsWith("-") && !line.startsWith("---")) {
198
- expected.get(curFile).removed.push(line.slice(1));
199
- }
200
- }
171
+ const expected = parseDiff(diff); // path -> { added: [], removed: [], isBinary: bool }
201
172
 
202
173
  // --- 3. Pick the findings candidate that covers the expected diff -----------
203
174
  let findings = null;
@@ -232,33 +203,30 @@ if (!findings || bestScore <= 0) {
232
203
  // removal into a pass. No change to the sensor: build-readback-request.js
233
204
  // still reports whole-file PRESENT/ABSENT honestly; only this judge gets
234
205
  // smarter.
235
- function oldTreeLines(path) {
236
- // Exact whole-line contents of <path> at <base>; [] when the base is the
237
- // empty tree or the file did not exist there (a new file has no removed
238
- // lines, so the exemption is vacuous for it).
239
- if (base === EMPTY_TREE) return [];
240
- let text;
241
- try {
242
- text = execFileSync("git", ["-C", repoPath, "show", `${base}:${path}`], {
243
- encoding: "utf8", maxBuffer: 4 * 1024 * 1024,
244
- });
245
- } catch {
246
- return [];
247
- }
248
- if (text === "") return [];
249
- const lines = text.split("\n");
250
- if (lines[lines.length - 1] === "") lines.pop(); // drop the trailing-newline artifact
251
- return lines;
252
- }
253
- const oldTreeCounts = new Map(); // path -> Map(line -> occurrences in old tree)
254
- function oldCount(path, line) {
255
- let counts = oldTreeCounts.get(path);
256
- if (counts === undefined) {
257
- counts = new Map();
258
- for (const l of oldTreeLines(path)) counts.set(l, (counts.get(l) || 0) + 1);
259
- oldTreeCounts.set(path, counts);
260
- }
261
- return counts.get(line) || 0;
206
+ // Collision exemption (2026-09-15, task 00bca4b8) — decided once by the
207
+ // shared lib/publish-content.js so the verifier and the unknown-recovery
208
+ // classifier can never disagree about what a diff proves:
209
+ //
210
+ // - a removed diff line that also occurs verbatim in untouched code has
211
+ // zero discriminating power — its presence in the new file cannot tell
212
+ // "old block removed" from "old block present";
213
+ // - an added line that already occurs in the old tree is reported PRESENT
214
+ // whether or not the new hunk actually landed — same zero power.
215
+ //
216
+ // Exempting a line can never turn a missed change into a pass: an exempted
217
+ // line must survive in untouched code no matter what. No change to the
218
+ // sensor: build-readback-request.js still reports whole-file PRESENT/ABSENT
219
+ // honestly; only this judge gets smarter. The occurrence counts are
220
+ // computed from `git show <base>:<path>` (never the working tree) by the
221
+ // shared makeOldCounter; the diff may remove the same line more than once
222
+ // and the budget accounting lives in the shared discriminatingLines.
223
+ const oldCount = makeOldCounter(repoPath, base);
224
+ const disc = discriminatingLines(expected, oldCount);
225
+ const discAdded = new Map(); // path -> Set(line): discriminating added lines
226
+ const discRemoved = new Map(); // path -> Set(line): discriminating removed lines
227
+ for (const [p, d] of disc) {
228
+ discAdded.set(p, new Set(d.addedDisc));
229
+ discRemoved.set(p, new Set(d.removedDisc));
262
230
  }
263
231
 
264
232
  let exempted = 0;
@@ -277,18 +245,14 @@ for (const [path, exp] of expected) {
277
245
  "the diff carries no added/removed lines for this file, so the line-based read-back checked nothing — " +
278
246
  "provenance NOT stamped; human verification needed");
279
247
  }
280
- // (2026-09-16, critic finding 5) Added-line collisions: an added line
281
- // that already occurs verbatim in the old tree is reported PRESENT
282
- // whether or not the new hunk actually landed — zero discriminating
283
- // power (the sensor is membership-only, not count-sensitive). Exempt
284
- // such lines from the pass criteria, symmetric to the removed-side
285
- // exemption below — and require at least one discriminating added line
286
- // per file, or the added-side check is vacuous (finding 1's class).
248
+ // (2026-09-16, critic finding 5) Added-line collisions: require at least
249
+ // one discriminating added line per file, or the added-side check is
250
+ // vacuous (finding 1's class).
287
251
  let discriminatingAdded = 0;
288
252
  for (const line of exp.added) {
289
253
  const v = found.added.get(line);
290
254
  if (v === undefined) terminal("unreadable-result", `no ADDED finding for line in ${path}: ${line.slice(0, 60)}`);
291
- if (oldCount(path, line) > 0) {
255
+ if (!discAdded.get(path).has(line)) {
292
256
  exempted += 1; // colliding pre-existing line: zero signal, cannot fail a good publish
293
257
  continue;
294
258
  }
@@ -301,15 +265,11 @@ for (const [path, exp] of expected) {
301
265
  "their PRESENT findings cannot tell \"hunk landed\" from \"hunk dropped\" — " +
302
266
  "provenance NOT stamped; human verification needed");
303
267
  }
304
- // The diff may remove the same line more than once; the old tree must
305
- // account for every removal before a line counts as a collision.
306
- const removedBudget = new Map();
307
- for (const line of exp.removed) removedBudget.set(line, (removedBudget.get(line) || 0) + 1);
308
268
  for (const line of exp.removed) {
309
269
  const v = found.removed.get(line);
310
270
  if (v === undefined) terminal("unreadable-result", `no REMOVED finding for line in ${path}: ${line.slice(0, 60)}`);
311
271
  if (v === "ABSENT") continue;
312
- if (oldCount(path, line) > removedBudget.get(line)) {
272
+ if (!discRemoved.get(path).has(line)) {
313
273
  exempted += 1; // colliding pre-existing line: zero signal, cannot fail a good publish
314
274
  continue;
315
275
  }
@@ -328,6 +288,63 @@ if (head !== commit) {
328
288
  terminal("superseded", `HEAD is ${head}, not ${commit} — a newer publish supersedes this one`);
329
289
  }
330
290
 
291
+ // --- 5b. Manifest freshness (design §1.9) ------------------------------------
292
+ // The stamp certifies that a NEW build landed for this attempt — not a
293
+ // replayed manifest. The workflow snapshots the pre-trigger manifest into
294
+ // the submitted ledger entry; require (a) the current manifest's built_at
295
+ // advanced past the trigger, and (b) its content_sha256 differs from the
296
+ // pre-trigger baseline (a new build identity). A missing baseline fails
297
+ // closed — without it, freshness cannot be proven.
298
+ // 2026-09-18, blocker 4: bind by exact (task_id, commit, attempt) when the
299
+ // attempt is known — the original and retry attempts share a commit, and
300
+ // positional selection is luck. Within one attempt, select the OLDEST
301
+ // submitted entry (the trigger issuance; per blocker 2, audit observations
302
+ // are now "build-observed", not "submitted").
303
+ let triggerTs = null;
304
+ let manifestBefore = null;
305
+ try {
306
+ const ledgerText = readFileSync(join(crewHome, ".publish-ledger", slug + ".jsonl"), "utf8");
307
+ const entries = ledgerText.split("\n").filter((l) => l.trim()).map((l) => JSON.parse(l));
308
+ // Collect all submitted entries matching (task_id, commit[, attempt]).
309
+ const candidates = entries.filter((e) =>
310
+ e.task_id === taskId &&
311
+ e.outcome === "submitted" &&
312
+ e.commit === commit &&
313
+ (attempt == null || e.attempt === attempt)
314
+ );
315
+ if (candidates.length > 0) {
316
+ // Oldest first: the trigger issuance is the first submitted entry for
317
+ // the attempt (observations are "build-observed" per blocker 2).
318
+ candidates.sort((a, b) => (a.ts || "").localeCompare(b.ts || ""));
319
+ const e = candidates[0];
320
+ triggerTs = e.ts || null;
321
+ if (e.manifest_before && typeof e.manifest_before === "object") manifestBefore = e.manifest_before;
322
+ }
323
+ } catch (e) {
324
+ terminal("read-back-unavailable", `cannot read publish ledger for freshness check: ${e.message}`);
325
+ }
326
+ if (!triggerTs) {
327
+ terminal("read-back-unavailable", `no submitted ledger entry for commit ${commit} — the trigger instant is unknowable; freshness cannot be proven`);
328
+ }
329
+ if (!manifestBefore || typeof manifestBefore.content_sha256 !== "string") {
330
+ terminal("unverifiable-content", `no pre-trigger manifest baseline in the submitted ledger entry — freshness cannot be proven; provenance NOT stamped; human verification needed`);
331
+ }
332
+ let currentManifest = null;
333
+ try {
334
+ const manifestText = readFileSync(join(spacesRoot, slug, ".space-build", "manifest.json"), "utf8");
335
+ currentManifest = JSON.parse(manifestText);
336
+ } catch (e) {
337
+ terminal("read-back-unavailable", `cannot read current manifest for freshness check: ${e.message}`);
338
+ }
339
+ const currentBuiltAtMs = currentManifest && currentManifest.built_at ? Date.parse(currentManifest.built_at) : NaN;
340
+ const triggerMs = Date.parse(triggerTs);
341
+ if (!(currentBuiltAtMs > triggerMs)) {
342
+ terminal("unverifiable-content", `current manifest built_at ${currentManifest && currentManifest.built_at} did not advance past trigger ${triggerTs} — no new build for this attempt; provenance NOT stamped`);
343
+ }
344
+ if (typeof currentManifest.content_sha256 !== "string" || currentManifest.content_sha256 === manifestBefore.content_sha256) {
345
+ terminal("unverifiable-content", `current manifest content_sha256 is unchanged from the pre-trigger baseline — a replayed manifest, not a new build; provenance NOT stamped; human verification needed`);
346
+ }
347
+
331
348
  // --- 6. Build-ID correlation (informational; the stamp certifies content) ---
332
349
  let buildNote = "build-id-unobserved";
333
350
  if (buildAgentId && resultText.includes(buildAgentId)) buildNote = `build-id-correlated ${buildAgentId}`;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "muse-crew",
3
- "version": "0.13.3",
4
- "description": "Opinionated orchestration for Muse \u2014 workflows, identities, and tooling for autonomous software development.",
3
+ "version": "0.14.0",
4
+ "description": "Opinionated orchestration for Muse — workflows, identities, and tooling for autonomous software development.",
5
5
  "license": "UNLICENSED",
6
6
  "private": false,
7
7
  "repository": {
@@ -35,7 +35,62 @@ You are the dispatch trigger for Muse Crew. Run the authoritative dispatcher wor
35
35
  - If the acknowledge fails (no reservation exists), DO NOT LAUNCH — the dispatcher did not acquire this task. This is a safety invariant.
36
36
  - The launched workflow self-claims the task and clears the reservation as its first actions. If the task was already claimed or is done, the claim fails closed and the run stands down quietly — this is the mechanical duplicate protection, not an error.
37
37
 
38
- If the dispatcher returned no claims or the claims array is empty, log NO_DISPATCH and CONTINUE to Step 4.5 — do NOT exit. Verification (4.5) and evidence (6) run independently of dispatch claims. A no-claims tick must still verify parked publishes and deliver evidence. (Fixed 2026-09-14: the old "exit on NO_DISPATCH" skipped verification permanently.)
38
+ If the dispatcher returned no claims or the claims array is empty, log NO_DISPATCH and CONTINUE to Step 4.4 — do NOT exit. Unknown-recovery (4.4), verification (4.5) and evidence (6) run independently of dispatch claims. A no-claims tick must still recover parked publishes, verify them, and deliver evidence. (Fixed 2026-09-14: the old "exit on NO_DISPATCH" skipped verification permanently.)
39
+
40
+ 4.4. **Unknown-recovery (docs/publish-unknown-recovery.md):** tasks parked with "Publish outcome unknown" are the receipt-less trigger gap (2026-09-18, blocker 15). The note is the state machine; deterministic code owns every transition; you are the ferry (scan → classifier → record / retry protocol). Run this BEFORE Step 4.5 so a re-triggered build's mirrored verification-requested is visible to the verification scan on a later tick. Log every list the scan returns.
41
+ - Scan (code): `node {crewHome}/lib/crew-api.js --crew-home {crewHome} scan-publish-unknown`
42
+ Returns `{ waiting: [...], due: [...], retry_due: [...], mirrored: [...], skipped: [...] }`. `waiting` = parked < 30m — leave alone, no claim. `mirrored`/`skipped` = already handled by code — log only, take no action.
43
+ - For each entry in `due`: classify (code), then record the decision (code). Use the entry's fields verbatim.
44
+ - `node {crewHome}/lib/classify-publish-absence.js --repo-path "<repo_path>" --commit <commit> --base <base> --slug "<slug>" --trigger-ts "<trigger_ts>" --park-ts "<park_ts>" --task-id <task_id> --manifest-before '<manifest_before JSON>'`
45
+ Serialize the entry's `manifest_before` field to compact JSON for the `--manifest-before` value; when the entry's `manifest_before` is null, omit the flag (the classifier falls back to the time-based advance check and the Step 4.5 verifier fails closed without a baseline).
46
+ Exit 0 prints the decision JSON (`{ ok:true, decision, reasons, details }`) — or `{ ok:false, ... }` for a semantic non-verdict. Exit non-zero (usage/git/IO failure) → log the stderr line and LEAVE THE TASK: do NOT record anything; the claim expires and the next tick re-claims. A classifier crash is never a verdict.
47
+ - `node {crewHome}/lib/crew-api.js --crew-home {crewHome} record-unknown-classification --json '{"task_id": "<task_id>", "claim_expiry": "<claim_expiry>", "decision": <classifier stdout JSON>}'`
48
+ Paste the classifier's stdout verbatim as the decision value. If it returns `recorded: false` → log the reason and continue (another tick owns the task). The routes are the code's — log the returned decision and `routed` value:
49
+ verified → `publish: verification-requested` (no re-trigger); provably-dropped → `publish: dropped` (queues the retry protocol); applied-not-built / ambiguous → `publish: ambiguous` (terminal); superseded → `publish: superseded` (terminal); deferred → nothing (the claim expires; the next scan re-claims).
50
+ - Never invent a decision. Never re-trigger the edit for a `due` entry — classification only.
51
+ - For each entry in `retry_due` (a provably-dropped edit with a free retry budget — exactly one retry per task, enforced from note history; the scan only emits `retry_due` when no `publish: retry-issued` exists in the task's history): run the deterministic retry protocol via `lib/retry-publish.js`. The command owns all lock handling, journaling, diff generation, ledger writes, and cleanup with try/finally lock release — you only mediate the unavoidable `artifact_edit` tool call.
52
+
53
+ Phase 1 — prepare (deterministic):
54
+ `node {crewHome}/lib/retry-publish.js --crew-home {crewHome} --task-id <task_id> --phase prepare`
55
+ - Exit 0 with `{"ok": true, "outcome": "ready", ...}`: the merge lock is HELD. The output carries `diff_path`, `diff_sha256`, and `slug`. Proceed to the trigger below.
56
+ - Exit 0 with `{"ok": true, "outcome": "superseded"|"lock-held"|"not-dropped", ...}`: terminal for this tick. Do NOT proceed to the trigger. If `outcome` is `not-dropped`, the output carries the full `classification` — pass it to `record-retry-recheck`: `node {crewHome}/lib/crew-api.js --crew-home {crewHome} record-retry-recheck --json '{"task_id": "<task_id>", "decision": <classification JSON>}'` and STOP for this task.
57
+ - Exit non-zero: the command failed (lock released via try/finally). Log the stderr line and STOP — do NOT trigger.
58
+
59
+ Trigger (agent-mediated — the ONLY prose-owned step):
60
+ Spawn ONE child (subagent) whose first instruction loads the artifact namespace (`tool_search.load_tool_namespace` with paths `["artifact"]`), with this exact brief — the same shape as the workflow's first attempt:
61
+ "The change to apply is the unified diff in the file \"<diff_path>\" (sha256 <diff_sha256>).
62
+ 1. Verify the file: run sha256sum on it. If the printed hash is not exactly <diff_sha256>, STOP and end your turn — do not call artifact_edit.
63
+ 2. Read the file's full content.
64
+ 3. Call artifact_edit with slug \"<slug>\" and verbatim_request:
65
+ 'Apply the following change to your source tree, then rebuild and deploy.
66
+
67
+ UNIFIED DIFF (relative to your source tree):
68
+ ```diff
69
+ <the full content of the verified file, pasted verbatim>
70
+ ```
71
+
72
+ Rules:
73
+ - For each file in the diff, apply its hunks to the same path in your source tree (use git apply or equivalent).
74
+ - For a new file (--- /dev/null), create it with the added (+) lines as its full content.
75
+ - For a deleted file (+++ /dev/null), delete it.
76
+ - If any hunk does not apply cleanly, STOP and report the failure — do not improvise or skip hunks.
77
+ - Do not make any other source changes.
78
+ - After applying, rebuild and deploy.'
79
+ Edit-request contract (read carefully):
80
+ - Call artifact_edit exactly once with the slug and verbatim_request above. Never retry the edit yourself: if the edit is not accepted, do NOT call artifact_edit again — end your turn.
81
+ - If artifact_edit explicitly refuses the edit (the call is rejected — e.g. the artifact does not exist), do NOT call artifact_edit again: end your turn with exactly one line and nothing else: ARTIFACT_EDIT_REFUSED: <the refusal text, one line>.
82
+ - If artifact_edit is not available after the load, do NOT improvise — end your turn.
83
+ - You do NOT call setprovenance, artifact_inspect, or post-deploy yourself.
84
+ No report is needed: do not return JSON, do not summarize what you did, do not echo the diff. End your turn after the artifact_edit call."
85
+ Wait for the child to finish. The child's final text is your only signal.
86
+
87
+ Phase 2 — complete (deterministic):
88
+ `node {crewHome}/lib/retry-publish.js --crew-home {crewHome} --task-id <task_id> --phase complete --edit-outcome <accepted|refused> [--refusal-text "<first line>"]`
89
+ - If the child's final text starts with `ARTIFACT_EDIT_REFUSED`: `--edit-outcome refused --refusal-text "<first line of the refusal>"`.
90
+ - Otherwise: `--edit-outcome accepted`.
91
+ - The command writes the outcome notes, appends the `-retry1` ledger entry (with the pre-retry `manifest_before` baseline), mirrors verification-requested, and RELEASES the lock in a finally block. Exit 0 means the protocol completed.
92
+
93
+ - Crash rules (fail closed, never re-trigger blind): if this tick dies anywhere in the retry protocol, the note journal is the recovery state. `publish: retry-intended` without `retry-issued` → the next scan mirrors `verification-requested` (never re-triggers). `publish: retry-issued` without a mirrored request → the next scan mirrors it. A leaked merge lock expires after 600s. The next tick's scan never emits `retry_due` for a task with `retry-issued` in history — exactly one retry, enforced from notes, not from the ledger.
39
94
 
40
95
  4.5. **Parent publish verification (docs/publish-verification.md):** The publisher parks instead of stamping provenance; the parent — this tick, the live root agent — verifies content and stamps. Deterministic code detects, reads back, and certifies; you are only the ferry between the deterministic steps (scan → sensor → verifier).
41
96
  - Scan (code): `node {crewHome}/lib/crew-api.js --crew-home {crewHome} scan-verification-pending`
@@ -46,7 +101,7 @@ You are the dispatch trigger for Muse Crew. Run the authoritative dispatcher wor
46
101
  `node {crewHome}/lib/readback-disk.js --repo-path "<repo_path>" --commit <commit> --base <base> --slug "<deploy_slug>" --task-id <task_id> > /tmp/readback-<task_id>.txt 2> /tmp/readback-<task_id>.err`
47
102
  Use the entry's `repo_path` and `deploy_slug` verbatim. If the sensor exits 0, the result file holds the findings block — hand it to the verify step below. If it exits non-zero, do NOT save or use stdout: log `publish: verification-procedural-error <commit> <first line of the .err file>` and leave the task parked — the next tick retries. A sensor failure is procedural (the read could not be performed), never a content verdict. Do NOT judge content yourself, and do NOT stamp provenance.
48
103
  - **Verify (code):** `node {crewHome}/lib/verify-publish.js --crew-home {crewHome} --task-id <task_id> --commit <commit> --base <base> --repo-path "<repo_path>" --slug "<deploy_slug>" --crew-release <crew_release> --project-id <project_id> --inspection-id <task_id>-disk --result-file /tmp/readback-<task_id>.txt`
49
- Pass `--build-agent-id <id>` from the entry's `build_agent_id` when it is present. Use the SAME `<base>` the sensor ran with and the entry's `project_id` verbatim. The verifier parses the findings, compares mechanically against the base..commit diff, checks supersession, and stamps only on a match. Its terminal verdicts (`publish: verified` → task re-queued; `publish: verification-failed` → stays parked) are final — log them and continue.
104
+ Pass `--build-agent-id <id>` from the entry's `build_agent_id` when it is present. Pass `--attempt <attempt>` from the entry's `attempt` when it is present (2026-09-18, blocker 4: the verifier binds the ledger entry by exact (task_id, commit, attempt) — without the attempt, a retry's trigger is indistinguishable from the original's). Use the SAME `<base>` the sensor ran with and the entry's `project_id` verbatim. The verifier parses the findings, compares mechanically against the base..commit diff, checks supersession, and stamps only on a match. Its terminal verdicts (`publish: verified` → task re-queued; `publish: verification-failed` → stays parked) are final — log them and continue.
50
105
  - Never stamp provenance from prose. Never infer a verdict from an inspector's summary text. The verify script's machine-checked comparison is the only certification.
51
106
 
52
107
  5. **Monitor launched workflows until terminal (stay-alive — 2026-09-13):** The platform ties async workflow `agent()` authorization to the launcher's lifetime: if THIS tick ends while a workflow is still running, the workflow's next `agent()` call fails with "subagent bootstrap is no longer authorized" / "subagent reservation owner is terminal". Prevention beats recovery here, so this tick is configured with a 90-minute execution timeout (`timeout_secs: 5400` in seed/crons.json) and you MUST stay alive until every launched run reaches a terminal state. Do not exit early while a launched run is still `running` — your death is what kills it.
@@ -428,6 +428,7 @@ async function recordPublishLedger(entry, rework) {
428
428
  attempt: entry.attempt || null,
429
429
  agent_id: entry.agent_id || null,
430
430
  applied_report: entry.applied_report || null,
431
+ manifest_before: entry.manifest_before || null,
431
432
  outcome: entry.outcome,
432
433
  detail: entry.detail || ""
433
434
  });
@@ -1549,20 +1550,36 @@ while (i < STEPS.length) {
1549
1550
  // See docs/decisions/publish-path.md#durable-evidence-snapshot: snapshot the audit-dir listing BEFORE the trigger; fallback diffs before/after.
1550
1551
  var auditDirsBeforeTrigger = [];
1551
1552
  var auditBeforeOk = false;
1553
+ // Pre-trigger baselines (design §1.9): the audit-dir listing and the
1554
+ // manifest snapshot. The verified-path freshness check compares the
1555
+ // post-trigger manifest against the baseline (built_at advance +
1556
+ // content_sha256 change). Best-effort, never gates — a missing
1557
+ // manifest baseline fails the verified path closed.
1558
+ var preTriggerManifest = null;
1552
1559
  try {
1553
1560
  var auditBefore = await agent(
1554
- "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
1555
- "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1556
- "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1557
- { key: attemptKey("publish-audit-before-" + taskId, totalReworkCount), label: "Snapshotting audit dirs before rebuild trigger",
1558
- schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1561
+ "Capture pre-trigger baselines for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshots, never gates).\n" +
1562
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null; echo ---MANIFEST---; cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/.space-build/manifest.json 2>/dev/null\n" +
1563
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when missing>\", \"manifest\": \"<the manifest's full text, or empty string when missing/unreadable>\" } and nothing else.",
1564
+ { key: attemptKey("publish-baseline-before-" + taskId, totalReworkCount), label: "Snapshotting baselines before rebuild trigger",
1565
+ schema: { type: "object", properties: { dirs: { type: "string" }, manifest: { type: "string" } }, required: ["dirs", "manifest"] } }
1559
1566
  );
1560
1567
  auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1568
+ var manifestText = String((auditBefore && auditBefore.manifest) || "").trim();
1569
+ if (manifestText) {
1570
+ var manifestJson = JSON.parse(manifestText);
1571
+ preTriggerManifest = {
1572
+ built_at: typeof manifestJson.built_at === "string" ? manifestJson.built_at : null,
1573
+ content_sha256: typeof manifestJson.content_sha256 === "string" ? manifestJson.content_sha256 : null,
1574
+ };
1575
+ }
1576
+ // Arm only after both baselines parse.
1561
1577
  auditBeforeOk = true;
1562
- log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
1578
+ log("Publish pre-trigger baselines for task " + taskId + ": " + auditDirsBeforeTrigger.length + " audit dirs, manifest " + (preTriggerManifest ? "built_at=" + preTriggerManifest.built_at : "none"));
1563
1579
  } catch (auditBeforeErr) {
1564
- log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1580
+ log("Publish pre-trigger baselines failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1565
1581
  }
1582
+
1566
1583
  // See docs/decisions/publish-path.md#fire-and-forget-trigger: the trigger child returns immediately; the workflow owns observation and verdict.
1567
1584
  var publishToolsOk = false;
1568
1585
  var publishToolsMissing = false;
@@ -1767,6 +1784,7 @@ while (i < STEPS.length) {
1767
1784
  attempt: rebuildAttemptKey,
1768
1785
  agent_id: rebuildAgentId,
1769
1786
  applied_report: publishAppliedObservation,
1787
+ manifest_before: preTriggerManifest,
1770
1788
  outcome: "submitted",
1771
1789
  detail: "fire-and-forget trigger; build receipt captured by workflow-owned build-state observation (pre/post-trigger diff)"
1772
1790
  }, totalReworkCount);
@@ -1837,7 +1855,8 @@ while (i < STEPS.length) {
1837
1855
  attempt: rebuildAttemptKey,
1838
1856
  agent_id: null,
1839
1857
  applied_report: publishAppliedObservation,
1840
- outcome: "submitted",
1858
+ manifest_before: preTriggerManifest,
1859
+ outcome: "build-observed",
1841
1860
  detail: "durable audit evidence shows a build completed during the attempt window (no receipt agent_id — attribution by window, not identity; receipt poll bypassed (verdict decided), routed to parent verification)"
1842
1861
  }, totalReworkCount);
1843
1862
  log("Publish verdict LANDED for task " + taskId + ": a completed build was observed during the attempt window — receipt poll bypassed (verdict decided, no receipt to chain to), routing directly to parent verification.");
@@ -2109,7 +2128,8 @@ while (i < STEPS.length) {
2109
2128
  attempt: rebuildAttemptKey,
2110
2129
  agent_id: rebuildAgentId,
2111
2130
  applied_report: publishAppliedObservation,
2112
- outcome: "submitted",
2131
+ manifest_before: preTriggerManifest,
2132
+ outcome: "build-observed",
2113
2133
  detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
2114
2134
  }, totalReworkCount);
2115
2135
  } else if (auditOkAfterPoll === false) {
@@ -2953,9 +2973,10 @@ while (i < STEPS.length) {
2953
2973
  // Success contract (2026-09-18, H2): the parent asked for exactly
2954
2974
  // one build for this publish. The build landed — do NOT republish: a
2955
2975
  // duplicate build would re-publish the same change. content_check=pending
2956
- // means the parent's independent read-back has not happened yet; this
2976
+ // means parent read-back is pending; this
2957
2977
  // park is NOT proof the content is correct.
2958
2978
  return await parkTask("publish: verification-requested " + mergeCommitForPublish +
2979
+ " attempt=" + rebuildAttemptKey +
2959
2980
  " Do NOT republish: a duplicate build would re-publish the same change. " +
2960
2981
  "Artifact build landed, post-deploy finalized, provenance not stamped — the crew has not yet independently confirmed the live artifact contains exactly the change; waiting on the manual read-back in docs/publish-verification.md. " +
2961
2982
  "Appendix: build=" + (rebuildAgentId || "agent_id unobserved") + "; provenance=unstamped; content_check=pending.");
@@ -70,7 +70,7 @@ let telemetryBuffer = [];
70
70
  async function telemetryStart(workflowName) {
71
71
  try {
72
72
  const out = await agent(crewCmd("record-run-start", {
73
- task_id: taskId, workflow: workflowName, launched_by: "cron"
73
+ task_id: taskId, workflow: workflowName, launched_by: inputs.launched_by
74
74
  }), { key: "telemetry-start", label: "Recording run start" });
75
75
  const parsed = typeof out === "string" ? JSON.parse(out) : out;
76
76
  if (parsed && parsed.run_id) telemetryRunId = parsed.run_id;
@@ -487,6 +487,7 @@ async function recordPublishLedger(entry, rework) {
487
487
  attempt: entry.attempt || null,
488
488
  agent_id: entry.agent_id || null,
489
489
  applied_report: entry.applied_report || null,
490
+ manifest_before: entry.manifest_before || null,
490
491
  outcome: entry.outcome,
491
492
  detail: entry.detail || ""
492
493
  });
@@ -1455,20 +1456,36 @@ while (i < STEPS.length) {
1455
1456
  // See docs/decisions/publish-path.md#durable-evidence-snapshot: snapshot the audit-dir listing BEFORE the trigger; fallback diffs before/after.
1456
1457
  var auditDirsBeforeTrigger = [];
1457
1458
  var auditBeforeOk = false;
1459
+ // Pre-trigger baselines (design §1.9): the audit-dir listing and the
1460
+ // manifest snapshot. The verified-path freshness check compares the
1461
+ // post-trigger manifest against the baseline (built_at advance +
1462
+ // content_sha256 change). Best-effort, never gates — a missing
1463
+ // manifest baseline fails the verified path closed.
1464
+ var preTriggerManifest = null;
1458
1465
  try {
1459
1466
  var auditBefore = await agent(
1460
- "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
1461
- "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1462
- "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1463
- { key: attemptKey("publish-audit-before-" + taskId, reworkCount), label: "Snapshotting audit dirs before rebuild trigger",
1464
- schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1467
+ "Capture pre-trigger baselines for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshots, never gates).\n" +
1468
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null; echo ---MANIFEST---; cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/.space-build/manifest.json 2>/dev/null\n" +
1469
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when missing>\", \"manifest\": \"<the manifest's full text, or empty string when missing/unreadable>\" } and nothing else.",
1470
+ { key: attemptKey("publish-baseline-before-" + taskId, reworkCount), label: "Snapshotting baselines before rebuild trigger",
1471
+ schema: { type: "object", properties: { dirs: { type: "string" }, manifest: { type: "string" } }, required: ["dirs", "manifest"] } }
1465
1472
  );
1466
1473
  auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1474
+ var manifestText = String((auditBefore && auditBefore.manifest) || "").trim();
1475
+ if (manifestText) {
1476
+ var manifestJson = JSON.parse(manifestText);
1477
+ preTriggerManifest = {
1478
+ built_at: typeof manifestJson.built_at === "string" ? manifestJson.built_at : null,
1479
+ content_sha256: typeof manifestJson.content_sha256 === "string" ? manifestJson.content_sha256 : null,
1480
+ };
1481
+ }
1482
+ // Arm only after both baselines parse.
1467
1483
  auditBeforeOk = true;
1468
- log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
1484
+ log("Publish pre-trigger baselines for task " + taskId + ": " + auditDirsBeforeTrigger.length + " audit dirs, manifest " + (preTriggerManifest ? "built_at=" + preTriggerManifest.built_at : "none"));
1469
1485
  } catch (auditBeforeErr) {
1470
- log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1486
+ log("Publish pre-trigger baselines failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1471
1487
  }
1488
+
1472
1489
  // See docs/decisions/publish-path.md#fire-and-forget-trigger: the trigger child returns immediately; the workflow owns observation and verdict.
1473
1490
  var publishToolsOk = false;
1474
1491
  var publishToolsMissing = false;
@@ -1673,6 +1690,7 @@ while (i < STEPS.length) {
1673
1690
  attempt: rebuildAttemptKey,
1674
1691
  agent_id: rebuildAgentId,
1675
1692
  applied_report: publishAppliedObservation,
1693
+ manifest_before: preTriggerManifest,
1676
1694
  outcome: "submitted",
1677
1695
  detail: "fire-and-forget trigger; build receipt captured by workflow-owned build-state observation (pre/post-trigger diff)"
1678
1696
  }, reworkCount);
@@ -1743,7 +1761,8 @@ while (i < STEPS.length) {
1743
1761
  attempt: rebuildAttemptKey,
1744
1762
  agent_id: null,
1745
1763
  applied_report: publishAppliedObservation,
1746
- outcome: "submitted",
1764
+ manifest_before: preTriggerManifest,
1765
+ outcome: "build-observed",
1747
1766
  detail: "durable audit evidence shows a build completed during the attempt window (no receipt agent_id — attribution by window, not identity; receipt poll bypassed (verdict decided), routed to parent verification)"
1748
1767
  }, reworkCount);
1749
1768
  log("Publish verdict LANDED for task " + taskId + ": a completed build was observed during the attempt window — receipt poll bypassed (verdict decided, no receipt to chain to), routing directly to parent verification.");
@@ -2015,7 +2034,8 @@ while (i < STEPS.length) {
2015
2034
  attempt: rebuildAttemptKey,
2016
2035
  agent_id: rebuildAgentId,
2017
2036
  applied_report: publishAppliedObservation,
2018
- outcome: "submitted",
2037
+ manifest_before: preTriggerManifest,
2038
+ outcome: "build-observed",
2019
2039
  detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
2020
2040
  }, reworkCount);
2021
2041
  } else if (auditOkAfterPoll === false) {
@@ -2552,6 +2572,7 @@ while (i < STEPS.length) {
2552
2572
  // means the parent's independent read-back has not happened yet; this
2553
2573
  // park is NOT proof the content is correct.
2554
2574
  return await parkTask("publish: verification-requested " + mergeCommitForPublish +
2575
+ " attempt=" + rebuildAttemptKey +
2555
2576
  " Do NOT republish: a duplicate build would re-publish the same change. " +
2556
2577
  "Artifact build landed, post-deploy finalized, provenance not stamped — the crew has not yet independently confirmed the live artifact contains exactly the change; waiting on the manual read-back in docs/publish-verification.md. " +
2557
2578
  "Appendix: build=" + (rebuildAgentId || "agent_id unobserved") + "; provenance=unstamped; content_check=pending.");