muse-crew 0.13.3 → 0.14.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/publish-unknown-recovery.md +113 -0
- package/lib/AGENTS.md +7 -2
- package/lib/advance-publish-base.js +74 -28
- package/lib/classify-publish-absence.js +451 -0
- package/lib/crew-api.js +584 -21
- package/lib/publish-content.js +154 -0
- package/lib/retry-publish.js +370 -0
- package/lib/scaffold-crew.js +129 -0
- package/lib/verify-publish.js +101 -84
- package/package.json +2 -2
- package/seed/cron-body-template.md +57 -2
- package/workflows/bugfix.js +38 -23
- package/workflows/chore.js +37 -22
- package/workflows/crew-dispatch.js +43 -8
- package/workflows/crew-init.js +333 -78
- package/workflows/crew-uninstall.js +57 -10
- package/workflows/standard.js +35 -16
- package/workflows/upgrade.js +28 -14
package/lib/verify-publish.js
CHANGED
|
@@ -39,9 +39,18 @@
|
|
|
39
39
|
|
|
40
40
|
import { execFileSync } from "node:child_process";
|
|
41
41
|
import { readFileSync } from "node:fs";
|
|
42
|
+
import { homedir } from "node:os";
|
|
42
43
|
import { join } from "node:path";
|
|
43
|
-
|
|
44
|
-
|
|
44
|
+
// Shared content-judgment primitives (2026-09-18, blocker 15): the same
|
|
45
|
+
// diff parser, findings parser, and collision-exemption rules the
|
|
46
|
+
// unknown-recovery classifier uses — one judgment, never two copies.
|
|
47
|
+
import {
|
|
48
|
+
EMPTY_TREE,
|
|
49
|
+
parseDiff,
|
|
50
|
+
parseFindings,
|
|
51
|
+
makeOldCounter,
|
|
52
|
+
discriminatingLines,
|
|
53
|
+
} from "./publish-content.js";
|
|
45
54
|
|
|
46
55
|
function arg(name, required = true) {
|
|
47
56
|
const i = process.argv.indexOf(name);
|
|
@@ -67,6 +76,8 @@ const resultFile = arg("--result-file");
|
|
|
67
76
|
const crewRelease = arg("--crew-release");
|
|
68
77
|
const buildAgentId = arg("--build-agent-id", false);
|
|
69
78
|
const projectId = arg("--project-id");
|
|
79
|
+
const attempt = arg("--attempt", false); // 2026-09-18, blocker 4: exact attempt binding
|
|
80
|
+
const spacesRoot = arg("--spaces-root", false) || join(homedir(), "workspace", "ts-spaces");
|
|
70
81
|
|
|
71
82
|
if (!/^[0-9a-f]{40}$/.test(commit)) fail("usage", "commit must be a 40-char hex sha.", 2);
|
|
72
83
|
if (!/^[0-9a-f]{40}$/.test(base)) fail("usage", "base must be a 40-char hex sha.", 2);
|
|
@@ -135,33 +146,7 @@ try {
|
|
|
135
146
|
// ADDED: <line> :: PRESENT|ABSENT
|
|
136
147
|
// REMOVED: <line> :: PRESENT|ABSENT
|
|
137
148
|
// END_FILE
|
|
138
|
-
|
|
139
|
-
const findings = new Map(); // path -> { added: Map(line->verdict), removed: Map(line->verdict) }
|
|
140
|
-
let cur = null;
|
|
141
|
-
let malformed = null;
|
|
142
|
-
for (const rawLine of text.split("\n")) {
|
|
143
|
-
const line = rawLine.trimEnd();
|
|
144
|
-
if (line.startsWith("FILE: ")) {
|
|
145
|
-
cur = { added: new Map(), removed: new Map() };
|
|
146
|
-
findings.set(line.slice(6).trim(), cur);
|
|
147
|
-
} else if (line === "END_FILE") {
|
|
148
|
-
cur = null;
|
|
149
|
-
} else if (cur && (line.startsWith("ADDED: ") || line.startsWith("REMOVED: "))) {
|
|
150
|
-
const kind = line.startsWith("ADDED: ") ? "added" : "removed";
|
|
151
|
-
const rest = line.slice(kind === "added" ? 7 : 9);
|
|
152
|
-
const sep = rest.lastIndexOf(" :: ");
|
|
153
|
-
if (sep < 0) { malformed = `malformed finding line: ${line.slice(0, 80)}`; break; }
|
|
154
|
-
const content = rest.slice(0, sep);
|
|
155
|
-
const verdict = rest.slice(sep + 4).trim();
|
|
156
|
-
if (verdict !== "PRESENT" && verdict !== "ABSENT") {
|
|
157
|
-
malformed = `bad verdict: ${verdict}`;
|
|
158
|
-
break;
|
|
159
|
-
}
|
|
160
|
-
cur[kind].set(content, verdict);
|
|
161
|
-
}
|
|
162
|
-
}
|
|
163
|
-
return { findings, malformed };
|
|
164
|
-
}
|
|
149
|
+
// (parsed by the shared parseFindings in lib/publish-content.js)
|
|
165
150
|
|
|
166
151
|
// --- 2. Expected diff from git (never from the builder's report) -----------
|
|
167
152
|
// The expected change is the publish delta base..commit, using the SAME
|
|
@@ -183,21 +168,7 @@ try {
|
|
|
183
168
|
} catch (e) {
|
|
184
169
|
terminal("read-back-unavailable", `git diff failed: ${e.message}`);
|
|
185
170
|
}
|
|
186
|
-
const expected =
|
|
187
|
-
let curFile = null;
|
|
188
|
-
for (const line of diff.split("\n")) {
|
|
189
|
-
if (line.startsWith("diff --git")) {
|
|
190
|
-
const m = line.match(/^diff --git a\/(.+) b\/(.+)$/);
|
|
191
|
-
curFile = m ? m[2] : "unknown";
|
|
192
|
-
expected.set(curFile, { added: [], removed: [], isBinary: false });
|
|
193
|
-
} else if (curFile && line.startsWith("Binary files ")) {
|
|
194
|
-
expected.get(curFile).isBinary = true;
|
|
195
|
-
} else if (curFile && line.startsWith("+") && !line.startsWith("+++")) {
|
|
196
|
-
expected.get(curFile).added.push(line.slice(1));
|
|
197
|
-
} else if (curFile && line.startsWith("-") && !line.startsWith("---")) {
|
|
198
|
-
expected.get(curFile).removed.push(line.slice(1));
|
|
199
|
-
}
|
|
200
|
-
}
|
|
171
|
+
const expected = parseDiff(diff); // path -> { added: [], removed: [], isBinary: bool }
|
|
201
172
|
|
|
202
173
|
// --- 3. Pick the findings candidate that covers the expected diff -----------
|
|
203
174
|
let findings = null;
|
|
@@ -232,33 +203,30 @@ if (!findings || bestScore <= 0) {
|
|
|
232
203
|
// removal into a pass. No change to the sensor: build-readback-request.js
|
|
233
204
|
// still reports whole-file PRESENT/ABSENT honestly; only this judge gets
|
|
234
205
|
// smarter.
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
const
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
oldTreeCounts.set(path, counts);
|
|
260
|
-
}
|
|
261
|
-
return counts.get(line) || 0;
|
|
206
|
+
// Collision exemption (2026-09-15, task 00bca4b8) — decided once by the
|
|
207
|
+
// shared lib/publish-content.js so the verifier and the unknown-recovery
|
|
208
|
+
// classifier can never disagree about what a diff proves:
|
|
209
|
+
//
|
|
210
|
+
// - a removed diff line that also occurs verbatim in untouched code has
|
|
211
|
+
// zero discriminating power — its presence in the new file cannot tell
|
|
212
|
+
// "old block removed" from "old block present";
|
|
213
|
+
// - an added line that already occurs in the old tree is reported PRESENT
|
|
214
|
+
// whether or not the new hunk actually landed — same zero power.
|
|
215
|
+
//
|
|
216
|
+
// Exempting a line can never turn a missed change into a pass: an exempted
|
|
217
|
+
// line must survive in untouched code no matter what. No change to the
|
|
218
|
+
// sensor: build-readback-request.js still reports whole-file PRESENT/ABSENT
|
|
219
|
+
// honestly; only this judge gets smarter. The occurrence counts are
|
|
220
|
+
// computed from `git show <base>:<path>` (never the working tree) by the
|
|
221
|
+
// shared makeOldCounter; the diff may remove the same line more than once
|
|
222
|
+
// and the budget accounting lives in the shared discriminatingLines.
|
|
223
|
+
const oldCount = makeOldCounter(repoPath, base);
|
|
224
|
+
const disc = discriminatingLines(expected, oldCount);
|
|
225
|
+
const discAdded = new Map(); // path -> Set(line): discriminating added lines
|
|
226
|
+
const discRemoved = new Map(); // path -> Set(line): discriminating removed lines
|
|
227
|
+
for (const [p, d] of disc) {
|
|
228
|
+
discAdded.set(p, new Set(d.addedDisc));
|
|
229
|
+
discRemoved.set(p, new Set(d.removedDisc));
|
|
262
230
|
}
|
|
263
231
|
|
|
264
232
|
let exempted = 0;
|
|
@@ -277,18 +245,14 @@ for (const [path, exp] of expected) {
|
|
|
277
245
|
"the diff carries no added/removed lines for this file, so the line-based read-back checked nothing — " +
|
|
278
246
|
"provenance NOT stamped; human verification needed");
|
|
279
247
|
}
|
|
280
|
-
// (2026-09-16, critic finding 5) Added-line collisions:
|
|
281
|
-
//
|
|
282
|
-
//
|
|
283
|
-
// power (the sensor is membership-only, not count-sensitive). Exempt
|
|
284
|
-
// such lines from the pass criteria, symmetric to the removed-side
|
|
285
|
-
// exemption below — and require at least one discriminating added line
|
|
286
|
-
// per file, or the added-side check is vacuous (finding 1's class).
|
|
248
|
+
// (2026-09-16, critic finding 5) Added-line collisions: require at least
|
|
249
|
+
// one discriminating added line per file, or the added-side check is
|
|
250
|
+
// vacuous (finding 1's class).
|
|
287
251
|
let discriminatingAdded = 0;
|
|
288
252
|
for (const line of exp.added) {
|
|
289
253
|
const v = found.added.get(line);
|
|
290
254
|
if (v === undefined) terminal("unreadable-result", `no ADDED finding for line in ${path}: ${line.slice(0, 60)}`);
|
|
291
|
-
if (
|
|
255
|
+
if (!discAdded.get(path).has(line)) {
|
|
292
256
|
exempted += 1; // colliding pre-existing line: zero signal, cannot fail a good publish
|
|
293
257
|
continue;
|
|
294
258
|
}
|
|
@@ -301,15 +265,11 @@ for (const [path, exp] of expected) {
|
|
|
301
265
|
"their PRESENT findings cannot tell \"hunk landed\" from \"hunk dropped\" — " +
|
|
302
266
|
"provenance NOT stamped; human verification needed");
|
|
303
267
|
}
|
|
304
|
-
// The diff may remove the same line more than once; the old tree must
|
|
305
|
-
// account for every removal before a line counts as a collision.
|
|
306
|
-
const removedBudget = new Map();
|
|
307
|
-
for (const line of exp.removed) removedBudget.set(line, (removedBudget.get(line) || 0) + 1);
|
|
308
268
|
for (const line of exp.removed) {
|
|
309
269
|
const v = found.removed.get(line);
|
|
310
270
|
if (v === undefined) terminal("unreadable-result", `no REMOVED finding for line in ${path}: ${line.slice(0, 60)}`);
|
|
311
271
|
if (v === "ABSENT") continue;
|
|
312
|
-
if (
|
|
272
|
+
if (!discRemoved.get(path).has(line)) {
|
|
313
273
|
exempted += 1; // colliding pre-existing line: zero signal, cannot fail a good publish
|
|
314
274
|
continue;
|
|
315
275
|
}
|
|
@@ -328,6 +288,63 @@ if (head !== commit) {
|
|
|
328
288
|
terminal("superseded", `HEAD is ${head}, not ${commit} — a newer publish supersedes this one`);
|
|
329
289
|
}
|
|
330
290
|
|
|
291
|
+
// --- 5b. Manifest freshness (design §1.9) ------------------------------------
|
|
292
|
+
// The stamp certifies that a NEW build landed for this attempt — not a
|
|
293
|
+
// replayed manifest. The workflow snapshots the pre-trigger manifest into
|
|
294
|
+
// the submitted ledger entry; require (a) the current manifest's built_at
|
|
295
|
+
// advanced past the trigger, and (b) its content_sha256 differs from the
|
|
296
|
+
// pre-trigger baseline (a new build identity). A missing baseline fails
|
|
297
|
+
// closed — without it, freshness cannot be proven.
|
|
298
|
+
// 2026-09-18, blocker 4: bind by exact (task_id, commit, attempt) when the
|
|
299
|
+
// attempt is known — the original and retry attempts share a commit, and
|
|
300
|
+
// positional selection is luck. Within one attempt, select the OLDEST
|
|
301
|
+
// submitted entry (the trigger issuance; per blocker 2, audit observations
|
|
302
|
+
// are now "build-observed", not "submitted").
|
|
303
|
+
let triggerTs = null;
|
|
304
|
+
let manifestBefore = null;
|
|
305
|
+
try {
|
|
306
|
+
const ledgerText = readFileSync(join(crewHome, ".publish-ledger", slug + ".jsonl"), "utf8");
|
|
307
|
+
const entries = ledgerText.split("\n").filter((l) => l.trim()).map((l) => JSON.parse(l));
|
|
308
|
+
// Collect all submitted entries matching (task_id, commit[, attempt]).
|
|
309
|
+
const candidates = entries.filter((e) =>
|
|
310
|
+
e.task_id === taskId &&
|
|
311
|
+
e.outcome === "submitted" &&
|
|
312
|
+
e.commit === commit &&
|
|
313
|
+
(attempt == null || e.attempt === attempt)
|
|
314
|
+
);
|
|
315
|
+
if (candidates.length > 0) {
|
|
316
|
+
// Oldest first: the trigger issuance is the first submitted entry for
|
|
317
|
+
// the attempt (observations are "build-observed" per blocker 2).
|
|
318
|
+
candidates.sort((a, b) => (a.ts || "").localeCompare(b.ts || ""));
|
|
319
|
+
const e = candidates[0];
|
|
320
|
+
triggerTs = e.ts || null;
|
|
321
|
+
if (e.manifest_before && typeof e.manifest_before === "object") manifestBefore = e.manifest_before;
|
|
322
|
+
}
|
|
323
|
+
} catch (e) {
|
|
324
|
+
terminal("read-back-unavailable", `cannot read publish ledger for freshness check: ${e.message}`);
|
|
325
|
+
}
|
|
326
|
+
if (!triggerTs) {
|
|
327
|
+
terminal("read-back-unavailable", `no submitted ledger entry for commit ${commit} — the trigger instant is unknowable; freshness cannot be proven`);
|
|
328
|
+
}
|
|
329
|
+
if (!manifestBefore || typeof manifestBefore.content_sha256 !== "string") {
|
|
330
|
+
terminal("unverifiable-content", `no pre-trigger manifest baseline in the submitted ledger entry — freshness cannot be proven; provenance NOT stamped; human verification needed`);
|
|
331
|
+
}
|
|
332
|
+
let currentManifest = null;
|
|
333
|
+
try {
|
|
334
|
+
const manifestText = readFileSync(join(spacesRoot, slug, ".space-build", "manifest.json"), "utf8");
|
|
335
|
+
currentManifest = JSON.parse(manifestText);
|
|
336
|
+
} catch (e) {
|
|
337
|
+
terminal("read-back-unavailable", `cannot read current manifest for freshness check: ${e.message}`);
|
|
338
|
+
}
|
|
339
|
+
const currentBuiltAtMs = currentManifest && currentManifest.built_at ? Date.parse(currentManifest.built_at) : NaN;
|
|
340
|
+
const triggerMs = Date.parse(triggerTs);
|
|
341
|
+
if (!(currentBuiltAtMs > triggerMs)) {
|
|
342
|
+
terminal("unverifiable-content", `current manifest built_at ${currentManifest && currentManifest.built_at} did not advance past trigger ${triggerTs} — no new build for this attempt; provenance NOT stamped`);
|
|
343
|
+
}
|
|
344
|
+
if (typeof currentManifest.content_sha256 !== "string" || currentManifest.content_sha256 === manifestBefore.content_sha256) {
|
|
345
|
+
terminal("unverifiable-content", `current manifest content_sha256 is unchanged from the pre-trigger baseline — a replayed manifest, not a new build; provenance NOT stamped; human verification needed`);
|
|
346
|
+
}
|
|
347
|
+
|
|
331
348
|
// --- 6. Build-ID correlation (informational; the stamp certifies content) ---
|
|
332
349
|
let buildNote = "build-id-unobserved";
|
|
333
350
|
if (buildAgentId && resultText.includes(buildAgentId)) buildNote = `build-id-correlated ${buildAgentId}`;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "muse-crew",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Opinionated orchestration for Muse
|
|
3
|
+
"version": "0.14.1",
|
|
4
|
+
"description": "Opinionated orchestration for Muse — workflows, identities, and tooling for autonomous software development.",
|
|
5
5
|
"license": "UNLICENSED",
|
|
6
6
|
"private": false,
|
|
7
7
|
"repository": {
|
|
@@ -35,7 +35,62 @@ You are the dispatch trigger for Muse Crew. Run the authoritative dispatcher wor
|
|
|
35
35
|
- If the acknowledge fails (no reservation exists), DO NOT LAUNCH — the dispatcher did not acquire this task. This is a safety invariant.
|
|
36
36
|
- The launched workflow self-claims the task and clears the reservation as its first actions. If the task was already claimed or is done, the claim fails closed and the run stands down quietly — this is the mechanical duplicate protection, not an error.
|
|
37
37
|
|
|
38
|
-
If the dispatcher returned no claims or the claims array is empty, log NO_DISPATCH and CONTINUE to Step 4.
|
|
38
|
+
If the dispatcher returned no claims or the claims array is empty, log NO_DISPATCH and CONTINUE to Step 4.4 — do NOT exit. Unknown-recovery (4.4), verification (4.5) and evidence (6) run independently of dispatch claims. A no-claims tick must still recover parked publishes, verify them, and deliver evidence. (Fixed 2026-09-14: the old "exit on NO_DISPATCH" skipped verification permanently.)
|
|
39
|
+
|
|
40
|
+
4.4. **Unknown-recovery (docs/publish-unknown-recovery.md):** tasks parked with "Publish outcome unknown" are the receipt-less trigger gap (2026-09-18, blocker 15). The note is the state machine; deterministic code owns every transition; you are the ferry (scan → classifier → record / retry protocol). Run this BEFORE Step 4.5 so a re-triggered build's mirrored verification-requested is visible to the verification scan on a later tick. Log every list the scan returns.
|
|
41
|
+
- Scan (code): `node {crewHome}/lib/crew-api.js --crew-home {crewHome} scan-publish-unknown`
|
|
42
|
+
Returns `{ waiting: [...], due: [...], retry_due: [...], mirrored: [...], skipped: [...] }`. `waiting` = parked < 30m — leave alone, no claim. `mirrored`/`skipped` = already handled by code — log only, take no action.
|
|
43
|
+
- For each entry in `due`: classify (code), then record the decision (code). Use the entry's fields verbatim.
|
|
44
|
+
- `node {crewHome}/lib/classify-publish-absence.js --repo-path "<repo_path>" --commit <commit> --base <base> --slug "<slug>" --trigger-ts "<trigger_ts>" --park-ts "<park_ts>" --task-id <task_id> --manifest-before '<manifest_before JSON>'`
|
|
45
|
+
Serialize the entry's `manifest_before` field to compact JSON for the `--manifest-before` value; when the entry's `manifest_before` is null, omit the flag (the classifier falls back to the time-based advance check and the Step 4.5 verifier fails closed without a baseline).
|
|
46
|
+
Exit 0 prints the decision JSON (`{ ok:true, decision, reasons, details }`) — or `{ ok:false, ... }` for a semantic non-verdict. Exit non-zero (usage/git/IO failure) → log the stderr line and LEAVE THE TASK: do NOT record anything; the claim expires and the next tick re-claims. A classifier crash is never a verdict.
|
|
47
|
+
- `node {crewHome}/lib/crew-api.js --crew-home {crewHome} record-unknown-classification --json '{"task_id": "<task_id>", "claim_expiry": "<claim_expiry>", "decision": <classifier stdout JSON>}'`
|
|
48
|
+
Paste the classifier's stdout verbatim as the decision value. If it returns `recorded: false` → log the reason and continue (another tick owns the task). The routes are the code's — log the returned decision and `routed` value:
|
|
49
|
+
verified → `publish: verification-requested` (no re-trigger); provably-dropped → `publish: dropped` (queues the retry protocol); applied-not-built / ambiguous → `publish: ambiguous` (terminal); superseded → `publish: superseded` (terminal); deferred → nothing (the claim expires; the next scan re-claims).
|
|
50
|
+
- Never invent a decision. Never re-trigger the edit for a `due` entry — classification only.
|
|
51
|
+
- For each entry in `retry_due` (a provably-dropped edit with a free retry budget — exactly one retry per task, enforced from note history; the scan only emits `retry_due` when no `publish: retry-issued` exists in the task's history): run the deterministic retry protocol via `lib/retry-publish.js`. The command owns all lock handling, journaling, diff generation, ledger writes, and cleanup with try/finally lock release — you only mediate the unavoidable `artifact_edit` tool call.
|
|
52
|
+
|
|
53
|
+
Phase 1 — prepare (deterministic):
|
|
54
|
+
`node {crewHome}/lib/retry-publish.js --crew-home {crewHome} --task-id <task_id> --phase prepare`
|
|
55
|
+
- Exit 0 with `{"ok": true, "outcome": "ready", ...}`: the merge lock is HELD. The output carries `diff_path`, `diff_sha256`, and `slug`. Proceed to the trigger below.
|
|
56
|
+
- Exit 0 with `{"ok": true, "outcome": "superseded"|"lock-held"|"not-dropped", ...}`: terminal for this tick. Do NOT proceed to the trigger. If `outcome` is `not-dropped`, the output carries the full `classification` — pass it to `record-retry-recheck`: `node {crewHome}/lib/crew-api.js --crew-home {crewHome} record-retry-recheck --json '{"task_id": "<task_id>", "decision": <classification JSON>}'` and STOP for this task.
|
|
57
|
+
- Exit non-zero: the command failed (lock released via try/finally). Log the stderr line and STOP — do NOT trigger.
|
|
58
|
+
|
|
59
|
+
Trigger (agent-mediated — the ONLY prose-owned step):
|
|
60
|
+
Spawn ONE child (subagent) whose first instruction loads the artifact namespace (`tool_search.load_tool_namespace` with paths `["artifact"]`), with this exact brief — the same shape as the workflow's first attempt:
|
|
61
|
+
"The change to apply is the unified diff in the file \"<diff_path>\" (sha256 <diff_sha256>).
|
|
62
|
+
1. Verify the file: run sha256sum on it. If the printed hash is not exactly <diff_sha256>, STOP and end your turn — do not call artifact_edit.
|
|
63
|
+
2. Read the file's full content.
|
|
64
|
+
3. Call artifact_edit with slug \"<slug>\" and verbatim_request:
|
|
65
|
+
'Apply the following change to your source tree, then rebuild and deploy.
|
|
66
|
+
|
|
67
|
+
UNIFIED DIFF (relative to your source tree):
|
|
68
|
+
```diff
|
|
69
|
+
<the full content of the verified file, pasted verbatim>
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Rules:
|
|
73
|
+
- For each file in the diff, apply its hunks to the same path in your source tree (use git apply or equivalent).
|
|
74
|
+
- For a new file (--- /dev/null), create it with the added (+) lines as its full content.
|
|
75
|
+
- For a deleted file (+++ /dev/null), delete it.
|
|
76
|
+
- If any hunk does not apply cleanly, STOP and report the failure — do not improvise or skip hunks.
|
|
77
|
+
- Do not make any other source changes.
|
|
78
|
+
- After applying, rebuild and deploy.'
|
|
79
|
+
Edit-request contract (read carefully):
|
|
80
|
+
- Call artifact_edit exactly once with the slug and verbatim_request above. Never retry the edit yourself: if the edit is not accepted, do NOT call artifact_edit again — end your turn.
|
|
81
|
+
- If artifact_edit explicitly refuses the edit (the call is rejected — e.g. the artifact does not exist), do NOT call artifact_edit again: end your turn with exactly one line and nothing else: ARTIFACT_EDIT_REFUSED: <the refusal text, one line>.
|
|
82
|
+
- If artifact_edit is not available after the load, do NOT improvise — end your turn.
|
|
83
|
+
- You do NOT call setprovenance, artifact_inspect, or post-deploy yourself.
|
|
84
|
+
No report is needed: do not return JSON, do not summarize what you did, do not echo the diff. End your turn after the artifact_edit call."
|
|
85
|
+
Wait for the child to finish. The child's final text is your only signal.
|
|
86
|
+
|
|
87
|
+
Phase 2 — complete (deterministic):
|
|
88
|
+
`node {crewHome}/lib/retry-publish.js --crew-home {crewHome} --task-id <task_id> --phase complete --edit-outcome <accepted|refused> [--refusal-text "<first line>"]`
|
|
89
|
+
- If the child's final text starts with `ARTIFACT_EDIT_REFUSED`: `--edit-outcome refused --refusal-text "<first line of the refusal>"`.
|
|
90
|
+
- Otherwise: `--edit-outcome accepted`.
|
|
91
|
+
- The command writes the outcome notes, appends the `-retry1` ledger entry (with the pre-retry `manifest_before` baseline), mirrors verification-requested, and RELEASES the lock in a finally block. Exit 0 means the protocol completed.
|
|
92
|
+
|
|
93
|
+
- Crash rules (fail closed, never re-trigger blind): if this tick dies anywhere in the retry protocol, the note journal is the recovery state. `publish: retry-intended` without `retry-issued` → the next scan mirrors `verification-requested` (never re-triggers). `publish: retry-issued` without a mirrored request → the next scan mirrors it. A leaked merge lock expires after 600s. The next tick's scan never emits `retry_due` for a task with `retry-issued` in history — exactly one retry, enforced from notes, not from the ledger.
|
|
39
94
|
|
|
40
95
|
4.5. **Parent publish verification (docs/publish-verification.md):** The publisher parks instead of stamping provenance; the parent — this tick, the live root agent — verifies content and stamps. Deterministic code detects, reads back, and certifies; you are only the ferry between the deterministic steps (scan → sensor → verifier).
|
|
41
96
|
- Scan (code): `node {crewHome}/lib/crew-api.js --crew-home {crewHome} scan-verification-pending`
|
|
@@ -46,7 +101,7 @@ You are the dispatch trigger for Muse Crew. Run the authoritative dispatcher wor
|
|
|
46
101
|
`node {crewHome}/lib/readback-disk.js --repo-path "<repo_path>" --commit <commit> --base <base> --slug "<deploy_slug>" --task-id <task_id> > /tmp/readback-<task_id>.txt 2> /tmp/readback-<task_id>.err`
|
|
47
102
|
Use the entry's `repo_path` and `deploy_slug` verbatim. If the sensor exits 0, the result file holds the findings block — hand it to the verify step below. If it exits non-zero, do NOT save or use stdout: log `publish: verification-procedural-error <commit> <first line of the .err file>` and leave the task parked — the next tick retries. A sensor failure is procedural (the read could not be performed), never a content verdict. Do NOT judge content yourself, and do NOT stamp provenance.
|
|
48
103
|
- **Verify (code):** `node {crewHome}/lib/verify-publish.js --crew-home {crewHome} --task-id <task_id> --commit <commit> --base <base> --repo-path "<repo_path>" --slug "<deploy_slug>" --crew-release <crew_release> --project-id <project_id> --inspection-id <task_id>-disk --result-file /tmp/readback-<task_id>.txt`
|
|
49
|
-
Pass `--build-agent-id <id>` from the entry's `build_agent_id` when it is present. Use the SAME `<base>` the sensor ran with and the entry's `project_id` verbatim. The verifier parses the findings, compares mechanically against the base..commit diff, checks supersession, and stamps only on a match. Its terminal verdicts (`publish: verified` → task re-queued; `publish: verification-failed` → stays parked) are final — log them and continue.
|
|
104
|
+
Pass `--build-agent-id <id>` from the entry's `build_agent_id` when it is present. Pass `--attempt <attempt>` from the entry's `attempt` when it is present (2026-09-18, blocker 4: the verifier binds the ledger entry by exact (task_id, commit, attempt) — without the attempt, a retry's trigger is indistinguishable from the original's). Use the SAME `<base>` the sensor ran with and the entry's `project_id` verbatim. The verifier parses the findings, compares mechanically against the base..commit diff, checks supersession, and stamps only on a match. Its terminal verdicts (`publish: verified` → task re-queued; `publish: verification-failed` → stays parked) are final — log them and continue.
|
|
50
105
|
- Never stamp provenance from prose. Never infer a verdict from an inspector's summary text. The verify script's machine-checked comparison is the only certification.
|
|
51
106
|
|
|
52
107
|
5. **Monitor launched workflows until terminal (stay-alive — 2026-09-13):** The platform ties async workflow `agent()` authorization to the launcher's lifetime: if THIS tick ends while a workflow is still running, the workflow's next `agent()` call fails with "subagent bootstrap is no longer authorized" / "subagent reservation owner is terminal". Prevention beats recovery here, so this tick is configured with a 90-minute execution timeout (`timeout_secs: 5400` in seed/crons.json) and you MUST stay alive until every launched run reaches a terminal state. Do not exit early while a launched run is still `running` — your death is what kills it.
|
package/workflows/bugfix.js
CHANGED
|
@@ -132,7 +132,7 @@ if (!taskId) {
|
|
|
132
132
|
// See docs/decisions/workflow-core.md#closeout-envelope: the work agent returns the runtime's native envelope; verdict extracted mechanically.
|
|
133
133
|
const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
|
|
134
134
|
function extractVerdict(workerText) {
|
|
135
|
-
|
|
135
|
+
// The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
|
|
136
136
|
// your report with the verdict). This ignores literal VERDICT strings echoed
|
|
137
137
|
// from the worker's instructions (which contain quoted examples). Fail
|
|
138
138
|
// closed if: no verdict found, the last verdict is not in the trailing 100
|
|
@@ -428,6 +428,7 @@ async function recordPublishLedger(entry, rework) {
|
|
|
428
428
|
attempt: entry.attempt || null,
|
|
429
429
|
agent_id: entry.agent_id || null,
|
|
430
430
|
applied_report: entry.applied_report || null,
|
|
431
|
+
manifest_before: entry.manifest_before || null,
|
|
431
432
|
outcome: entry.outcome,
|
|
432
433
|
detail: entry.detail || ""
|
|
433
434
|
});
|
|
@@ -441,9 +442,7 @@ async function recordPublishLedger(entry, rework) {
|
|
|
441
442
|
label: "Recording publish attempt in ledger",
|
|
442
443
|
schema: { type: "object", properties: { result: { type: "string" } }, required: ["result"] } }
|
|
443
444
|
);
|
|
444
|
-
|
|
445
|
-
log("Publish ledger: outcome '" + entry.outcome + "' for task " + taskId +
|
|
446
|
-
(ok ? " recorded." : " NOT confirmed (" + ((res && res.result) || "no output") + ")"));
|
|
445
|
+
log("Noted publish outcome '" + entry.outcome + "' for task " + taskId + " in ledger");
|
|
447
446
|
} catch (e) {
|
|
448
447
|
log("Publish ledger: write failed for task " + taskId + " (non-fatal, observability only): " + (e && e.message ? e.message : e));
|
|
449
448
|
}
|
|
@@ -1549,20 +1548,36 @@ while (i < STEPS.length) {
|
|
|
1549
1548
|
// See docs/decisions/publish-path.md#durable-evidence-snapshot: snapshot the audit-dir listing BEFORE the trigger; fallback diffs before/after.
|
|
1550
1549
|
var auditDirsBeforeTrigger = [];
|
|
1551
1550
|
var auditBeforeOk = false;
|
|
1551
|
+
// Pre-trigger baselines (design §1.9): the audit-dir listing and the
|
|
1552
|
+
// manifest snapshot. The verified-path freshness check compares the
|
|
1553
|
+
// post-trigger manifest against the baseline (built_at advance +
|
|
1554
|
+
// content_sha256 change). Best-effort, never gates — a missing
|
|
1555
|
+
// manifest baseline fails the verified path closed.
|
|
1556
|
+
var preTriggerManifest = null;
|
|
1552
1557
|
try {
|
|
1553
1558
|
var auditBefore = await agent(
|
|
1554
|
-
"
|
|
1555
|
-
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
|
|
1556
|
-
"Return JSON { \"dirs\": \"<newline-separated names, empty string when
|
|
1557
|
-
{ key: attemptKey("publish-
|
|
1558
|
-
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1559
|
+
"Capture pre-trigger baselines for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshots, never gates).\n" +
|
|
1560
|
+
"Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null; echo ---MANIFEST---; cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/.space-build/manifest.json 2>/dev/null\n" +
|
|
1561
|
+
"Return JSON { \"dirs\": \"<newline-separated names, empty string when missing>\", \"manifest\": \"<the manifest's full text, or empty string when missing/unreadable>\" } and nothing else.",
|
|
1562
|
+
{ key: attemptKey("publish-baseline-before-" + taskId, totalReworkCount), label: "Snapshotting baselines before rebuild trigger",
|
|
1563
|
+
schema: { type: "object", properties: { dirs: { type: "string" }, manifest: { type: "string" } }, required: ["dirs", "manifest"] } }
|
|
1559
1564
|
);
|
|
1560
1565
|
auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1566
|
+
var manifestText = String((auditBefore && auditBefore.manifest) || "").trim();
|
|
1567
|
+
if (manifestText) {
|
|
1568
|
+
var manifestJson = JSON.parse(manifestText);
|
|
1569
|
+
preTriggerManifest = {
|
|
1570
|
+
built_at: typeof manifestJson.built_at === "string" ? manifestJson.built_at : null,
|
|
1571
|
+
content_sha256: typeof manifestJson.content_sha256 === "string" ? manifestJson.content_sha256 : null,
|
|
1572
|
+
};
|
|
1573
|
+
}
|
|
1574
|
+
// Arm only after both baselines parse.
|
|
1561
1575
|
auditBeforeOk = true;
|
|
1562
|
-
log("Publish
|
|
1576
|
+
log("Publish pre-trigger baselines for task " + taskId + ": " + auditDirsBeforeTrigger.length + " audit dirs, manifest " + (preTriggerManifest ? "built_at=" + preTriggerManifest.built_at : "none"));
|
|
1563
1577
|
} catch (auditBeforeErr) {
|
|
1564
|
-
log("Publish
|
|
1578
|
+
log("Publish pre-trigger baselines failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
|
|
1565
1579
|
}
|
|
1580
|
+
|
|
1566
1581
|
// See docs/decisions/publish-path.md#fire-and-forget-trigger: the trigger child returns immediately; the workflow owns observation and verdict.
|
|
1567
1582
|
var publishToolsOk = false;
|
|
1568
1583
|
var publishToolsMissing = false;
|
|
@@ -1767,6 +1782,7 @@ while (i < STEPS.length) {
|
|
|
1767
1782
|
attempt: rebuildAttemptKey,
|
|
1768
1783
|
agent_id: rebuildAgentId,
|
|
1769
1784
|
applied_report: publishAppliedObservation,
|
|
1785
|
+
manifest_before: preTriggerManifest,
|
|
1770
1786
|
outcome: "submitted",
|
|
1771
1787
|
detail: "fire-and-forget trigger; build receipt captured by workflow-owned build-state observation (pre/post-trigger diff)"
|
|
1772
1788
|
}, totalReworkCount);
|
|
@@ -1837,7 +1853,8 @@ while (i < STEPS.length) {
|
|
|
1837
1853
|
attempt: rebuildAttemptKey,
|
|
1838
1854
|
agent_id: null,
|
|
1839
1855
|
applied_report: publishAppliedObservation,
|
|
1840
|
-
|
|
1856
|
+
manifest_before: preTriggerManifest,
|
|
1857
|
+
outcome: "build-observed",
|
|
1841
1858
|
detail: "durable audit evidence shows a build completed during the attempt window (no receipt agent_id — attribution by window, not identity; receipt poll bypassed (verdict decided), routed to parent verification)"
|
|
1842
1859
|
}, totalReworkCount);
|
|
1843
1860
|
log("Publish verdict LANDED for task " + taskId + ": a completed build was observed during the attempt window — receipt poll bypassed (verdict decided, no receipt to chain to), routing directly to parent verification.");
|
|
@@ -2109,7 +2126,8 @@ while (i < STEPS.length) {
|
|
|
2109
2126
|
attempt: rebuildAttemptKey,
|
|
2110
2127
|
agent_id: rebuildAgentId,
|
|
2111
2128
|
applied_report: publishAppliedObservation,
|
|
2112
|
-
|
|
2129
|
+
manifest_before: preTriggerManifest,
|
|
2130
|
+
outcome: "build-observed",
|
|
2113
2131
|
detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
|
|
2114
2132
|
}, totalReworkCount);
|
|
2115
2133
|
} else if (auditOkAfterPoll === false) {
|
|
@@ -2172,11 +2190,11 @@ while (i < STEPS.length) {
|
|
|
2172
2190
|
if (!postDeploy.deployed) {
|
|
2173
2191
|
return await parkTask("Post-deploy failed after a skipped publish: " + (postDeploy.output || "no output") + ". Nothing was published; worktree cleanup and lock state unknown — human attention needed.");
|
|
2174
2192
|
}
|
|
2175
|
-
log("Publish skipped cleanly for task " + taskId + " (no lock held) — post-deploy
|
|
2193
|
+
log("Publish skipped cleanly for task " + taskId + " (no lock held) — post-deploy step finished");
|
|
2176
2194
|
} else {
|
|
2177
2195
|
if (publishFailure) {
|
|
2178
2196
|
return await parkTask(publishFailure + (postDeploy.deployed
|
|
2179
|
-
? " Post-deploy
|
|
2197
|
+
? " Post-deploy step finished (cleanup status unknown)."
|
|
2180
2198
|
: " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
|
|
2181
2199
|
}
|
|
2182
2200
|
if (!publishBuildLanded) {
|
|
@@ -2189,7 +2207,7 @@ while (i < STEPS.length) {
|
|
|
2189
2207
|
// commit (design §1.5: "the trigger was sent for commit <short-sha>").
|
|
2190
2208
|
if (publishUnknownFields) publishUnknownFields.commitShortSha = mergeCommitShortForPublish;
|
|
2191
2209
|
return await parkTask(composeUnattributedParkReason(publishUnknownFields) + (postDeploy.deployed
|
|
2192
|
-
? " Post-deploy
|
|
2210
|
+
? " Post-deploy step finished (cleanup status unknown)."
|
|
2193
2211
|
: " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
|
|
2194
2212
|
}
|
|
2195
2213
|
if (!postDeploy.deployed) {
|
|
@@ -2218,7 +2236,7 @@ while (i < STEPS.length) {
|
|
|
2218
2236
|
// artifact_edit would trigger a duplicate build.
|
|
2219
2237
|
if (publishSkippedNoLock) {
|
|
2220
2238
|
instructions = "Publish was skipped deterministically by the workflow before your step — do NOT call artifact_edit, artifact_status, setprovenance, or post-deploy yourself; doing so would disturb the finalized state.\n\n" +
|
|
2221
|
-
"Integrate reported MERGED_EMPTY (the task branch had no commits ahead of main), so no merge lock was taken and there is nothing to ship.
|
|
2239
|
+
"Integrate reported MERGED_EMPTY (the task branch had no commits ahead of main), so no merge lock was taken and there is nothing to ship. Post-deploy step finished per its return — do NOT run post-deploy yourself.\n\n" +
|
|
2222
2240
|
"Write plain prose describing the skip, then on its own line: VERDICT: PASS\n" +
|
|
2223
2241
|
"The VERDICT line must be the last line of your report.";
|
|
2224
2242
|
} else {
|
|
@@ -2811,11 +2829,7 @@ while (i < STEPS.length) {
|
|
|
2811
2829
|
{ key: attemptKey("publish-provenance-refresh-" + taskId, totalReworkCount), label: "Refreshing dashboard provenance after crew release",
|
|
2812
2830
|
schema: { type: "object", properties: { refreshed: { type: "boolean" }, reason: { type: "string" }, crew_release: { type: "string" }, published_at: { type: "string" } }, required: ["refreshed"] } }
|
|
2813
2831
|
);
|
|
2814
|
-
if (provRefresh.refreshed) {
|
|
2815
|
-
log("Provenance refreshed for task " + taskId + ": crew_release " + provRefresh.crew_release);
|
|
2816
|
-
} else if (provRefresh.reason === "no-record") {
|
|
2817
|
-
log("Provenance refresh skipped for task " + taskId + ": no existing record to refresh (fresh instance — the dashboard's first artifact publish will create it)");
|
|
2818
|
-
} else {
|
|
2832
|
+
if (!provRefresh.refreshed && provRefresh.reason !== "no-record") {
|
|
2819
2833
|
return await parkTask("Provenance refresh failed after a verified npm publish (crew_release " + (provRefresh.crew_release || "unknown") + "). The release is live but the dashboard record is stale — fail-closed.");
|
|
2820
2834
|
}
|
|
2821
2835
|
} catch (e) {
|
|
@@ -2953,9 +2967,10 @@ while (i < STEPS.length) {
|
|
|
2953
2967
|
// Success contract (2026-09-18, H2): the parent asked for exactly
|
|
2954
2968
|
// one build for this publish. The build landed — do NOT republish: a
|
|
2955
2969
|
// duplicate build would re-publish the same change. content_check=pending
|
|
2956
|
-
// means
|
|
2970
|
+
// means parent read-back is pending; this
|
|
2957
2971
|
// park is NOT proof the content is correct.
|
|
2958
2972
|
return await parkTask("publish: verification-requested " + mergeCommitForPublish +
|
|
2973
|
+
" attempt=" + rebuildAttemptKey +
|
|
2959
2974
|
" Do NOT republish: a duplicate build would re-publish the same change. " +
|
|
2960
2975
|
"Artifact build landed, post-deploy finalized, provenance not stamped — the crew has not yet independently confirmed the live artifact contains exactly the change; waiting on the manual read-back in docs/publish-verification.md. " +
|
|
2961
2976
|
"Appendix: build=" + (rebuildAgentId || "agent_id unobserved") + "; provenance=unstamped; content_check=pending.");
|