muse-crew 0.13.3 → 0.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,6 +17,13 @@ const force = inputs.force;
17
17
 
18
18
  if (!crewHome) throw new Error("crewHome is required — e.g. ~/workspace/.crew");
19
19
 
20
+ // sq(p) — single-quote a path for safe interpolation into shell commands the
21
+ // agents run. A bare double-quoted path breaks on spaces, $, backticks, or
22
+ // quotes; single-quote wrapping is always safe.
23
+ function sq(p) {
24
+ return "'" + String(p).split("'").join("'\\''") + "'";
25
+ }
26
+
20
27
  // ── Pure decision helpers ──────────────────────────────────────────────
21
28
  // The agent is a sensor, not a judge: it reports raw facts (HOME, expanded
22
29
  // paths, existence, in-progress counts) and the verdicts below are computed
@@ -110,8 +117,8 @@ if (gateFacts.homeExists) {
110
117
  "Report whether this crew has live work. Do not judge — just report.\n\n" +
111
118
  "crewHome: " + crewHomeExpanded + "\n\n" +
112
119
  "Steps (run in shell):\n" +
113
- "1. If \"" + crewHomeExpanded + "/crew-state.db\" exists, run:\n" +
114
- " sqlite3 \"" + crewHomeExpanded + "/crew-state.db\" \"SELECT COUNT(*) FROM tasks WHERE state='in_progress';\"\n" +
120
+ "1. If " + sq(crewHomeExpanded + "/crew-state.db") + " exists, run:\n" +
121
+ " sqlite3 " + sq(crewHomeExpanded + "/crew-state.db") + " \"SELECT COUNT(*) FROM tasks WHERE state='in_progress';\"\n" +
115
122
  " Report the number. When the database is missing, the count is 0.\n" +
116
123
  "2. Return JSON { homeExists: true, inProgress: <number> }.",
117
124
  {
@@ -149,7 +156,7 @@ try {
149
156
  "Remove this crew instance's scheduler cron jobs — and nothing else.\n\n" +
150
157
  "crewHome: " + crewHomeExpanded + "\n\n" +
151
158
  "Steps:\n" +
152
- "1. Registry: if \"" + crewHomeExpanded + "/.cron-registry.json\" exists, read it\n" +
159
+ "1. Registry: if " + sq(crewHomeExpanded + "/.cron-registry.json") + " exists, read it\n" +
153
160
  " and take its crons[].id list as registry candidates.\n" +
154
161
  "2. Discovery: call cron_list. For every job whose id starts with 'crew-poll-',\n" +
155
162
  " call cron_view and keep it as a discovery candidate when the job body\n" +
@@ -203,8 +210,29 @@ try {
203
210
  if (!cronsResult.passed || !cronsResult.summary || !Array.isArray(cronsResult.summary.removed)) {
204
211
  return { __hatchWorkflowControl: "blocked", result: { blocked_reason: "Cron removal failed", message: "The cron-removal agent did not pass. No further teardown was attempted; the crew home is untouched." } };
205
212
  }
206
- var removedDesc = cronsResult.summary.removed.map(function (r) { return r.id + "=" + r.action; }).join(", ");
207
- log("Crons: " + (removedDesc || "(none belonged to this crew)"));
213
+ // ── Razor witness: the removal list above is the agent's report, not a
214
+ // fact. Re-read the cron registry independently and only assert removal for
215
+ // ids the re-read confirms absent. Agent returns are testimony, never
216
+ // evidence. Non-blocking: a surviving id is logged loudly, not asserted away.
217
+ var removedEntries = cronsResult.summary.removed;
218
+ var cronsGone = [], cronsPresent = [], cronsUnverified = [];
219
+ try {
220
+ var cronListOut = String(await agent(
221
+ "Call cron_list and return its full output verbatim. Do not summarize, interpret, or filter it.",
222
+ { key: "uninstall-crons-verify", label: "Re-reading cron registry after removal" }
223
+ ) || "");
224
+ removedEntries.forEach(function (r) {
225
+ if (cronListOut.indexOf(r.id) === -1) { cronsGone.push(r.id); }
226
+ else { cronsPresent.push(r.id); }
227
+ });
228
+ } catch (e) {
229
+ cronsUnverified = removedEntries.map(function (r) { return r.id; });
230
+ }
231
+ var cronBits = [];
232
+ if (cronsGone.length) { cronBits.push("removed (registry re-read confirms gone): " + cronsGone.join(", ")); }
233
+ if (cronsPresent.length) { cronBits.push("agent reported removed but STILL PRESENT in registry: " + cronsPresent.join(", ")); }
234
+ if (cronsUnverified.length) { cronBits.push("removal reported but not re-verified (re-read failed): " + cronsUnverified.join(", ")); }
235
+ log("Crons: " + (cronBits.join("; ") || "(none belonged to this crew)"));
208
236
 
209
237
  // ── Delete the crew home directory ─────────────────────────────────────
210
238
  phase("home");
@@ -217,10 +245,10 @@ if (gateFacts.homeExists) {
217
245
  "crewHome: " + crewHomeExpanded + "\n\n" +
218
246
  "Steps:\n" +
219
247
  "1. Best-effort worktree hygiene (failures do not block): if\n" +
220
- " \"" + crewHomeExpanded + "/crew-state.db\" exists, list the repo_path values\n" +
248
+ " " + sq(crewHomeExpanded + "/crew-state.db") + " exists, list the repo_path values\n" +
221
249
  " from its projects table and run 'git worktree prune' in each. Ignore errors.\n" +
222
- "2. Run: rm -rf \"" + crewHomeExpanded + "\"\n" +
223
- "3. Verify: test ! -e \"" + crewHomeExpanded + "\" — fail closed (passed: false)\n" +
250
+ "2. Run: rm -rf " + sq(crewHomeExpanded) + "\n" +
251
+ "3. Verify: test ! -e " + sq(crewHomeExpanded) + " — fail closed (passed: false)\n" +
224
252
  " if the path still exists.\n" +
225
253
  "4. Return { passed: true, summary: { removed: true } }.",
226
254
  {
@@ -248,8 +276,27 @@ if (gateFacts.homeExists) {
248
276
  if (!homeResult.passed || !homeResult.summary || homeResult.summary.removed !== true) {
249
277
  return { __hatchWorkflowControl: "blocked", result: { blocked_reason: "Crew home removal failed", message: "The crew home could not be deleted. Its crons were already removed; delete " + crewHomeExpanded + " manually." } };
250
278
  }
251
- homeRemoved = true;
252
- log("Crew home deleted: " + crewHomeExpanded);
279
+ // ── Razor witness: the agent's `removed` flag is testimony, not evidence.
280
+ // Confirm the directory is actually gone before claiming deletion. Gone is
281
+ // asserted only when the re-read says DIR_GONE without DIR_PRESENT (the
282
+ // check command itself names both markers, so requiring the absence of
283
+ // DIR_PRESENT keeps a mere echo of the command from counting as proof).
284
+ var homeGone = false;
285
+ try {
286
+ var homeCheckOut = String(await agent(
287
+ "Run in shell and return the stdout verbatim:\ntest -d " + sq(crewHomeExpanded) + " && echo DIR_PRESENT || echo DIR_GONE",
288
+ { key: "uninstall-home-verify", label: "Re-checking crew home directory after deletion" }
289
+ ) || "");
290
+ homeGone = homeCheckOut.indexOf("DIR_GONE") !== -1 && homeCheckOut.indexOf("DIR_PRESENT") === -1;
291
+ } catch (e) {
292
+ homeGone = false;
293
+ }
294
+ if (homeGone) {
295
+ homeRemoved = true;
296
+ log("Crew home deleted: " + crewHomeExpanded + " (directory re-read confirms it is gone)");
297
+ } else {
298
+ log("Crew home was reported deleted by the agent, but the directory is still present (or the re-check failed): " + crewHomeExpanded);
299
+ }
253
300
  } else {
254
301
  log("Crew home already gone — nothing to delete.");
255
302
  }
@@ -440,6 +440,7 @@ async function recordPublishLedger(entry, rework) {
440
440
  attempt: entry.attempt || null,
441
441
  agent_id: entry.agent_id || null,
442
442
  applied_report: entry.applied_report || null,
443
+ manifest_before: entry.manifest_before || null,
443
444
  outcome: entry.outcome,
444
445
  detail: entry.detail || ""
445
446
  });
@@ -453,9 +454,7 @@ async function recordPublishLedger(entry, rework) {
453
454
  label: "Recording publish attempt in ledger",
454
455
  schema: { type: "object", properties: { result: { type: "string" } }, required: ["result"] } }
455
456
  );
456
- var ok = !!(res && res.result && res.result.indexOf("LEDGER_OK") !== -1);
457
- log("Publish ledger: outcome '" + entry.outcome + "' for task " + taskId +
458
- (ok ? " recorded." : " NOT confirmed (" + ((res && res.result) || "no output") + ")"));
457
+ log("Noted publish outcome '" + entry.outcome + "' for task " + taskId + " in ledger");
459
458
  } catch (e) {
460
459
  log("Publish ledger: write failed for task " + taskId + " (non-fatal, observability only): " + (e && e.message ? e.message : e));
461
460
  }
@@ -1442,20 +1441,36 @@ while (i < STEPS.length) {
1442
1441
  // See docs/decisions/publish-path.md#durable-evidence-snapshot: snapshot the audit-dir listing BEFORE the trigger; fallback diffs before/after.
1443
1442
  var auditDirsBeforeTrigger = [];
1444
1443
  var auditBeforeOk = false;
1444
+ // Pre-trigger baselines (design §1.9): the audit-dir listing and the
1445
+ // manifest snapshot. The verified-path freshness check compares the
1446
+ // post-trigger manifest against the baseline (built_at advance +
1447
+ // content_sha256 change). Best-effort, never gates — a missing
1448
+ // manifest baseline fails the verified path closed.
1449
+ var preTriggerManifest = null;
1445
1450
  try {
1446
1451
  var auditBefore = await agent(
1447
- "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
1448
- "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null\n" +
1449
- "Return JSON { \"dirs\": \"<newline-separated names, empty string when the audits directory does not exist or is empty>\" } and nothing else.",
1450
- { key: attemptKey("publish-audit-before-" + taskId, totalReworkCount), label: "Snapshotting audit dirs before rebuild trigger",
1451
- schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1452
+ "Capture pre-trigger baselines for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshots, never gates).\n" +
1453
+ "Run: ls -1 ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/ 2>/dev/null; echo ---MANIFEST---; cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/.space-build/manifest.json 2>/dev/null\n" +
1454
+ "Return JSON { \"dirs\": \"<newline-separated names, empty string when missing>\", \"manifest\": \"<the manifest's full text, or empty string when missing/unreadable>\" } and nothing else.",
1455
+ { key: attemptKey("publish-baseline-before-" + taskId, totalReworkCount), label: "Snapshotting baselines before rebuild trigger",
1456
+ schema: { type: "object", properties: { dirs: { type: "string" }, manifest: { type: "string" } }, required: ["dirs", "manifest"] } }
1452
1457
  );
1453
1458
  auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1459
+ var manifestText = String((auditBefore && auditBefore.manifest) || "").trim();
1460
+ if (manifestText) {
1461
+ var manifestJson = JSON.parse(manifestText);
1462
+ preTriggerManifest = {
1463
+ built_at: typeof manifestJson.built_at === "string" ? manifestJson.built_at : null,
1464
+ content_sha256: typeof manifestJson.content_sha256 === "string" ? manifestJson.content_sha256 : null,
1465
+ };
1466
+ }
1467
+ // Arm only after both baselines parse.
1454
1468
  auditBeforeOk = true;
1455
- log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
1469
+ log("Publish pre-trigger baselines for task " + taskId + ": " + auditDirsBeforeTrigger.length + " audit dirs, manifest " + (preTriggerManifest ? "built_at=" + preTriggerManifest.built_at : "none"));
1456
1470
  } catch (auditBeforeErr) {
1457
- log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1471
+ log("Publish pre-trigger baselines failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1458
1472
  }
1473
+
1459
1474
  // See docs/decisions/publish-path.md#fire-and-forget-trigger: the trigger child returns immediately; the workflow owns observation and verdict.
1460
1475
  var publishToolsOk = false;
1461
1476
  var publishToolsMissing = false;
@@ -1660,6 +1675,7 @@ while (i < STEPS.length) {
1660
1675
  attempt: rebuildAttemptKey,
1661
1676
  agent_id: rebuildAgentId,
1662
1677
  applied_report: publishAppliedObservation,
1678
+ manifest_before: preTriggerManifest,
1663
1679
  outcome: "submitted",
1664
1680
  detail: "fire-and-forget trigger; build receipt captured by workflow-owned build-state observation (pre/post-trigger diff)"
1665
1681
  }, totalReworkCount);
@@ -1730,7 +1746,8 @@ while (i < STEPS.length) {
1730
1746
  attempt: rebuildAttemptKey,
1731
1747
  agent_id: null,
1732
1748
  applied_report: publishAppliedObservation,
1733
- outcome: "submitted",
1749
+ manifest_before: preTriggerManifest,
1750
+ outcome: "build-observed",
1734
1751
  detail: "durable audit evidence shows a build completed during the attempt window (no receipt agent_id — attribution by window, not identity; receipt poll bypassed (verdict decided), routed to parent verification)"
1735
1752
  }, totalReworkCount);
1736
1753
  log("Publish verdict LANDED for task " + taskId + ": a completed build was observed during the attempt window — receipt poll bypassed (verdict decided, no receipt to chain to), routing directly to parent verification.");
@@ -2002,7 +2019,8 @@ while (i < STEPS.length) {
2002
2019
  attempt: rebuildAttemptKey,
2003
2020
  agent_id: rebuildAgentId,
2004
2021
  applied_report: publishAppliedObservation,
2005
- outcome: "submitted",
2022
+ manifest_before: preTriggerManifest,
2023
+ outcome: "build-observed",
2006
2024
  detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestAuditDirAfterPoll + ", report ok=true); routed to parent verification"
2007
2025
  }, totalReworkCount);
2008
2026
  } else if (auditOkAfterPoll === false) {
@@ -2065,11 +2083,11 @@ while (i < STEPS.length) {
2065
2083
  if (!postDeploy.deployed) {
2066
2084
  return await parkTask("Post-deploy failed after a skipped publish: " + (postDeploy.output || "no output") + ". Nothing was published; worktree cleanup and lock state unknown — human attention needed.");
2067
2085
  }
2068
- log("Publish skipped cleanly for task " + taskId + " (no lock held) — post-deploy finalized cleanup");
2086
+ log("Publish skipped cleanly for task " + taskId + " (no lock held) — post-deploy step finished");
2069
2087
  } else {
2070
2088
  if (publishFailure) {
2071
2089
  return await parkTask(publishFailure + (postDeploy.deployed
2072
- ? " Post-deploy finalized cleanup."
2090
+ ? " Post-deploy step finished (cleanup status unknown)."
2073
2091
  : " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
2074
2092
  }
2075
2093
  if (!publishBuildLanded) {
@@ -2082,7 +2100,7 @@ while (i < STEPS.length) {
2082
2100
  // commit (design §1.5: "the trigger was sent for commit <short-sha>").
2083
2101
  if (publishUnknownFields) publishUnknownFields.commitShortSha = mergeCommitShortForPublish;
2084
2102
  return await parkTask(composeUnattributedParkReason(publishUnknownFields) + (postDeploy.deployed
2085
- ? " Post-deploy finalized cleanup."
2103
+ ? " Post-deploy step finished (cleanup status unknown)."
2086
2104
  : " Post-deploy also failed (" + (postDeploy.output || "no output") + ") — worktree and lock state unknown."));
2087
2105
  }
2088
2106
  if (!postDeploy.deployed) {
@@ -2111,7 +2129,7 @@ while (i < STEPS.length) {
2111
2129
  // artifact_edit would trigger a duplicate build.
2112
2130
  if (publishSkippedNoLock) {
2113
2131
  instructions = "Publish was skipped deterministically by the workflow before your step — do NOT call artifact_edit, artifact_status, setprovenance, or post-deploy yourself; doing so would disturb the finalized state.\n\n" +
2114
- "Integrate reported MERGED_EMPTY (the task branch had no commits ahead of main), so no merge lock was taken and there is nothing to ship. The workflow finalized cleanup via post-deploy.\n\n" +
2132
+ "Integrate reported MERGED_EMPTY (the task branch had no commits ahead of main), so no merge lock was taken and there is nothing to ship. Post-deploy step finished per its return — do NOT run post-deploy yourself.\n\n" +
2115
2133
  "Write plain prose describing the skip, then on its own line: VERDICT: PASS\n" +
2116
2134
  "The VERDICT line must be the last line of your report.";
2117
2135
  } else {
@@ -2740,6 +2758,7 @@ while (i < STEPS.length) {
2740
2758
  // means the parent's independent read-back has not happened yet; this
2741
2759
  // park is NOT proof the content is correct.
2742
2760
  return await parkTask("publish: verification-requested " + mergeCommitForPublish +
2761
+ " attempt=" + rebuildAttemptKey +
2743
2762
  " Do NOT republish: a duplicate build would re-publish the same change. " +
2744
2763
  "Artifact build landed, post-deploy finalized, provenance not stamped — the crew has not yet independently confirmed the live artifact contains exactly the change; waiting on the manual read-back in docs/publish-verification.md. " +
2745
2764
  "Appendix: build=" + (rebuildAgentId || "agent_id unobserved") + "; provenance=unstamped; content_check=pending.");
@@ -79,6 +79,12 @@ const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_
79
79
  // (2026-09-16: the silent default let a run resolve against the dev home).
80
80
  if (!inputs.crewHome) throw new Error("crewHome is required — pass the crew home explicitly; no default");
81
81
  const crewHome = inputs.crewHome;
82
+ // sq(p) — single-quote a path for safe interpolation into shell commands the
83
+ // agents run. A bare double-quoted path breaks on spaces, $, backticks, or
84
+ // quotes; single-quote wrapping is always safe.
85
+ function sq(p) {
86
+ return "'" + String(p).split("'").join("'\\''") + "'";
87
+ }
82
88
  // Crew API: the workflow calls the crew-owned CLI, not the dashboard.
83
89
  // The CLI implements the API.md contract against $CREW_HOME/crew-state.db.
84
90
  const CREW_API_SRC = crewHome + "/current/lib/crew-api.js";
@@ -91,7 +97,7 @@ let CREW_API = CREW_API_SRC;
91
97
  // returns the stdout verbatim (the CLI emits JSON on stdout).
92
98
  function crewCmd(command, args) {
93
99
  var json = JSON.stringify(args || {}).replace(/'/g, "'\\''");
94
- return "node " + CREW_API + " --crew-home " + crewHome + " " + command + " --json '" + json + "'";
100
+ return "node " + sq(CREW_API) + " --crew-home " + sq(crewHome) + " " + command + " --json '" + json + "'";
95
101
  }
96
102
  // Telemetry (2026-09-13): structured run timeline. The workflow mints a
97
103
  // telemetry run_id at startup via record-run-start (the API generates it
@@ -104,7 +110,7 @@ let telemetryBuffer = [];
104
110
  async function telemetryStart(workflowName) {
105
111
  try {
106
112
  const out = await agent(crewCmd("record-run-start", {
107
- task_id: taskId, workflow: workflowName, launched_by: "cron"
113
+ task_id: taskId, workflow: workflowName, launched_by: inputs.launched_by
108
114
  }), { key: "telemetry-start", label: "Recording run start" });
109
115
  const parsed = typeof out === "string" ? JSON.parse(out) : out;
110
116
  if (parsed && parsed.run_id) telemetryRunId = parsed.run_id;
@@ -178,7 +184,7 @@ const LAUNCH_PROJECT_ID = inputs.project_id || "";
178
184
  // parks fail-closed at Triage.
179
185
  const REPO_PATH = projectConfig.repo_path || "";
180
186
  // Env prefix baked into every lifecycle invocation the agents run.
181
- const LIFECYCLE_ENV = "CREW_HOME=" + crewHome + " CREW_REPO=" + REPO_PATH + " ";
187
+ const LIFECYCLE_ENV = "CREW_HOME=" + sq(crewHome) + " CREW_REPO=" + REPO_PATH + " ";
182
188
  // The stable release script — deploys, reports the live release.
183
189
  const DEPLOY_SCRIPT = crewHome + "/crew-release.sh";
184
190
 
@@ -195,7 +201,7 @@ if (!taskId) {
195
201
  function pinLifecycle(key) {
196
202
  return agent(
197
203
  "Snapshot lifecycle scripts for version pinning.\n" +
198
- "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
204
+ "Run: mkdir -p " + sq(RUN_LIB) + " && cp " + sq(LIFECYCLE_SRC) + " " + sq(LIFECYCLE) + " && cp " + sq(MERGE_LOCK_SRC) + " " + sq(MERGE_LOCK) + " && cp " + sq(PUBLISH_NPM_SRC) + " " + sq(PUBLISH_NPM) + " && cp " + sq(CREW_API_SRC) + " " + sq(CREW_API_PINNED) + " && cp " + sq(SCHEMA_SQL_SRC) + " " + sq(SCHEMA_SQL_PINNED) + " && cp " + sq(CLASSIFY_SURFACE_SRC) + " " + sq(CLASSIFY_SURFACE) + " && chmod +x " + sq(LIFECYCLE) + " " + sq(MERGE_LOCK) + " " + sq(PUBLISH_NPM) + " && ls -1 " + sq(RUN_LIB) + "\n" +
199
205
  "Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
200
206
  { key: key, label: "Pinning lifecycle scripts",
201
207
  schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
@@ -258,7 +264,7 @@ async function computeUpgradePlan(evidenceKey, noteIdentity) {
258
264
  (sourceIsRepo
259
265
  ? "2. The workflow was told source is repo. Check that " + REPO_PATH + "/workflows is a directory AND " + REPO_PATH + "/lib/crew-release.sh exists — repo_ok is true only if both hold. Run: git -C " + REPO_PATH + " rev-parse HEAD and capture the sha as head (empty string if the command fails).\n"
260
266
  : "2. The workflow was told source is not repo. Skip all repo checks: return repo_ok false and head as an empty string.\n") +
261
- "3. Run: " + DEPLOY_SCRIPT + " current " + crewHome + " — capture the full trimmed stdout as current.\n" +
267
+ "3. Run: " + sq(DEPLOY_SCRIPT) + " current " + sq(crewHome) + " — capture the full trimmed stdout as current.\n" +
262
268
  "Return JSON { \"source_line\": \"<verbatim>\", \"repo_ok\": <bool>, \"head\": \"<sha or empty>\", \"current\": \"<trimmed stdout>\" } and nothing else.",
263
269
  {
264
270
  key: evidenceKey,
@@ -583,8 +589,8 @@ while (i < STEPS.length) {
583
589
  const stagingDir = crewHome + "/.upgrade-staging/" + upgradePlan.version;
584
590
  const deployShell =
585
591
  (upgradePlan.source === "repo")
586
- ? DEPLOY_SCRIPT + " deploy " + REPO_PATH + " " + crewHome
587
- : "STAGING=\"" + stagingDir + "\" && mkdir -p \"$STAGING\" && cd \"$STAGING\" && npm init -y >/dev/null 2>&1 && npm install muse-crew@" + upgradePlan.version;
592
+ ? sq(DEPLOY_SCRIPT) + " deploy " + REPO_PATH + " " + sq(crewHome)
593
+ : "STAGING=" + sq(stagingDir) + " && mkdir -p \"$STAGING\" && cd \"$STAGING\" && npm init -y >/dev/null 2>&1 && npm install muse-crew@" + upgradePlan.version;
588
594
  const deploySteps =
589
595
  "1. Run the install (npm source only):\n" + (upgradePlan.source === "repo" ? " (skipped — repo source has nothing to install)\n" : " " + deployShell + "\n") +
590
596
  "2. Run the deploy:\n " + (upgradePlan.source === "repo"
@@ -594,10 +600,10 @@ while (i < STEPS.length) {
594
600
  // so the deploy source is the installed package root, not the
595
601
  // staging root. Do not rely on a $STAGING shell variable persisting
596
602
  // between separate shell invocations.
597
- : DEPLOY_SCRIPT + " deploy \"" + stagingDir + "/node_modules/muse-crew\" " + crewHome) +
603
+ : sq(DEPLOY_SCRIPT) + " deploy " + sq(stagingDir + "/node_modules/muse-crew") + " " + sq(crewHome)) +
598
604
  " — capture ALL of the command's output and its exit code (run the command, then echo EXIT_CODE=$?).\n" +
599
605
  (upgradePlan.source === "npm"
600
- ? "3. Remove the staging dir: rm -rf \"" + stagingDir + "\" — best-effort, ALWAYS, even when the deploy fails. Log whether the removal succeeded. Never let cleanup change the deploy outcome.\n"
606
+ ? "3. Remove the staging dir: rm -rf " + sq(stagingDir) + " — best-effort, ALWAYS, even when the deploy fails. Log whether the removal succeeded. Never let cleanup change the deploy outcome.\n"
601
607
  : "") +
602
608
  "Do not run git checkout, git pull, or any repo-mutating command. Do not publish to npm — the npm source only INSTALLS the published package. Do not touch the scheduler.";
603
609
  var deployResult;
@@ -630,9 +636,12 @@ while (i < STEPS.length) {
630
636
  await recordPhase("Deploy", step.identity, activeSessionId, "failed", msg);
631
637
  return await parkTask(msg);
632
638
  }
633
- log("Deploy succeeded for task " + taskId + " — target " + upgradePlan.target);
639
+ // Razor: exit 0 is the deploy agent's testimony, not proof the release
640
+ // swap landed. Success is claimed only after Verify's independent checks
641
+ // pass below.
642
+ log("Deploy agent returned exit 0 for task " + taskId + " — target " + upgradePlan.target + " (Verify checks pending)");
634
643
  await recordPhase("Deploy", step.identity, activeSessionId, "completed",
635
- "deployed " + upgradePlan.target + " (" + upgradePlan.source + ")\ndeploy output:\n" + deployOutput.slice(0, 1500));
644
+ "deploy command exit 0 for " + upgradePlan.target + " (" + upgradePlan.source + ") — Verify checks pending\ndeploy output:\n" + deployOutput.slice(0, 1500));
636
645
  i++;
637
646
  continue;
638
647
  }
@@ -663,7 +672,7 @@ while (i < STEPS.length) {
663
672
  curCheck = await agent(
664
673
  TOOL_CHECK_PREAMBLE +
665
674
  "Read the live release identity.\n" +
666
- "Run: " + DEPLOY_SCRIPT + " current " + crewHome + "\n" +
675
+ "Run: " + sq(DEPLOY_SCRIPT) + " current " + sq(crewHome) + "\n" +
667
676
  "Return JSON { \"current\": \"<trimmed stdout>\" } and nothing else.",
668
677
  {
669
678
  key: "verify-current",
@@ -695,7 +704,7 @@ while (i < STEPS.length) {
695
704
  regCheck = await agent(
696
705
  TOOL_CHECK_PREAMBLE +
697
706
  "Read the deployed registry.\n" +
698
- "Run: cat " + crewHome + "/workflows/registry.json\n" +
707
+ "Run: cat " + sq(crewHome + "/workflows/registry.json") + "\n" +
699
708
  "Return JSON { \"raw\": \"<the file's full content, verbatim>\" } and nothing else.",
700
709
  {
701
710
  key: "verify-registry",
@@ -740,7 +749,7 @@ while (i < STEPS.length) {
740
749
  gsCheck = await agent(
741
750
  TOOL_CHECK_PREAMBLE +
742
751
  "Check the crew database is readable by the OLD release's CLI. Use the PINNED path below — NOT " + crewHome + "/current — because the workflow swapped `current` under itself at Deploy and the live path may already point at the new release.\n" +
743
- "Run: node " + CREW_API_PINNED + " --crew-home " + crewHome + " get-state --json '{}'; echo EXIT_CODE=$?\n" +
752
+ "Run: node " + sq(CREW_API_PINNED) + " --crew-home " + sq(crewHome) + " get-state --json '{}'; echo EXIT_CODE=$?\n" +
744
753
  "Return JSON { \"exit\": <the exit code as an integer> } and nothing else.",
745
754
  {
746
755
  key: "verify-get-state",
@@ -762,6 +771,11 @@ while (i < STEPS.length) {
762
771
  }
763
772
  log("Verify check 3 passed for task " + taskId + ": pinned crew-api.js get-state exits 0");
764
773
 
774
+ // The success claim lives here now: all three independent checks re-read
775
+ // the world and confirm the swap — the deploy agent's exit 0 alone never
776
+ // proved it.
777
+ log("Deploy succeeded for task " + taskId + " — target " + target + " (Verify: live release, registry, and pinned CLI all confirm)");
778
+
765
779
  await agent(
766
780
  "Log the completed upgrade.\n" +
767
781
  "Run in shell and return the stdout verbatim:\n" + crewCmd("log-event", {