muse-crew 0.7.20 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,794 @@
1
+ export const meta = {
2
+ name: "crew-upgrade",
3
+ description: "Upgrade the crew itself: deploy a new release and hand over to it on the next tick.",
4
+ phases: ["Triage", "Deploy", "Verify"],
5
+ steps: [
6
+ { name: "Triage", identity: "sage" },
7
+ { name: "Deploy", identity: "wren" },
8
+ { name: "Verify", identity: "wren" }
9
+ ],
10
+ reworkTarget: "Deploy"
11
+ };
12
+
13
+ // ── Pure helpers (upgrade.js) ─────────────────────────────────────────
14
+ // No I/O, no clock, no randomness. Tests extract these by balanced-brace
15
+ // matching (the merge-record.test.js pattern), so they are plain top-level
16
+ // function declarations with no unbalanced braces inside string literals.
17
+
18
+ // The upgrade source comes from a `source:` line in the task description:
19
+ // source: repo — upgrade to the task project's repo HEAD (default when the line is absent)
20
+ // source: npm@<x.y.z> — upgrade to the published npm package muse-crew@<x.y.z>
21
+ // Returns the trimmed token after `source:`, or "repo" when no line is present.
22
+ function parseUpgradeSource(description) {
23
+ var m = /^source:\s*(\S+)/im.exec(description || "");
24
+ return m ? m[1].trim() : "repo";
25
+ }
26
+
27
+ // npm versions are exactly x.y.z — no tags, no ranges, no dist-tags.
28
+ function isValidNpmVersion(v) {
29
+ return /^\d+\.\d+\.\d+$/.test(v || "");
30
+ }
31
+
32
+ // Resolve the release identity for a source. Matches crew-release.sh:
33
+ // git source deploys as the HEAD sha, npm source as pkg-<version>.
34
+ function upgradeTarget(source, version, repoHead) {
35
+ if (source === "repo") {
36
+ if (!/^[0-9a-f]{40}$/.test(repoHead || "")) {
37
+ return { ok: false, error: "repo HEAD is not a 40-char hex sha: '" + (repoHead || "") + "'" };
38
+ }
39
+ return { ok: true, target: repoHead };
40
+ }
41
+ if (source === "npm") {
42
+ if (!isValidNpmVersion(version)) {
43
+ return { ok: false, error: "invalid npm version: '" + (version || "") + "' — expected x.y.z" };
44
+ }
45
+ return { ok: true, target: "pkg-" + version };
46
+ }
47
+ return { ok: false, error: "unknown upgrade source: '" + source + "'" };
48
+ }
49
+
50
+ // Idempotency: exact string equality of current and target release identities.
51
+ function decideNoOp(current, target) {
52
+ return current === target;
53
+ }
54
+
55
+ const inputs = args ?? {};
56
+ const taskId = inputs.task_id;
57
+ const taskTitle = inputs.task_title || "";
58
+ const taskDescription = inputs.task_description || "";
59
+ const startStepIndex = inputs.start_step_index || 0;
60
+
61
+ // Dispatcher-carried facts for the claim-time persist: the workflow name the
62
+ // dispatcher actually resolved and launched, and whether the task record had
63
+ // no workflow (the standard fallback). The launched workflow writes the
64
+ // resolution back via updatetask in the self-claim below.
65
+ const RESOLVED_WORKFLOW = inputs.resolved_workflow || null;
66
+ const WORKFLOW_WAS_NULL = inputs.workflow_was_null === true;
67
+ // One-shot recovery routing: the dispatcher sets inputs.next_phase when it
68
+ // routes this run via an explicit recover-task redirect. The value is
69
+ // consumed (cleared) atomically by the successful self-claim below:
70
+ // claim-task takes expected_next_phase and clears the matching next_phase in
71
+ // the same transaction as the winning session insert, so no platform death
72
+ // can slip between claim and consumption and replay the routing. A stale or
73
+ // superseded routing survives — only an exact match clears.
74
+ const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_phase.length > 0) ? inputs.next_phase : null;
75
+
76
+ // crewHome is required — the dispatcher always passes it (crew-dispatch.js
77
+ // throws without it). Fail closed instead of silently defaulting to the dev
78
+ // home: a missing home is a loud error, a wrong home is silent corruption
79
+ // (2026-09-16: the silent default let a run resolve against the dev home).
80
+ if (!inputs.crewHome) throw new Error("crewHome is required — pass the crew home explicitly; no default");
81
+ const crewHome = inputs.crewHome;
82
+ // Crew API: the workflow calls the crew-owned CLI, not the dashboard.
83
+ // The CLI implements the API.md contract against $CREW_HOME/crew-state.db.
84
+ const CREW_API_SRC = crewHome + "/current/lib/crew-api.js";
85
+ // Pinned at pinLifecycle: after the pin, CREW_API points into RUN_LIB so a
86
+ // mid-flight release swap cannot change the CLI under a running workflow.
87
+ // Load-bearing here: this workflow swaps `current` under itself at Deploy.
88
+ let CREW_API = CREW_API_SRC;
89
+ // Build a shell command invoking the CLI. Args are JSON-encoded and
90
+ // single-quote-wrapped for safe shell passing. The agent runs this and
91
+ // returns the stdout verbatim (the CLI emits JSON on stdout).
92
+ function crewCmd(command, args) {
93
+ var json = JSON.stringify(args || {}).replace(/'/g, "'\\''");
94
+ return "node " + CREW_API + " --crew-home " + crewHome + " " + command + " --json '" + json + "'";
95
+ }
96
+ // Telemetry (2026-09-13): structured run timeline. The workflow mints a
97
+ // telemetry run_id at startup via record-run-start (the API generates it
98
+ // server-side — workflow JS never touches the clock or randomness) and
99
+ // records milestone events. Timestamps come from SQLite datetime('now').
100
+ // Fire-and-forget: telemetry must never break the run (no schema — the
101
+ // run-5/run-6 fire-and-forget rule).
102
+ let telemetryRunId = null;
103
+ let telemetryBuffer = [];
104
+ async function telemetryStart(workflowName) {
105
+ try {
106
+ const out = await agent(crewCmd("record-run-start", {
107
+ task_id: taskId, workflow: workflowName, launched_by: "cron"
108
+ }), { key: "telemetry-start", label: "Recording run start" });
109
+ const parsed = typeof out === "string" ? JSON.parse(out) : out;
110
+ if (parsed && parsed.run_id) telemetryRunId = parsed.run_id;
111
+ } catch (e) {
112
+ log("Telemetry start failed (non-fatal): " + (e && e.message ? e.message : e));
113
+ }
114
+ }
115
+ async function telemetryEvent(eventName, detail) {
116
+ // Buffer non-critical events; flush at critical points.
117
+ if (!telemetryRunId) return;
118
+ telemetryBuffer.push({ event_name: eventName, detail: detail || "" });
119
+ // Auto-flush if buffer gets large (avoid unbounded memory).
120
+ if (telemetryBuffer.length >= 10) {
121
+ await telemetryFlush();
122
+ }
123
+ }
124
+ async function telemetryFlush() {
125
+ if (!telemetryRunId || telemetryBuffer.length === 0) return;
126
+ const events = telemetryBuffer.splice(0, telemetryBuffer.length);
127
+ try {
128
+ // Send all buffered events in a single agent() call via a batch command.
129
+ await agent(crewCmd("record-run-events-batch", {
130
+ run_id: telemetryRunId, task_id: taskId, events: events
131
+ }), { key: "telemetry-flush", label: "Flushing " + events.length + " telemetry events" });
132
+ } catch (e) {
133
+ log("Telemetry flush failed (non-fatal): " + events.length + " events lost");
134
+ }
135
+ }
136
+ async function telemetryEnd(status) {
137
+ if (!telemetryRunId) return;
138
+ // Flush any buffered events before recording the end.
139
+ await telemetryFlush();
140
+ try {
141
+ await agent(crewCmd("record-run-end", {
142
+ run_id: telemetryRunId, status: status
143
+ }), { key: "telemetry-end", label: "Recording run end" });
144
+ } catch (e) {
145
+ log("Telemetry end failed (non-fatal)");
146
+ }
147
+ }
148
+ const ORCH_PATH = crewHome + "/.orchestration";
149
+ // Pin lifecycle scripts to this run
150
+ const LIFECYCLE_SRC = crewHome + "/lib/worktree-lifecycle.sh";
151
+ const MERGE_LOCK_SRC = crewHome + "/lib/merge-lock.sh";
152
+ const RUN_LIB = crewHome + "/.pins/" + taskId; // persistent disk, NOT /tmp (tmpfs wiped by cell reboots — canary b5efd1b1)
153
+ const LIFECYCLE = RUN_LIB + "/worktree-lifecycle.sh";
154
+ const MERGE_LOCK = RUN_LIB + "/merge-lock.sh";
155
+ const PUBLISH_NPM_SRC = crewHome + "/lib/publish-npm.sh";
156
+ const PUBLISH_NPM = RUN_LIB + "/publish-npm.sh";
157
+ const CREW_API_PINNED = RUN_LIB + "/crew-api.js";
158
+ const SCHEMA_SQL_SRC = crewHome + "/lib/schema.sql";
159
+ const SCHEMA_SQL_PINNED = RUN_LIB + "/schema.sql";
160
+ // The five basenames the pin step must materialize — asserted mechanically
161
+ // by workflow code from the verbatim listing, never from agent prose.
162
+ const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED].map(function (p) { return p.split("/").pop(); });
163
+
164
+ // Project config — passed by dispatcher, falls back to dashboard defaults
165
+ const projectConfig = inputs.project_config || {};
166
+ // Project this run was dispatched for — the mid-run project-change guard
167
+ // compares the task's live project against this on every phase boundary.
168
+ const LAUNCH_PROJECT_ID = inputs.project_id || "";
169
+ // REPO_PATH is READ-ONLY in this workflow: it is used only to resolve
170
+ // `git rev-parse HEAD` for the repo upgrade source. Never written to, never
171
+ // a worktree target. Unlike chore.js there is no startup throw here: the
172
+ // npm source needs no repo, and a missing repo_path for the repo source
173
+ // parks fail-closed at Triage.
174
+ const REPO_PATH = projectConfig.repo_path || "";
175
+ // Env prefix baked into every lifecycle invocation the agents run.
176
+ const LIFECYCLE_ENV = "CREW_HOME=" + crewHome + " CREW_REPO=" + REPO_PATH + " ";
177
+ // The stable release script — deploys, reports the live release.
178
+ const DEPLOY_SCRIPT = crewHome + "/crew-release.sh";
179
+
180
+ if (!taskId) {
181
+ throw new Error("task_id is required in args");
182
+ }
183
+
184
+ // pinLifecycle(key) — snapshot the lifecycle scripts into RUN_LIB and return
185
+ // the verbatim `ls -1` listing so WORKFLOW CODE asserts the five pinned
186
+ // basenames; the agent cannot self-certify. (The pin step was the one place
187
+ // the workflows trusted agent prose: task 24be1cd6 walked to Publish on an
188
+ // empty pin dir.) Byte-identical across standard/bugfix/chore — pinned by
189
+ // tests/pin-location.test.js.
190
+ function pinLifecycle(key) {
191
+ return agent(
192
+ "Snapshot lifecycle scripts for version pinning.\n" +
193
+ "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
194
+ "Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
195
+ { key: key, label: "Pinning lifecycle scripts",
196
+ schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
197
+ );
198
+ }
199
+ // parsePinListing(result) — basenames from a pin/verify listing.
200
+ // Byte-identical across standard/bugfix/chore — pinned by
201
+ // tests/pin-location.test.js.
202
+ function parsePinListing(result) {
203
+ return (result && result.listing ? result.listing : "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
204
+ }
205
+ // TOOL_CHECK_PREAMBLE - every work agent runs this first. The artifact tool
206
+ // namespace is deferred for workflow children: present but invisible until
207
+ // the child loads it via tool_search.load_tool_namespace (bug 3472bf36 root
208
+ // cause, verified 2026-09-11 by direct probe: 5/5 children self-loaded it;
209
+ // the "non-deterministic platform flake" was children never being told to
210
+ // load it). The two signal lines are the ONLY machine-read tool-availability
211
+ // evidence - the workflow never guesses from English prose.
212
+ // Byte-identical across standard/bugfix/chore/docs - pinned by
213
+ // tests/artifact-tools.test.js.
214
+ var TOOL_CHECK_PREAMBLE =
215
+ "TOOL CHECK (do this first, before any other work):\n" +
216
+ "1. Call tool_search.load_tool_namespace with paths [\"artifact\"].\n" +
217
+ "2. Write exactly one line: artifact_tools: ok - or artifact_tools: missing if the call failed or the tool does not exist.\n" +
218
+ "3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
219
+ "Then do the assignment below.\n\n";
220
+
221
+ // STEPS inline — export const meta is parsed as metadata, not a runtime binding
222
+ const STEPS = [
223
+ { name: "Triage", identity: "sage" },
224
+ { name: "Deploy", identity: "wren" },
225
+ { name: "Verify", identity: "wren" }
226
+ ];
227
+ const REWORK_STEP = "Deploy"; // meta.reworkTarget; kept in the phase loop, but every failure parks fail-closed
228
+ let reworkCount = 0; // never incremented: no Review phase, so rework never routes in practice
229
+ let i = startStepIndex;
230
+ // Triage computes these; later phases consume them. A resumed run (the
231
+ // dispatcher launches at the next step after a completed phase, so Triage
232
+ // never re-executes) reconstructs the plan deterministically through
233
+ // computeUpgradePlan before Deploy/Verify — the plan is pure evidence +
234
+ // mechanical decisions, so re-gathering it is safe.
235
+ let upgradePlan = null; // { source, version, target, current }
236
+ let skipDeploy = false; // idempotent no-op: current == target
237
+
238
+ // Gather upgrade evidence and compute the plan mechanically. The agent does
239
+ // shell I/O only and returns a schema'd object; WORKFLOW CODE re-parses the
240
+ // `source:` line and makes every decision — the agent's reading is evidence,
241
+ // never the decision. Used by Triage and, on resumed runs, to reconstruct
242
+ // the plan before Deploy/Verify.
243
+ async function computeUpgradePlan(evidenceKey, noteIdentity) {
244
+ const srcToken = parseUpgradeSource(taskDescription);
245
+ const sourceIsRepo = (srcToken === "repo");
246
+ var triageResult;
247
+ try {
248
+ triageResult = await agent(
249
+ TOOL_CHECK_PREAMBLE +
250
+ "Collect upgrade evidence for a crew self-upgrade task. You do shell I/O only — every decision is made by the workflow from the values you return.\n" +
251
+ "Task description:\n" + taskDescription + "\n\n" +
252
+ "1. Find the first line of the task description matching /^source:/im. Echo it verbatim as source_line (empty string if no such line exists).\n" +
253
+ (sourceIsRepo
254
+ ? "2. The workflow was told source is repo. Check that " + REPO_PATH + "/workflows is a directory AND " + REPO_PATH + "/lib/crew-release.sh exists — repo_ok is true only if both hold. Run: git -C " + REPO_PATH + " rev-parse HEAD and capture the sha as head (empty string if the command fails).\n"
255
+ : "2. The workflow was told source is not repo. Skip all repo checks: return repo_ok false and head as an empty string.\n") +
256
+ "3. Run: " + DEPLOY_SCRIPT + " current " + crewHome + " — capture the full trimmed stdout as current.\n" +
257
+ "Return JSON { \"source_line\": \"<verbatim>\", \"repo_ok\": <bool>, \"head\": \"<sha or empty>\", \"current\": \"<trimmed stdout>\" } and nothing else.",
258
+ {
259
+ key: evidenceKey,
260
+ label: "Collecting upgrade evidence",
261
+ schema: {
262
+ type: "object",
263
+ properties: {
264
+ source_line: { type: "string" },
265
+ repo_ok: { type: "boolean" },
266
+ head: { type: "string" },
267
+ current: { type: "string" }
268
+ },
269
+ required: ["source_line", "repo_ok", "head", "current"]
270
+ }
271
+ }
272
+ );
273
+ } catch (e) {
274
+ return { ok: false, message: "Triage evidence collection failed: " + (e && e.message ? e.message : e) + ". Fail-closed." };
275
+ }
276
+ const sourceLine = (triageResult.source_line || "").trim();
277
+ const current = (triageResult.current || "").trim();
278
+ const head = (triageResult.head || "").trim();
279
+
280
+ // Workflow-side decisions only.
281
+ let source = null, version = null;
282
+ if (srcToken === "repo") {
283
+ source = "repo";
284
+ } else {
285
+ const msrc = /^npm@(\S+)$/.exec(srcToken);
286
+ if (msrc && isValidNpmVersion(msrc[1])) { source = "npm"; version = msrc[1]; }
287
+ }
288
+ if (!source) {
289
+ return { ok: false, message: "Triage rejected: unrecognized upgrade source line: '" + sourceLine + "' — expected 'source: repo' or 'source: npm@<x.y.z>'" };
290
+ }
291
+ if (source === "repo") {
292
+ // Fail closed on a missing repo_path — never silently fall back to
293
+ // another checkout (the canary's wrong-repo Build, 2026-09-11). The
294
+ // dispatcher skips unconfigured projects; this is the backstop for
295
+ // direct launches.
296
+ if (!REPO_PATH) {
297
+ return { ok: false, message: "Triage rejected: source is repo but project '" + (inputs.project_id || "unknown") + "' has no repo_path configured — set it via updateproject before dispatching tasks." };
298
+ }
299
+ if (!triageResult.repo_ok) {
300
+ return { ok: false, message: "Triage rejected: repo_path '" + REPO_PATH + "' does not look like a crew repo (workflows dir or lib/crew-release.sh missing)." };
301
+ }
302
+ }
303
+ const t = upgradeTarget(source, version, head);
304
+ if (!t.ok) {
305
+ return { ok: false, message: "Triage rejected: " + t.error };
306
+ }
307
+ const noOp = decideNoOp(current, t.target);
308
+ log("Upgrade plan for task " + taskId + ": " + current + " -> " + t.target + " (" + source + ")" + (noOp ? " — idempotent no-op" : ""));
309
+ await agent(
310
+ "Log the upgrade plan.\n" +
311
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("log-event", {
312
+ task_id: taskId, type: "note", identity: noteIdentity,
313
+ message: "upgrade plan: " + current + " -> " + t.target + " (" + source + ")" + (noOp ? " — no-op" : "")
314
+ }) + "\n",
315
+ { key: evidenceKey + "-plan-note", label: "Logging upgrade plan" }
316
+ );
317
+ return { ok: true, plan: { source: source, version: version, target: t.target, current: current }, skipDeploy: noOp };
318
+ }
319
+
320
+ // Park the task for human attention and end the run. "blocked" is never
321
+ // manually authored — the dashboard derives it mechanically from unmet
322
+ // dependencies — so a workflow outcome that needs a human parks the task
323
+ // instead. Parking is one atomic dashboard action (parktask): the parked
324
+ // state and the explanatory note land in one transaction, never half.
325
+ // The dispatcher skips parked tasks; a human moving parked→todo
326
+ // mechanically resets the retry counters. Returns the workflow result
327
+ // envelope the launcher sees. If the park call itself fails, the run
328
+ // reports "failed" (retryable) so the next tick re-attempts the park —
329
+ // a lost park is never reported as parked.
330
+ // Terminal cleanup: the run's last act at every park/fail boundary. A run
331
+ // that parks or fails must not leak its worktree, branch, or merge lock.
332
+ // The lifecycle's terminal-cleanup releases the lock unconditionally and
333
+ // reclaims the worktree+branch ONLY when the task branch is fully merged
334
+ // into main (then it is redundant); unmerged work is preserved for the
335
+ // human by design. This workflow never acquires the lock or creates a
336
+ // worktree, so the merge-lock release and branch reclamation are harmless
337
+ // no-ops here — the pair is kept verbatim anyway. Fire-and-forget with one
338
+ // bounded retry — the merge-lock lease expiry and the orphan sweep are the
339
+ // backstop for a dead transport.
340
+ async function terminalCleanup() {
341
+ for (var attempt = 1; attempt <= 2; attempt++) {
342
+ try {
343
+ await agent(
344
+ "Run in shell and return the stdout verbatim:\n" + LIFECYCLE_ENV + " terminal-cleanup " + taskId,
345
+ { key: "terminal-cleanup" + (attempt > 1 ? "-retry" : ""),
346
+ label: "Terminal cleanup (merged-branch reclamation)" + (attempt > 1 ? " (retry)" : "") }
347
+ );
348
+ return;
349
+ } catch (cleanupErr) {
350
+ log("Terminal cleanup attempt " + attempt + " failed for task " + taskId + ": " + (cleanupErr && cleanupErr.message ? cleanupErr.message : cleanupErr));
351
+ }
352
+ }
353
+ log("Terminal cleanup exhausted for task " + taskId + " — merge-lock lease expiry and orphan sweep are the backstop");
354
+ }
355
+ async function parkTask(reason) {
356
+ log("Parking task " + taskId + " for human attention: " + reason);
357
+ var parkMessage = ("Parked: " + reason).slice(0, 1000);
358
+ try {
359
+ await agent(
360
+ "Park this task for human attention.\n" +
361
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("park-task", { task_id: taskId, message: parkMessage }) + "\n" +
362
+ "The parked state is the human-attention signal — the dispatcher skips parked tasks.",
363
+ { key: "park-task", label: "Parking task for human attention" }
364
+ );
365
+ } catch (parkErr) {
366
+ log("PARK FAILED for task " + taskId + ": " + (parkErr && parkErr.message ? parkErr.message : parkErr) + " — park did not land, reporting failed so the next tick retries");
367
+ await terminalCleanup();
368
+ return { status: "failed", task_id: taskId, reason: "park failed: " + reason, park_failed: true };
369
+ }
370
+ await terminalCleanup();
371
+ await telemetryEnd("parked");
372
+ return { status: "parked", task_id: taskId, reason: reason };
373
+ }
374
+ // recordPhase(stepName, identity, sessionId, status, notes) — the crew-api
375
+ // composite: session + event in one transaction. One call per phase.
376
+ async function recordPhase(stepName, identity, sessionId, status, notes) {
377
+ await agent(
378
+ "Update session and log event.\n" +
379
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
380
+ task_id: taskId,
381
+ session: { id: sessionId, task_id: taskId, identity: identity, step: stepName, status: status, notes: notes },
382
+ event: { task_id: taskId, type: status, identity: identity, message: stepName + " " + status + " by " + identity }
383
+ }) + "\n",
384
+ { key: "record-" + stepName, label: "Recording " + stepName + " result" }
385
+ );
386
+ }
387
+
388
+ // Telemetry: mark run start before anything else (pin timing is the
389
+ // 2026-09-13 mystery — the pin agent took 15m with no visibility).
390
+ await telemetryStart("crew-upgrade");
391
+
392
+ // ── Pin lifecycle scripts ────────────────────────────────────────────
393
+ // Copy lifecycle scripts into a per-task temp dir so this run is immune
394
+ // to upgrades that land while it's in flight. Verified mechanically:
395
+ // workflow code asserts the five basenames from the verbatim listing —
396
+ // the agent cannot self-certify. Any miss parks the task before Triage.
397
+ // Load-bearing in this workflow: Deploy swaps `current` under the run, so
398
+ // every crew-api call after the pin goes through the OLD release's CLI.
399
+ await telemetryEvent("pin_start");
400
+ const initialPins = parsePinListing(await pinLifecycle("pin-lifecycle"));
401
+ await telemetryEvent("pin_end", "pinned " + initialPins.length + " files");
402
+ const missingInitialPins = PIN_BASENAMES.filter(function (b) { return initialPins.indexOf(b) === -1; });
403
+ if (missingInitialPins.length > 0) {
404
+ return await parkTask("Lifecycle pin incomplete before Triage — missing " + missingInitialPins.join(", ") + " in " + RUN_LIB + ".");
405
+ }
406
+ log("Lifecycle scripts pinned to " + RUN_LIB);
407
+ // From here on, every crew-api.js invocation uses the pinned copy: immune
408
+ // to a release swap landing mid-flight.
409
+ CREW_API = CREW_API_PINNED;
410
+ log("Crew API pinned to " + CREW_API);
411
+
412
+ while (i < STEPS.length) {
413
+ const step = STEPS[i];
414
+ const isFirstClaim = (i === startStepIndex && reworkCount === 0);
415
+
416
+ phase(step.name);
417
+ log(step.name + " step (" + step.identity + ") for task " + taskId);
418
+ await telemetryEvent("phase_start", step.name);
419
+
420
+ // ── Project-change guard ───────────────────────────────────────────
421
+ // A task moved to another project mid-run must not keep working in the
422
+ // old project's repo. Re-read the task's project at every phase
423
+ // boundary: if it differs from the project this run was dispatched for,
424
+ // abort the stale run (failed at the rework target) so the dispatcher
425
+ // re-launches the step with the new project's config. No worktree to
426
+ // clean up here — this workflow never creates one.
427
+ if (LAUNCH_PROJECT_ID) {
428
+ const projectCheck = await agent(
429
+ "Read this task's current project from the crew API.\n" +
430
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("get-state", { events_limit: 1 }) + "\n" +
431
+ "Find the task with id \"" + taskId + "\" in the returned tasks array.\n" +
432
+ "Return exactly { \"project\": \"<the task's project field, or empty string if absent>\" } and nothing else.",
433
+ {
434
+ key: "project-check-" + step.name,
435
+ label: "Checking project before " + step.name,
436
+ schema: { type: "object", properties: { project: { type: "string" } }, required: ["project"] }
437
+ }
438
+ );
439
+ const currentProject = (projectCheck && projectCheck.project) ? projectCheck.project : LAUNCH_PROJECT_ID;
440
+ if (currentProject !== LAUNCH_PROJECT_ID) {
441
+ const abortMessage = "Task project changed mid-run from '" + LAUNCH_PROJECT_ID + "' to '" + currentProject + "' — aborting stale run. The dispatcher will re-launch from " + REWORK_STEP + " with the new project context.";
442
+ log(abortMessage);
443
+ await agent(
444
+ "Abort the stale run.\n" +
445
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("record-phase", {
446
+ task_id: taskId,
447
+ session: { task_id: taskId, identity: step.identity, step: REWORK_STEP, status: "failed",
448
+ notes: abortMessage + " No worktree exists — this workflow never creates one." },
449
+ event: { task_id: taskId, type: "failed", identity: step.identity, message: abortMessage }
450
+ }) + "\n",
451
+ { key: "abort-project-change", label: "Aborting stale run (project changed)" }
452
+ );
453
+ return { status: "failed", task_id: taskId, reason: abortMessage };
454
+ }
455
+ }
456
+
457
+ // Self-claim — the dispatcher only recommends; the launched workflow claims
458
+ // the task as its first action, so a claim can never exist without a launched
459
+ // agent behind it. If another run claimed the task first (two poll ticks
460
+ // raced in the window before this run's claim), claimtask returns
461
+ // claimed:false and this run stands down as a duplicate.
462
+ let activeSessionId;
463
+ if (isFirstClaim) {
464
+ var firstClaimUpdateArgs = { id: taskId, state: "in_progress" };
465
+ if (WORKFLOW_WAS_NULL && RESOLVED_WORKFLOW) firstClaimUpdateArgs.workflow = RESOLVED_WORKFLOW;
466
+ const claimResult = await agent(
467
+ "Claim this task for the " + step.name + " step.\n" +
468
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("update-task", firstClaimUpdateArgs) + "\n" +
469
+ "Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started", ...(NEXT_PHASE_ROUTED ? { expected_next_phase: NEXT_PHASE_ROUTED } : {}) }) + "\n" +
470
+ "If the claim response has claimed=true, then run in shell and return the stdout verbatim:\n" + crewCmd("clear-reservation", { task_id: taskId }) + "\n" +
471
+ "Do not interpret the claim response. It already contains an explicit \"claimed\" field — copy it verbatim.\n" +
472
+ "Return { claimed: <verbatim>, session_id: \"<...>\" }. If claimed is false there is no session_id; return { claimed: false, session_id: \"\" }.",
473
+ {
474
+ key: "claim-" + step.name,
475
+ label: "Claiming " + step.name,
476
+ schema: {
477
+ type: "object",
478
+ properties: { claimed: { type: "boolean" }, session_id: { type: "string" } },
479
+ required: ["claimed", "session_id"]
480
+ }
481
+ }
482
+ );
483
+ if (!claimResult.claimed) {
484
+ log("Standing down — task " + taskId + " was already claimed by another run");
485
+ await telemetryEnd("duplicate");
486
+ return { status: "duplicate", task_id: taskId, reason: "task already claimed by another run" };
487
+ }
488
+ activeSessionId = claimResult.session_id;
489
+ await telemetryEvent("claim", step.name + " claimed");
490
+ // Flush telemetry before proceeding — the claim is a critical point.
491
+ // If a subsequent agent() fails, we want the claim event persisted.
492
+ await telemetryFlush();
493
+ } else {
494
+ const claimResult = await agent(
495
+ "Claim a session for this task step.\n" +
496
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started" }) + "\n" +
497
+ "Return the session_id from the response.",
498
+ {
499
+ key: "claim-" + step.name,
500
+ label: "Claiming " + step.name,
501
+ schema: {
502
+ type: "object",
503
+ properties: { session_id: { type: "string" } },
504
+ required: ["session_id"]
505
+ }
506
+ }
507
+ );
508
+ activeSessionId = claimResult.session_id;
509
+ }
510
+
511
+ // ── Triage (sage) ──────────────────────────────────────────────────
512
+ // The agent does shell I/O only and returns a schema'd object. WORKFLOW
513
+ // CODE re-parses the `source:` line and makes every decision — the
514
+ // agent's reading is evidence, never the decision.
515
+ // ── Triage (sage) ──────────────────────────────────────────────────
516
+ // The agent does shell I/O only and returns a schema'd object. WORKFLOW
517
+ // CODE re-parses the `source:` line and makes every decision — the
518
+ // agent's reading is evidence, never the decision.
519
+ if (step.name === "Triage") {
520
+ const g = await computeUpgradePlan("triage-evidence", step.identity);
521
+ if (!g.ok) {
522
+ await recordPhase("Triage", step.identity, activeSessionId, "rejected", g.message);
523
+ return await parkTask(g.message);
524
+ }
525
+ upgradePlan = g.plan;
526
+ const t = upgradePlan.target;
527
+ const source = upgradePlan.source;
528
+ if (g.skipDeploy) {
529
+ skipDeploy = true;
530
+ log("Idempotent no-op for task " + taskId + ": already on " + t + " — skipping Deploy");
531
+ await agent(
532
+ "Log the idempotent no-op.\n" +
533
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("log-event", {
534
+ task_id: taskId, type: "note", identity: step.identity,
535
+ message: "already on " + t + " — no-op"
536
+ }) + "\n",
537
+ { key: "upgrade-noop-note", label: "Logging idempotent no-op" }
538
+ );
539
+ await recordPhase("Triage", step.identity, activeSessionId, "completed",
540
+ "already on " + t + " — no-op; Deploy skipped, Verify re-checks and passes trivially");
541
+ } else {
542
+ await recordPhase("Triage", step.identity, activeSessionId, "completed",
543
+ "upgrade plan: " + upgradePlan.current + " -> " + t + " (" + source + ")");
544
+ }
545
+ i++;
546
+ continue;
547
+ }
548
+
549
+ // ── Deploy (wren) ──────────────────────────────────────────────────
550
+ // Single command through the STABLE $CREW_HOME/crew-release.sh: it
551
+ // self-updates from the source, validates workflow syntax, builds the
552
+ // registry, and swaps `current` atomically. deploy is documented as
553
+ // needing no cross-step lock, so Deploy takes no merge lock. No retries,
554
+ // no auto-rollback — any non-zero exit parks with the exact output and
555
+ // names `crew-release.sh rollback` as the human recovery path.
556
+ if (step.name === "Deploy") {
557
+ if (skipDeploy) {
558
+ log("Deploy skipped for task " + taskId + " — idempotent no-op, already on " + upgradePlan.target);
559
+ await recordPhase("Deploy", step.identity, activeSessionId, "completed",
560
+ "Deploy skipped — already on target " + upgradePlan.target + " (no-op)");
561
+ i++;
562
+ continue;
563
+ }
564
+ if (!upgradePlan) {
565
+ // Resumed run: the dispatcher launches at the next step after a
566
+ // completed Triage, so Triage never re-executes in this process.
567
+ // Reconstruct the plan deterministically — it is pure evidence +
568
+ // mechanical decisions, so re-gathering it is safe. Fail closed only
569
+ // when the evidence itself cannot be gathered.
570
+ log("Deploy reached with no upgrade plan — reconstructing on resumed run.");
571
+ const rg = await computeUpgradePlan("triage-evidence-reconstruct", step.identity);
572
+ if (!rg.ok) {
573
+ return await parkTask("Deploy reached with no upgrade plan and reconstruction failed: " + rg.message + " Fail-closed.");
574
+ }
575
+ upgradePlan = rg.plan;
576
+ skipDeploy = rg.skipDeploy;
577
+ }
578
+ const stagingDir = crewHome + "/.upgrade-staging/" + upgradePlan.version;
579
+ const deployShell =
580
+ (upgradePlan.source === "repo")
581
+ ? DEPLOY_SCRIPT + " deploy " + REPO_PATH + " " + crewHome
582
+ : "STAGING=\"" + stagingDir + "\" && mkdir -p \"$STAGING\" && cd \"$STAGING\" && npm init -y >/dev/null 2>&1 && npm install muse-crew@" + upgradePlan.version;
583
+ const deploySteps =
584
+ "1. Run the install (npm source only):\n" + (upgradePlan.source === "repo" ? " (skipped — repo source has nothing to install)\n" : " " + deployShell + "\n") +
585
+ "2. Run the deploy:\n " + (upgradePlan.source === "repo"
586
+ ? deployShell
587
+ // Explicit validated package-root path — crew-release.sh reads
588
+ // workflows/, lib/ and package.json from the directory it is given,
589
+ // so the deploy source is the installed package root, not the
590
+ // staging root. Do not rely on a $STAGING shell variable persisting
591
+ // between separate shell invocations.
592
+ : DEPLOY_SCRIPT + " deploy \"" + stagingDir + "/node_modules/muse-crew\" " + crewHome) +
593
+ " — capture ALL of the command's output and its exit code (run the command, then echo EXIT_CODE=$?).\n" +
594
+ (upgradePlan.source === "npm"
595
+ ? "3. Remove the staging dir: rm -rf \"" + stagingDir + "\" — best-effort, ALWAYS, even when the deploy fails. Log whether the removal succeeded. Never let cleanup change the deploy outcome.\n"
596
+ : "") +
597
+ "Do not run git checkout, git pull, or any repo-mutating command. Do not publish to npm — the npm source only INSTALLS the published package. Do not touch the scheduler.";
598
+ var deployResult;
599
+ try {
600
+ deployResult = await agent(
601
+ TOOL_CHECK_PREAMBLE +
602
+ "Run the crew upgrade deploy. This step performs the release swap — the workflow decides everything from the values you return.\n" +
603
+ deploySteps + "\n" +
604
+ "Return JSON { \"exit\": <the deploy command's exit code as an integer>, \"output\": \"<the deploy command's full output, trimmed>\" } and nothing else.",
605
+ {
606
+ key: "deploy-run",
607
+ label: "Deploying crew upgrade (" + upgradePlan.source + ")",
608
+ schema: {
609
+ type: "object",
610
+ properties: {
611
+ exit: { type: "number" },
612
+ output: { type: "string" }
613
+ },
614
+ required: ["exit", "output"]
615
+ }
616
+ }
617
+ );
618
+ } catch (e) {
619
+ return await parkTask("Deploy agent call failed: " + (e && e.message ? e.message : e) + ". Fail-closed — the deploy outcome is unknown; human recovery path: " + DEPLOY_SCRIPT + " rollback " + crewHome);
620
+ }
621
+ const deployExit = deployResult.exit;
622
+ const deployOutput = (deployResult.output || "").trim();
623
+ if (deployExit !== 0) {
624
+ const msg = "Deploy failed (exit " + deployExit + "). Human recovery path: " + DEPLOY_SCRIPT + " rollback " + crewHome + ". Deploy output: " + deployOutput;
625
+ await recordPhase("Deploy", step.identity, activeSessionId, "failed", msg);
626
+ return await parkTask(msg);
627
+ }
628
+ log("Deploy succeeded for task " + taskId + " — target " + upgradePlan.target);
629
+ await recordPhase("Deploy", step.identity, activeSessionId, "completed",
630
+ "deployed " + upgradePlan.target + " (" + upgradePlan.source + ")\ndeploy output:\n" + deployOutput.slice(0, 1500));
631
+ i++;
632
+ continue;
633
+ }
634
+
635
+ // ── Verify (wren — mechanical, no LLM judgment) ─────────────────────
636
+ // Three mechanical checks, each a schema'd agent() shell call. The agent
637
+ // never writes "looks good" — the WORKFLOW evaluates pass/fail from the
638
+ // returned values. Handover is automatic: the next poll tick launches the
639
+ // dispatcher through the `current` symlink, i.e. the new release.
640
+ if (step.name === "Verify") {
641
+ if (!upgradePlan) {
642
+ // Resumed run at Verify (Deploy completed in a prior run):
643
+ // reconstruct the plan deterministically, as in Deploy.
644
+ log("Verify reached with no upgrade plan — reconstructing on resumed run.");
645
+ const rg = await computeUpgradePlan("triage-evidence-reconstruct", step.identity);
646
+ if (!rg.ok) {
647
+ return await parkTask("Verify reached with no upgrade plan and reconstruction failed: " + rg.message + " Fail-closed.");
648
+ }
649
+ upgradePlan = rg.plan;
650
+ skipDeploy = rg.skipDeploy;
651
+ }
652
+ const target = upgradePlan.target;
653
+ const old = upgradePlan.current;
654
+
655
+ // Check 1: the live release identity equals the target.
656
+ var curCheck;
657
+ try {
658
+ curCheck = await agent(
659
+ TOOL_CHECK_PREAMBLE +
660
+ "Read the live release identity.\n" +
661
+ "Run: " + DEPLOY_SCRIPT + " current " + crewHome + "\n" +
662
+ "Return JSON { \"current\": \"<trimmed stdout>\" } and nothing else.",
663
+ {
664
+ key: "verify-current",
665
+ label: "Verifying live release",
666
+ schema: {
667
+ type: "object",
668
+ properties: { current: { type: "string" } },
669
+ required: ["current"]
670
+ }
671
+ }
672
+ );
673
+ } catch (e) {
674
+ return await parkTask("Verify failed — check 1 (crew-release.sh current): agent call failed (" + (e && e.message ? e.message : e) + "). Fail-closed.");
675
+ }
676
+ const liveCurrent = (curCheck.current || "").trim();
677
+ if (liveCurrent !== target) {
678
+ const msg = "Verify failed — check 1 (crew-release.sh current): live release '" + liveCurrent + "' != target '" + target + "'";
679
+ await recordPhase("Verify", step.identity, activeSessionId, "rejected", msg);
680
+ return await parkTask(msg);
681
+ }
682
+ log("Verify check 1 passed for task " + taskId + ": current == target (" + target + ")");
683
+
684
+ // Check 2: the deployed registry parses as JSON. For the repo source it
685
+ // must contain the "upgrade" key (the release was built from a repo
686
+ // containing this workflow); npm releases predating the upgrade
687
+ // workflow are still valid upgrades and are only required to parse.
688
+ var regCheck;
689
+ try {
690
+ regCheck = await agent(
691
+ TOOL_CHECK_PREAMBLE +
692
+ "Read the deployed registry.\n" +
693
+ "Run: cat " + crewHome + "/workflows/registry.json\n" +
694
+ "Return JSON { \"raw\": \"<the file's full content, verbatim>\" } and nothing else.",
695
+ {
696
+ key: "verify-registry",
697
+ label: "Reading deployed registry",
698
+ schema: {
699
+ type: "object",
700
+ properties: { raw: { type: "string" } },
701
+ required: ["raw"]
702
+ }
703
+ }
704
+ );
705
+ } catch (e) {
706
+ return await parkTask("Verify failed — check 2 (registry.json): agent call failed (" + (e && e.message ? e.message : e) + "). Fail-closed.");
707
+ }
708
+ let regOk = false, regHasUpgrade = false;
709
+ try {
710
+ const regParsed = JSON.parse(regCheck.raw || "");
711
+ regOk = true;
712
+ regHasUpgrade = !!(regParsed && regParsed.upgrade);
713
+ } catch (e) {
714
+ regOk = false;
715
+ }
716
+ if (!regOk) {
717
+ const msg = "Verify failed — check 2 (registry.json): " + crewHome + "/workflows/registry.json does not parse as JSON";
718
+ await recordPhase("Verify", step.identity, activeSessionId, "rejected", msg);
719
+ return await parkTask(msg);
720
+ }
721
+ if (upgradePlan.source === "repo" && !regHasUpgrade) {
722
+ const msg = "Verify failed — check 2 (registry.json): parses as JSON but has no \"upgrade\" key — the deployed release was not built from a repo containing this workflow";
723
+ await recordPhase("Verify", step.identity, activeSessionId, "rejected", msg);
724
+ return await parkTask(msg);
725
+ }
726
+ log("Verify check 2 passed for task " + taskId + ": registry.json parses" + (upgradePlan.source === "repo" ? " and carries the upgrade key" : ""));
727
+
728
+ // Check 3: the crew DB is readable by the OLD release's CLI. Use the
729
+ // PINNED copy explicitly: the workflow swapped `current` under itself
730
+ // at Deploy, so the live path may already point at the new release.
731
+ // The new release's CLI is exercised by the next tick's dispatcher,
732
+ // not here.
733
+ var gsCheck;
734
+ try {
735
+ gsCheck = await agent(
736
+ TOOL_CHECK_PREAMBLE +
737
+ "Check the crew database is readable by the OLD release's CLI. Use the PINNED path below — NOT " + crewHome + "/current — because the workflow swapped `current` under itself at Deploy and the live path may already point at the new release.\n" +
738
+ "Run: node " + CREW_API_PINNED + " --crew-home " + crewHome + " get-state --json '{}'; echo EXIT_CODE=$?\n" +
739
+ "Return JSON { \"exit\": <the exit code as an integer> } and nothing else.",
740
+ {
741
+ key: "verify-get-state",
742
+ label: "Verifying crew DB readability (pinned CLI)",
743
+ schema: {
744
+ type: "object",
745
+ properties: { exit: { type: "number" } },
746
+ required: ["exit"]
747
+ }
748
+ }
749
+ );
750
+ } catch (e) {
751
+ return await parkTask("Verify failed — check 3 (pinned crew-api.js get-state): agent call failed (" + (e && e.message ? e.message : e) + "). Fail-closed.");
752
+ }
753
+ if (gsCheck.exit !== 0) {
754
+ const msg = "Verify failed — check 3 (pinned crew-api.js get-state): exit " + gsCheck.exit + " — the crew DB is not readable by the pinned old-release CLI";
755
+ await recordPhase("Verify", step.identity, activeSessionId, "rejected", msg);
756
+ return await parkTask(msg);
757
+ }
758
+ log("Verify check 3 passed for task " + taskId + ": pinned crew-api.js get-state exits 0");
759
+
760
+ await agent(
761
+ "Log the completed upgrade.\n" +
762
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("log-event", {
763
+ task_id: taskId, type: "note", identity: step.identity,
764
+ message: "upgrade: " + old + " -> " + target + " (" + upgradePlan.source + ")"
765
+ }) + "\n",
766
+ { key: "upgrade-done-note", label: "Logging completed upgrade" }
767
+ );
768
+ await recordPhase("Verify", step.identity, activeSessionId, "completed",
769
+ "all three checks passed: current == " + target + ", registry.json valid" +
770
+ (upgradePlan.source === "repo" ? " (carries upgrade key)" : "") +
771
+ ", pinned crew-api.js get-state exits 0" +
772
+ (skipDeploy ? " (idempotent no-op — Deploy skipped)" : ""));
773
+ i++;
774
+ continue;
775
+ }
776
+
777
+ // Unreachable: STEPS is fixed, but fail closed on an unknown step.
778
+ return await parkTask("Unknown step '" + step.name + "' in crew-upgrade. Fail-closed.");
779
+ }
780
+
781
+ await agent(
782
+ "Mark this task as done.\n" +
783
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("update-task", { id: taskId, state: "done" }) + "\n" +
784
+ "Then run in shell and return the stdout verbatim:\n" + crewCmd("log-event", {
785
+ task_id: taskId, type: "completed",
786
+ message: "Crew self-upgrade complete: " + upgradePlan.current + " -> " + upgradePlan.target + " (" + upgradePlan.source + "). Handover is automatic: the next poll tick launches the dispatcher through the `current` symlink, i.e. the new release."
787
+ }),
788
+ { key: "task-done", label: "Completing task: " + taskTitle }
789
+ );
790
+
791
+ log("Crew self-upgrade complete for task " + taskId + ": " + upgradePlan.current + " -> " + upgradePlan.target);
792
+ await telemetryEnd("completed");
793
+ await agent("Clean up pinned lifecycle scripts: rm -rf " + RUN_LIB, { key: "cleanup-pins", label: "Cleaning pinned scripts" });
794
+ return { status: "ok", task_id: taskId, message: "Crew self-upgrade complete: " + upgradePlan.current + " -> " + upgradePlan.target + " (" + upgradePlan.source + ")" };