muse-crew 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -72,8 +72,14 @@ const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
72
72
  const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
73
73
  const NOTE_VOCAB_SRC = crewHome + "/current/lib/publish-note-vocabulary.js";
74
74
  const NOTE_VOCAB = RUN_LIB + "/publish-note-vocabulary.js";
75
+ const QA_DEPLOY_SRC = crewHome + "/current/lib/qa-deploy.mjs";
76
+ const QA_DEPLOY = RUN_LIB + "/qa-deploy.mjs";
77
+ const SERVE_ARTIFACT_SRC = crewHome + "/current/lib/serve-artifact.js";
78
+ const SERVE_ARTIFACT = RUN_LIB + "/serve-artifact.js";
79
+ const QA_DB_SRC = crewHome + "/current/lib/qa-db.js";
80
+ const QA_DB = RUN_LIB + "/qa-db.js";
75
81
  // See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
76
- const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB].map(function (p) { return p.split("/").pop(); });
82
+ const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB, QA_DEPLOY, SERVE_ARTIFACT, QA_DB].map(function (p) { return p.split("/").pop(); });
77
83
 
78
84
  // Project config — passed by dispatcher, falls back to dashboard defaults
79
85
  const projectConfig = inputs.project_config || {};
@@ -265,7 +271,7 @@ function attemptKey(base, reworkCount) {
265
271
  function pinLifecycle(key) {
266
272
  return agent(
267
273
  "Snapshot lifecycle scripts for version pinning.\n" +
268
- "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
274
+ "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && cp " + QA_DEPLOY_SRC + " " + QA_DEPLOY + " && cp " + SERVE_ARTIFACT_SRC + " " + SERVE_ARTIFACT + " && cp " + QA_DB_SRC + " " + QA_DB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
269
275
  "Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
270
276
  { key: key, label: "Pinning lifecycle scripts",
271
277
  schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
@@ -563,17 +569,28 @@ function decideIntegrateRetry(o) {
563
569
  if (rec.malformed) return "park";
564
570
  return ancestor ? "skip-to-publish" : "proceed";
565
571
  }
566
- // QA version gate (issue #3): S=footer build, R=required. S>=R proves the
567
- // bundle is at/past the merge (single-branch: version order = containment).
568
- // Pure — pinned by tests/version-gate.test.js. Null = unreadable/missing.
569
- function decideVersionGate(S, R) {
570
- if (!(R > 0)) return { proceed: false, reason: "qa-version-missing",
571
- remedy: "no build version recorded — re-run Integrate or record dashboard-version." };
572
- if (!(S > 0)) return { proceed: false, reason: "qa-version-unreadable",
573
- remedy: "no readable build number in bundle footer — rebuild from versioned source." };
574
- if (S >= R) return { proceed: true };
575
- return { proceed: false, reason: "qa-bundle-stale",
576
- remedy: "bundle build " + S + " behind required " + R + " — builder did not rebuild from current source." };
572
+ // parseDeployResult: pure, fails closed. Ferry returns qa-deploy.mjs stdout
573
+ // verbatim; the marker line is the only machine-read signal.
574
+ function parseDeployResult(output) {
575
+ var m = /^DEPLOY_OK ([0-9a-f]{7})$/m.exec(String(output || "").trim());
576
+ if (m) return { ok: true, hash: m[1] };
577
+ var f = /^DEPLOY_FAIL ([a-z-]+):(.{1,200})$/m.exec(String(output || "").trim());
578
+ if (f) return { ok: false, reason: f[1] + ": " + f[2].trim() };
579
+ return { ok: false, reason: "unrecognized deploy output" };
580
+ }
581
+ // QA_ENV_DIR: pipeline-owned, under the crew home. Never the web artifact dir.
582
+ var QA_ENV_DIR = crewHome + "/qa-envs/" + LAUNCH_PROJECT_ID;
583
+ // runDeployFerry: thin mechanical agent, one pinned command, stdout verbatim
584
+ function runDeployFerry(mode) {
585
+ var cmd = "node " + QA_DEPLOY + " --repo " + REPO_PATH + " --qa-dir " + QA_ENV_DIR +
586
+ " --task " + taskId + " --crew-home " + crewHome + " --crew-api " + CREW_API_PINNED +
587
+ " --serve-artifact " + SERVE_ARTIFACT + (mode === "--ensure" ? " --ensure" : "") + " 2>&1";
588
+ return agent(
589
+ "Run in shell and return the stdout verbatim:\n" + cmd + "\n" +
590
+ "Return JSON { \"output\": \"<the command's full stdout, trimmed>\" } and nothing else.",
591
+ { key: "qa-deploy-" + mode.replace(/-/g, "") + "-" + taskId, label: "Deploying QA environment",
592
+ schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
593
+ );
577
594
  }
578
595
  // On a dispatcher retry resumed at Integrate, the run-local releaseDecision
579
596
  // is null (Build doesn't re-run). Hydrate it from the merge record so the
@@ -1367,25 +1384,28 @@ while (i < STEPS.length) {
1367
1384
  // verdicts.jsonl). No parent verdict gate remains.
1368
1385
  var qaArtifact = false;
1369
1386
  var qaTerminal = false;
1370
- var qaRequiredVersion = null;
1387
+ var qaBundleHash = null;
1371
1388
  if (step.name === "QA") {
1372
1389
  var qaExp = (await resolveExperiential()) === "yes";
1373
1390
  qaArtifact = qaExp && SURFACE_ARTIFACT;
1374
1391
  qaTerminal = qaExp && SURFACE_TERMINAL;
1375
1392
  if (qaArtifact && projectConfig.versioned_build === true) {
1376
- var vfile = projectConfig.version_file || "client/src/buildNumber.ts";
1377
- var vgPre = await agent(
1378
- "Return ONLY JSON. 1. Run, return stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId }) + "\n" +
1379
- "R = integer after `dashboard-version: ` in newest matching note. 2. If none, run, return stdout verbatim:\n" + crewCmd("get-provenance", { project_id: LAUNCH_PROJECT_ID }) + "\n" +
1380
- "R = BUILD_NUMBER from: cd " + REPO_PATH + " && git show <source_commit>:" + vfile + " | grep -o 'BUILD_NUMBER = [0-9]*'. If unknown: {\"ok\":false,\"park\":\"qa-version-missing: no build version recorded — re-run Integrate or record dashboard-version.\"}. " +
1381
- "3. cd " + REPO_PATH + " && git fetch origin && V=$(grep -o 'BUILD_NUMBER = [0-9]*' " + vfile + " | grep -o '[0-9]*'|head -1); if [ $V -lt R ]; then git pull --ff-only; V=$(grep -o 'BUILD_NUMBER = [0-9]*' " + vfile + " | grep -o '[0-9]*'|head -1); fi; " +
1382
- "if [ $V -lt R ]: {\"ok\":false,\"park\":\"qa-bundle-stale: checkout \" + V + \" < required \" + R + \" — sync past the merge, re-run QA.\"}. 4. Else {\"ok\":true,\"R\":R}.",
1383
- { key: "qa-version-pre-" + taskId, label: "QA version-gate pre-check",
1384
- schema: { type: "object", required: ["ok"],
1385
- properties: { ok: { type: "boolean" }, R: { type: "integer" }, park: { type: "string" } } } }
1386
- );
1387
- if (!vgPre.ok) return await parkTask(vgPre.park);
1388
- qaRequiredVersion = vgPre.R;
1393
+ // ensure-deployed: idempotent precondition. Fast path when bundle is
1394
+ // current. Failed deploy = operational failure (retryable), never a park.
1395
+ var ensureOut = "";
1396
+ try {
1397
+ var ensureResult = await runDeployFerry("--ensure");
1398
+ ensureOut = (ensureResult.output || "").trim();
1399
+ } catch (e) {
1400
+ ensureOut = "";
1401
+ }
1402
+ var ensureParsed = parseDeployResult(ensureOut);
1403
+ if (!ensureParsed.ok) {
1404
+ log("QA ensure-deployed failed for task " + taskId + ": " + ensureParsed.reason + " — returning failed for dispatcher retry");
1405
+ return { status: "failed", task_id: taskId, reason: "QA ensure-deployed failed: " + ensureParsed.reason };
1406
+ }
1407
+ qaBundleHash = ensureParsed.hash;
1408
+ log("QA ensure-deployed ok for task " + taskId + ": bundle " + qaBundleHash);
1389
1409
  }
1390
1410
  }
1391
1411
 
@@ -1715,14 +1735,6 @@ while (i < STEPS.length) {
1715
1735
  "R5-PUSH (manual R5 resolution only — the normal path pushed inline). Run: " + LIFECYCLE_ENV + LIFECYCLE + " push-target " + taskId + "\n" +
1716
1736
  "PUSHED — report the merged hash (detached prints PUSHED: origin/main (refspec HEAD:main)), VERDICT: PASS; PUSH_SKIPPED — no push attempted (no record + no lock); NO_REMOTE_PUSH — no remote; VERDICT: PASS; ERROR or CONFLICT — report it, VERDICT: FAIL.\n" +
1717
1737
  "NEVER force-push.\n\n" +
1718
- (projectConfig.versioned_build === true ?
1719
- "VERSION BUMP (issue #3): after MERGED+PUSHED, bump so QA can prove the bundle contains this fix — QA parks without it.\n" +
1720
- "1. cd " + REPO_PATH + " && git fetch origin; BR=<branch-from-integration-target>; F=" + (projectConfig.version_file || "client/src/buildNumber.ts") + "\n" +
1721
- "2. N=$(git show origin/$BR:$F | grep -o 'BUILD_NUMBER = [0-9]*' | grep -o '[0-9]*'); if missing/unparsable: VERDICT: FAIL.\n" +
1722
- "3. Edit $F: `export const BUILD_NUMBER = $N;` → `export const BUILD_NUMBER = $((N+1));` (keep header comment).\n" +
1723
- "4. git add $F && git commit -m \"build-number: $((N+1)) - QA version gate\" (SEPARATE commit, never amend).\n" +
1724
- "5. git push origin $BR; if rejected retry 3x (fetch, re-read N, re-bump, re-commit, push). NEVER force-push. Push MUST succeed or VERDICT: FAIL.\n" +
1725
- "6. Run, return stdout verbatim:\n" + crewCmd("log-event", { task_id: taskId, type: "note", identity: step.identity, message: "dashboard-version: <new> — QA must test a bundle built from source at or after the merge that recorded this (build <new> or later)." }) + "\n(substitute <new>).\n\n" : "") +
1726
1738
  "Report what happened at each step, ending with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
1727
1739
 
1728
1740
  } else if (step.name === "Publish") {
@@ -2176,8 +2188,6 @@ while (i < STEPS.length) {
2176
2188
  "c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
2177
2189
  "d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the dialog, then confirm it, then judge the result. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and continue with the mechanical checks.\n" +
2178
2190
  "e. Judge as a user against the task description: is the reported bug fixed AND is nothing else visibly broken? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
2179
- (projectConfig.versioned_build === true ?
2180
- "e2. VERSION (issue #3): footer shows `build <n>` — report `footer_build: <n>` on its own line, or `footer_build: unreadable`. Required — the gate cannot pass without it.\n" : "") +
2181
2191
  "f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
2182
2192
  "Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
2183
2193
  "MECHANICAL CHECKS:\n" +
@@ -2659,20 +2669,24 @@ while (i < STEPS.length) {
2659
2669
  // never a park. No automatic FAIL override: if the QA agent still
2660
2670
  // reports FAIL, it stands — finding attribution informs follow-up
2661
2671
  // filing only.
2662
- if (step.name === "QA") {
2663
- if (projectConfig.versioned_build === true && qaArtifact && qaRequiredVersion !== null) {
2664
- var vgM = /footer_build:\s*(\d+|unreadable)/i.exec(workerText || "");
2665
- var vgS = (vgM && /^\d+$/.test(vgM[1])) ? parseInt(vgM[1], 10) : null;
2666
- var vgD = decideVersionGate(vgS, qaRequiredVersion);
2667
- if (!vgD.proceed) {
2668
- return await parkTask(vgD.reason + ": " + vgD.remedy + " [built=" + vgS + " required=" + qaRequiredVersion + "]");
2669
- }
2670
- await agent(
2671
- "Run in shell and return the stdout verbatim:\n" + crewCmd("log-event", {
2672
- task_id: taskId, type: "note", identity: step.identity,
2673
- message: "version-check: built=" + vgS + " >= required=" + qaRequiredVersion + " → testing now" }),
2674
- { key: "record-version-check-" + taskId, label: "Recording version-gate pass" }
2672
+ if (step.name === "QA" && qaBundleHash !== null) {
2673
+ // Closeout: bundle Hazel tested must match ensure-deployed hash from QA
2674
+ // start. Mismatch/unreadable = operational failure, never a park.
2675
+ var closeHash = "";
2676
+ try {
2677
+ var closeResult = await agent(
2678
+ "Run: node " + QA_DEPLOY + " --hash-only --qa-dir " + QA_ENV_DIR + " 2>&1\n" +
2679
+ "Return JSON { \"output\": \"<stdout, trimmed>\" } and nothing else.",
2680
+ { key: "qa-closeout-hash-" + taskId, label: "Reading QA bundle hash",
2681
+ schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
2675
2682
  );
2683
+ var chm = /BUNDLE_HASH ([0-9a-f]{40})/.exec(String(closeResult.output || ""));
2684
+ closeHash = chm ? chm[1].slice(0, 7) : "";
2685
+ } catch (e) {}
2686
+ if (closeHash !== qaBundleHash) {
2687
+ log("QA closeout hash mismatch for task " + taskId + " — marking failed for retry");
2688
+ stepResult.summary = (stepResult.summary || "") + "\nqa_closeout: FAILED — bundle hash " + (closeHash ? "changed during QA" : "unreadable");
2689
+ passed = false;
2676
2690
  }
2677
2691
  var contentFindings = extractContentFindings(workerText);
2678
2692
  if (!contentFindings.ok) {
@@ -2803,6 +2817,26 @@ while (i < STEPS.length) {
2803
2817
  }
2804
2818
  if (/^VERIFIED:/m.test(integrateVerifyOut)) {
2805
2819
  log("Integrate verified for task " + taskId + ": task branch tip is an ancestor of the integration target");
2820
+ // QA deploy: after verified merge, ensure QA env serves a bundle from
2821
+ // merged source. Failed deploy = operational failure, never a park.
2822
+ if (projectConfig.versioned_build === true && SURFACE_ARTIFACT) {
2823
+ var deployOut = "";
2824
+ try {
2825
+ var deployResult = await runDeployFerry("");
2826
+ deployOut = (deployResult.output || "").trim();
2827
+ } catch (e) {
2828
+ deployOut = "";
2829
+ }
2830
+ var deployParsed = parseDeployResult(deployOut);
2831
+ if (deployParsed.ok) {
2832
+ log("QA deploy ok for task " + taskId + ": bundle " + deployParsed.hash);
2833
+ stepResult.summary = (stepResult.summary || "") + "\ndeploy: ok " + deployParsed.hash;
2834
+ } else {
2835
+ log("QA deploy failed for task " + taskId + ": " + deployParsed.reason + " — marking failed for retry");
2836
+ stepResult.summary = (stepResult.summary || "") + "\ndeploy: FAILED — " + deployParsed.reason;
2837
+ passed = false;
2838
+ }
2839
+ }
2806
2840
  } else {
2807
2841
  var integrateVerifyReason = integrateVerifyOut
2808
2842
  ? integrateVerifyOut.split("\n")[0].slice(0, 200)
@@ -136,8 +136,14 @@ const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
136
136
  const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
137
137
  const NOTE_VOCAB_SRC = crewHome + "/current/lib/publish-note-vocabulary.js";
138
138
  const NOTE_VOCAB = RUN_LIB + "/publish-note-vocabulary.js";
139
+ const QA_DEPLOY_SRC = crewHome + "/current/lib/qa-deploy.mjs";
140
+ const QA_DEPLOY = RUN_LIB + "/qa-deploy.mjs";
141
+ const SERVE_ARTIFACT_SRC = crewHome + "/current/lib/serve-artifact.js";
142
+ const SERVE_ARTIFACT = RUN_LIB + "/serve-artifact.js";
143
+ const QA_DB_SRC = crewHome + "/current/lib/qa-db.js";
144
+ const QA_DB = RUN_LIB + "/qa-db.js";
139
145
  // See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
140
- const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB].map(function (p) { return p.split("/").pop(); });
146
+ const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB, QA_DEPLOY, SERVE_ARTIFACT, QA_DB].map(function (p) { return p.split("/").pop(); });
141
147
 
142
148
  // Project config — passed by dispatcher, falls back to dashboard defaults
143
149
  const projectConfig = inputs.project_config || {};
@@ -275,7 +281,7 @@ function attemptKey(base, reworkCount) {
275
281
  function pinLifecycle(key) {
276
282
  return agent(
277
283
  "Snapshot lifecycle scripts for version pinning.\n" +
278
- "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
284
+ "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && cp " + QA_DEPLOY_SRC + " " + QA_DEPLOY + " && cp " + SERVE_ARTIFACT_SRC + " " + SERVE_ARTIFACT + " && cp " + QA_DB_SRC + " " + QA_DB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
279
285
  "Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
280
286
  { key: key, label: "Pinning lifecycle scripts",
281
287
  schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
@@ -477,10 +477,9 @@ for (var pi = 0; pi < projects.length; pi++) {
477
477
  // terminal | null=unclassified). Carried alongside deploy_type — it is a
478
478
  // separate axis, not a redeclaration of the deployment target.
479
479
  environment_type: proj.environment_type || null,
480
- // QA staleness gate (emojimanegg1/muse-crew#3): versioned_build opts the
481
- // project into the Integrate version bump + QA bundle-freshness gate;
482
- // version_file is the repo-relative path of the TS file exporting
483
- // BUILD_NUMBER (null when unversioned).
480
+ // QA deploy (replaces the issue #3 version gate): versioned_build opts the
481
+ // project into the post-Integrate QA deploy + pre-QA ensure-deployed;
482
+ // version_file is retained for compatibility but no longer consumed.
484
483
  versioned_build: !!proj.versioned_build,
485
484
  version_file: proj.version_file || null
486
485
  };
@@ -1055,7 +1054,11 @@ for (var p = 0; p < toProcess.length; p++) {
1055
1054
  };
1056
1055
 
1057
1056
  log("Recommended " + iworkflow + " for \"" + itask.title + "\" [" + taskProject + "] at step " + nextStepName);
1058
- results.push({ task_id: itask.id, workflow: iworkflow, step: nextStepName, action: "recommended", scriptPath: scriptPath, args: launchArgs });
1057
+ // executor tag (Piece 1, 2026-09-26): the sandboxed dispatcher claims
1058
+ // "sandbox"; the worker-layer dispatcher (lib/crew-dispatch-worker.js)
1059
+ // claims "worker". Piece 2 routes launches on this tag. The shadow
1060
+ // comparison ignores it.
1061
+ results.push({ task_id: itask.id, workflow: iworkflow, step: nextStepName, action: "recommended", scriptPath: scriptPath, args: launchArgs, executor: "sandbox" });
1059
1062
 
1060
1063
  } // end for (per-task loop)
1061
1064
  } // end processing block
@@ -80,8 +80,14 @@ const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
80
80
  const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
81
81
  const NOTE_VOCAB_SRC = crewHome + "/current/lib/publish-note-vocabulary.js";
82
82
  const NOTE_VOCAB = RUN_LIB + "/publish-note-vocabulary.js";
83
+ const QA_DEPLOY_SRC = crewHome + "/current/lib/qa-deploy.mjs";
84
+ const QA_DEPLOY = RUN_LIB + "/qa-deploy.mjs";
85
+ const SERVE_ARTIFACT_SRC = crewHome + "/current/lib/serve-artifact.js";
86
+ const SERVE_ARTIFACT = RUN_LIB + "/serve-artifact.js";
87
+ const QA_DB_SRC = crewHome + "/current/lib/qa-db.js";
88
+ const QA_DB = RUN_LIB + "/qa-db.js";
83
89
  // See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
84
- const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB].map(function (p) { return p.split("/").pop(); });
90
+ const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB, QA_DEPLOY, SERVE_ARTIFACT, QA_DB].map(function (p) { return p.split("/").pop(); });
85
91
 
86
92
  // Project config — passed by dispatcher, falls back to dashboard defaults
87
93
  const projectConfig = inputs.project_config || {};
@@ -273,7 +279,7 @@ function attemptKey(base, reworkCount) {
273
279
  function pinLifecycle(key) {
274
280
  return agent(
275
281
  "Snapshot lifecycle scripts for version pinning.\n" +
276
- "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
282
+ "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && cp " + QA_DEPLOY_SRC + " " + QA_DEPLOY + " && cp " + SERVE_ARTIFACT_SRC + " " + SERVE_ARTIFACT + " && cp " + QA_DB_SRC + " " + QA_DB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
277
283
  "Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
278
284
  { key: key, label: "Pinning lifecycle scripts",
279
285
  schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
@@ -571,17 +577,37 @@ function decideIntegrateRetry(o) {
571
577
  if (rec.malformed) return "park";
572
578
  return ancestor ? "skip-to-publish" : "proceed";
573
579
  }
574
- // QA version gate (issue #3): S=footer build, R=required. S>=R proves the
575
- // bundle is at/past the merge (single-branch: version order = containment).
576
- // Pure — pinned by tests/version-gate.test.js. Null = unreadable/missing.
577
- function decideVersionGate(S, R) {
578
- if (!(R > 0)) return { proceed: false, reason: "qa-version-missing",
579
- remedy: "no build version recorded — re-run Integrate or record dashboard-version." };
580
- if (!(S > 0)) return { proceed: false, reason: "qa-version-unreadable",
581
- remedy: "no readable build number in bundle footer — rebuild from versioned source." };
582
- if (S >= R) return { proceed: true };
583
- return { proceed: false, reason: "qa-bundle-stale",
584
- remedy: "bundle build " + S + " behind required " + R + " — builder did not rebuild from current source." };
580
+ // QA deploy result parser (replaces the version gate, issue #3): the ferry
581
+ // returns qa-deploy.mjs's stdout verbatim; the marker line is the only
582
+ // machine-read signal. Pure — fails closed on anything unrecognized.
583
+ // Returns { ok, hash } or { ok:false, reason }.
584
+ function parseDeployResult(output) {
585
+ var m = /^DEPLOY_OK ([0-9a-f]{7})$/m.exec(String(output || "").trim());
586
+ if (m) return { ok: true, hash: m[1] };
587
+ var f = /^DEPLOY_FAIL ([a-z-]+):(.{1,200})$/m.exec(String(output || "").trim());
588
+ if (f) return { ok: false, reason: f[1] + ": " + f[2].trim() };
589
+ return { ok: false, reason: "unrecognized deploy output" };
590
+ }
591
+ // QA environment dir — pipeline-owned, under the crew home. The web artifact
592
+ // directory (~/workspace/ts-spaces/<slug>/) is never built by hand (its own
593
+ // AGENTS.md forbids it); this separate dir is cloned, synced, built, and
594
+ // validated by qa-deploy.mjs. Named by project id (deploy_slug is an
595
+ // artifact-system concern; the QA env is ours).
596
+ var QA_ENV_DIR = crewHome + "/qa-envs/" + LAUNCH_PROJECT_ID;
597
+ // Run the deploy ferry: a thin mechanical agent that runs one exact pinned
598
+ // command and returns stdout verbatim. No identity, no judgment — the
599
+ // module self-records the full evidence via the Crew API; the ferry carries
600
+ // only the marker line.
601
+ function runDeployFerry(mode) {
602
+ var cmd = "node " + QA_DEPLOY + " --repo " + REPO_PATH + " --qa-dir " + QA_ENV_DIR +
603
+ " --task " + taskId + " --crew-home " + crewHome + " --crew-api " + CREW_API_PINNED +
604
+ " --serve-artifact " + SERVE_ARTIFACT + (mode === "--ensure" ? " --ensure" : "") + " 2>&1";
605
+ return agent(
606
+ "Run in shell and return the stdout verbatim:\n" + cmd + "\n" +
607
+ "Return JSON { \"output\": \"<the command's full stdout, trimmed>\" } and nothing else.",
608
+ { key: "qa-deploy-" + mode.replace(/-/g, "") + "-" + taskId, label: "Deploying QA environment",
609
+ schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
610
+ );
585
611
  }
586
612
  // On a dispatcher retry resumed at Integrate, the run-local releaseDecision
587
613
  // is null (Build doesn't re-run). Hydrate it from the merge record so the
@@ -1310,23 +1336,27 @@ while (i < STEPS.length) {
1310
1336
  // experiential-loop guard (a PASS with missing experiential evidence is
1311
1337
  // never terminal).
1312
1338
  var qaExperiential = false;
1339
+ var qaBundleHash = null;
1313
1340
  if (step.name === "QA") {
1314
1341
  qaExperiential = (await resolveExperiential()) === "yes" && SURFACE_CLASSIFIED;
1315
- var qaRequiredVersion = null;
1316
1342
  if (qaExperiential && SURFACE_ARTIFACT && projectConfig.versioned_build === true) {
1317
- var vfileStd = projectConfig.version_file || "client/src/buildNumber.ts";
1318
- var vgPreStd = await agent(
1319
- "Return ONLY JSON. 1. Run, return stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId }) + "\n" +
1320
- "R = integer after `dashboard-version: ` in newest matching note. 2. If none, run, return stdout verbatim:\n" + crewCmd("get-provenance", { project_id: LAUNCH_PROJECT_ID }) + "\n" +
1321
- "R = BUILD_NUMBER from: cd " + REPO_PATH + " && git show <source_commit>:" + vfileStd + " | grep -o 'BUILD_NUMBER = [0-9]*'. If unknown: {\"ok\":false,\"park\":\"qa-version-missing: no build version recorded — re-run Integrate or record dashboard-version.\"}. " +
1322
- "3. cd " + REPO_PATH + " && git fetch origin && V=$(grep -o 'BUILD_NUMBER = [0-9]*' " + vfileStd + " | grep -o '[0-9]*'|head -1); if [ $V -lt R ]; then git pull --ff-only; V=$(grep -o 'BUILD_NUMBER = [0-9]*' " + vfileStd + " | grep -o '[0-9]*'|head -1); fi; " +
1323
- "if [ $V -lt R ]: {\"ok\":false,\"park\":\"qa-bundle-stale: checkout \" + V + \" < required \" + R + \" — sync past the merge, re-run QA.\"}. 4. Else {\"ok\":true,\"R\":R}.",
1324
- { key: "qa-version-pre-" + taskId, label: "QA version-gate pre-check",
1325
- schema: { type: "object", required: ["ok"],
1326
- properties: { ok: { type: "boolean" }, R: { type: "integer" }, park: { type: "string" } } } }
1327
- );
1328
- if (!vgPreStd.ok) return await parkTask(vgPreStd.park);
1329
- qaRequiredVersion = vgPreStd.R;
1343
+ // ensure-deployed: idempotent precondition. Fast path when the bundle
1344
+ // is already current. Failed deploy = operational failure (retryable),
1345
+ // never a park. The module self-records the bundle hash to the event log.
1346
+ var ensureOut = "";
1347
+ try {
1348
+ var ensureResult = await runDeployFerry("--ensure");
1349
+ ensureOut = (ensureResult.output || "").trim();
1350
+ } catch (e) {
1351
+ ensureOut = "";
1352
+ }
1353
+ var ensureParsed = parseDeployResult(ensureOut);
1354
+ if (!ensureParsed.ok) {
1355
+ log("QA ensure-deployed failed for task " + taskId + ": " + ensureParsed.reason + " — returning failed for dispatcher retry");
1356
+ return { status: "failed", task_id: taskId, reason: "QA ensure-deployed failed: " + ensureParsed.reason };
1357
+ }
1358
+ qaBundleHash = ensureParsed.hash;
1359
+ log("QA ensure-deployed ok for task " + taskId + ": bundle " + qaBundleHash);
1330
1360
  }
1331
1361
  }
1332
1362
 
@@ -1604,14 +1634,6 @@ while (i < STEPS.length) {
1604
1634
  "R5-PUSH (manual R5 resolution only — the normal path pushed inline). Run: " + LIFECYCLE_ENV + LIFECYCLE + " push-target " + taskId + "\n" +
1605
1635
  "PUSHED — report the merged hash (detached prints PUSHED: origin/main (refspec HEAD:main)), VERDICT: PASS; PUSH_SKIPPED — no push attempted (no record + no lock); NO_REMOTE_PUSH — no remote; VERDICT: PASS; ERROR or CONFLICT — report it, VERDICT: FAIL.\n" +
1606
1636
  "NEVER force-push.\n\n" +
1607
- (projectConfig.versioned_build === true ?
1608
- "VERSION BUMP (issue #3): after MERGED+PUSHED, bump so QA can prove the bundle contains this fix — QA parks without it.\n" +
1609
- "1. cd " + REPO_PATH + " && git fetch origin; BR=<branch-from-integration-target>; F=" + (projectConfig.version_file || "client/src/buildNumber.ts") + "\n" +
1610
- "2. N=$(git show origin/$BR:$F | grep -o 'BUILD_NUMBER = [0-9]*' | grep -o '[0-9]*'); if missing/unparsable: VERDICT: FAIL.\n" +
1611
- "3. Edit $F: `export const BUILD_NUMBER = $N;` → `export const BUILD_NUMBER = $((N+1));` (keep header comment).\n" +
1612
- "4. git add $F && git commit -m \"build-number: $((N+1)) - QA version gate\" (SEPARATE commit, never amend).\n" +
1613
- "5. git push origin $BR; if rejected retry 3x (fetch, re-read N, re-bump, re-commit, push). NEVER force-push. Push MUST succeed or VERDICT: FAIL.\n" +
1614
- "6. Run, return stdout verbatim:\n" + crewCmd("log-event", { task_id: taskId, type: "note", identity: step.identity, message: "dashboard-version: <new> — QA must test a bundle built from source at or after the merge that recorded this (build <new> or later)." }) + "\n(substitute <new>).\n\n" : "") +
1615
1637
  "Report what happened at each step, ending with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
1616
1638
 
1617
1639
  } else if (step.name === "Publish") {
@@ -2065,8 +2087,6 @@ while (i < STEPS.length) {
2065
2087
  "c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
2066
2088
  "d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the deck, then click Study, then judge the study view. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and judge what you did reach.\n" +
2067
2089
  "e. Judge as a user against the task description: does the change render correctly? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
2068
- (projectConfig.versioned_build === true ?
2069
- "e2. VERSION (issue #3): footer shows `build <n>` — report `footer_build: <n>` on its own line, or `footer_build: unreadable`. Required — the gate cannot pass without it.\n" : "") +
2070
2090
  "f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
2071
2091
  "Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
2072
2092
  "STEP 2: Verify data integrity via the crew API.\n" +
@@ -2411,20 +2431,24 @@ while (i < STEPS.length) {
2411
2431
  // never a park. No automatic FAIL override: if the QA agent still
2412
2432
  // reports FAIL, it stands — finding attribution informs follow-up
2413
2433
  // filing only.
2414
- if (step.name === "QA") {
2415
- if (projectConfig.versioned_build === true && SURFACE_ARTIFACT && qaRequiredVersion !== null) {
2416
- var vgM = /footer_build:\s*(\d+|unreadable)/i.exec(workerText || "");
2417
- var vgS = (vgM && /^\d+$/.test(vgM[1])) ? parseInt(vgM[1], 10) : null;
2418
- var vgD = decideVersionGate(vgS, qaRequiredVersion);
2419
- if (!vgD.proceed) {
2420
- return await parkTask(vgD.reason + ": " + vgD.remedy + " [built=" + vgS + " required=" + qaRequiredVersion + "]");
2421
- }
2422
- await agent(
2423
- "Run in shell and return the stdout verbatim:\n" + crewCmd("log-event", {
2424
- task_id: taskId, type: "note", identity: step.identity,
2425
- message: "version-check: built=" + vgS + " >= required=" + qaRequiredVersion + " → testing now" }),
2426
- { key: "record-version-check-" + taskId, label: "Recording version-gate pass" }
2434
+ if (step.name === "QA" && qaBundleHash !== null) {
2435
+ // Closeout: the bundle Hazel tested must match the ensure-deployed hash
2436
+ // from QA start. Mismatch/unreadable = operational failure, never a park.
2437
+ var closeHash = "";
2438
+ try {
2439
+ var closeResult = await agent(
2440
+ "Run: node " + QA_DEPLOY + " --hash-only --qa-dir " + QA_ENV_DIR + " 2>&1\n" +
2441
+ "Return JSON { \"output\": \"<stdout, trimmed>\" } and nothing else.",
2442
+ { key: "qa-closeout-hash-" + taskId, label: "Reading QA bundle hash",
2443
+ schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
2427
2444
  );
2445
+ var hm = /BUNDLE_HASH ([0-9a-f]{40})/.exec(String(closeResult.output || ""));
2446
+ closeHash = hm ? hm[1].slice(0, 7) : "";
2447
+ } catch (e) {}
2448
+ if (closeHash !== qaBundleHash) {
2449
+ log("QA closeout hash mismatch for task " + taskId + " — marking failed for retry");
2450
+ stepResult.summary = (stepResult.summary || "") + "\nqa_closeout: FAILED — bundle hash " + (closeHash ? "changed during QA" : "unreadable");
2451
+ passed = false;
2428
2452
  }
2429
2453
  var contentFindings = extractContentFindings(workerText);
2430
2454
  if (!contentFindings.ok) {
@@ -2555,6 +2579,29 @@ while (i < STEPS.length) {
2555
2579
  }
2556
2580
  if (/^VERIFIED:/m.test(integrateVerifyOut)) {
2557
2581
  log("Integrate verified for task " + taskId + ": task branch tip is an ancestor of the integration target");
2582
+ // QA deploy (replaces the version gate): after a verified merge, ensure
2583
+ // the QA environment serves a bundle built from merged source. Runs for
2584
+ // versioned_build projects on a classified artifact surface; experiential
2585
+ // tasks are the ones Hazel serves the bundle to. A failed deploy is an
2586
+ // operational failure (retryable under the dispatcher's cap), never a park.
2587
+ if (projectConfig.versioned_build === true && SURFACE_ARTIFACT) {
2588
+ var deployOut = "";
2589
+ try {
2590
+ var deployResult = await runDeployFerry("");
2591
+ deployOut = (deployResult.output || "").trim();
2592
+ } catch (e) {
2593
+ deployOut = "";
2594
+ }
2595
+ var deployParsed = parseDeployResult(deployOut);
2596
+ if (deployParsed.ok) {
2597
+ log("QA deploy ok for task " + taskId + ": bundle " + deployParsed.hash);
2598
+ stepResult.summary = (stepResult.summary || "") + "\ndeploy: ok " + deployParsed.hash;
2599
+ } else {
2600
+ log("QA deploy failed for task " + taskId + ": " + deployParsed.reason + " — marking failed for retry");
2601
+ stepResult.summary = (stepResult.summary || "") + "\ndeploy: FAILED — " + deployParsed.reason;
2602
+ passed = false;
2603
+ }
2604
+ }
2558
2605
  } else {
2559
2606
  var integrateVerifyReason = integrateVerifyOut
2560
2607
  ? integrateVerifyOut.split("\n")[0].slice(0, 200)