muse-crew 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +14 -0
- package/lib/AGENTS.md +3 -0
- package/lib/compare-dispatch-shadow.js +150 -0
- package/lib/crew-api.js +31 -0
- package/lib/crew-dispatch-worker.js +1059 -0
- package/lib/qa-deploy.mjs +498 -0
- package/lib/schema.sql +19 -0
- package/lib/spawn-boundary.js +101 -0
- package/package.json +1 -1
- package/seed/cron-body-template.md +8 -0
- package/workflows/bugfix.js +84 -50
- package/workflows/chore.js +8 -2
- package/workflows/crew-dispatch.js +8 -5
- package/workflows/standard.js +97 -50
package/workflows/bugfix.js
CHANGED
|
@@ -72,8 +72,14 @@ const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
|
|
|
72
72
|
const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
|
|
73
73
|
const NOTE_VOCAB_SRC = crewHome + "/current/lib/publish-note-vocabulary.js";
|
|
74
74
|
const NOTE_VOCAB = RUN_LIB + "/publish-note-vocabulary.js";
|
|
75
|
+
const QA_DEPLOY_SRC = crewHome + "/current/lib/qa-deploy.mjs";
|
|
76
|
+
const QA_DEPLOY = RUN_LIB + "/qa-deploy.mjs";
|
|
77
|
+
const SERVE_ARTIFACT_SRC = crewHome + "/current/lib/serve-artifact.js";
|
|
78
|
+
const SERVE_ARTIFACT = RUN_LIB + "/serve-artifact.js";
|
|
79
|
+
const QA_DB_SRC = crewHome + "/current/lib/qa-db.js";
|
|
80
|
+
const QA_DB = RUN_LIB + "/qa-db.js";
|
|
75
81
|
// See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
|
|
76
|
-
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB].map(function (p) { return p.split("/").pop(); });
|
|
82
|
+
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB, QA_DEPLOY, SERVE_ARTIFACT, QA_DB].map(function (p) { return p.split("/").pop(); });
|
|
77
83
|
|
|
78
84
|
// Project config — passed by dispatcher, falls back to dashboard defaults
|
|
79
85
|
const projectConfig = inputs.project_config || {};
|
|
@@ -265,7 +271,7 @@ function attemptKey(base, reworkCount) {
|
|
|
265
271
|
function pinLifecycle(key) {
|
|
266
272
|
return agent(
|
|
267
273
|
"Snapshot lifecycle scripts for version pinning.\n" +
|
|
268
|
-
"Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
|
|
274
|
+
"Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && cp " + QA_DEPLOY_SRC + " " + QA_DEPLOY + " && cp " + SERVE_ARTIFACT_SRC + " " + SERVE_ARTIFACT + " && cp " + QA_DB_SRC + " " + QA_DB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
|
|
269
275
|
"Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
|
|
270
276
|
{ key: key, label: "Pinning lifecycle scripts",
|
|
271
277
|
schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
|
|
@@ -563,17 +569,28 @@ function decideIntegrateRetry(o) {
|
|
|
563
569
|
if (rec.malformed) return "park";
|
|
564
570
|
return ancestor ? "skip-to-publish" : "proceed";
|
|
565
571
|
}
|
|
566
|
-
//
|
|
567
|
-
//
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
if (
|
|
571
|
-
|
|
572
|
-
if (
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
572
|
+
// parseDeployResult: pure, fails closed. Ferry returns qa-deploy.mjs stdout
|
|
573
|
+
// verbatim; the marker line is the only machine-read signal.
|
|
574
|
+
function parseDeployResult(output) {
|
|
575
|
+
var m = /^DEPLOY_OK ([0-9a-f]{7})$/m.exec(String(output || "").trim());
|
|
576
|
+
if (m) return { ok: true, hash: m[1] };
|
|
577
|
+
var f = /^DEPLOY_FAIL ([a-z-]+):(.{1,200})$/m.exec(String(output || "").trim());
|
|
578
|
+
if (f) return { ok: false, reason: f[1] + ": " + f[2].trim() };
|
|
579
|
+
return { ok: false, reason: "unrecognized deploy output" };
|
|
580
|
+
}
|
|
581
|
+
// QA_ENV_DIR: pipeline-owned, under the crew home. Never the web artifact dir.
|
|
582
|
+
var QA_ENV_DIR = crewHome + "/qa-envs/" + LAUNCH_PROJECT_ID;
|
|
583
|
+
// runDeployFerry: thin mechanical agent, one pinned command, stdout verbatim
|
|
584
|
+
function runDeployFerry(mode) {
|
|
585
|
+
var cmd = "node " + QA_DEPLOY + " --repo " + REPO_PATH + " --qa-dir " + QA_ENV_DIR +
|
|
586
|
+
" --task " + taskId + " --crew-home " + crewHome + " --crew-api " + CREW_API_PINNED +
|
|
587
|
+
" --serve-artifact " + SERVE_ARTIFACT + (mode === "--ensure" ? " --ensure" : "") + " 2>&1";
|
|
588
|
+
return agent(
|
|
589
|
+
"Run in shell and return the stdout verbatim:\n" + cmd + "\n" +
|
|
590
|
+
"Return JSON { \"output\": \"<the command's full stdout, trimmed>\" } and nothing else.",
|
|
591
|
+
{ key: "qa-deploy-" + mode.replace(/-/g, "") + "-" + taskId, label: "Deploying QA environment",
|
|
592
|
+
schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
|
|
593
|
+
);
|
|
577
594
|
}
|
|
578
595
|
// On a dispatcher retry resumed at Integrate, the run-local releaseDecision
|
|
579
596
|
// is null (Build doesn't re-run). Hydrate it from the merge record so the
|
|
@@ -1367,25 +1384,28 @@ while (i < STEPS.length) {
|
|
|
1367
1384
|
// verdicts.jsonl). No parent verdict gate remains.
|
|
1368
1385
|
var qaArtifact = false;
|
|
1369
1386
|
var qaTerminal = false;
|
|
1370
|
-
var
|
|
1387
|
+
var qaBundleHash = null;
|
|
1371
1388
|
if (step.name === "QA") {
|
|
1372
1389
|
var qaExp = (await resolveExperiential()) === "yes";
|
|
1373
1390
|
qaArtifact = qaExp && SURFACE_ARTIFACT;
|
|
1374
1391
|
qaTerminal = qaExp && SURFACE_TERMINAL;
|
|
1375
1392
|
if (qaArtifact && projectConfig.versioned_build === true) {
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1380
|
-
|
|
1381
|
-
|
|
1382
|
-
|
|
1383
|
-
|
|
1384
|
-
|
|
1385
|
-
|
|
1386
|
-
)
|
|
1387
|
-
|
|
1388
|
-
|
|
1393
|
+
// ensure-deployed: idempotent precondition. Fast path when bundle is
|
|
1394
|
+
// current. Failed deploy = operational failure (retryable), never a park.
|
|
1395
|
+
var ensureOut = "";
|
|
1396
|
+
try {
|
|
1397
|
+
var ensureResult = await runDeployFerry("--ensure");
|
|
1398
|
+
ensureOut = (ensureResult.output || "").trim();
|
|
1399
|
+
} catch (e) {
|
|
1400
|
+
ensureOut = "";
|
|
1401
|
+
}
|
|
1402
|
+
var ensureParsed = parseDeployResult(ensureOut);
|
|
1403
|
+
if (!ensureParsed.ok) {
|
|
1404
|
+
log("QA ensure-deployed failed for task " + taskId + ": " + ensureParsed.reason + " — returning failed for dispatcher retry");
|
|
1405
|
+
return { status: "failed", task_id: taskId, reason: "QA ensure-deployed failed: " + ensureParsed.reason };
|
|
1406
|
+
}
|
|
1407
|
+
qaBundleHash = ensureParsed.hash;
|
|
1408
|
+
log("QA ensure-deployed ok for task " + taskId + ": bundle " + qaBundleHash);
|
|
1389
1409
|
}
|
|
1390
1410
|
}
|
|
1391
1411
|
|
|
@@ -1715,14 +1735,6 @@ while (i < STEPS.length) {
|
|
|
1715
1735
|
"R5-PUSH (manual R5 resolution only — the normal path pushed inline). Run: " + LIFECYCLE_ENV + LIFECYCLE + " push-target " + taskId + "\n" +
|
|
1716
1736
|
"PUSHED — report the merged hash (detached prints PUSHED: origin/main (refspec HEAD:main)), VERDICT: PASS; PUSH_SKIPPED — no push attempted (no record + no lock); NO_REMOTE_PUSH — no remote; VERDICT: PASS; ERROR or CONFLICT — report it, VERDICT: FAIL.\n" +
|
|
1717
1737
|
"NEVER force-push.\n\n" +
|
|
1718
|
-
(projectConfig.versioned_build === true ?
|
|
1719
|
-
"VERSION BUMP (issue #3): after MERGED+PUSHED, bump so QA can prove the bundle contains this fix — QA parks without it.\n" +
|
|
1720
|
-
"1. cd " + REPO_PATH + " && git fetch origin; BR=<branch-from-integration-target>; F=" + (projectConfig.version_file || "client/src/buildNumber.ts") + "\n" +
|
|
1721
|
-
"2. N=$(git show origin/$BR:$F | grep -o 'BUILD_NUMBER = [0-9]*' | grep -o '[0-9]*'); if missing/unparsable: VERDICT: FAIL.\n" +
|
|
1722
|
-
"3. Edit $F: `export const BUILD_NUMBER = $N;` → `export const BUILD_NUMBER = $((N+1));` (keep header comment).\n" +
|
|
1723
|
-
"4. git add $F && git commit -m \"build-number: $((N+1)) - QA version gate\" (SEPARATE commit, never amend).\n" +
|
|
1724
|
-
"5. git push origin $BR; if rejected retry 3x (fetch, re-read N, re-bump, re-commit, push). NEVER force-push. Push MUST succeed or VERDICT: FAIL.\n" +
|
|
1725
|
-
"6. Run, return stdout verbatim:\n" + crewCmd("log-event", { task_id: taskId, type: "note", identity: step.identity, message: "dashboard-version: <new> — QA must test a bundle built from source at or after the merge that recorded this (build <new> or later)." }) + "\n(substitute <new>).\n\n" : "") +
|
|
1726
1738
|
"Report what happened at each step, ending with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
1727
1739
|
|
|
1728
1740
|
} else if (step.name === "Publish") {
|
|
@@ -2176,8 +2188,6 @@ while (i < STEPS.length) {
|
|
|
2176
2188
|
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
2177
2189
|
"d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the dialog, then confirm it, then judge the result. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and continue with the mechanical checks.\n" +
|
|
2178
2190
|
"e. Judge as a user against the task description: is the reported bug fixed AND is nothing else visibly broken? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
|
|
2179
|
-
(projectConfig.versioned_build === true ?
|
|
2180
|
-
"e2. VERSION (issue #3): footer shows `build <n>` — report `footer_build: <n>` on its own line, or `footer_build: unreadable`. Required — the gate cannot pass without it.\n" : "") +
|
|
2181
2191
|
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
2182
2192
|
"Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
|
|
2183
2193
|
"MECHANICAL CHECKS:\n" +
|
|
@@ -2659,20 +2669,24 @@ while (i < STEPS.length) {
|
|
|
2659
2669
|
// never a park. No automatic FAIL override: if the QA agent still
|
|
2660
2670
|
// reports FAIL, it stands — finding attribution informs follow-up
|
|
2661
2671
|
// filing only.
|
|
2662
|
-
if (step.name === "QA") {
|
|
2663
|
-
|
|
2664
|
-
|
|
2665
|
-
|
|
2666
|
-
|
|
2667
|
-
|
|
2668
|
-
|
|
2669
|
-
|
|
2670
|
-
|
|
2671
|
-
|
|
2672
|
-
task_id: taskId, type: "note", identity: step.identity,
|
|
2673
|
-
message: "version-check: built=" + vgS + " >= required=" + qaRequiredVersion + " → testing now" }),
|
|
2674
|
-
{ key: "record-version-check-" + taskId, label: "Recording version-gate pass" }
|
|
2672
|
+
if (step.name === "QA" && qaBundleHash !== null) {
|
|
2673
|
+
// Closeout: bundle Hazel tested must match ensure-deployed hash from QA
|
|
2674
|
+
// start. Mismatch/unreadable = operational failure, never a park.
|
|
2675
|
+
var closeHash = "";
|
|
2676
|
+
try {
|
|
2677
|
+
var closeResult = await agent(
|
|
2678
|
+
"Run: node " + QA_DEPLOY + " --hash-only --qa-dir " + QA_ENV_DIR + " 2>&1\n" +
|
|
2679
|
+
"Return JSON { \"output\": \"<stdout, trimmed>\" } and nothing else.",
|
|
2680
|
+
{ key: "qa-closeout-hash-" + taskId, label: "Reading QA bundle hash",
|
|
2681
|
+
schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
|
|
2675
2682
|
);
|
|
2683
|
+
var chm = /BUNDLE_HASH ([0-9a-f]{40})/.exec(String(closeResult.output || ""));
|
|
2684
|
+
closeHash = chm ? chm[1].slice(0, 7) : "";
|
|
2685
|
+
} catch (e) {}
|
|
2686
|
+
if (closeHash !== qaBundleHash) {
|
|
2687
|
+
log("QA closeout hash mismatch for task " + taskId + " — marking failed for retry");
|
|
2688
|
+
stepResult.summary = (stepResult.summary || "") + "\nqa_closeout: FAILED — bundle hash " + (closeHash ? "changed during QA" : "unreadable");
|
|
2689
|
+
passed = false;
|
|
2676
2690
|
}
|
|
2677
2691
|
var contentFindings = extractContentFindings(workerText);
|
|
2678
2692
|
if (!contentFindings.ok) {
|
|
@@ -2803,6 +2817,26 @@ while (i < STEPS.length) {
|
|
|
2803
2817
|
}
|
|
2804
2818
|
if (/^VERIFIED:/m.test(integrateVerifyOut)) {
|
|
2805
2819
|
log("Integrate verified for task " + taskId + ": task branch tip is an ancestor of the integration target");
|
|
2820
|
+
// QA deploy: after verified merge, ensure QA env serves a bundle from
|
|
2821
|
+
// merged source. Failed deploy = operational failure, never a park.
|
|
2822
|
+
if (projectConfig.versioned_build === true && SURFACE_ARTIFACT) {
|
|
2823
|
+
var deployOut = "";
|
|
2824
|
+
try {
|
|
2825
|
+
var deployResult = await runDeployFerry("");
|
|
2826
|
+
deployOut = (deployResult.output || "").trim();
|
|
2827
|
+
} catch (e) {
|
|
2828
|
+
deployOut = "";
|
|
2829
|
+
}
|
|
2830
|
+
var deployParsed = parseDeployResult(deployOut);
|
|
2831
|
+
if (deployParsed.ok) {
|
|
2832
|
+
log("QA deploy ok for task " + taskId + ": bundle " + deployParsed.hash);
|
|
2833
|
+
stepResult.summary = (stepResult.summary || "") + "\ndeploy: ok " + deployParsed.hash;
|
|
2834
|
+
} else {
|
|
2835
|
+
log("QA deploy failed for task " + taskId + ": " + deployParsed.reason + " — marking failed for retry");
|
|
2836
|
+
stepResult.summary = (stepResult.summary || "") + "\ndeploy: FAILED — " + deployParsed.reason;
|
|
2837
|
+
passed = false;
|
|
2838
|
+
}
|
|
2839
|
+
}
|
|
2806
2840
|
} else {
|
|
2807
2841
|
var integrateVerifyReason = integrateVerifyOut
|
|
2808
2842
|
? integrateVerifyOut.split("\n")[0].slice(0, 200)
|
package/workflows/chore.js
CHANGED
|
@@ -136,8 +136,14 @@ const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
|
|
|
136
136
|
const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
|
|
137
137
|
const NOTE_VOCAB_SRC = crewHome + "/current/lib/publish-note-vocabulary.js";
|
|
138
138
|
const NOTE_VOCAB = RUN_LIB + "/publish-note-vocabulary.js";
|
|
139
|
+
const QA_DEPLOY_SRC = crewHome + "/current/lib/qa-deploy.mjs";
|
|
140
|
+
const QA_DEPLOY = RUN_LIB + "/qa-deploy.mjs";
|
|
141
|
+
const SERVE_ARTIFACT_SRC = crewHome + "/current/lib/serve-artifact.js";
|
|
142
|
+
const SERVE_ARTIFACT = RUN_LIB + "/serve-artifact.js";
|
|
143
|
+
const QA_DB_SRC = crewHome + "/current/lib/qa-db.js";
|
|
144
|
+
const QA_DB = RUN_LIB + "/qa-db.js";
|
|
139
145
|
// See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
|
|
140
|
-
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB].map(function (p) { return p.split("/").pop(); });
|
|
146
|
+
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB, QA_DEPLOY, SERVE_ARTIFACT, QA_DB].map(function (p) { return p.split("/").pop(); });
|
|
141
147
|
|
|
142
148
|
// Project config — passed by dispatcher, falls back to dashboard defaults
|
|
143
149
|
const projectConfig = inputs.project_config || {};
|
|
@@ -275,7 +281,7 @@ function attemptKey(base, reworkCount) {
|
|
|
275
281
|
function pinLifecycle(key) {
|
|
276
282
|
return agent(
|
|
277
283
|
"Snapshot lifecycle scripts for version pinning.\n" +
|
|
278
|
-
"Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
|
|
284
|
+
"Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && cp " + QA_DEPLOY_SRC + " " + QA_DEPLOY + " && cp " + SERVE_ARTIFACT_SRC + " " + SERVE_ARTIFACT + " && cp " + QA_DB_SRC + " " + QA_DB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
|
|
279
285
|
"Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
|
|
280
286
|
{ key: key, label: "Pinning lifecycle scripts",
|
|
281
287
|
schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
|
|
@@ -477,10 +477,9 @@ for (var pi = 0; pi < projects.length; pi++) {
|
|
|
477
477
|
// terminal | null=unclassified). Carried alongside deploy_type — it is a
|
|
478
478
|
// separate axis, not a redeclaration of the deployment target.
|
|
479
479
|
environment_type: proj.environment_type || null,
|
|
480
|
-
// QA
|
|
481
|
-
// project into the Integrate
|
|
482
|
-
// version_file is
|
|
483
|
-
// BUILD_NUMBER (null when unversioned).
|
|
480
|
+
// QA deploy (replaces the issue #3 version gate): versioned_build opts the
|
|
481
|
+
// project into the post-Integrate QA deploy + pre-QA ensure-deployed;
|
|
482
|
+
// version_file is retained for compatibility but no longer consumed.
|
|
484
483
|
versioned_build: !!proj.versioned_build,
|
|
485
484
|
version_file: proj.version_file || null
|
|
486
485
|
};
|
|
@@ -1055,7 +1054,11 @@ for (var p = 0; p < toProcess.length; p++) {
|
|
|
1055
1054
|
};
|
|
1056
1055
|
|
|
1057
1056
|
log("Recommended " + iworkflow + " for \"" + itask.title + "\" [" + taskProject + "] at step " + nextStepName);
|
|
1058
|
-
|
|
1057
|
+
// executor tag (Piece 1, 2026-09-26): the sandboxed dispatcher claims
|
|
1058
|
+
// "sandbox"; the worker-layer dispatcher (lib/crew-dispatch-worker.js)
|
|
1059
|
+
// claims "worker". Piece 2 routes launches on this tag. The shadow
|
|
1060
|
+
// comparison ignores it.
|
|
1061
|
+
results.push({ task_id: itask.id, workflow: iworkflow, step: nextStepName, action: "recommended", scriptPath: scriptPath, args: launchArgs, executor: "sandbox" });
|
|
1059
1062
|
|
|
1060
1063
|
} // end for (per-task loop)
|
|
1061
1064
|
} // end processing block
|
package/workflows/standard.js
CHANGED
|
@@ -80,8 +80,14 @@ const CLASSIFY_SURFACE_SRC = crewHome + "/current/lib/classify-surface.js";
|
|
|
80
80
|
const CLASSIFY_SURFACE = RUN_LIB + "/classify-surface.js";
|
|
81
81
|
const NOTE_VOCAB_SRC = crewHome + "/current/lib/publish-note-vocabulary.js";
|
|
82
82
|
const NOTE_VOCAB = RUN_LIB + "/publish-note-vocabulary.js";
|
|
83
|
+
const QA_DEPLOY_SRC = crewHome + "/current/lib/qa-deploy.mjs";
|
|
84
|
+
const QA_DEPLOY = RUN_LIB + "/qa-deploy.mjs";
|
|
85
|
+
const SERVE_ARTIFACT_SRC = crewHome + "/current/lib/serve-artifact.js";
|
|
86
|
+
const SERVE_ARTIFACT = RUN_LIB + "/serve-artifact.js";
|
|
87
|
+
const QA_DB_SRC = crewHome + "/current/lib/qa-db.js";
|
|
88
|
+
const QA_DB = RUN_LIB + "/qa-db.js";
|
|
83
89
|
// See docs/decisions/qa-reproduce.md#pin-basenames: the pin step materializes the required scripts.
|
|
84
|
-
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB].map(function (p) { return p.split("/").pop(); });
|
|
90
|
+
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED, COMPUTE_DIFF, CLASSIFY_SURFACE, NOTE_VOCAB, QA_DEPLOY, SERVE_ARTIFACT, QA_DB].map(function (p) { return p.split("/").pop(); });
|
|
85
91
|
|
|
86
92
|
// Project config — passed by dispatcher, falls back to dashboard defaults
|
|
87
93
|
const projectConfig = inputs.project_config || {};
|
|
@@ -273,7 +279,7 @@ function attemptKey(base, reworkCount) {
|
|
|
273
279
|
function pinLifecycle(key) {
|
|
274
280
|
return agent(
|
|
275
281
|
"Snapshot lifecycle scripts for version pinning.\n" +
|
|
276
|
-
"Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
|
|
282
|
+
"Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && cp " + COMPUTE_DIFF_SRC + " " + COMPUTE_DIFF + " && cp " + CLASSIFY_SURFACE_SRC + " " + CLASSIFY_SURFACE + " && cp " + NOTE_VOCAB_SRC + " " + NOTE_VOCAB + " && cp " + QA_DEPLOY_SRC + " " + QA_DEPLOY + " && cp " + SERVE_ARTIFACT_SRC + " " + SERVE_ARTIFACT + " && cp " + QA_DB_SRC + " " + QA_DB + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
|
|
277
283
|
"Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
|
|
278
284
|
{ key: key, label: "Pinning lifecycle scripts",
|
|
279
285
|
schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
|
|
@@ -571,17 +577,37 @@ function decideIntegrateRetry(o) {
|
|
|
571
577
|
if (rec.malformed) return "park";
|
|
572
578
|
return ancestor ? "skip-to-publish" : "proceed";
|
|
573
579
|
}
|
|
574
|
-
// QA version gate
|
|
575
|
-
//
|
|
576
|
-
// Pure —
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
if (
|
|
581
|
-
|
|
582
|
-
if (
|
|
583
|
-
return {
|
|
584
|
-
|
|
580
|
+
// QA deploy result parser (replaces the version gate, issue #3): the ferry
|
|
581
|
+
// returns qa-deploy.mjs's stdout verbatim; the marker line is the only
|
|
582
|
+
// machine-read signal. Pure — fails closed on anything unrecognized.
|
|
583
|
+
// Returns { ok, hash } or { ok:false, reason }.
|
|
584
|
+
function parseDeployResult(output) {
|
|
585
|
+
var m = /^DEPLOY_OK ([0-9a-f]{7})$/m.exec(String(output || "").trim());
|
|
586
|
+
if (m) return { ok: true, hash: m[1] };
|
|
587
|
+
var f = /^DEPLOY_FAIL ([a-z-]+):(.{1,200})$/m.exec(String(output || "").trim());
|
|
588
|
+
if (f) return { ok: false, reason: f[1] + ": " + f[2].trim() };
|
|
589
|
+
return { ok: false, reason: "unrecognized deploy output" };
|
|
590
|
+
}
|
|
591
|
+
// QA environment dir — pipeline-owned, under the crew home. The web artifact
|
|
592
|
+
// directory (~/workspace/ts-spaces/<slug>/) is never built by hand (its own
|
|
593
|
+
// AGENTS.md forbids it); this separate dir is cloned, synced, built, and
|
|
594
|
+
// validated by qa-deploy.mjs. Named by project id (deploy_slug is an
|
|
595
|
+
// artifact-system concern; the QA env is ours).
|
|
596
|
+
var QA_ENV_DIR = crewHome + "/qa-envs/" + LAUNCH_PROJECT_ID;
|
|
597
|
+
// Run the deploy ferry: a thin mechanical agent that runs one exact pinned
|
|
598
|
+
// command and returns stdout verbatim. No identity, no judgment — the
|
|
599
|
+
// module self-records the full evidence via the Crew API; the ferry carries
|
|
600
|
+
// only the marker line.
|
|
601
|
+
function runDeployFerry(mode) {
|
|
602
|
+
var cmd = "node " + QA_DEPLOY + " --repo " + REPO_PATH + " --qa-dir " + QA_ENV_DIR +
|
|
603
|
+
" --task " + taskId + " --crew-home " + crewHome + " --crew-api " + CREW_API_PINNED +
|
|
604
|
+
" --serve-artifact " + SERVE_ARTIFACT + (mode === "--ensure" ? " --ensure" : "") + " 2>&1";
|
|
605
|
+
return agent(
|
|
606
|
+
"Run in shell and return the stdout verbatim:\n" + cmd + "\n" +
|
|
607
|
+
"Return JSON { \"output\": \"<the command's full stdout, trimmed>\" } and nothing else.",
|
|
608
|
+
{ key: "qa-deploy-" + mode.replace(/-/g, "") + "-" + taskId, label: "Deploying QA environment",
|
|
609
|
+
schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
|
|
610
|
+
);
|
|
585
611
|
}
|
|
586
612
|
// On a dispatcher retry resumed at Integrate, the run-local releaseDecision
|
|
587
613
|
// is null (Build doesn't re-run). Hydrate it from the merge record so the
|
|
@@ -1310,23 +1336,27 @@ while (i < STEPS.length) {
|
|
|
1310
1336
|
// experiential-loop guard (a PASS with missing experiential evidence is
|
|
1311
1337
|
// never terminal).
|
|
1312
1338
|
var qaExperiential = false;
|
|
1339
|
+
var qaBundleHash = null;
|
|
1313
1340
|
if (step.name === "QA") {
|
|
1314
1341
|
qaExperiential = (await resolveExperiential()) === "yes" && SURFACE_CLASSIFIED;
|
|
1315
|
-
var qaRequiredVersion = null;
|
|
1316
1342
|
if (qaExperiential && SURFACE_ARTIFACT && projectConfig.versioned_build === true) {
|
|
1317
|
-
|
|
1318
|
-
|
|
1319
|
-
|
|
1320
|
-
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1325
|
-
|
|
1326
|
-
|
|
1327
|
-
);
|
|
1328
|
-
if (!
|
|
1329
|
-
|
|
1343
|
+
// ensure-deployed: idempotent precondition. Fast path when the bundle
|
|
1344
|
+
// is already current. Failed deploy = operational failure (retryable),
|
|
1345
|
+
// never a park. The module self-records the bundle hash to the event log.
|
|
1346
|
+
var ensureOut = "";
|
|
1347
|
+
try {
|
|
1348
|
+
var ensureResult = await runDeployFerry("--ensure");
|
|
1349
|
+
ensureOut = (ensureResult.output || "").trim();
|
|
1350
|
+
} catch (e) {
|
|
1351
|
+
ensureOut = "";
|
|
1352
|
+
}
|
|
1353
|
+
var ensureParsed = parseDeployResult(ensureOut);
|
|
1354
|
+
if (!ensureParsed.ok) {
|
|
1355
|
+
log("QA ensure-deployed failed for task " + taskId + ": " + ensureParsed.reason + " — returning failed for dispatcher retry");
|
|
1356
|
+
return { status: "failed", task_id: taskId, reason: "QA ensure-deployed failed: " + ensureParsed.reason };
|
|
1357
|
+
}
|
|
1358
|
+
qaBundleHash = ensureParsed.hash;
|
|
1359
|
+
log("QA ensure-deployed ok for task " + taskId + ": bundle " + qaBundleHash);
|
|
1330
1360
|
}
|
|
1331
1361
|
}
|
|
1332
1362
|
|
|
@@ -1604,14 +1634,6 @@ while (i < STEPS.length) {
|
|
|
1604
1634
|
"R5-PUSH (manual R5 resolution only — the normal path pushed inline). Run: " + LIFECYCLE_ENV + LIFECYCLE + " push-target " + taskId + "\n" +
|
|
1605
1635
|
"PUSHED — report the merged hash (detached prints PUSHED: origin/main (refspec HEAD:main)), VERDICT: PASS; PUSH_SKIPPED — no push attempted (no record + no lock); NO_REMOTE_PUSH — no remote; VERDICT: PASS; ERROR or CONFLICT — report it, VERDICT: FAIL.\n" +
|
|
1606
1636
|
"NEVER force-push.\n\n" +
|
|
1607
|
-
(projectConfig.versioned_build === true ?
|
|
1608
|
-
"VERSION BUMP (issue #3): after MERGED+PUSHED, bump so QA can prove the bundle contains this fix — QA parks without it.\n" +
|
|
1609
|
-
"1. cd " + REPO_PATH + " && git fetch origin; BR=<branch-from-integration-target>; F=" + (projectConfig.version_file || "client/src/buildNumber.ts") + "\n" +
|
|
1610
|
-
"2. N=$(git show origin/$BR:$F | grep -o 'BUILD_NUMBER = [0-9]*' | grep -o '[0-9]*'); if missing/unparsable: VERDICT: FAIL.\n" +
|
|
1611
|
-
"3. Edit $F: `export const BUILD_NUMBER = $N;` → `export const BUILD_NUMBER = $((N+1));` (keep header comment).\n" +
|
|
1612
|
-
"4. git add $F && git commit -m \"build-number: $((N+1)) - QA version gate\" (SEPARATE commit, never amend).\n" +
|
|
1613
|
-
"5. git push origin $BR; if rejected retry 3x (fetch, re-read N, re-bump, re-commit, push). NEVER force-push. Push MUST succeed or VERDICT: FAIL.\n" +
|
|
1614
|
-
"6. Run, return stdout verbatim:\n" + crewCmd("log-event", { task_id: taskId, type: "note", identity: step.identity, message: "dashboard-version: <new> — QA must test a bundle built from source at or after the merge that recorded this (build <new> or later)." }) + "\n(substitute <new>).\n\n" : "") +
|
|
1615
1637
|
"Report what happened at each step, ending with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
1616
1638
|
|
|
1617
1639
|
} else if (step.name === "Publish") {
|
|
@@ -2065,8 +2087,6 @@ while (i < STEPS.length) {
|
|
|
2065
2087
|
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
2066
2088
|
"d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the deck, then click Study, then judge the study view. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and judge what you did reach.\n" +
|
|
2067
2089
|
"e. Judge as a user against the task description: does the change render correctly? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
|
|
2068
|
-
(projectConfig.versioned_build === true ?
|
|
2069
|
-
"e2. VERSION (issue #3): footer shows `build <n>` — report `footer_build: <n>` on its own line, or `footer_build: unreadable`. Required — the gate cannot pass without it.\n" : "") +
|
|
2070
2090
|
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
2071
2091
|
"Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
|
|
2072
2092
|
"STEP 2: Verify data integrity via the crew API.\n" +
|
|
@@ -2411,20 +2431,24 @@ while (i < STEPS.length) {
|
|
|
2411
2431
|
// never a park. No automatic FAIL override: if the QA agent still
|
|
2412
2432
|
// reports FAIL, it stands — finding attribution informs follow-up
|
|
2413
2433
|
// filing only.
|
|
2414
|
-
if (step.name === "QA") {
|
|
2415
|
-
|
|
2416
|
-
|
|
2417
|
-
|
|
2418
|
-
|
|
2419
|
-
|
|
2420
|
-
|
|
2421
|
-
|
|
2422
|
-
|
|
2423
|
-
|
|
2424
|
-
task_id: taskId, type: "note", identity: step.identity,
|
|
2425
|
-
message: "version-check: built=" + vgS + " >= required=" + qaRequiredVersion + " → testing now" }),
|
|
2426
|
-
{ key: "record-version-check-" + taskId, label: "Recording version-gate pass" }
|
|
2434
|
+
if (step.name === "QA" && qaBundleHash !== null) {
|
|
2435
|
+
// Closeout: the bundle Hazel tested must match the ensure-deployed hash
|
|
2436
|
+
// from QA start. Mismatch/unreadable = operational failure, never a park.
|
|
2437
|
+
var closeHash = "";
|
|
2438
|
+
try {
|
|
2439
|
+
var closeResult = await agent(
|
|
2440
|
+
"Run: node " + QA_DEPLOY + " --hash-only --qa-dir " + QA_ENV_DIR + " 2>&1\n" +
|
|
2441
|
+
"Return JSON { \"output\": \"<stdout, trimmed>\" } and nothing else.",
|
|
2442
|
+
{ key: "qa-closeout-hash-" + taskId, label: "Reading QA bundle hash",
|
|
2443
|
+
schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
|
|
2427
2444
|
);
|
|
2445
|
+
var hm = /BUNDLE_HASH ([0-9a-f]{40})/.exec(String(closeResult.output || ""));
|
|
2446
|
+
closeHash = hm ? hm[1].slice(0, 7) : "";
|
|
2447
|
+
} catch (e) {}
|
|
2448
|
+
if (closeHash !== qaBundleHash) {
|
|
2449
|
+
log("QA closeout hash mismatch for task " + taskId + " — marking failed for retry");
|
|
2450
|
+
stepResult.summary = (stepResult.summary || "") + "\nqa_closeout: FAILED — bundle hash " + (closeHash ? "changed during QA" : "unreadable");
|
|
2451
|
+
passed = false;
|
|
2428
2452
|
}
|
|
2429
2453
|
var contentFindings = extractContentFindings(workerText);
|
|
2430
2454
|
if (!contentFindings.ok) {
|
|
@@ -2555,6 +2579,29 @@ while (i < STEPS.length) {
|
|
|
2555
2579
|
}
|
|
2556
2580
|
if (/^VERIFIED:/m.test(integrateVerifyOut)) {
|
|
2557
2581
|
log("Integrate verified for task " + taskId + ": task branch tip is an ancestor of the integration target");
|
|
2582
|
+
// QA deploy (replaces the version gate): after a verified merge, ensure
|
|
2583
|
+
// the QA environment serves a bundle built from merged source. Runs for
|
|
2584
|
+
// versioned_build projects on a classified artifact surface; experiential
|
|
2585
|
+
// tasks are the ones Hazel serves the bundle to. A failed deploy is an
|
|
2586
|
+
// operational failure (retryable under the dispatcher's cap), never a park.
|
|
2587
|
+
if (projectConfig.versioned_build === true && SURFACE_ARTIFACT) {
|
|
2588
|
+
var deployOut = "";
|
|
2589
|
+
try {
|
|
2590
|
+
var deployResult = await runDeployFerry("");
|
|
2591
|
+
deployOut = (deployResult.output || "").trim();
|
|
2592
|
+
} catch (e) {
|
|
2593
|
+
deployOut = "";
|
|
2594
|
+
}
|
|
2595
|
+
var deployParsed = parseDeployResult(deployOut);
|
|
2596
|
+
if (deployParsed.ok) {
|
|
2597
|
+
log("QA deploy ok for task " + taskId + ": bundle " + deployParsed.hash);
|
|
2598
|
+
stepResult.summary = (stepResult.summary || "") + "\ndeploy: ok " + deployParsed.hash;
|
|
2599
|
+
} else {
|
|
2600
|
+
log("QA deploy failed for task " + taskId + ": " + deployParsed.reason + " — marking failed for retry");
|
|
2601
|
+
stepResult.summary = (stepResult.summary || "") + "\ndeploy: FAILED — " + deployParsed.reason;
|
|
2602
|
+
passed = false;
|
|
2603
|
+
}
|
|
2604
|
+
}
|
|
2558
2605
|
} else {
|
|
2559
2606
|
var integrateVerifyReason = integrateVerifyOut
|
|
2560
2607
|
? integrateVerifyOut.split("\n")[0].slice(0, 200)
|