muse-crew 0.17.2 → 0.17.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +11 -0
- package/docs/decisions/composition-machinery.md +180 -0
- package/docs/decisions/publish-path.md +6 -0
- package/docs/decisions/workflow-core.md +5 -4
- package/lib/AGENTS.md +2 -1
- package/lib/bugfix/phases/build.js +168 -0
- package/lib/bugfix/phases/capture.js +170 -0
- package/lib/bugfix/phases/integrate.js +165 -0
- package/lib/bugfix/phases/map.js +129 -0
- package/lib/bugfix/phases/publish.js +592 -0
- package/lib/bugfix/phases/qa.js +356 -0
- package/lib/bugfix/phases/reproduce.js +254 -0
- package/lib/bugfix/phases/review.js +292 -0
- package/lib/bugfix/phases/triage.js +69 -0
- package/lib/chore/CONTRACT.md +181 -0
- package/lib/chore/DISPOSITION.md +98 -0
- package/lib/chore/extract.js +204 -0
- package/lib/chore/phase-lib.js +885 -0
- package/lib/chore/phases/build.js +110 -0
- package/lib/chore/phases/capture.js +82 -0
- package/lib/chore/phases/integrate.js +109 -0
- package/lib/chore/phases/map.js +81 -0
- package/lib/chore/phases/publish.js +543 -0
- package/lib/chore/phases/review.js +255 -0
- package/lib/chore/phases/triage.js +57 -0
- package/lib/chore/prompts/evidence-gatherer.js +41 -0
- package/lib/chore/prompts/evidence-gatherer.schema.json +1 -0
- package/lib/chore/prompts/tool-check.js +15 -0
- package/lib/chore/prompts/trailers.js +56 -0
- package/lib/chore/prompts/verdict-reask.js +28 -0
- package/lib/chore/prompts/verdict-reask.schema.json +1 -0
- package/lib/chore/prompts/work-agent.js +52 -0
- package/lib/chore/prompts/work-agent.schema.json +1 -0
- package/lib/chore/spawn-keys.js +44 -0
- package/lib/chore/spawn-vocab.js +87 -0
- package/lib/chore-run.js +538 -0
- package/lib/chore-tick.js +289 -0
- package/lib/crew-api.js +273 -0
- package/lib/crew-dispatch-worker.js +27 -7
- package/lib/crew-release.sh +7 -2
- package/lib/extract.js +252 -0
- package/lib/prompts/tool-check.js +18 -0
- package/lib/prompts/trailers.js +59 -0
- package/lib/prompts/verdict-reask.js +31 -0
- package/lib/prompts/verdict-reask.schema.json +1 -0
- package/lib/prompts/work-agent.js +56 -0
- package/lib/prompts/work-agent.schema.json +1 -0
- package/lib/reap-spawns.js +407 -0
- package/lib/schema.sql +12 -1
- package/lib/spawn-keys.js +47 -0
- package/lib/spawn-step.js +572 -0
- package/lib/standard/phases/build.js +120 -0
- package/lib/standard/phases/capture.js +163 -0
- package/lib/standard/phases/integrate.js +172 -0
- package/lib/standard/phases/map.js +119 -0
- package/lib/standard/phases/publish.js +565 -0
- package/lib/standard/phases/qa.js +399 -0
- package/lib/standard/phases/review.js +281 -0
- package/lib/standard/phases/triage.js +64 -0
- package/lib/test-detached-integrate.sh +47 -0
- package/lib/workflow-driver.js +605 -0
- package/lib/workflow-lib.js +1012 -0
- package/lib/workflow-spec.js +187 -0
- package/lib/worktree-lifecycle.sh +55 -3
- package/package.json +1 -1
- package/seed/cron-body-template.md +61 -9
- package/workflows/bugfix.js +17 -17
- package/workflows/chore.js +16 -16
- package/workflows/docs.js +14 -11
- package/workflows/standard.js +16 -16
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
// lib/bugfix/phases/qa.js — Bugfix QA phase (worker layer).
|
|
2
|
+
//
|
|
3
|
+
// Hazel does final QA, code-blind. Derived from workflows/bugfix.js (the
|
|
4
|
+
// Bugfix QA step).
|
|
5
|
+
//
|
|
6
|
+
// Contract:
|
|
7
|
+
// - Read: task description; prior phase events (context); experiential
|
|
8
|
+
// flag (from Triage); Build's release decision (for the npm publish
|
|
9
|
+
// backstop); Triage's terminal_targets (terminal QA).
|
|
10
|
+
// - qaArtifact/qaTerminal branch the instructions (experiential see-act
|
|
11
|
+
// vs CLI loops vs state verification).
|
|
12
|
+
// - Verdict gate (experiential only): read-ooda-verdict.js cross-checks
|
|
13
|
+
// the prose VERDICT: line against verdict.json. Disagreement, missing,
|
|
14
|
+
// corrupt, or reason-less FAIL → record "failed", return FAILED
|
|
15
|
+
// (dispatcher retry — never rework routing).
|
|
16
|
+
// - Experiential loop unavailable on PASS → park fail-closed (a PASS
|
|
17
|
+
// without experiential evidence is never terminal).
|
|
18
|
+
// - CONTENT-FINDINGS block: machine-read at closeout; each finding
|
|
19
|
+
// classified against the publish diff file set. task-change → file a
|
|
20
|
+
// bugfix; otherwise record with attribution. Missing/malformed block →
|
|
21
|
+
// record "failed", return FAILED (fail-closed, never "no findings").
|
|
22
|
+
// - QA rejection → shared Review/QA budget (spec.reworkPhases). Exhausted
|
|
23
|
+
// → park (worktree preserved). Otherwise REWIND to Build.
|
|
24
|
+
// - QA PASS → ADVANCE with next:null (driver closeout marks the task done).
|
|
25
|
+
//
|
|
26
|
+
// Phase I/O contract (Phase D):
|
|
27
|
+
// read: task description, events, experiential flag, Build release
|
|
28
|
+
// decision, Triage terminal_targets, verdict.json, publish diff
|
|
29
|
+
// write: session (completed | rejected | failed), event, park row,
|
|
30
|
+
// follow-up tasks (content findings)
|
|
31
|
+
// out: ADVANCE (next:null → done) | REWIND → Build | PARK | FAILED |
|
|
32
|
+
// NEED_SPAWN / STANDBY (boundary)
|
|
33
|
+
|
|
34
|
+
import {
|
|
35
|
+
runWorkBoundary, recordPhase, buildEventPreamble, summarizeReport,
|
|
36
|
+
ensureClaimed, log, latestSessionNotes, crewApi, runCmd,
|
|
37
|
+
parkTask, logEvent,
|
|
38
|
+
} from "../../workflow-lib.js";
|
|
39
|
+
import { readFileSync } from "fs";
|
|
40
|
+
import {
|
|
41
|
+
extractReleaseDecision, extractVerdict, extractContentFindings,
|
|
42
|
+
classifyContentFinding,
|
|
43
|
+
} from "../../extract.js";
|
|
44
|
+
|
|
45
|
+
export const PHASE = { name: "QA", identity: "hazel" };
|
|
46
|
+
|
|
47
|
+
var MAX_REWORK = 2;
|
|
48
|
+
|
|
49
|
+
// buildInstructions — verbatim from workflows/bugfix.js (variable
|
|
50
|
+
// references remapped; logic and prose unchanged).
|
|
51
|
+
// ctx: { env, taskDesc, phaseCtx, hazelPhaseCtx }.
|
|
52
|
+
export function buildInstructions(ctx) {
|
|
53
|
+
var env = ctx.env;
|
|
54
|
+
// Backstop for merge-time versioning: when the accepted Build summary
|
|
55
|
+
// declared release: yes, QA verifies the registry actually moved. A silent
|
|
56
|
+
// publish skip becomes a loud QA failure with evidence, not a pass.
|
|
57
|
+
var npmPublishCheck = (env.publishType === "npm" && ctx.releaseDecision && ctx.releaseDecision.release === "yes")
|
|
58
|
+
? "NPM PUBLISH CHECK: the accepted Build report declared release: yes, so this run's Publish phase must have published — UNLESS it was skipped deterministically on an empty-diff Integrate. Find this task's recorded Publish result: run in shell and return the stdout verbatim:\n" + crewCmdString(env, "get-state", { events_limit: 1 }) + "\n, then find the session for this task_id with step \"Publish\" (status completed) in the returned sessions array and read its session notes (the Publish agent's summary — event history does NOT carry it).\n" +
|
|
59
|
+
"CHECK THE SKIP PATH FIRST: if the notes contain a line matching skipped: no-lock-held, Publish was skipped deterministically — Integrate reported MERGED_EMPTY (no commits ahead of the integration target), so no merge lock was taken and there was nothing to ship. Verify the notes contain that skip-marker line and do NOT contain a line matching published: muse-crew@. Do NOT run npm view and do NOT demand registry movement — nothing was supposed to ship. Report 'npm publish check: Publish skipped deterministically (empty-diff Integrate — nothing to ship)' and PASS this check.\n" +
|
|
60
|
+
"Only when the notes contain no skip marker must the publish have landed — run the full verification below.\n" +
|
|
61
|
+
"Extract the line matching TARGET_VERSION=<new-version> computed as <base> + <scope> → <new-version> (the workflow appends it to the Publish notes, so it is always present). If the line is missing, report 'npm publish verification failed: Publish notes did not carry the computed target version', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
62
|
+
"Verify the bump SCOPE: the <scope> in that line MUST equal the accepted version_bump scope \"" + ctx.releaseDecision.version_bump + "\" — if it differs, report the mismatch, then end your report with exactly this line: VERDICT: FAIL. Verify the ARITHMETIC: <base> + <scope> must equal <new-version> (patch increments the last segment only, e.g. 0.3.0 + patch → 0.3.1; minor increments the middle and resets the last to 0; major increments the first and resets the rest to 0) — if the math is wrong, report it, then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
63
|
+
"Extract the published version from the notes line matching published: muse-crew@<version>. It MUST equal <new-version> from the TARGET_VERSION line. Then run: npm view muse-crew version. The registry version MUST equal <new-version>. If any of these checks fails, report 'npm publish verification failed: [details]', then end your report with exactly this line: VERDICT: FAIL.\n"
|
|
64
|
+
: "";
|
|
65
|
+
var instructions = "Final QA testing. You are CODE-BLIND — do NOT read source code.\n" +
|
|
66
|
+
"Public docs (API.md, README, published action schemas) are NOT source code — read them freely, exactly as a user would.\n" +
|
|
67
|
+
"DOCS GATE: If the fix is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n" +
|
|
68
|
+
"To test, run in shell and read the stdout JSON:\n" + crewCmdString(env, "get-state", {}) + "\nThis returns current sessions, events, and tasks.\n" +
|
|
69
|
+
"Verify the fix against the task description in the returned data: the state the bug corrupted should now read correctly, and the task's own expected behavior should hold.\n" +
|
|
70
|
+
"You can also check specific data with: node " + env.crewApi + " --crew-home " + env.crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
|
|
71
|
+
"Do NOT use artifact_inspect — it was removed by the platform (2026-09-14) and does not exist; do not substitute artifact.inspect (malfunction diagnosis, not an inspection tool).\n" +
|
|
72
|
+
"File follow-up tasks by running in shell (EXCEPT content findings — user-visible content/data goes in the CONTENT-FINDINGS block at the end of these instructions, never through create-task):\n" + crewCmdString(env, "create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n" +
|
|
73
|
+
npmPublishCheck +
|
|
74
|
+
"Report your test results as plain prose.\n" +
|
|
75
|
+
"End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails.";
|
|
76
|
+
// ctx.qaArtifact: Hazel runs the experiential see-act loop herself (STEP 1
|
|
77
|
+
// below) and owns the visual verdict — no parent capture protocol.
|
|
78
|
+
if (ctx.qaArtifact) {
|
|
79
|
+
instructions = "Read the shared UX bar FIRST: " + env.uxDoctrinePath + " — it is the bar the whole crew builds to, and your verdict judges against it point by point.\n\n" +
|
|
80
|
+
"You are code-blind QA. You NEVER read source files. Public docs are not source — read them as a user would.\n" +
|
|
81
|
+
"STEP 1: Experiential visual inspection — drive the fixed artifact as a user would, one browser step at a time, and verify the reported bug is actually fixed.\n" +
|
|
82
|
+
"You have a see-act driver: " + env.crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
|
|
83
|
+
"SESSION PROTOCOL (multi-step flows): one-shot invocations launch a fresh browser each time, so they cannot drive flows where a later step needs an earlier step's living page state (open a dialog, then confirm it — the second step needs the first step's page state). For those, hold ONE browser alive: node " + env.crewHome + "/current/lib/see-act.js session-start --name qa --url http://localhost:<N>/ --log " + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ooda-log.jsonl — launches Chromium once in a detached daemon (its control server binds 127.0.0.1 only; the daemon exits after 10 min with no act). The --log path is the SAME ooda-log.jsonl the QA guard reads: every session-act appends its own JSON line in the exact step schema (step, attempt, action, args, exit, screenshot, observation), so you do NOT call append-ooda-step.js for session acts — READ each act's JSON result (the screenshot field is the path to READ with your read tool) and decide the next action. Then drive it: node " + env.crewHome + "/current/lib/see-act.js session-act --name qa --attempt \"1\" <aria|shot|click|scroll|type> [args] — one action inside the living page. Prefix EVERY session-act invocation with SEE_ACT_ARCHIVE_DIR=" + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one). When done: node " + env.crewHome + "/current/lib/see-act.js session-end --name qa — closes the browser, prints the session summary, and keeps the session dir as evidence. If session-start exits 3 with a \"NOT POSSIBLE: <reason>\" string, the environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
|
|
84
|
+
"Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
|
|
85
|
+
"a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + env.publishSlug + "/client/dist && test -f ~/workspace/ts-spaces/" + env.publishSlug + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + env.publishSlug + " and continue with the mechanical checks.\n" +
|
|
86
|
+
"b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + env.crewHome + " setsid nohup node " + env.crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + env.publishSlug + " --port 0 --tag " + env.taskId + "-qa > /tmp/qa-server-" + env.taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + env.taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + env.taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + env.taskId + ".log) and continue with the mechanical checks.\n" +
|
|
87
|
+
"c. Bounded see-act loop, at most 8 steps: re-run the reproduction steps for the reported bug — does it still occur? Then check the surrounding views for regressions. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
|
|
88
|
+
"c2. After EVERY one-shot see-act invocation, append it to the OODA report: node " + env.crewHome + "/current/lib/append-ooda-step.js --log " + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). (Session acts are self-logging — session-act already appended its line; for those, READ the result instead of re-logging.) Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
|
|
89
|
+
"c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + env.crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + env.crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
|
|
90
|
+
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ node " + env.crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ node " + env.crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
91
|
+
"d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the dialog, then confirm it, then judge the result. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and continue with the mechanical checks.\n" +
|
|
92
|
+
"e. Judge as a user against the task description: is the reported bug fixed AND is nothing else visibly broken? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
|
|
93
|
+
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + env.taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
94
|
+
"Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
|
|
95
|
+
"MECHANICAL CHECKS:\n" +
|
|
96
|
+
"DOCS GATE: If the fix is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n" +
|
|
97
|
+
"To test, run in shell and read the stdout JSON:\n" + crewCmdString(env, "get-state", {}) + "\nThis returns current sessions, events, and tasks.\n" +
|
|
98
|
+
"Verify the fix against the task description in the returned data: the state the bug corrupted should now read correctly, and the task's own expected behavior should hold.\n" +
|
|
99
|
+
"You can also check specific data with: node " + env.crewApi + " --crew-home " + env.crewHome + " get-events --json '{\"task_id\":\"<the task id>\"}'.\n" +
|
|
100
|
+
npmPublishCheck +
|
|
101
|
+
"File follow-up tasks by running in shell (EXCEPT content findings — user-visible content/data goes in the CONTENT-FINDINGS block at the end of these instructions, never through create-task):\n" + crewCmdString(env, "create-task", { title: "<short title>", description: "<details>", project: "<project id>", workflow: "bugfix", filed_by: "hazel" }) + "\n(substitute the real values for the placeholders).\n\n" +
|
|
102
|
+
"BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
|
|
103
|
+
"Report your test results as plain prose.\n" +
|
|
104
|
+
"End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails on the visual or the mechanical checks. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + env.crewHome + "/current/lib/write-ooda-verdict.js --dir " + env.crewHome + "/task-evidence/" + env.taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the reported bug, fixed>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL verdict must carry a machine-readable reason: the workflow closeout cross-checks verdict.json against your prose VERDICT line, and an unreasoned or contradictory verdict fails the phase (never routes to rework).";
|
|
105
|
+
} else if (ctx.qaTerminal) {
|
|
106
|
+
// See docs/decisions/qa-reproduce.md#terminal-qa: Hazel drives CLI transcripts for terminal surfaces.
|
|
107
|
+
var btermEvidence = env.crewHome + "/task-evidence/" + env.taskId + "/postchange";
|
|
108
|
+
var btermTargetsLine = ctx.terminalTargets || "not declared — derive from --help and the task description";
|
|
109
|
+
instructions = "You are code-blind QA. You NEVER read source files.\n" +
|
|
110
|
+
"Public docs (API.md, README) are NOT source code — read them freely, exactly as a user would.\n" +
|
|
111
|
+
"Read the shared UX bar FIRST: " + env.uxDoctrinePath + " — it is the bar the whole crew builds to, and your verdict judges against it point by point.\n\n" +
|
|
112
|
+
"STEP 1: Experiential terminal inspection — drive the fixed CLI as a user would, one command at a time, and verify the reported bug is actually fixed.\n" +
|
|
113
|
+
"a. The CLI under test lives in " + env.repoPath + " (the merged change is on the integration target there). Start like a new user: run --help. You may RUN the CLI; you may never READ its source.\n" +
|
|
114
|
+
"b. Terminal targets for this task: " + btermTargetsLine + ".\n" +
|
|
115
|
+
"c. Bounded terminal loop — at most 8 commands. For each target: run it RIGHT (the happy path — the reported bug's scenario first), then run it WRONG on purpose (bad flags, missing args, nonexistent files, empty input, contradictory flags). Error quality is half the grade: every failure must exit non-zero, say what went wrong in plain language, and tell the user the fix. A raw stack trace shown to a user is a defect — file it as one.\n" +
|
|
116
|
+
"d. Evidence: capture EVERY invocation as a transcript. Run: mkdir -p " + btermEvidence + "\n" +
|
|
117
|
+
" For each command: <cmd> > " + btermEvidence + "/<nn>-<short-slug>.txt 2>&1; echo \"exit=$?\" >> " + btermEvidence + "/<nn>-<short-slug>.txt (number them 01, 02, ...). Then READ the transcript before judging it — an unread transcript is not evidence.\n" +
|
|
118
|
+
"e. Log each step to the OODA log — run in shell, one command per step:\n" +
|
|
119
|
+
" node " + env.crewHome + "/current/lib/append-ooda-step.js --log " + btermEvidence + "/ooda-log.jsonl --attempt \"1\" --step <N> --action terminal --exit <code> --transcript " + btermEvidence + "/<nn>-<short-slug>.txt --args '{\"cmd\":\"<the exact command>\"}' --observation \"<1-2 sentences: what the output said and what you concluded>\"\n" +
|
|
120
|
+
" Steps are strictly monotonic within an attempt (1, 2, 3, ...). A rerun is a NEW attempt (\"2\", \"3\", ...) — never overwrite attempt 1. If the CLI will not run at all, log the step with --exit 3 and NOT POSSIBLE in the observation — never fabricate a transcript.\n" +
|
|
121
|
+
"f. Compare against the pre-change baseline transcripts in " + env.crewHome + "/task-evidence/" + env.taskId + "/baseline/ — every finding cites its baseline and post-change transcripts by step number.\n" +
|
|
122
|
+
"Then continue with the mechanical checks below. Your VERDICT covers both the experiential and the mechanical checks.\n\n" +
|
|
123
|
+
"STEP 2: Verify data integrity via the crew API.\n" +
|
|
124
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmdString(env, "get-state", { events_limit: 1 }) + "\n" +
|
|
125
|
+
"Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
|
|
126
|
+
"DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale for a public-affecting change, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n\n" +
|
|
127
|
+
"STEP 3: File follow-up tasks for any related issues you discover.\n" +
|
|
128
|
+
"For each issue (EXCEPT content findings — user-visible content/data goes in the CONTENT-FINDINGS block at the end of these instructions, never through create-task), run in shell:\n" +
|
|
129
|
+
"node " + env.crewApi + " --crew-home " + env.crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + env.projectId + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
130
|
+
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
131
|
+
"BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL. When the baseline is terminal transcripts, confirm every terminal target you judged has a baseline transcript: a target with no pre-change transcript is an evidence gap — name it in --missing, never invent the baseline.\n\n" +
|
|
132
|
+
"Report your test results as plain prose. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. End your report with exactly one line: VERDICT: PASS if testing passes, VERDICT: FAIL if it fails on the terminal or the mechanical checks. First ensure the OODA log exists even if you logged zero steps (touch " + btermEvidence + "/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + env.crewHome + "/current/lib/write-ooda-verdict.js --dir " + btermEvidence + " --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the reported bug, fixed>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL verdict must carry a machine-readable reason: the workflow closeout cross-checks verdict.json against your prose VERDICT line, and an unreasoned or contradictory verdict fails the phase (never routes to rework).";
|
|
133
|
+
}
|
|
134
|
+
if (env.publishType === "artifact") {
|
|
135
|
+
instructions = "PROVENANCE CHECK (this project publishes to a dashboard artifact).\n" +
|
|
136
|
+
"Provenance is crew-owned state: read it from the Crew API, never from the artifact's own getprovenance action (a different, non-authoritative store).\n" +
|
|
137
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmdString(env, "get-provenance", { project_id: env.projectId }) + "\n" +
|
|
138
|
+
"If provenance is null, report 'provenance missing — publish did not stamp source/crew release', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
139
|
+
"Run: cd " + env.repoPath + " && git rev-parse HEAD — call this LIVE_HEAD.\n" +
|
|
140
|
+
"Run: test -d " + env.crewHome + "/releases/<provenance.crew_release> (substitute the real stamped hash; do not run the literal placeholder). If the directory does not exist, report 'provenance mismatch: crew_release [value from get-provenance] not found in release registry', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
141
|
+
"If provenance.source_commit equals LIVE_HEAD, the source check passes.\n" +
|
|
142
|
+
"Otherwise the check is NOT failed yet: the parent stamps provenance AFTER post-deploy (docs/publish-verification.md), and post-deploy may commit an artifact-builder staging commit (\"rebuild: <task_id>\"), so LIVE_HEAD may sit ahead of the stamped commit ONLY IF every commit in between is such a rebuild marker. Verify exactly:\n" +
|
|
143
|
+
"1. Run: cd " + env.repoPath + " && git merge-base --is-ancestor <provenance.source_commit> LIVE_HEAD && echo ANCESTOR_OK (substitute the real stamped hash and LIVE_HEAD; do not run the literal placeholders). If this command fails, report 'provenance mismatch: stamped source_commit is not an ancestor of live HEAD', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
144
|
+
"2. Run: cd " + env.repoPath + " && git log --format=%s <provenance.source_commit>..LIVE_HEAD (substitute real values). Every subject line MUST start with \"rebuild: \". If any line does not, report 'provenance mismatch: live HEAD moved past the stamped commit with non-rebuild source commits: [paste the offending subject lines]', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
|
|
145
|
+
instructions;
|
|
146
|
+
}
|
|
147
|
+
// Room #26 blocker 38 (2026-09-21): content-finding attribution
|
|
148
|
+
// protocol. Hazel reports user-visible content/data observations in a
|
|
149
|
+
// machine-readable CONTENT-FINDINGS block; the workflow classifies each
|
|
150
|
+
// against the task's publish diff and files follow-ups itself — Hazel
|
|
151
|
+
// never files bugfixes for content findings directly, and
|
|
152
|
+
// environment-attributable content never fails her verdict alone.
|
|
153
|
+
var contentFindingsProtocol =
|
|
154
|
+
"CONTENT FINDINGS (machine-read — room #26 blocker 38): if you observe user-visible content or data that looks wrong, stale, or out of place (decks, cards, rows, text, files in the artifact — anything this task did not obviously produce), report it here — do NOT file a follow-up task for it yourself and do NOT fail your verdict for it alone. Emit exactly one line in this shape, immediately BEFORE your final VERDICT line (the verdict stays the last line of your report):\n" +
|
|
155
|
+
"CONTENT-FINDINGS: [{\"subject\": \"<repo-relative file path, or the literal live-data for content in the artifact's runtime data stores>\", \"observation\": \"<what you saw, one line>\"}, ...]\n" +
|
|
156
|
+
"Use subject \"live-data\" for anything you saw in the running artifact (UI content, database rows, uploaded files) — you are code-blind and cannot name its file. Use a repo-relative file path only when you know the content lives in a specific file (e.g. from the task description or public docs). When you saw no such content, emit the empty array: CONTENT-FINDINGS: [].\n" +
|
|
157
|
+
"The workflow classifies each finding against the task's publish diff: a finding whose subject file is in the diff is attributable to this task's change and the workflow files a bugfix for it; anything else is recorded with its attribution (environment-attributable, or unknown when the diff is unavailable) — the finding itself never spawns an artifact bugfix. No finding is ever dropped for looking like test residue: it is classified by attribution and recorded with it. Your VERDICT judges this task's change; content you cannot attribute to it is not a failure of this task.\n";
|
|
158
|
+
// The content-findings block is machine-read at closeout (blocker 38).
|
|
159
|
+
// Hazel emits it immediately before the VERDICT line, so the verdict
|
|
160
|
+
// keeps its trailing-window contract with extractVerdict.
|
|
161
|
+
instructions += "\n" + contentFindingsProtocol;
|
|
162
|
+
return instructions;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
export async function runPhase(ctx) {
|
|
166
|
+
var env = ctx.env, state = ctx.state;
|
|
167
|
+
|
|
168
|
+
var claimed = await ensureClaimed(env, state, PHASE);
|
|
169
|
+
if (claimed.type !== "CLAIMED") return claimed;
|
|
170
|
+
|
|
171
|
+
// Experiential dispatch (source: qaExp = resolveExperiential() === "yes").
|
|
172
|
+
var qaExp = (state.experiential === true);
|
|
173
|
+
ctx.qaArtifact = qaExp && env.surfaceArtifact;
|
|
174
|
+
ctx.qaTerminal = qaExp && env.surfaceTerminal;
|
|
175
|
+
ctx.qaBundleHash = null; // set below when versioned (source parity)
|
|
176
|
+
|
|
177
|
+
// Versioned-artifact bundle hash for findings classification (source:
|
|
178
|
+
// qaBundleHash from the Integrate QA deploy, when versioned_build).
|
|
179
|
+
if (ctx.qaArtifact && env.versionedBuild === true) {
|
|
180
|
+
try {
|
|
181
|
+
var integNotes = await latestSessionNotes(env, "Integrate", "completed");
|
|
182
|
+
var dm = /^deploy: ok ([0-9a-f]{7})$/m.exec(integNotes || "");
|
|
183
|
+
ctx.qaBundleHash = dm ? dm[1] : null;
|
|
184
|
+
} catch (e) { ctx.qaBundleHash = null; }
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
// Build's release decision (npm publish backstop).
|
|
188
|
+
var buildNotes = await latestSessionNotes(env, "Build", "completed");
|
|
189
|
+
ctx.releaseDecision = extractReleaseDecision(buildNotes);
|
|
190
|
+
|
|
191
|
+
// Triage's terminal_targets (terminal QA).
|
|
192
|
+
var triageNotes = await latestSessionNotes(env, "Triage", "completed");
|
|
193
|
+
var ttm = /^terminal_targets:\s*(.+)/im.exec(triageNotes || "");
|
|
194
|
+
ctx.terminalTargets = ttm ? ttm[1].trim().slice(0, 300) : "";
|
|
195
|
+
|
|
196
|
+
ctx.taskDesc = env.taskDescription;
|
|
197
|
+
// phaseCtx and hazelPhaseCtx: the source builds these from prior events.
|
|
198
|
+
// The worker layer passes the event preamble (which the boundary injects);
|
|
199
|
+
// the instruction text references ctx.phaseCtx/ctx.hazelPhaseCtx for the
|
|
200
|
+
// surface-specific context blocks.
|
|
201
|
+
ctx.phaseCtx = "";
|
|
202
|
+
ctx.hazelPhaseCtx = "";
|
|
203
|
+
|
|
204
|
+
var boundary = await runWorkBoundary(env, state, {
|
|
205
|
+
phase: PHASE.name,
|
|
206
|
+
identity: PHASE.identity,
|
|
207
|
+
instructions: buildInstructions(ctx),
|
|
208
|
+
eventPreamble: buildEventPreamble(env, PHASE.name),
|
|
209
|
+
crewApiLine: true,
|
|
210
|
+
verdictStep: true,
|
|
211
|
+
});
|
|
212
|
+
if (boundary.type !== "BOUNDARY_DONE") return boundary;
|
|
213
|
+
var workerText = boundary.workerText;
|
|
214
|
+
var verdictPassed = boundary.verdictPassed === true;
|
|
215
|
+
|
|
216
|
+
// QA verdict.json closeout gate (experiential paths only): the prose
|
|
217
|
+
// VERDICT: line is the routing signal; verdict.json is cross-checked
|
|
218
|
+
// before any rework routing.
|
|
219
|
+
if ((ctx.qaArtifact || ctx.qaTerminal) && boundary.verdictPassed !== null) {
|
|
220
|
+
var qaVerdictDir = env.taskEvidenceDir + "/postchange";
|
|
221
|
+
var qaVerdictExpect = verdictPassed ? "PASS" : "FAIL";
|
|
222
|
+
var qaVerdictOut = "";
|
|
223
|
+
try {
|
|
224
|
+
var qv = await runCmd(["node", env.crewHome + "/current/lib/read-ooda-verdict.js",
|
|
225
|
+
"--dir", qaVerdictDir, "--expect", qaVerdictExpect]);
|
|
226
|
+
qaVerdictOut = String(qv.stdout || "").trim();
|
|
227
|
+
} catch (e) {
|
|
228
|
+
qaVerdictOut = "";
|
|
229
|
+
}
|
|
230
|
+
var qaVerdictGate = null;
|
|
231
|
+
try {
|
|
232
|
+
var qaVerdictLines = qaVerdictOut.split("\n");
|
|
233
|
+
qaVerdictGate = JSON.parse(qaVerdictLines[qaVerdictLines.length - 1]);
|
|
234
|
+
} catch (e) {
|
|
235
|
+
qaVerdictGate = null;
|
|
236
|
+
}
|
|
237
|
+
if (!qaVerdictGate || qaVerdictGate.ok !== true) {
|
|
238
|
+
var qaGateCode = (qaVerdictGate && qaVerdictGate.code) ? qaVerdictGate.code : "unreadable";
|
|
239
|
+
var qaGateDetail = (qaVerdictGate && qaVerdictGate.error) ? qaVerdictGate.error : (qaVerdictOut ? qaVerdictOut.slice(0, 200) : "cross-checker produced no usable output");
|
|
240
|
+
log("QA verdict.json closeout gate failed (" + qaGateCode + "): " + qaGateDetail + " — marking QA failed for retry, NOT routing to rework");
|
|
241
|
+
await recordPhase(env, {
|
|
242
|
+
task_id: env.taskId,
|
|
243
|
+
session: {
|
|
244
|
+
id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity,
|
|
245
|
+
step: PHASE.name, status: "failed",
|
|
246
|
+
notes: "QA verdict.json closeout gate failed (" + qaGateCode + "): " + qaGateDetail + ". The prose VERDICT line said " + qaVerdictExpect + " but verdict.json is missing, corrupt, contradictory, or (for FAIL) carries no machine-readable reason. Unreasoned or contradictory verdicts never route to rework; phase failed for retry",
|
|
247
|
+
},
|
|
248
|
+
event: {
|
|
249
|
+
task_id: env.taskId, type: "failed", identity: PHASE.identity,
|
|
250
|
+
message: "QA verdict.json closeout gate failed (" + qaGateCode + ") — verdict unreasoned or contradictory, phase failed, dispatcher will retry QA",
|
|
251
|
+
},
|
|
252
|
+
});
|
|
253
|
+
return { type: "FAILED" };
|
|
254
|
+
}
|
|
255
|
+
log("QA verdict.json closeout gate passed: verdict.json agrees with prose VERDICT: " + qaVerdictExpect);
|
|
256
|
+
|
|
257
|
+
// Experiential loop guard: a PASS with missing experiential evidence
|
|
258
|
+
// parks fail-closed — never done.
|
|
259
|
+
var qaLoopSurface = env.surfaceTerminal ? "terminal" : "visual";
|
|
260
|
+
var qaLoopReason = env.surfaceTerminal ? "qa-terminal-loop-unavailable" : "qa-visual-loop-unavailable";
|
|
261
|
+
var qaLoopUnavailable = env.surfaceTerminal ? qaVerdictGate.terminal_loop_unavailable : qaVerdictGate.visual_loop_unavailable;
|
|
262
|
+
if (verdictPassed === true && qaLoopUnavailable === true) {
|
|
263
|
+
log("QA " + qaLoopSurface + " loop unavailable — parking fail-closed (unattributable_reason=" + qaLoopReason + "), never done");
|
|
264
|
+
return await parkTask(env, "QA " + qaLoopSurface + " loop unavailable (unattributable_reason=" + qaLoopReason + "): " +
|
|
265
|
+
(env.surfaceTerminal
|
|
266
|
+
? "the terminal loop could not run — the CLI would not execute, and verdict.json records missing terminal evidence. A PASS without experiential evidence is never terminal. Human attention needed: check the project's runtime dependencies at " + env.repoPath + ", then re-queue QA."
|
|
267
|
+
: "the see-act browser loop could not run — verdict.json records missing visual evidence. A PASS without experiential evidence is never terminal. Human attention needed: repair the crew home's dependency symlink ($CREW_HOME/node_modules) or the npm install, then re-queue QA."));
|
|
268
|
+
}
|
|
269
|
+
log("QA " + qaLoopSurface + "-loop guard passed: experiential evidence present");
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
// Content-findings attribution (blocker 38): Hazel's CONTENT-FINDINGS
|
|
273
|
+
// block is machine-read at closeout; each finding is classified against
|
|
274
|
+
// the task's publish diff and filed/recorded by the workflow.
|
|
275
|
+
var contentFindings = extractContentFindings(workerText);
|
|
276
|
+
if (!contentFindings.ok) {
|
|
277
|
+
log("QA CONTENT-FINDINGS unreadable (" + contentFindings.reason + ") — marking failed for retry");
|
|
278
|
+
await recordPhase(env, {
|
|
279
|
+
task_id: env.taskId,
|
|
280
|
+
session: {
|
|
281
|
+
id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity,
|
|
282
|
+
step: PHASE.name, status: "failed",
|
|
283
|
+
notes: "QA report had no machine-readable CONTENT-FINDINGS block (" + contentFindings.reason + "); phase failed for retry",
|
|
284
|
+
},
|
|
285
|
+
event: {
|
|
286
|
+
task_id: env.taskId, type: "failed", identity: PHASE.identity,
|
|
287
|
+
message: "QA CONTENT-FINDINGS block unreadable (" + contentFindings.reason + ") — phase failed, dispatcher will retry",
|
|
288
|
+
},
|
|
289
|
+
});
|
|
290
|
+
return { type: "FAILED" };
|
|
291
|
+
}
|
|
292
|
+
// The publish-diff file set, persisted at Publish. Unreadable →
|
|
293
|
+
// attribution unknown (never environment-attributable, never a bugfix).
|
|
294
|
+
var qaDiffFiles = [];
|
|
295
|
+
var qaDiffFilesOk = false;
|
|
296
|
+
try {
|
|
297
|
+
var qaDiffFilesRaw = readFileSync(env.crewHome + "/.publish-diffs/" + env.taskId + ".files.json", "utf8").trim();
|
|
298
|
+
if (qaDiffFilesRaw && qaDiffFilesRaw !== "DIFF_FILES_MISSING") {
|
|
299
|
+
var qaDiffParsed = JSON.parse(qaDiffFilesRaw);
|
|
300
|
+
if (Array.isArray(qaDiffParsed)) {
|
|
301
|
+
qaDiffFiles = qaDiffParsed;
|
|
302
|
+
qaDiffFilesOk = true;
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
} catch (e) {
|
|
306
|
+
log("QA diff file list unreadable: " + String((e && e.message) || e).slice(0, 200));
|
|
307
|
+
}
|
|
308
|
+
if (!qaDiffFilesOk) {
|
|
309
|
+
log("QA publish diff file list unavailable — content findings record attribution unknown");
|
|
310
|
+
}
|
|
311
|
+
for (var cfi = 0; cfi < contentFindings.findings.length; cfi++) {
|
|
312
|
+
var cf = contentFindings.findings[cfi];
|
|
313
|
+
var cfc = qaDiffFilesOk ? classifyContentFinding(cf, qaDiffFiles)
|
|
314
|
+
: { attribution: "unknown", reason: "publish diff file set unavailable — cannot attribute" };
|
|
315
|
+
if (cfc.attribution === "task-change") {
|
|
316
|
+
log("QA content finding attributable to task change — filing bugfix: " + cf.subject + " :: " + String(cf.observation).slice(0, 120));
|
|
317
|
+
await crewApi(env, "create-task", {
|
|
318
|
+
title: "QA content finding: " + String(cf.observation).slice(0, 120),
|
|
319
|
+
description: "Content finding from QA on task " + env.taskId + " (subject: " + cf.subject + ", classified task-change: " + cfc.reason + "). Observation: " + cf.observation,
|
|
320
|
+
project: env.projectId, workflow: "bugfix", filed_by: "hazel",
|
|
321
|
+
});
|
|
322
|
+
} else {
|
|
323
|
+
log("QA content finding " + cfc.attribution + " — recording, no bugfix: " + cf.subject + " :: " + String(cf.observation).slice(0, 120));
|
|
324
|
+
await logEvent(env, {
|
|
325
|
+
task_id: env.taskId, type: "note", identity: PHASE.identity,
|
|
326
|
+
message: "content-finding: " + cfc.attribution + " — subject: " + cf.subject + " — " + cf.observation + " (" + cfc.reason + ")",
|
|
327
|
+
});
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
// Closeout.
|
|
332
|
+
var status = verdictPassed ? "completed" : "rejected";
|
|
333
|
+
await recordPhase(env, {
|
|
334
|
+
task_id: env.taskId,
|
|
335
|
+
session: {
|
|
336
|
+
id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity,
|
|
337
|
+
step: PHASE.name, status: status, notes: summarizeReport(workerText),
|
|
338
|
+
},
|
|
339
|
+
event: {
|
|
340
|
+
task_id: env.taskId, type: status, identity: PHASE.identity,
|
|
341
|
+
message: "QA " + status + " by " + PHASE.identity,
|
|
342
|
+
},
|
|
343
|
+
});
|
|
344
|
+
|
|
345
|
+
if (!verdictPassed) {
|
|
346
|
+
// Shared Review/QA rework budget (spec.reworkPhases).
|
|
347
|
+
if (state.reworkCount >= MAX_REWORK) {
|
|
348
|
+
log("Shared rework budget exhausted for task " + env.taskId + " — worktree preserved at " + env.worktreePreservedHint + " for manual inspection");
|
|
349
|
+
return await parkTask(env, "Exceeded shared rework budget (" + MAX_REWORK + " total rework attempts across Review and QA) after QA rejection. Worktree preserved.");
|
|
350
|
+
}
|
|
351
|
+
log("QA rejected — bouncing to Build (rework #" + (state.reworkCount + 1) + " of " + MAX_REWORK + ")");
|
|
352
|
+
return { type: "REWIND", next: "Build" };
|
|
353
|
+
}
|
|
354
|
+
// QA PASS is terminal: the driver closeout marks the task done.
|
|
355
|
+
return { type: "ADVANCE", next: null };
|
|
356
|
+
}
|