muse-crew 0.17.2 → 0.17.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +11 -0
- package/docs/decisions/composition-machinery.md +180 -0
- package/docs/decisions/publish-path.md +6 -0
- package/docs/decisions/workflow-core.md +5 -4
- package/lib/AGENTS.md +2 -1
- package/lib/bugfix/phases/build.js +168 -0
- package/lib/bugfix/phases/capture.js +170 -0
- package/lib/bugfix/phases/integrate.js +165 -0
- package/lib/bugfix/phases/map.js +129 -0
- package/lib/bugfix/phases/publish.js +592 -0
- package/lib/bugfix/phases/qa.js +356 -0
- package/lib/bugfix/phases/reproduce.js +254 -0
- package/lib/bugfix/phases/review.js +292 -0
- package/lib/bugfix/phases/triage.js +69 -0
- package/lib/chore/CONTRACT.md +181 -0
- package/lib/chore/DISPOSITION.md +98 -0
- package/lib/chore/extract.js +204 -0
- package/lib/chore/phase-lib.js +885 -0
- package/lib/chore/phases/build.js +110 -0
- package/lib/chore/phases/capture.js +82 -0
- package/lib/chore/phases/integrate.js +109 -0
- package/lib/chore/phases/map.js +81 -0
- package/lib/chore/phases/publish.js +543 -0
- package/lib/chore/phases/review.js +255 -0
- package/lib/chore/phases/triage.js +57 -0
- package/lib/chore/prompts/evidence-gatherer.js +41 -0
- package/lib/chore/prompts/evidence-gatherer.schema.json +1 -0
- package/lib/chore/prompts/tool-check.js +15 -0
- package/lib/chore/prompts/trailers.js +56 -0
- package/lib/chore/prompts/verdict-reask.js +28 -0
- package/lib/chore/prompts/verdict-reask.schema.json +1 -0
- package/lib/chore/prompts/work-agent.js +52 -0
- package/lib/chore/prompts/work-agent.schema.json +1 -0
- package/lib/chore/spawn-keys.js +44 -0
- package/lib/chore/spawn-vocab.js +87 -0
- package/lib/chore-run.js +538 -0
- package/lib/chore-tick.js +289 -0
- package/lib/crew-api.js +273 -0
- package/lib/crew-dispatch-worker.js +27 -7
- package/lib/crew-release.sh +7 -2
- package/lib/extract.js +252 -0
- package/lib/prompts/tool-check.js +18 -0
- package/lib/prompts/trailers.js +59 -0
- package/lib/prompts/verdict-reask.js +31 -0
- package/lib/prompts/verdict-reask.schema.json +1 -0
- package/lib/prompts/work-agent.js +56 -0
- package/lib/prompts/work-agent.schema.json +1 -0
- package/lib/reap-spawns.js +407 -0
- package/lib/schema.sql +12 -1
- package/lib/spawn-keys.js +47 -0
- package/lib/spawn-step.js +572 -0
- package/lib/standard/phases/build.js +120 -0
- package/lib/standard/phases/capture.js +163 -0
- package/lib/standard/phases/integrate.js +172 -0
- package/lib/standard/phases/map.js +119 -0
- package/lib/standard/phases/publish.js +565 -0
- package/lib/standard/phases/qa.js +399 -0
- package/lib/standard/phases/review.js +281 -0
- package/lib/standard/phases/triage.js +64 -0
- package/lib/test-detached-integrate.sh +47 -0
- package/lib/workflow-driver.js +605 -0
- package/lib/workflow-lib.js +1012 -0
- package/lib/workflow-spec.js +187 -0
- package/lib/worktree-lifecycle.sh +55 -3
- package/package.json +1 -1
- package/seed/cron-body-template.md +61 -9
- package/workflows/bugfix.js +17 -17
- package/workflows/chore.js +16 -16
- package/workflows/docs.js +14 -11
- package/workflows/standard.js +16 -16
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
// lib/standard/phases/qa.js — QA (hazel) phase module (sandbox exit,
|
|
2
|
+
// Phase C Piece 1). Import-safe: no side effects on import, bare `node`
|
|
3
|
+
// exits 0.
|
|
4
|
+
//
|
|
5
|
+
// Reads: claim/session context, the ensure-deployed bundle hash (0.16.0 QA
|
|
6
|
+
// deploy pipeline), the release decision from durable Build notes (npm
|
|
7
|
+
// backstop), the publish-diff file list persisted by Publish, the OODA
|
|
8
|
+
// verdict record, the terminal_targets declaration from Triage notes.
|
|
9
|
+
// Writes: completed/rejected/failed session + event, bugfix tasks for
|
|
10
|
+
// task-attributable content findings (filed_by hazel), note events for
|
|
11
|
+
// environment/unknown-attribution findings.
|
|
12
|
+
// Transitions: ADVANCE (terminal, next null) on pass. Verdict FAIL records
|
|
13
|
+
// "rejected" and rewinds to Build against the SHARED Review/QA rework
|
|
14
|
+
// budget (MAX_TOTAL_REWORK = 2, derived durably from rejected events).
|
|
15
|
+
// FAILED on operational failure (unreadable verdict, hash mismatch,
|
|
16
|
+
// unreadable CONTENT-FINDINGS). PARK on missing experiential evidence
|
|
17
|
+
// (a PASS without experiential evidence is never terminal).
|
|
18
|
+
//
|
|
19
|
+
// 0.16.0 note: the numeric qa-bundle-stale / BUILD_NUMBER gate was
|
|
20
|
+
// deliberately removed from the source (commit b7319ca). Freshness is now
|
|
21
|
+
// constructive: Integrate rebuilds the QA bundle from merged source via
|
|
22
|
+
// qa-deploy.mjs, QA ensure-deploys, and the closeout verifies the bundle
|
|
23
|
+
// hash. No numbers, no comparison.
|
|
24
|
+
|
|
25
|
+
import {
|
|
26
|
+
runWorkBoundary, recordPhase, buildEventPreamble, summarizeReport,
|
|
27
|
+
closeoutPassed, ensureClaimed, latestSessionNotes, crewApi, runCmd,
|
|
28
|
+
resolveExperiential, parseDeployResult, parkTask, crewCmdString, log,
|
|
29
|
+
} from "../../workflow-lib.js";
|
|
30
|
+
import {
|
|
31
|
+
extractContentFindings, classifyContentFinding, extractReleaseDecision,
|
|
32
|
+
extractMarkerLines,
|
|
33
|
+
} from "../../extract.js";
|
|
34
|
+
import { join } from "path";
|
|
35
|
+
import { readFileSync, readdirSync, statSync } from "fs";
|
|
36
|
+
|
|
37
|
+
export const PHASE = { name: "QA", identity: "hazel" };
|
|
38
|
+
const MAX_TOTAL_REWORK = 2;
|
|
39
|
+
|
|
40
|
+
// buildInstructions — pure function of ctx. The QA facts are stashed in
|
|
41
|
+
// ctx.state.memo.qaFacts by runPhase before the call (keeps the signature
|
|
42
|
+
// pure). Instruction text verbatim from workflows/standard.js (QA branch:
|
|
43
|
+
// artifact / terminal / generic surfaces, npm publish backstop, and the
|
|
44
|
+
// machine-readable CONTENT-FINDINGS protocol).
|
|
45
|
+
export function buildInstructions(ctx) {
|
|
46
|
+
var env = ctx.env;
|
|
47
|
+
var facts = (ctx.state.memo && ctx.state.memo.qaFacts) || {};
|
|
48
|
+
var taskId = env.taskId;
|
|
49
|
+
var crewHome = env.crewHome;
|
|
50
|
+
var crewApiCmd = env.crewApiPinned;
|
|
51
|
+
|
|
52
|
+
var contentFindingsProtocol =
|
|
53
|
+
"CONTENT FINDINGS (machine-read — room #26 blocker 38): if you observe user-visible content or data that looks wrong, stale, or out of place (decks, cards, rows, text, files in the artifact — anything this task did not obviously produce), report it here — do NOT file a follow-up task for it yourself and do NOT fail your verdict for it alone. Emit exactly one line in this shape, immediately BEFORE your final VERDICT line (the verdict stays the last line of your report):\n" +
|
|
54
|
+
"CONTENT-FINDINGS: [{\"subject\": \"<repo-relative file path, or the literal live-data for content in the artifact's runtime data stores>\", \"observation\": \"<what you saw, one line>\"}, ...]\n" +
|
|
55
|
+
"Use subject \"live-data\" for anything you saw in the running artifact (UI content, database rows, uploaded files) — you are code-blind and cannot name its file. Use a repo-relative file path only when you know the content lives in a specific file (e.g. from the task description or public docs). When you saw no such content, emit the empty array: CONTENT-FINDINGS: [].\n" +
|
|
56
|
+
"The workflow classifies each finding against the task's publish diff: a finding whose subject file is in the diff is attributable to this task's change and the workflow files a bugfix for it; anything else is recorded with its attribution (environment-attributable, or unknown when the diff is unavailable) — the finding itself never spawns an artifact bugfix. No finding is ever dropped for looking like test residue: it is classified by attribution and recorded with it. Your VERDICT judges this task's change; content you cannot attribute to it is not a failure of this task.\n";
|
|
57
|
+
|
|
58
|
+
var npmPublishCheck = (env.publishType === "npm" && facts.releaseDecision && facts.releaseDecision.release === "yes")
|
|
59
|
+
? "NPM PUBLISH CHECK: the accepted Build report declared release: yes, so this run's Publish phase must have published — UNLESS it was skipped deterministically on an empty-diff Integrate. Find this task's recorded Publish result: run in shell and return the stdout verbatim:\n" + crewCmdString(env, "get-state", { events_limit: 1 }) + "\n, then find the session for this task_id with step \"Publish\" (status completed) in the returned sessions array and read its session notes (the Publish agent's summary — event history does NOT carry it).\n" +
|
|
60
|
+
"CHECK THE SKIP PATH FIRST: if the notes contain a line matching skipped: no-lock-held, Publish was skipped deterministically — Integrate reported MERGED_EMPTY (no commits ahead of the integration target), so no merge lock was taken and there was nothing to ship. Verify the notes contain that skip-marker line and do NOT contain a line matching published: muse-crew@. Do NOT run npm view and do NOT demand registry movement — nothing was supposed to ship. Report 'npm publish check: Publish skipped deterministically (empty-diff Integrate — nothing to ship)' and PASS this check.\n" +
|
|
61
|
+
"Only when the notes contain no skip marker must the publish have landed — run the full verification below.\n" +
|
|
62
|
+
"Extract the line matching TARGET_VERSION=<new-version> computed as <base> + <scope> → <new-version> (the workflow appends it to the Publish notes, so it is always present). If the line is missing, report 'npm publish verification failed: Publish notes did not carry the computed target version', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
63
|
+
"Verify the bump SCOPE: the <scope> in that line MUST equal the accepted version_bump scope \"" + facts.releaseDecision.version_bump + "\" — if it differs, report the mismatch, then end your report with exactly this line: VERDICT: FAIL. Verify the ARITHMETIC: <base> + <scope> must equal <new-version> (patch increments the last segment only, e.g. 0.3.0 + patch → 0.3.1; minor increments the middle and resets the last to 0; major increments the first and resets the rest to 0) — if the math is wrong, report it, then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
64
|
+
"Extract the published version from the notes line matching published: muse-crew@<version>. It MUST equal <new-version> from the TARGET_VERSION line. Then run: npm view muse-crew version. The registry version MUST equal <new-version>. If any of these checks fails, report 'npm publish verification failed: [details]', then end your report with exactly this line: VERDICT: FAIL.\n"
|
|
65
|
+
: "";
|
|
66
|
+
|
|
67
|
+
var instructions;
|
|
68
|
+
if (env.surfaceArtifact) {
|
|
69
|
+
instructions =
|
|
70
|
+
"Read the shared UX bar FIRST: " + env.uxDoctrinePath + " — it is the bar the whole crew builds to, and your verdict judges against it point by point.\n\n" +
|
|
71
|
+
"You are code-blind QA. You NEVER read source files.\n" +
|
|
72
|
+
"Public docs (API.md, README, published action schemas) are NOT source code — read them freely, exactly as a user would.\n\n" +
|
|
73
|
+
"STEP 1: Experiential visual inspection — drive the artifact as a user would, one browser step at a time.\n" +
|
|
74
|
+
"You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
|
|
75
|
+
"SESSION PROTOCOL (multi-step flows): one-shot invocations launch a fresh browser each time, so they cannot drive flows where a later step needs an earlier step's living page state (open a deck, then click Study — the second step needs the first step's page state). For those, hold ONE browser alive: node " + crewHome + "/current/lib/see-act.js session-start --name qa --url http://localhost:<N>/ --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — launches Chromium once in a detached daemon (its control server binds 127.0.0.1 only; the daemon exits after 10 min with no act). The --log path is the SAME ooda-log.jsonl the QA guard reads: every session-act appends its own JSON line in the exact step schema (step, attempt, action, args, exit, screenshot, observation), so you do NOT call append-ooda-step.js for session acts — READ each act's JSON result (the screenshot field is the path to READ with your read tool) and decide the next action. Then drive it: node " + crewHome + "/current/lib/see-act.js session-act --name qa --attempt \"1\" <aria|shot|click|scroll|type> [args] — one action inside the living page. Prefix EVERY session-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one). When done: node " + crewHome + "/current/lib/see-act.js session-end --name qa — closes the browser, prints the session summary, and keeps the session dir as evidence. If session-start exits 3 with a \"NOT POSSIBLE: <reason>\" string, the environment cannot drive a browser: report NOT POSSIBLE: <reason> and judge what you can.\n" +
|
|
76
|
+
"Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
|
|
77
|
+
"a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + env.publishSlug + "/client/dist && test -f ~/workspace/ts-spaces/" + env.publishSlug + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + env.publishSlug + " and continue with the mechanical checks.\n" +
|
|
78
|
+
"b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + env.publishSlug + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
|
|
79
|
+
"c. Bounded see-act loop, at most 8 steps: drive the artifact as a user would, one browser step at a time. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
|
|
80
|
+
"c2. After EVERY one-shot see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). (Session acts are self-logging — session-act already appended its line; for those, READ the result instead of re-logging.) Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
|
|
81
|
+
"c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
|
|
82
|
+
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
83
|
+
"d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the deck, then click Study, then judge the study view. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and judge what you did reach.\n" +
|
|
84
|
+
"e. Judge as a user against the task description: does the change render correctly? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
|
|
85
|
+
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
86
|
+
"Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
|
|
87
|
+
"STEP 2: Verify data integrity via the crew API.\n" +
|
|
88
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmdString(env, "get-state", { events_limit: 1 }) + "\n" +
|
|
89
|
+
"Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
|
|
90
|
+
"DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale for a public-affecting change, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n\n" +
|
|
91
|
+
"PROVENANCE CHECK: Run in shell and return the stdout verbatim:\n" + crewCmdString(env, "get-provenance", { project_id: env.projectId }) + "\n" +
|
|
92
|
+
"If provenance is null, report 'provenance missing — publish did not stamp source/crew release', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
93
|
+
"Run: cd " + env.repoPath + " && git rev-parse HEAD — call this LIVE_HEAD.\n" +
|
|
94
|
+
"Run: test -d " + crewHome + "/releases/<provenance.crew_release> (substitute the real stamped hash; do not run the literal placeholder). If the directory does not exist, FAIL: { \"passed\": false, \"summary\": \"provenance mismatch: crew_release [value from get-provenance] not found in release registry\" }.\n" +
|
|
95
|
+
"If provenance.source_commit equals LIVE_HEAD, the source check passes — continue to STEP 3.\n" +
|
|
96
|
+
"Otherwise the check is NOT failed yet: the parent stamps provenance AFTER post-deploy (docs/publish-verification.md), and post-deploy may commit an artifact-builder staging commit (\"rebuild: <task_id>\"), so LIVE_HEAD may sit ahead of the stamped commit ONLY IF every commit in between is such a rebuild marker. Verify exactly:\n" +
|
|
97
|
+
"1. Run: cd " + env.repoPath + " && git merge-base --is-ancestor <provenance.source_commit> LIVE_HEAD && echo ANCESTOR_OK (substitute the real stamped hash and LIVE_HEAD; do not run the literal placeholders). If this command fails, FAIL: { \"passed\": false, \"summary\": \"provenance mismatch: stamped source_commit is not an ancestor of live HEAD\" }.\n" +
|
|
98
|
+
"2. Run: cd " + env.repoPath + " && git log --format=%s <provenance.source_commit>..LIVE_HEAD (substitute real values). Every subject line MUST start with \"rebuild: \". If any line does not, report 'provenance mismatch: live HEAD moved past the stamped commit with non-rebuild source commits: [paste the offending subject lines]', then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
99
|
+
"If both pass, the source check passes — the only drift since the stamp is builder staging output committed by post-deploy. Continue to STEP 3.\n\n" +
|
|
100
|
+
"STEP 3: File follow-up tasks for any related issues you discover — EXCEPT content findings (user-visible content/data): those go in the CONTENT-FINDINGS block at the end of these instructions, never through create-task. The workflow files follow-ups for attributable content itself.\n" +
|
|
101
|
+
"For each issue, run in shell:\n" +
|
|
102
|
+
"node " + crewApiCmd + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + env.projectId + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
103
|
+
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
104
|
+
"BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL.\n\n" +
|
|
105
|
+
"Report back in plain prose — what checks you ran and their results. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + crewHome + "/task-evidence/" + taskId + "/postchange/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<what the task required>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten).";
|
|
106
|
+
} else if (env.surfaceTerminal) {
|
|
107
|
+
var termEvidence = crewHome + "/task-evidence/" + taskId + "/postchange";
|
|
108
|
+
var termTargetsLine = facts.terminalTargets || "not declared — derive from --help and the task description";
|
|
109
|
+
instructions = "You are code-blind QA. You NEVER read source files.\n" +
|
|
110
|
+
"Public docs (API.md, README) are NOT source code — read them freely, exactly as a user would.\n" +
|
|
111
|
+
"Read the shared UX bar FIRST: " + env.uxDoctrinePath + " — it is the bar the whole crew builds to, and your verdict judges against it point by point.\n\n" +
|
|
112
|
+
"STEP 1: Experiential terminal inspection — drive the CLI as a user would, one command at a time.\n" +
|
|
113
|
+
"a. The CLI under test lives in " + env.repoPath + " (the merged change is on the integration target there). Start like a new user: run --help. You may RUN the CLI; you may never READ its source.\n" +
|
|
114
|
+
"b. Terminal targets for this task: " + termTargetsLine + ".\n" +
|
|
115
|
+
"c. Bounded terminal loop — at most 8 commands. For each target: run it RIGHT (the happy path), then run it WRONG on purpose (bad flags, missing args, nonexistent files, empty input, contradictory flags). Error quality is half the grade: every failure must exit non-zero, say what went wrong in plain language, and tell the user the fix. A raw stack trace shown to a user is a defect — file it as one.\n" +
|
|
116
|
+
"d. Evidence: capture EVERY invocation as a transcript. Run: mkdir -p " + termEvidence + "\n" +
|
|
117
|
+
" For each command: <cmd> > " + termEvidence + "/<nn>-<short-slug>.txt 2>&1; echo \"exit=$?\" >> " + termEvidence + "/<nn>-<short-slug>.txt (number them 01, 02, ...). Then READ the transcript before judging it — an unread transcript is not evidence.\n" +
|
|
118
|
+
"e. Log each step to the OODA log — run in shell, one command per step:\n" +
|
|
119
|
+
" node " + crewHome + "/current/lib/append-ooda-step.js --log " + termEvidence + "/ooda-log.jsonl --attempt \"1\" --step <N> --action terminal --exit <code> --transcript " + termEvidence + "/<nn>-<short-slug>.txt --args '{\"cmd\":\"<the exact command>\"}' --observation \"<1-2 sentences: what the output said and what you concluded>\"\n" +
|
|
120
|
+
" Steps are strictly monotonic within an attempt (1, 2, 3, ...). A rerun is a NEW attempt (\"2\", \"3\", ...) — never overwrite attempt 1. If the CLI will not run at all, log the step with --exit 3 and NOT POSSIBLE in the observation — never fabricate a transcript.\n" +
|
|
121
|
+
"f. Compare against the pre-change baseline transcripts in " + crewHome + "/task-evidence/" + taskId + "/baseline/ — every finding cites its baseline and post-change transcripts by step number.\n" +
|
|
122
|
+
"Then continue with the mechanical checks below. Your VERDICT covers both the experiential and the mechanical checks.\n\n" +
|
|
123
|
+
"STEP 2: Verify data integrity via the crew API.\n" +
|
|
124
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmdString(env, "get-state", { events_limit: 1 }) + "\n" +
|
|
125
|
+
"Use the returned tasks, sessions, and events to check the task's data-level effects.\n" +
|
|
126
|
+
"DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale for a public-affecting change, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n\n" +
|
|
127
|
+
"STEP 3: File follow-up tasks for any related issues you discover — EXCEPT content findings (user-visible content/data): those go in the CONTENT-FINDINGS block at the end of these instructions, never through create-task. The workflow files follow-ups for attributable content itself.\n" +
|
|
128
|
+
"For each issue, run in shell:\n" +
|
|
129
|
+
"node " + crewApiCmd + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + env.projectId + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
130
|
+
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
131
|
+
npmPublishCheck +
|
|
132
|
+
"BASELINE SANITY: in the event history you fetched, the task's note events must contain a message starting with `baseline: captured` or `baseline: none`. If no message starts with either prefix, report 'baseline evidence missing at QA — the Map gate was bypassed', then end your report with exactly this line: VERDICT: FAIL. When the baseline is terminal transcripts, confirm every terminal target you judged has a baseline transcript: a target with no pre-change transcript is an evidence gap — name it in --missing, never invent the baseline.\n\n" +
|
|
133
|
+
"Report back in plain prose — what checks you ran and their results. Checks you could not run are evidence gaps, not silent drops: name every one in --missing — unknown is neither PASS nor FAIL. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL. First ensure the OODA log exists even if you logged zero steps (touch " + termEvidence + "/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + crewHome + "/current/lib/write-ooda-verdict.js --dir " + termEvidence + " --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<what the task required>\" --actual \"<what you observed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten).";
|
|
134
|
+
} else {
|
|
135
|
+
instructions = "Test from a user's perspective. You are CODE-BLIND — do NOT read source code.\n" +
|
|
136
|
+
"Public docs (API.md, README) are NOT source code — read them freely, exactly as a user would.\n" +
|
|
137
|
+
"Verify the change is working as described in the task.\n" +
|
|
138
|
+
"DOCS GATE: If the change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), verify the public docs describe it. If public docs are missing or stale, report 'public docs missing/stale for [the change]', then end your report with exactly this line: VERDICT: FAIL. QA always fails when public-affecting changes lack public docs. Guide/tutorial gaps are lower priority — file a follow-up task for those instead of failing.\n" +
|
|
139
|
+
"File follow-up tasks for related issues found (EXCEPT content findings — user-visible content/data goes in the CONTENT-FINDINGS block at the end of these instructions, never through create-task) by running in shell:\n" +
|
|
140
|
+
"node " + crewApiCmd + " --crew-home " + crewHome + " create-task --json '{\"title\": \"<issue title>\", \"description\": \"<issue details>\", \"project\": \"" + env.projectId + "\", \"workflow\": \"bugfix\", \"filed_by\": \"hazel\"}'\n" +
|
|
141
|
+
"(replace <issue title> and <issue details> with the real values).\n\n" +
|
|
142
|
+
npmPublishCheck +
|
|
143
|
+
"Report back in plain prose — what you tested and found. End your report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
144
|
+
}
|
|
145
|
+
// The content-findings block is machine-read at closeout (blocker 38).
|
|
146
|
+
// Hazel emits it immediately before the VERDICT line, so the verdict
|
|
147
|
+
// keeps its trailing-window contract with extractVerdict.
|
|
148
|
+
instructions += "\n" + contentFindingsProtocol;
|
|
149
|
+
return instructions;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
// ensureDeployed — idempotent precondition for experiential artifact QA on
|
|
153
|
+
// versioned projects: the QA environment serves a bundle built from merged
|
|
154
|
+
// source. Fast path when the bundle is already current. A failed deploy is
|
|
155
|
+
// an operational failure (retryable), never a park. Direct qa-deploy.mjs
|
|
156
|
+
// call — no agent ferry. The module self-records the bundle hash to the
|
|
157
|
+
// event log.
|
|
158
|
+
async function ensureDeployed(env) {
|
|
159
|
+
var qaEnvDir = env.crewHome + "/qa-envs/" + env.projectId;
|
|
160
|
+
var out = "";
|
|
161
|
+
try {
|
|
162
|
+
out = String((await runCmd(["node", join(env.runLib, "qa-deploy.mjs"),
|
|
163
|
+
"--repo", env.repoPath, "--qa-dir", qaEnvDir,
|
|
164
|
+
"--task", env.taskId, "--crew-home", env.crewHome, "--crew-api", env.crewApiPinned,
|
|
165
|
+
"--serve-artifact", join(env.runLib, "serve-artifact.js"), "--ensure"], {})).stdout || "").trim();
|
|
166
|
+
} catch (e) {
|
|
167
|
+
out = "";
|
|
168
|
+
}
|
|
169
|
+
var parsed = parseDeployResult(out);
|
|
170
|
+
if (!parsed.ok) {
|
|
171
|
+
return { failed: parsed.reason };
|
|
172
|
+
}
|
|
173
|
+
log("QA ensure-deployed ok for task " + env.taskId + ": bundle " + parsed.hash);
|
|
174
|
+
return { ok: true, hash: parsed.hash };
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// runLoopGuard — the experiential-loop guard (qa-reproduce.md): a PASS with
|
|
178
|
+
// missing experiential evidence parks fail-closed, never completes. Reads
|
|
179
|
+
// the machine-readable OODA verdict record directly.
|
|
180
|
+
async function runLoopGuard(env) {
|
|
181
|
+
var surface = env.surfaceTerminal ? "terminal" : "visual";
|
|
182
|
+
var reason = env.surfaceTerminal ? "qa-terminal-loop-unavailable" : "qa-visual-loop-unavailable";
|
|
183
|
+
var reader = join(env.runLib, "read-ooda-verdict.js");
|
|
184
|
+
try {
|
|
185
|
+
statSync(reader);
|
|
186
|
+
} catch (e) {
|
|
187
|
+
// The reader is a release-lib script, not a pinned script.
|
|
188
|
+
reader = env.crewHome + "/current/lib/read-ooda-verdict.js";
|
|
189
|
+
}
|
|
190
|
+
var out = "";
|
|
191
|
+
try {
|
|
192
|
+
out = String((await runCmd(["node", reader, "--dir", env.taskEvidenceDir + "/postchange", "--expect", "PASS"], {})).stdout || "").trim();
|
|
193
|
+
} catch (e) {
|
|
194
|
+
out = "";
|
|
195
|
+
}
|
|
196
|
+
// The script prints exactly one JSON line; the parsed object is the gate
|
|
197
|
+
// signal.
|
|
198
|
+
var gate = null;
|
|
199
|
+
try {
|
|
200
|
+
var lines = out.split("\n");
|
|
201
|
+
gate = JSON.parse(lines[lines.length - 1]);
|
|
202
|
+
} catch (e) {
|
|
203
|
+
gate = null;
|
|
204
|
+
}
|
|
205
|
+
if (!gate || gate.ok !== true) {
|
|
206
|
+
return { parked: await parkTask(env, "QA " + surface + " loop unverifiable (unattributable_reason=" + reason + "): lib/read-ooda-verdict.js could not confirm the QA verdict record — a PASS without a machine-readable experiential record is never terminal. Human attention needed.") };
|
|
207
|
+
}
|
|
208
|
+
var unavailable = env.surfaceTerminal ? gate.terminal_loop_unavailable : gate.visual_loop_unavailable;
|
|
209
|
+
if (unavailable === true) {
|
|
210
|
+
var why = env.surfaceTerminal
|
|
211
|
+
? "the terminal loop could not run — the CLI would not execute, and verdict.json records missing terminal evidence. A PASS without experiential evidence is never terminal. Human attention needed: check the project's runtime dependencies at " + env.repoPath + ", then re-queue QA."
|
|
212
|
+
: "the see-act browser loop could not run — verdict.json records missing visual evidence. A PASS without experiential evidence is never terminal. Human attention needed: repair the crew home's dependency symlink ($CREW_HOME/node_modules) or the npm install, then re-queue QA.";
|
|
213
|
+
return { parked: await parkTask(env, "QA " + surface + " loop unavailable (unattributable_reason=" + reason + "): " + why) };
|
|
214
|
+
}
|
|
215
|
+
log("QA " + surface + "-loop guard passed: experiential evidence present");
|
|
216
|
+
return { ok: true };
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
// closeoutHashCheck — the bundle Hazel tested must match the ensure-deployed
|
|
220
|
+
// hash from QA start. Mismatch/unreadable = operational failure, never a park.
|
|
221
|
+
async function closeoutHashCheck(env, qaBundleHash) {
|
|
222
|
+
var qaEnvDir = env.crewHome + "/qa-envs/" + env.projectId;
|
|
223
|
+
var out = "";
|
|
224
|
+
try {
|
|
225
|
+
out = String((await runCmd(["node", join(env.runLib, "qa-deploy.mjs"), "--hash-only", "--qa-dir", qaEnvDir], {})).stdout || "");
|
|
226
|
+
} catch (e) {
|
|
227
|
+
out = "";
|
|
228
|
+
}
|
|
229
|
+
var hm = /BUNDLE_HASH ([0-9a-f]{40})/.exec(out);
|
|
230
|
+
var closeHash = hm ? hm[1].slice(0, 7) : "";
|
|
231
|
+
if (closeHash !== qaBundleHash) {
|
|
232
|
+
return { ok: false, reason: "bundle hash " + (closeHash ? "changed during QA" : "unreadable") };
|
|
233
|
+
}
|
|
234
|
+
return { ok: true };
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
// closeoutContentFindings — blocker 38: Hazel's CONTENT-FINDINGS block is
|
|
238
|
+
// extracted deterministically and each finding is classified against the
|
|
239
|
+
// task's publish-diff file set (persisted at Publish). Attributable
|
|
240
|
+
// findings become bugfix tasks filed by the workflow; environment/unknown
|
|
241
|
+
// findings are recorded as task note events — never suppressed, never a
|
|
242
|
+
// bugfix, never a park. No automatic FAIL override: if Hazel still reports
|
|
243
|
+
// FAIL, it stands — finding attribution informs follow-up filing only.
|
|
244
|
+
async function closeoutContentFindings(env, workerText) {
|
|
245
|
+
var taskId = env.taskId;
|
|
246
|
+
var contentFindings = extractContentFindings(workerText);
|
|
247
|
+
if (!contentFindings.ok) {
|
|
248
|
+
// Unknown attribution fails closed: a missing or malformed
|
|
249
|
+
// CONTENT-FINDINGS block blocks the phase for retry — the way an
|
|
250
|
+
// unreadable VERDICT line does. It must never degrade to "no
|
|
251
|
+
// findings": unknown attribution converted to an empty set silently
|
|
252
|
+
// drops the observations the prose grounds carry.
|
|
253
|
+
await recordPhase(env, {
|
|
254
|
+
task_id: taskId,
|
|
255
|
+
session: { id: undefined, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "failed",
|
|
256
|
+
notes: "QA report had no machine-readable CONTENT-FINDINGS block (" + contentFindings.reason + "); phase failed for retry" },
|
|
257
|
+
event: { task_id: taskId, type: "failed", identity: PHASE.identity, message: "QA CONTENT-FINDINGS block unreadable (" + contentFindings.reason + ") — phase failed, dispatcher will retry" },
|
|
258
|
+
});
|
|
259
|
+
return { failed: "QA worker report had no machine-readable CONTENT-FINDINGS block (" + contentFindings.reason + ")" };
|
|
260
|
+
}
|
|
261
|
+
// The publish-diff file set, persisted at Publish. Unreadable →
|
|
262
|
+
// attribution unknown: every finding records "unknown" with the reason
|
|
263
|
+
// naming the missing diff (never environment-attributable, never a
|
|
264
|
+
// bugfix). The QA verdict stands on its own.
|
|
265
|
+
var qaDiffFiles = [];
|
|
266
|
+
var qaDiffFilesOk = false;
|
|
267
|
+
try {
|
|
268
|
+
var raw = readFileSync(env.publishDiffsDir + "/" + taskId + ".files.json", "utf8").trim();
|
|
269
|
+
var parsed = JSON.parse(raw);
|
|
270
|
+
if (Array.isArray(parsed)) {
|
|
271
|
+
qaDiffFiles = parsed;
|
|
272
|
+
qaDiffFilesOk = true;
|
|
273
|
+
}
|
|
274
|
+
} catch (e) {
|
|
275
|
+
log("QA diff file list unreadable: " + String((e && e.message) || e).slice(0, 200));
|
|
276
|
+
}
|
|
277
|
+
if (!qaDiffFilesOk) {
|
|
278
|
+
log("QA publish diff file list unavailable — content findings record attribution unknown");
|
|
279
|
+
}
|
|
280
|
+
for (var i = 0; i < contentFindings.findings.length; i++) {
|
|
281
|
+
var cf = contentFindings.findings[i];
|
|
282
|
+
var cfc = qaDiffFilesOk ? classifyContentFinding(cf, qaDiffFiles)
|
|
283
|
+
: { attribution: "unknown", reason: "publish diff file set unavailable — cannot attribute" };
|
|
284
|
+
if (cfc.attribution === "task-change") {
|
|
285
|
+
log("QA content finding attributable to task change — filing bugfix: " + cf.subject + " :: " + cf.observation.slice(0, 120));
|
|
286
|
+
await crewApi(env, "create-task", {
|
|
287
|
+
title: "QA content finding: " + cf.observation.slice(0, 120),
|
|
288
|
+
description: "Content finding from QA on task " + taskId + " (subject: " + cf.subject + ", classified task-change: " + cfc.reason + "). Observation: " + cf.observation,
|
|
289
|
+
project: env.projectId, workflow: "bugfix", filed_by: "hazel",
|
|
290
|
+
});
|
|
291
|
+
} else {
|
|
292
|
+
// environment-attributable and unknown findings are recorded as task
|
|
293
|
+
// note events with their attribution — never suppressed, never a
|
|
294
|
+
// bugfix, never a park by themselves.
|
|
295
|
+
log("QA content finding " + cfc.attribution + " — recording, no bugfix: " + cf.subject + " :: " + cf.observation.slice(0, 120));
|
|
296
|
+
await crewApi(env, "log-event", {
|
|
297
|
+
task_id: taskId, type: "note",
|
|
298
|
+
message: "content-finding: " + cfc.attribution + " — subject: " + cf.subject + " — " + cf.observation + " (" + cfc.reason + ")",
|
|
299
|
+
});
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
return { ok: true };
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
export async function runPhase(ctx) {
|
|
306
|
+
var env = ctx.env, state = ctx.state;
|
|
307
|
+
var taskId = env.taskId;
|
|
308
|
+
var claimed = await ensureClaimed(env, state, PHASE);
|
|
309
|
+
if (claimed.type !== "CLAIMED") return claimed;
|
|
310
|
+
|
|
311
|
+
// Experiential routing: experiential tasks on a classified surface run
|
|
312
|
+
// the Hazel experiential QA prompt — Hazel drives the surface herself
|
|
313
|
+
// and owns the verdict through her OODA report.
|
|
314
|
+
var qaExperiential = (await resolveExperiential(env, state)) === "yes" && (env.surfaceArtifact || env.surfaceTerminal);
|
|
315
|
+
var qaBundleHash = null;
|
|
316
|
+
if (qaExperiential && env.surfaceArtifact && env.versionedBuild) {
|
|
317
|
+
var dep = await ensureDeployed(env);
|
|
318
|
+
if (dep.failed) {
|
|
319
|
+
await recordPhase(env, {
|
|
320
|
+
task_id: taskId,
|
|
321
|
+
session: { id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "failed",
|
|
322
|
+
notes: "QA ensure-deployed failed: " + dep.failed },
|
|
323
|
+
event: { task_id: taskId, type: "failed", identity: PHASE.identity, message: "QA ensure-deployed failed: " + dep.failed + " — phase failed, dispatcher will retry" },
|
|
324
|
+
});
|
|
325
|
+
return { type: "FAILED", reason: "QA ensure-deployed failed: " + dep.failed };
|
|
326
|
+
}
|
|
327
|
+
qaBundleHash = dep.hash;
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
// Cross-phase handoffs re-derived from durable session notes.
|
|
331
|
+
var releaseDecision = extractReleaseDecision(await latestSessionNotes(env, "Build", "completed"));
|
|
332
|
+
var triageMarkers = extractMarkerLines(await latestSessionNotes(env, "Triage", "completed"));
|
|
333
|
+
var ttMatch = /^terminal_targets:\s*(.+)$/im.exec(triageMarkers);
|
|
334
|
+
state.memo.qaFacts = {
|
|
335
|
+
qaExperiential: qaExperiential,
|
|
336
|
+
qaBundleHash: qaBundleHash,
|
|
337
|
+
releaseDecision: releaseDecision,
|
|
338
|
+
terminalTargets: ttMatch ? ttMatch[1].trim() : "",
|
|
339
|
+
};
|
|
340
|
+
|
|
341
|
+
var boundary = await runWorkBoundary(env, state, {
|
|
342
|
+
phase: PHASE.name, identity: PHASE.identity,
|
|
343
|
+
instructions: buildInstructions(ctx),
|
|
344
|
+
eventPreamble: buildEventPreamble(env, PHASE.name),
|
|
345
|
+
crewApiLine: true, verdictStep: true,
|
|
346
|
+
});
|
|
347
|
+
if (boundary.type !== "BOUNDARY_DONE") return boundary;
|
|
348
|
+
var workerText = boundary.workerText;
|
|
349
|
+
var passed = closeoutPassed(boundary) === true;
|
|
350
|
+
var summary = summarizeReport(workerText);
|
|
351
|
+
|
|
352
|
+
// Experiential-loop guard: a PASS with missing experiential evidence is
|
|
353
|
+
// never terminal.
|
|
354
|
+
if (passed && qaExperiential) {
|
|
355
|
+
var lg = await runLoopGuard(env);
|
|
356
|
+
if (lg.parked) return lg.parked;
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
// Closeout integrity (experiential artifact on versioned builds): the
|
|
360
|
+
// bundle Hazel tested must match the ensure-deployed hash, and her
|
|
361
|
+
// content findings are classified against the Publish diff file set.
|
|
362
|
+
// The QA verdict stands on its own and is never overridden by finding
|
|
363
|
+
// attribution.
|
|
364
|
+
if (qaBundleHash !== null) {
|
|
365
|
+
var hc = await closeoutHashCheck(env, qaBundleHash);
|
|
366
|
+
if (!hc.ok) {
|
|
367
|
+
log("QA closeout hash mismatch for task " + taskId + " — marking failed for retry");
|
|
368
|
+
summary += "\nqa_closeout: FAILED — bundle hash " + hc.reason;
|
|
369
|
+
passed = false;
|
|
370
|
+
}
|
|
371
|
+
var cfr = await closeoutContentFindings(env, workerText);
|
|
372
|
+
if (cfr.failed) {
|
|
373
|
+
return { type: "FAILED", reason: cfr.failed };
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
var status = passed ? "completed" : "rejected";
|
|
378
|
+
await recordPhase(env, {
|
|
379
|
+
task_id: taskId,
|
|
380
|
+
session: { id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: status, notes: summary },
|
|
381
|
+
event: { task_id: taskId, type: status, identity: PHASE.identity, message: PHASE.name + " " + status + " by " + PHASE.identity },
|
|
382
|
+
});
|
|
383
|
+
|
|
384
|
+
if (!passed) {
|
|
385
|
+
// Shared rework budget: Review and QA rejections draw from the SAME
|
|
386
|
+
// pool of 2 (derived durably by counting "rejected" events). The
|
|
387
|
+
// rejected event above is already recorded, so the new count is
|
|
388
|
+
// state.reworkCount + 1.
|
|
389
|
+
var newRework = state.reworkCount + 1;
|
|
390
|
+
if (newRework > MAX_TOTAL_REWORK) {
|
|
391
|
+
log("Shared rework budget exhausted for task " + taskId + " after QA rejection");
|
|
392
|
+
return await parkTask(env, "Exceeded shared rework budget (" + MAX_TOTAL_REWORK + " total rework attempts across Review and QA) after QA rejection.");
|
|
393
|
+
}
|
|
394
|
+
log("QA rejected — bouncing to Build (rework #" + newRework + " of " + MAX_TOTAL_REWORK + ")");
|
|
395
|
+
return { type: "REWIND", next: "Build", reason: "QA rejected — bouncing to Build (rework #" + newRework + " of " + MAX_TOTAL_REWORK + ")" };
|
|
396
|
+
}
|
|
397
|
+
log("QA complete for task " + taskId + " — workflow terminal");
|
|
398
|
+
return { type: "ADVANCE", next: null };
|
|
399
|
+
}
|