muse-crew 0.17.2 → 0.17.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +11 -0
- package/docs/decisions/composition-machinery.md +180 -0
- package/docs/decisions/publish-path.md +6 -0
- package/docs/decisions/workflow-core.md +5 -4
- package/lib/AGENTS.md +2 -1
- package/lib/bugfix/phases/build.js +168 -0
- package/lib/bugfix/phases/capture.js +170 -0
- package/lib/bugfix/phases/integrate.js +165 -0
- package/lib/bugfix/phases/map.js +129 -0
- package/lib/bugfix/phases/publish.js +592 -0
- package/lib/bugfix/phases/qa.js +356 -0
- package/lib/bugfix/phases/reproduce.js +254 -0
- package/lib/bugfix/phases/review.js +292 -0
- package/lib/bugfix/phases/triage.js +69 -0
- package/lib/chore/CONTRACT.md +181 -0
- package/lib/chore/DISPOSITION.md +98 -0
- package/lib/chore/extract.js +204 -0
- package/lib/chore/phase-lib.js +885 -0
- package/lib/chore/phases/build.js +110 -0
- package/lib/chore/phases/capture.js +82 -0
- package/lib/chore/phases/integrate.js +109 -0
- package/lib/chore/phases/map.js +81 -0
- package/lib/chore/phases/publish.js +543 -0
- package/lib/chore/phases/review.js +255 -0
- package/lib/chore/phases/triage.js +57 -0
- package/lib/chore/prompts/evidence-gatherer.js +41 -0
- package/lib/chore/prompts/evidence-gatherer.schema.json +1 -0
- package/lib/chore/prompts/tool-check.js +15 -0
- package/lib/chore/prompts/trailers.js +56 -0
- package/lib/chore/prompts/verdict-reask.js +28 -0
- package/lib/chore/prompts/verdict-reask.schema.json +1 -0
- package/lib/chore/prompts/work-agent.js +52 -0
- package/lib/chore/prompts/work-agent.schema.json +1 -0
- package/lib/chore/spawn-keys.js +44 -0
- package/lib/chore/spawn-vocab.js +87 -0
- package/lib/chore-run.js +538 -0
- package/lib/chore-tick.js +289 -0
- package/lib/crew-api.js +273 -0
- package/lib/crew-dispatch-worker.js +27 -7
- package/lib/crew-release.sh +7 -2
- package/lib/extract.js +252 -0
- package/lib/prompts/tool-check.js +18 -0
- package/lib/prompts/trailers.js +59 -0
- package/lib/prompts/verdict-reask.js +31 -0
- package/lib/prompts/verdict-reask.schema.json +1 -0
- package/lib/prompts/work-agent.js +56 -0
- package/lib/prompts/work-agent.schema.json +1 -0
- package/lib/reap-spawns.js +407 -0
- package/lib/schema.sql +12 -1
- package/lib/spawn-keys.js +47 -0
- package/lib/spawn-step.js +572 -0
- package/lib/standard/phases/build.js +120 -0
- package/lib/standard/phases/capture.js +163 -0
- package/lib/standard/phases/integrate.js +172 -0
- package/lib/standard/phases/map.js +119 -0
- package/lib/standard/phases/publish.js +565 -0
- package/lib/standard/phases/qa.js +399 -0
- package/lib/standard/phases/review.js +281 -0
- package/lib/standard/phases/triage.js +64 -0
- package/lib/test-detached-integrate.sh +47 -0
- package/lib/workflow-driver.js +605 -0
- package/lib/workflow-lib.js +1012 -0
- package/lib/workflow-spec.js +187 -0
- package/lib/worktree-lifecycle.sh +55 -3
- package/package.json +1 -1
- package/seed/cron-body-template.md +61 -9
- package/workflows/bugfix.js +17 -17
- package/workflows/chore.js +16 -16
- package/workflows/docs.js +14 -11
- package/workflows/standard.js +16 -16
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
// lib/bugfix/phases/reproduce.js — Bugfix Reproduce phase (worker layer).
|
|
2
|
+
//
|
|
3
|
+
// Hazel re-runs the bug at the bug's own layer with the terminal transcript
|
|
4
|
+
// protocol, then issues a machine-readable verdict. Derived from
|
|
5
|
+
// workflows/bugfix.js (the Bugfix Reproduce step).
|
|
6
|
+
//
|
|
7
|
+
// Contract:
|
|
8
|
+
// - Read: Triage layer (durable marker; falls back to task row
|
|
9
|
+
// `bug_layer`), task row (title, description), project surface
|
|
10
|
+
// (artifact | terminal | unclassified).
|
|
11
|
+
// - The phase is skipped when the bug is non-experiential... — no:
|
|
12
|
+
// Reproduce runs for every bug that reaches it. (Capture skips; Reproduce
|
|
13
|
+
// does not.)
|
|
14
|
+
// - Layer routing (source-faithful):
|
|
15
|
+
// artifact + terminal surface → terminal transcript protocol
|
|
16
|
+
// artifact + artifact/unclassified surface → rendered page flow
|
|
17
|
+
// engine → engine/CLI instructions
|
|
18
|
+
// docs → docs/CLI instructions
|
|
19
|
+
// - Wrong-layer guard (engine/docs only): fail closed if the attempt
|
|
20
|
+
// created new see-act frames, or if the pre-attempt baseline is
|
|
21
|
+
// unavailable.
|
|
22
|
+
// - Verdict: mechanical extraction with bounded re-ask; exhaustion is an
|
|
23
|
+
// operational FAILED (retryable), not a rejection.
|
|
24
|
+
// - Valid FAIL → record `rejected`, then park for PM attention
|
|
25
|
+
// ("Reproduction failed — needs PM attention"). Never advances.
|
|
26
|
+
// - Valid PASS → advance to Map.
|
|
27
|
+
// - The OODA agent writes verdict.md, transcript, and the OODA attempt
|
|
28
|
+
// log into the repro evidence dir.
|
|
29
|
+
//
|
|
30
|
+
// Phase I/O contract (Phase D):
|
|
31
|
+
// read: bug_layer marker | task.bug_layer, task title/description,
|
|
32
|
+
// project surface classification
|
|
33
|
+
// write: session (completed | rejected), event (completed | rejected),
|
|
34
|
+
// park row (on FAIL)
|
|
35
|
+
// out: ADVANCE → Map | PARK (on FAIL) | FAILED (guard / verdict
|
|
36
|
+
// exhaustion / transport)
|
|
37
|
+
|
|
38
|
+
import { readdirSync, readFileSync, writeFileSync, mkdirSync } from "fs";
|
|
39
|
+
import { join, dirname } from "path";
|
|
40
|
+
import {
|
|
41
|
+
runWorkBoundary, recordPhase, buildEventPreamble, summarizeReport,
|
|
42
|
+
ensureClaimed, log, resolveLayer, parkTask, crewCmdString,
|
|
43
|
+
} from "../../workflow-lib.js";
|
|
44
|
+
|
|
45
|
+
export const PHASE = { name: "Reproduce", identity: "hazel" };
|
|
46
|
+
|
|
47
|
+
// reproEvidenceDir — the OODA agent's evidence directory. Matches the
|
|
48
|
+
// source's reproEvidenceDir exactly.
|
|
49
|
+
function reproEvidenceDir(env) {
|
|
50
|
+
return join(env.crewHome, "task-evidence", env.taskId, "repro");
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// listReproFrames — the see-act frame listing for the wrong-layer guard.
|
|
54
|
+
// Returns an array of frame filenames, or null when the directory cannot
|
|
55
|
+
// be read. A missing directory means no frames exist (matches the source's
|
|
56
|
+
// `|| echo NONE` → empty-list behavior).
|
|
57
|
+
function listReproFrames(env) {
|
|
58
|
+
var dir = reproEvidenceDir(env);
|
|
59
|
+
var files;
|
|
60
|
+
try {
|
|
61
|
+
files = readdirSync(dir);
|
|
62
|
+
} catch (e) {
|
|
63
|
+
if (e && e.code === "ENOENT") return [];
|
|
64
|
+
return null;
|
|
65
|
+
}
|
|
66
|
+
return files.filter(function (f) { return /-(shot|click|scroll|type)-/.test(f); });
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// guardMarkerPath — durable attempt-scoped storage for the pre-spawn frame
|
|
70
|
+
// baseline. The worker layer is stateless across NEED_SPAWN re-entries (the
|
|
71
|
+
// driver re-invokes the phase in a fresh process), so the snapshot must live
|
|
72
|
+
// on disk, written BEFORE the first spawn and never replaced after the agent
|
|
73
|
+
// runs.
|
|
74
|
+
function guardMarkerPath(env) {
|
|
75
|
+
return join(reproEvidenceDir(env), ".guard-before.json");
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function readGuardMarker(env) {
|
|
79
|
+
try {
|
|
80
|
+
return JSON.parse(readFileSync(guardMarkerPath(env), "utf8"));
|
|
81
|
+
} catch (e) {
|
|
82
|
+
return null;
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function writeGuardMarker(env, marker) {
|
|
87
|
+
var p = guardMarkerPath(env);
|
|
88
|
+
try { mkdirSync(dirname(p), { recursive: true }); } catch (e) { /* best effort */ }
|
|
89
|
+
writeFileSync(p, JSON.stringify(marker));
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// buildInstructions — the four layer-routed instruction variants,
|
|
93
|
+
// verbatim from workflows/bugfix.js (variable references remapped to
|
|
94
|
+
// ctx.env; logic and prose unchanged). ctx: { env, state }.
|
|
95
|
+
export function buildInstructions(ctx) {
|
|
96
|
+
var env = ctx.env;
|
|
97
|
+
var reproLayer = ctx.reproLayer || "artifact";
|
|
98
|
+
if (reproLayer !== "engine" && reproLayer !== "docs") {
|
|
99
|
+
if (env.surfaceTerminal) {
|
|
100
|
+
var instructions = "Reproduce the bug from a user's perspective — by USING the CLI, not by reading source code. You are CODE-BLIND — do NOT read source code.\n" +
|
|
101
|
+
"This project's user-facing surface is a terminal: a command-line interface. Read the shared UX bar FIRST: " + env.uxDoctrinePath + " — it is what correct CLI behavior looks like.\n" +
|
|
102
|
+
"a. Work in the project checkout at " + env.repoPath + " — that exact checkout, not any other copy of the project on disk. Drive the CLI the way a user would: start with --help to discover the surface, then the commands the bug report implicates.\n" +
|
|
103
|
+
"b. Bounded CLI loop, at most 8 commands: drive the CLI to TRIGGER the reported bug. For EVERY command, save a transcript file under " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ named 01-<slug>.txt, 02-<slug>.txt, ... — each transcript holds the exact command line, its stdout, its stderr, and its exit code. Never invent output: a transcript you did not run is not evidence.\n" +
|
|
104
|
+
"c. After EVERY command, append it to the OODA report: node " + env.crewHome + "/current/lib/append-ooda-step.js --log " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ooda-log.jsonl --attempt \"1\" --step <N> --action terminal --exit <exit-code> --transcript <the transcript file you saved> --observation \"<1-2 sentences: what the command showed and what you concluded>\" — the script requires --transcript for terminal actions and rejects --screenshot (there is nothing to see). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what the output showed and what it means, not just what you ran. A command you ran but did not log is invisible to everyone after you. If the CLI will not run at all (missing runtime, broken install), log ONE step with --exit 3, --observation \"NOT POSSIBLE: <reason>\" and a transcript holding the failed invocation, then fall back to the data-level investigation at the end.\n" +
|
|
105
|
+
"Data-level fallback (only if the CLI loop above is NOT POSSIBLE): run in shell and read the stdout JSON:\n" + crewCmdString(env, "get-state", {}) + "\n" +
|
|
106
|
+
"This returns current sessions, events, and tasks — capture concrete evidence from the data you retrieve.\n" +
|
|
107
|
+
"Report your reproduction steps and evidence as plain prose.\n" +
|
|
108
|
+
"End your report with exactly one line: VERDICT: PASS if you reproduced the reported bug (your transcripts show the reported misbehavior), VERDICT: FAIL if you could not. --expected names the bug as reported (its CLI manifestation); --actual names what your commands actually showed. Checks you could not run are evidence gaps, not silent drops: name every one in --missing. First ensure the OODA log exists even if you logged zero steps (touch " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + env.crewHome + "/current/lib/write-ooda-verdict.js --dir " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the bug as reported — its CLI manifestation>\" --actual \"<what your commands actually showed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL or NOT_POSSIBLE verdict without a machine-readable --reason cannot be written — state the reason.";
|
|
109
|
+
} else {
|
|
110
|
+
instructions = "Reproduce the bug from a user's perspective — by USING the artifact, not by reading data. You are CODE-BLIND — do NOT read source code.\n" +
|
|
111
|
+
"You have a see-act driver: " + env.crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
|
|
112
|
+
"Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and fall back to the data-level investigation at the end.\n" +
|
|
113
|
+
"a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + env.publishSlug + "/client/dist && test -f ~/workspace/ts-spaces/" + env.publishSlug + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + env.publishSlug + " and fall back to the data-level investigation.\n" +
|
|
114
|
+
"b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + env.crewHome + " setsid nohup node " + env.crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + env.publishSlug + " --port 0 --tag " + env.taskId + "-repro > /tmp/repro-server-" + env.taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/repro-server-" + env.taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/repro-server-" + env.taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/repro-server-" + env.taskId + ".log) and fall back to the data-level investigation.\n" +
|
|
115
|
+
"c. Bounded see-act loop, at most 8 steps: drive the artifact to TRIGGER the reported bug. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
|
|
116
|
+
"c2. After EVERY see-act invocation, append it to the OODA report: node " + env.crewHome + "/current/lib/append-ooda-step.js --log " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you. If the server stops responding mid-loop (curl --max-time 3 http://localhost:<N>/ fails), restart it per (b).\n" +
|
|
117
|
+
"c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + env.crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + env.crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
|
|
118
|
+
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ node " + env.crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ node " + env.crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
119
|
+
"d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the bug needs such a sequence, report NOT POSSIBLE for that part and reproduce what you can.\n" +
|
|
120
|
+
"e. Every frame you captured is already archived under " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ and indexed in ooda-log.jsonl — that log plus the frames is your OODA report for this reproduction. A frame you did not read is not evidence. Loading, error, or blank frames never reproduce anything.\n" +
|
|
121
|
+
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + env.taskId + "-repro' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
122
|
+
"Data-level fallback (only if the browser loop above is NOT POSSIBLE): run in shell and read the stdout JSON:\n" + crewCmdString(env, "get-state", {}) + "\n" +
|
|
123
|
+
"This returns current sessions, events, and tasks — capture concrete evidence from the data you retrieve.\n" +
|
|
124
|
+
"Report your reproduction steps and evidence as plain prose.\n" +
|
|
125
|
+
"End your report with exactly one line: VERDICT: PASS if you reproduced the reported bug (your frames show the reported misbehavior), VERDICT: FAIL if you could not. --expected names the bug as reported (its visible manifestation); --actual names what your frames actually showed. Checks you could not run are evidence gaps, not silent drops: name every one in --missing. First ensure the OODA log exists even if you logged zero steps (touch " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ooda-log.jsonl — an empty log is honest, an absent one is a broken report). Also write the same verdict machine-readably: node " + env.crewHome + "/current/lib/write-ooda-verdict.js --dir " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the bug as reported — its visible manifestation>\" --actual \"<what your frames actually showed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL or NOT_POSSIBLE verdict without a machine-readable --reason cannot be written — state the reason.";
|
|
126
|
+
}
|
|
127
|
+
} else if (reproLayer === "engine") {
|
|
128
|
+
instructions = "REPRODUCE AT THE BUG'S LAYER. Triage classified this bug as layer: engine — it lives in the crew's own machinery (workflows, lib scripts, shell scripts, tests, scheduler), not in the rendered artifact. Reproduce it deterministically with shell commands in the repo checkout at " + env.repoPath + " — that exact checkout, not any other copy of the project on disk. You are NOT code-blind here: reading source to find the failing mechanism is expected.\n" +
|
|
129
|
+
"Do NOT start an artifact server. Do NOT invoke see-act.js or any browser loop — the experiential loop is for artifact-layer bugs only, and driving it here fails the phase loudly. If you cannot reproduce without a browser, say so and FAIL with a reason naming the layer you tried.\n" +
|
|
130
|
+
"Reproduce with the commands that match the bug: run the failing script directly, re-run the failing test suite (bash tests/run.sh), grep the workflow source, query scheduler/crew state via the crew API. Concrete evidence only: the command lines you ran, their verbatim stdout, and their exit codes.\n" +
|
|
131
|
+
"Report your reproduction steps and evidence as plain prose.\n" +
|
|
132
|
+
"End your report with exactly one line: VERDICT: PASS if you reproduced the reported bug (your command output shows the reported misbehavior), VERDICT: FAIL if you could not. --expected names the bug as reported; --actual names what your commands actually showed. Checks you could not run are evidence gaps, not silent drops: name every one in --missing. Also write the same verdict machine-readably: node " + env.crewHome + "/current/lib/write-ooda-verdict.js --dir " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the bug as reported>\" --actual \"<what your commands actually showed>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL or NOT_POSSIBLE verdict without a machine-readable --reason cannot be written — state the reason.";
|
|
133
|
+
} else {
|
|
134
|
+
instructions = "REPRODUCE AT THE BUG'S LAYER. Triage classified this bug as layer: docs — a documentation gap or error. Reproduce it by reading the file, not the browser: open the document in the repo checkout at " + env.repoPath + " (that exact checkout, not any other copy of the project on disk) and verify the reported gap or error is actually there (a missing section, a stale instruction, a wrong claim). Quote the exact lines you found — or their absence.\n" +
|
|
135
|
+
"Do NOT start an artifact server. Do NOT invoke see-act.js or any browser loop — the experiential loop is for artifact-layer bugs only, and driving it here fails the phase loudly. If you cannot verify the gap without a browser, say so and FAIL with a reason naming the layer you tried.\n" +
|
|
136
|
+
"Report your findings as plain prose: what the document says today versus what it should say.\n" +
|
|
137
|
+
"End your report with exactly one line: VERDICT: PASS if you confirmed the reported docs gap, VERDICT: FAIL if you could not. Also write the same verdict machine-readably: node " + env.crewHome + "/current/lib/write-ooda-verdict.js --dir " + env.crewHome + "/task-evidence/" + env.taskId + "/repro/ --attempt \"1\" --verdict <PASS|FAIL|NOT_POSSIBLE> --summary \"<one line>\" --expected \"<the bug as reported>\" --actual \"<what the document actually says>\" --missing '[\"honest evidence gap, if any\"]' [--reason \"<why it failed — REQUIRED and non-empty when verdict is FAIL or NOT_POSSIBLE; the script rejects a reason-less negative verdict with exit 2>\"] — this writes verdict.json (the latest verdict) and appends to verdicts.jsonl (the append-only ledger: every attempt's verdict is preserved, never overwritten). A FAIL or NOT_POSSIBLE verdict without a machine-readable --reason cannot be written — state the reason.";
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
return instructions;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// recordGuardFailure — the wrong-layer / baseline-unavailable closeout:
|
|
145
|
+
// operational FAILED (retryable), session "failed". Matches the source's
|
|
146
|
+
// BLOCKED control semantics in the composed vocabulary (the checkpoint's
|
|
147
|
+
// "operational FAILED").
|
|
148
|
+
async function recordGuardFailure(env, state, kind, detail) {
|
|
149
|
+
var reason = kind === "baseline-unavailable"
|
|
150
|
+
? "Reproduce wrong-layer guard baseline unavailable for task " + env.taskId + " — failing closed for retry"
|
|
151
|
+
: "Reproduce wrong-layer guard tripped for task " + env.taskId + ": see-act frames created during this attempt on a " + kind + "-layer bug (" + detail + ")";
|
|
152
|
+
log(reason);
|
|
153
|
+
await recordPhase(env, {
|
|
154
|
+
task_id: env.taskId,
|
|
155
|
+
session: {
|
|
156
|
+
id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity,
|
|
157
|
+
step: PHASE.name, status: "failed",
|
|
158
|
+
notes: "Reproduce wrong-layer guard: " + kind + (detail ? " (" + detail + ")" : ""),
|
|
159
|
+
},
|
|
160
|
+
event: { task_id: env.taskId, type: "failed", message: reason },
|
|
161
|
+
});
|
|
162
|
+
return { type: "FAILED", reason: reason, detail: detail || kind };
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
export async function runPhase(ctx) {
|
|
166
|
+
var env = ctx.env, state = ctx.state;
|
|
167
|
+
// The bug's layer routes the reproduction protocol. resolveLayer is the
|
|
168
|
+
// worker-layer equivalent of the source's layer marker resolution.
|
|
169
|
+
var reproLayer = await resolveLayer(env, state);
|
|
170
|
+
ctx.reproLayer = reproLayer;
|
|
171
|
+
log("Reproduce dispatching at the bug's layer for task " + env.taskId + ": " + reproLayer);
|
|
172
|
+
|
|
173
|
+
// Wrong-layer guard baseline (engine/docs only). The snapshot is taken
|
|
174
|
+
// BEFORE the first spawn and persisted in a durable marker file: the
|
|
175
|
+
// worker layer is stateless across NEED_SPAWN re-entries (each is a fresh
|
|
176
|
+
// driver invocation), so an in-memory baseline would be lost. The marker
|
|
177
|
+
// is never replaced after the agent runs — a fresh visit (first entry or
|
|
178
|
+
// dispatcher retry, both with no active session yet) takes a new snapshot;
|
|
179
|
+
// a NEED_SPAWN re-entry (active session already claimed) reuses it.
|
|
180
|
+
var guardApplicable = (reproLayer === "engine" || reproLayer === "docs");
|
|
181
|
+
var guardBefore = null; // null: not applicable, or baseline unavailable
|
|
182
|
+
if (guardApplicable) {
|
|
183
|
+
if (!state.activeSessionId) {
|
|
184
|
+
var snap = listReproFrames(env);
|
|
185
|
+
writeGuardMarker(env, snap === null ? { unavailable: true } : { frames: snap });
|
|
186
|
+
guardBefore = snap;
|
|
187
|
+
log("Reproduce guard baseline for task " + env.taskId + ": " +
|
|
188
|
+
(snap === null ? "unavailable — guard will fail closed" : snap.length + " pre-existing frame(s)"));
|
|
189
|
+
} else {
|
|
190
|
+
var marker = readGuardMarker(env);
|
|
191
|
+
if (marker && !marker.unavailable && Array.isArray(marker.frames)) {
|
|
192
|
+
guardBefore = marker.frames;
|
|
193
|
+
}
|
|
194
|
+
// else: marker missing or baseline unavailable — guardBefore stays
|
|
195
|
+
// null and the guard fails closed below.
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
var claimed = await ensureClaimed(env, state, PHASE);
|
|
200
|
+
if (claimed.type !== "CLAIMED") return claimed;
|
|
201
|
+
|
|
202
|
+
var boundary = await runWorkBoundary(env, state, {
|
|
203
|
+
phase: PHASE.name,
|
|
204
|
+
identity: PHASE.identity,
|
|
205
|
+
instructions: buildInstructions(ctx),
|
|
206
|
+
eventPreamble: buildEventPreamble(env, PHASE.name),
|
|
207
|
+
crewApiLine: true,
|
|
208
|
+
verdictStep: true,
|
|
209
|
+
});
|
|
210
|
+
if (boundary.type !== "BOUNDARY_DONE") return boundary;
|
|
211
|
+
|
|
212
|
+
var workerText = boundary.workerText;
|
|
213
|
+
var verdictPassed = boundary.verdictPassed === true;
|
|
214
|
+
|
|
215
|
+
// Wrong-layer guard (engine/docs only): only frames created during THIS
|
|
216
|
+
// attempt fail the guard — pre-existing frames are the baseline. A
|
|
217
|
+
// missing or unreadable baseline fails closed (fail-closed for retry).
|
|
218
|
+
if (guardApplicable) {
|
|
219
|
+
var after = listReproFrames(env);
|
|
220
|
+
if (guardBefore === null || after === null) {
|
|
221
|
+
return await recordGuardFailure(env, state, "baseline-unavailable", "");
|
|
222
|
+
}
|
|
223
|
+
var beforeSet = {};
|
|
224
|
+
for (var i = 0; i < guardBefore.length; i++) beforeSet[guardBefore[i]] = true;
|
|
225
|
+
var newFrames = [];
|
|
226
|
+
for (var j = 0; j < after.length; j++) {
|
|
227
|
+
if (!beforeSet[after[j]]) newFrames.push(after[j]);
|
|
228
|
+
}
|
|
229
|
+
if (newFrames.length > 0) {
|
|
230
|
+
return await recordGuardFailure(env, state, reproLayer, newFrames.slice(0, 5).join(", ").slice(0, 300));
|
|
231
|
+
}
|
|
232
|
+
log("Reproduce wrong-layer guard passed: no see-act frames from this attempt on " + reproLayer + "-layer task " + env.taskId);
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
// Closeout: record the session and event, then route. A valid FAIL is a
|
|
236
|
+
// rejection (the bug does not reproduce) followed by a park for PM
|
|
237
|
+
// attention — it never advances. A valid PASS advances to Map.
|
|
238
|
+
var status = verdictPassed ? "completed" : "rejected";
|
|
239
|
+
var summary = summarizeReport(workerText);
|
|
240
|
+
await recordPhase(env, {
|
|
241
|
+
task_id: env.taskId,
|
|
242
|
+
session: {
|
|
243
|
+
id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity,
|
|
244
|
+
step: PHASE.name, status: status, notes: summary,
|
|
245
|
+
},
|
|
246
|
+
event: { task_id: env.taskId, type: status, identity: PHASE.identity, message: "Reproduce " + status + " by " + PHASE.identity },
|
|
247
|
+
});
|
|
248
|
+
|
|
249
|
+
if (!verdictPassed) {
|
|
250
|
+
log("Reproduction failed for task " + env.taskId + " — parking for PM attention");
|
|
251
|
+
return await parkTask(env, "Reproduction failed — needs PM attention");
|
|
252
|
+
}
|
|
253
|
+
return { type: "ADVANCE", next: "Map" };
|
|
254
|
+
}
|
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
// lib/bugfix/phases/review.js — Bugfix Review phase (worker layer).
|
|
2
|
+
//
|
|
3
|
+
// Cass reviews the task branch cold against Mara's spec. Derived from
|
|
4
|
+
// workflows/bugfix.js (the Bugfix Review step).
|
|
5
|
+
//
|
|
6
|
+
// Contract:
|
|
7
|
+
// - Read: task branch state (mechanical git classification, re-done on
|
|
8
|
+
// every entry); accepted Build report (durable session notes:
|
|
9
|
+
// repo_diff:none claim, release decision); package.json version state
|
|
10
|
+
// (npm only, mechanical).
|
|
11
|
+
// - Mechanical gates (Cass never dispatched):
|
|
12
|
+
// * version-touched (npm): branch changed package.json `version`.
|
|
13
|
+
// * empty-no-work without a repo_diff:none claim: no change to review.
|
|
14
|
+
// Both synthesize a workflow-authored FAIL report flowing through the
|
|
15
|
+
// same verdict/closeout path as a real FAIL.
|
|
16
|
+
// - Branch classification and version check are direct lifecycle-script
|
|
17
|
+
// invocations (worker-layer equivalent of the source's agent ferry);
|
|
18
|
+
// unparseable/instrument failure retries in place (2x), exhaustion
|
|
19
|
+
// parks honestly (never asserts a positive claim from unparseable
|
|
20
|
+
// output).
|
|
21
|
+
// - Cass's instructions carry the branch-state-specific exam rule and the
|
|
22
|
+
// release-decision validation (npm only).
|
|
23
|
+
// - Structured verdict record: step, attempt (rework round), reviewer
|
|
24
|
+
// ("workflow" on mechanical FAIL), verdict, review_basis, grounds.
|
|
25
|
+
// - Rejection: shared Review/QA budget (spec.reworkPhases). Budget
|
|
26
|
+
// exhausted → park (worktree preserved). Otherwise REWIND to Build.
|
|
27
|
+
// - PASS → Integrate.
|
|
28
|
+
//
|
|
29
|
+
// Phase I/O contract (Phase D):
|
|
30
|
+
// read: git branch state, Build session notes (completed), package.json
|
|
31
|
+
// version state (npm)
|
|
32
|
+
// write: session (completed | rejected), event (completed | rejected),
|
|
33
|
+
// verdict row, park row (budget/instrument exhaustion)
|
|
34
|
+
// out: ADVANCE → Integrate | REWIND → Build | PARK |
|
|
35
|
+
// NEED_SPAWN / STANDBY / FAILED (boundary)
|
|
36
|
+
|
|
37
|
+
import {
|
|
38
|
+
runWorkBoundary, recordPhase, buildEventPreamble, summarizeReport,
|
|
39
|
+
ensureClaimed, log, latestSessionNotes, crewApi, lifecycle,
|
|
40
|
+
lifecycleEnvPrefix, parkTask,
|
|
41
|
+
} from "../../workflow-lib.js";
|
|
42
|
+
import {
|
|
43
|
+
parseClassifyFerry, parseVersionFerry, extractReleaseDecision, extractVerdict,
|
|
44
|
+
} from "../../extract.js";
|
|
45
|
+
|
|
46
|
+
export const PHASE = { name: "Review", identity: "cass" };
|
|
47
|
+
|
|
48
|
+
var MAX_REWORK = 2;
|
|
49
|
+
|
|
50
|
+
// buildInstructions — verbatim from workflows/bugfix.js (variable
|
|
51
|
+
// references remapped; logic and prose unchanged).
|
|
52
|
+
// ctx: { env, mapperSpec, releaseDecisionText, releaseDecision,
|
|
53
|
+
// alreadyMergedSha, reviewChangeExam, reviewBranchRule }.
|
|
54
|
+
export function buildInstructions(ctx) {
|
|
55
|
+
var env = ctx.env;
|
|
56
|
+
var instructions = "Review independently and cold. You have NOT seen any reasoning from the builder.\nDo NOT access the task dashboard, event log, or any comments. Your review is based solely on the spec and the code.\n\n" +
|
|
57
|
+
(ctx.mapperSpec ? "MAPPER'S SPEC (the builder was asked to implement exactly this):\n" + ctx.mapperSpec + "\n\n" : "Read the spec (from the task description or spec files under " + env.crewHome + "/).\n\n") +
|
|
58
|
+
ctx.reviewChangeExam +
|
|
59
|
+
"You can also read specific files in the worktree at:\n" +
|
|
60
|
+
env.worktreeHint + "/\n\n" +
|
|
61
|
+
"Check quality, correctness, and spec compliance.\n" +
|
|
62
|
+
(env.surfaceTerminal ? "TERMINAL UX REVIEW: judge the CLI surface against " + env.uxDoctrinePath + " — help accuracy, error quality, exit codes, output clarity. Reject when the bar is not met.\n" : "") +
|
|
63
|
+
(env.surfaceArtifact ? "ARTIFACT UX REVIEW: judge the rendered surface against " + env.uxDoctrinePath + " — alignment, spacing, hierarchy, composition, balance, finish, correctness. Reject when the bar is not met.\n" : "") +
|
|
64
|
+
"Check that public-affecting changes have matching public doc updates (API.md or the published API contract). If the docs are missing or inaccurate, report what is stale, then end your report with exactly this line: VERDICT: FAIL.\n" +
|
|
65
|
+
ctx.reviewBranchRule +
|
|
66
|
+
(env.publishType === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry, and versions are assigned at publish time — never in branches.\n" +
|
|
67
|
+
"MECHANICAL FACT (computed by the workflow from git — never by a reviewer): the task branch did not change package.json's `version` field. A branch that touches `version` fails Review mechanically before any reviewer is dispatched, so this is settled — do not re-check it.\n" +
|
|
68
|
+
"The accepted Build report declares: " + ctx.releaseDecisionText + ". " +
|
|
69
|
+
(ctx.releaseDecision
|
|
70
|
+
? "Validate this decision against the change: release must be 'yes' when the change is consumer-observable and 'no' when internal-only; the version_bump scope must fit the change (patch for fixes, minor for new behavior, major for breaking changes). If the decision is wrong or mis-scoped, report your notes, then end with exactly this line: VERDICT: FAIL."
|
|
71
|
+
: "The decision is missing or malformed — report 'Build report must end with release: yes|no and (when release is yes) version_bump: patch|minor|major lines', then end with exactly this line: VERDICT: FAIL.") + "\n" : "") +
|
|
72
|
+
"Write your review as plain prose — findings, then decision. End your report with exactly one line: VERDICT: PASS if the work passes, VERDICT: FAIL if it fails.";
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
return instructions;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// releaseDecisionText — the human-readable release decision (source:
|
|
79
|
+
// releaseDecisionText).
|
|
80
|
+
function releaseDecisionText(rd) {
|
|
81
|
+
if (!rd) return "no machine-readable release decision from the Build report";
|
|
82
|
+
return "release: " + rd.release + (rd.version_bump ? ", version_bump: " + rd.version_bump : " (no version_bump line)");
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
// classifyBranch — mechanical branch-state classification via the lifecycle
|
|
86
|
+
// script, run directly (worker-layer equivalent of the source's agent
|
|
87
|
+
// ferry). Retries in place on unparseable/instrument failure (2 retries);
|
|
88
|
+
// exhaustion returns {ok:false} and the caller parks honestly.
|
|
89
|
+
async function classifyBranch(env) {
|
|
90
|
+
var lastErr = "";
|
|
91
|
+
for (var attempt = 0; attempt <= 2; attempt++) {
|
|
92
|
+
var stdout, stderr;
|
|
93
|
+
try {
|
|
94
|
+
var res = await lifecycle(env, "classify-branch", [env.taskId]);
|
|
95
|
+
stdout = res.stdout; stderr = res.stderr;
|
|
96
|
+
} catch (e) {
|
|
97
|
+
lastErr = "classifier call failed: " + String((e && e.message) || e).slice(0, 120);
|
|
98
|
+
if (attempt < 2) log("classify-branch instrument failure (" + lastErr + ") — retrying in place (attempt " + (attempt + 1) + " of 2)");
|
|
99
|
+
continue;
|
|
100
|
+
}
|
|
101
|
+
var parsed = parseClassifyFerry({ stdout: stdout, stderr: stderr });
|
|
102
|
+
if (parsed.state !== null) {
|
|
103
|
+
if (parsed.diagField !== null) {
|
|
104
|
+
log("classify-branch DIAG pin (" + parsed.diagField + " field): " + parsed.diagText);
|
|
105
|
+
} else {
|
|
106
|
+
log("classify-branch: no DIAG: record-scan counter returned (scan unverifiable)");
|
|
107
|
+
}
|
|
108
|
+
return { ok: true, state: parsed.state };
|
|
109
|
+
}
|
|
110
|
+
lastErr = parsed.failReason;
|
|
111
|
+
if (attempt < 2) log("classify-branch unparseable (" + lastErr + ") — retrying in place (attempt " + (attempt + 1) + " of 2)");
|
|
112
|
+
}
|
|
113
|
+
return { ok: false, reason: lastErr };
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// checkVersion — mechanical package.json `version` check (npm only), same
|
|
117
|
+
// direct-execution shape as classifyBranch.
|
|
118
|
+
async function checkVersion(env) {
|
|
119
|
+
var lastErr = "";
|
|
120
|
+
for (var attempt = 0; attempt <= 2; attempt++) {
|
|
121
|
+
var stdout;
|
|
122
|
+
try {
|
|
123
|
+
var res = await lifecycle(env, "version-check", [env.taskId]);
|
|
124
|
+
stdout = res.stdout;
|
|
125
|
+
} catch (e) {
|
|
126
|
+
lastErr = "version-check call failed: " + String((e && e.message) || e).slice(0, 120);
|
|
127
|
+
if (attempt < 2) log("version-check instrument failure (" + lastErr + ") — retrying in place (attempt " + (attempt + 1) + " of 2)");
|
|
128
|
+
continue;
|
|
129
|
+
}
|
|
130
|
+
var parsed = parseVersionFerry({ stdout: stdout });
|
|
131
|
+
if (parsed.versionState !== null) return { ok: true, state: parsed.versionState };
|
|
132
|
+
lastErr = parsed.failReason;
|
|
133
|
+
if (attempt < 2) log("version-check unparseable (" + lastErr + ") — retrying in place (attempt " + (attempt + 1) + " of 2)");
|
|
134
|
+
}
|
|
135
|
+
return { ok: false, reason: lastErr };
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
export async function runPhase(ctx) {
|
|
139
|
+
var env = ctx.env, state = ctx.state;
|
|
140
|
+
|
|
141
|
+
var claimed = await ensureClaimed(env, state, PHASE);
|
|
142
|
+
if (claimed.type !== "CLAIMED") return claimed;
|
|
143
|
+
|
|
144
|
+
// 1. Branch classification — re-done on EVERY Review entry (rework
|
|
145
|
+
// rewinds here and the branch changed between rounds).
|
|
146
|
+
var bc = await classifyBranch(env);
|
|
147
|
+
if (!bc.ok) {
|
|
148
|
+
return await parkTask(env, "Branch state unclassifiable after 2 instrument retries (" + bc.reason + ") — the classifier instrument failed, not the branch. Human attention needed: inspect the task branch directly to determine its state — no trustworthy branch classification was produced.");
|
|
149
|
+
}
|
|
150
|
+
var branchState = bc.state;
|
|
151
|
+
var alreadyMergedSha = null;
|
|
152
|
+
if (branchState.indexOf("already-merged:") === 0) {
|
|
153
|
+
alreadyMergedSha = branchState.slice("already-merged:".length);
|
|
154
|
+
log("Branch state already-merged: " + alreadyMergedSha + " (DIAG pin present — tripwire satisfied) — Cass reviews the frozen merge diff");
|
|
155
|
+
} else {
|
|
156
|
+
log("Branch state: " + branchState);
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// 2. Accepted Build report facts (durable session notes).
|
|
160
|
+
var buildNotes = await latestSessionNotes(env, "Build", "completed");
|
|
161
|
+
var buildClaimedNoDiff = /repo_diff:\s*none/im.test(buildNotes);
|
|
162
|
+
var releaseDecision = extractReleaseDecision(buildNotes);
|
|
163
|
+
|
|
164
|
+
// 3. package.json `version` check (npm only) — mechanical git fact.
|
|
165
|
+
var versionTouched = false;
|
|
166
|
+
if (env.publishType === "npm") {
|
|
167
|
+
var vc = await checkVersion(env);
|
|
168
|
+
if (!vc.ok) {
|
|
169
|
+
return await parkTask(env, "package.json `version` state unclassifiable after 2 instrument retries (" + vc.reason + ") — the version-check instrument failed, not the branch. Human attention needed: inspect the task branch directly to determine its state — no trustworthy version classification was produced.");
|
|
170
|
+
}
|
|
171
|
+
versionTouched = (vc.state === "touched");
|
|
172
|
+
log("Review: package.json `version` " + (versionTouched ? "touched by the branch — mechanical FAIL, Cass not dispatched" : "untouched — mechanical check clean"));
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// 4. Mechanical FAIL gate — Cass is never dispatched; the synthetic
|
|
176
|
+
// report flows through the same verdict/closeout path as a real FAIL.
|
|
177
|
+
var mechanicalFailReason = versionTouched ? "version-touched"
|
|
178
|
+
: (branchState === "empty-no-work" && !buildClaimedNoDiff ? "empty-no-work" : null);
|
|
179
|
+
var mechanicalReviewFail = (mechanicalFailReason !== null);
|
|
180
|
+
|
|
181
|
+
var workerText;
|
|
182
|
+
var verdictPassed;
|
|
183
|
+
if (mechanicalReviewFail) {
|
|
184
|
+
log("Review: " + mechanicalFailReason + " — mechanical FAIL, Cass not dispatched");
|
|
185
|
+
workerText =
|
|
186
|
+
"MECHANICAL REVIEW VERDICT (written by the workflow — no reviewer was dispatched).\n" +
|
|
187
|
+
(mechanicalFailReason === "version-touched"
|
|
188
|
+
? "Package.json `version` (checked by the workflow from git): the task branch changed the `version` field. Versions are assigned at publish time — never in branches. Remove the version change.\n"
|
|
189
|
+
: "Branch state (classified by the workflow from git): empty-no-work — the task branch has no commits ahead of the integration target, " +
|
|
190
|
+
"no task-attributed merge on the target was verified, " +
|
|
191
|
+
"and the accepted Build report declared no `repo_diff: none` deliverable. " +
|
|
192
|
+
"No change was reviewed because there is no change to review. The branch is empty and no task-attributed merge was verified.\n") +
|
|
193
|
+
"Worktree: " + env.worktreeHint + "\n" +
|
|
194
|
+
"Recovery: commit the deliverable on the task branch in the worktree above and re-run Review. If the deliverable is genuinely runtime state outside the repo, declare `repo_diff: none` in the Build report instead of leaving the branch empty without a declaration.\n" +
|
|
195
|
+
"VERDICT: FAIL";
|
|
196
|
+
verdictPassed = false;
|
|
197
|
+
} else {
|
|
198
|
+
// 5. Branch-state-specific exam rule (verbatim from source).
|
|
199
|
+
var reviewChangeExam, reviewBranchRule;
|
|
200
|
+
if (branchState === "has-work") {
|
|
201
|
+
reviewChangeExam =
|
|
202
|
+
"Examine the code changes by running:\n" +
|
|
203
|
+
lifecycleEnvPrefix(env) + env.lifecycle + " inspect " + env.taskId + "\n\n" +
|
|
204
|
+
"The inspect output is authoritative: it prints the task branch's actual tip commit (TIP) and every commit ahead of the integration target (inspect prints the target name). Base your review ONLY on this output — do NOT run git log yourself to pick commits, and do NOT discuss commit hashes from any other source (they may come from stale rework rounds or a different repo).\n\n";
|
|
205
|
+
reviewBranchRule =
|
|
206
|
+
"MECHANICAL FACT (computed by the workflow from git — never by a reviewer): branch state = has-work. The branch has commits ahead of the integration target. Review their content; branch emptiness is settled and is not yours to judge.\n";
|
|
207
|
+
} else if (branchState.indexOf("already-merged:") === 0) {
|
|
208
|
+
reviewChangeExam =
|
|
209
|
+
"The change under review is the frozen merge " + alreadyMergedSha + " — it is already on the integration target (see MECHANICAL FACT below). Examine it by running:\n" +
|
|
210
|
+
"cd " + env.repoPath + " && git diff " + alreadyMergedSha + "^1 " + alreadyMergedSha + "\n\n" +
|
|
211
|
+
"That first-parent diff is the frozen, attributable change. Base your review ONLY on this diff — do NOT run git log to pick commits, and do NOT discuss commit hashes from any other source (they may come from stale rework rounds or a different repo).\n\n";
|
|
212
|
+
reviewBranchRule =
|
|
213
|
+
"MECHANICAL FACT (computed by the workflow from git — never by a reviewer): branch state = already-merged:" + alreadyMergedSha + ". The deliverable landed on the integration target via this task's own prior merge " + alreadyMergedSha + " (verified ancestor of the integration target). The branch is empty by design — emptiness is settled fact, not a finding.\n";
|
|
214
|
+
} else if (buildClaimedNoDiff) {
|
|
215
|
+
reviewChangeExam =
|
|
216
|
+
"The branch has no commits ahead of the integration target (see MECHANICAL FACT below) — there is no diff to examine. Judge the Build report's `repo_diff: none` claim on its plausibility.\n\n";
|
|
217
|
+
reviewBranchRule =
|
|
218
|
+
"MECHANICAL FACT (computed by the workflow from git — never by a reviewer): the branch has no commits ahead of the integration target and no task-attributed merge on the target was verified; the Build report declares `repo_diff: none` (no repo change). Approve ONLY if the task's deliverable is plausibly runtime state (e.g. a cron definition, scheduler change, or dashboard/config state created outside the repo). Otherwise report 'no commits ahead of the integration target and no plausible runtime-state deliverable — the branch is empty and no task-attributed merge was verified', then end your report with exactly this line: VERDICT: FAIL.\n";
|
|
219
|
+
} else {
|
|
220
|
+
reviewChangeExam = "";
|
|
221
|
+
reviewBranchRule = "";
|
|
222
|
+
}
|
|
223
|
+
ctx.mapperSpec = await latestSessionNotes(env, "Map", "completed");
|
|
224
|
+
ctx.releaseDecisionText = releaseDecisionText(releaseDecision);
|
|
225
|
+
ctx.releaseDecision = releaseDecision;
|
|
226
|
+
ctx.alreadyMergedSha = alreadyMergedSha;
|
|
227
|
+
ctx.reviewChangeExam = reviewChangeExam;
|
|
228
|
+
ctx.reviewBranchRule = reviewBranchRule;
|
|
229
|
+
|
|
230
|
+
var boundary = await runWorkBoundary(env, state, {
|
|
231
|
+
phase: PHASE.name,
|
|
232
|
+
identity: PHASE.identity,
|
|
233
|
+
instructions: buildInstructions(ctx),
|
|
234
|
+
eventPreamble: buildEventPreamble(env, PHASE.name),
|
|
235
|
+
crewApiLine: true,
|
|
236
|
+
verdictStep: true,
|
|
237
|
+
});
|
|
238
|
+
if (boundary.type !== "BOUNDARY_DONE") return boundary;
|
|
239
|
+
workerText = boundary.workerText;
|
|
240
|
+
verdictPassed = boundary.verdictPassed === true;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
// 6. Review basis (structured, for the verdict row and session notes).
|
|
244
|
+
var reviewBasis = mechanicalReviewFail ? "mechanical-fail" :
|
|
245
|
+
branchState.indexOf("already-merged:") === 0 ? "frozen-merge:" + alreadyMergedSha :
|
|
246
|
+
branchState === "has-work" ? "branch-diff" : "runtime-state-none";
|
|
247
|
+
var reviewBasisNote;
|
|
248
|
+
if (mechanicalReviewFail) {
|
|
249
|
+
reviewBasisNote = "REVIEW_BASIS: none — mechanical FAIL (" + mechanicalFailReason + "), no reviewer dispatched";
|
|
250
|
+
} else if (branchState.indexOf("already-merged:") === 0) {
|
|
251
|
+
reviewBasisNote = "REVIEW_BASIS: frozen merge " + alreadyMergedSha + " — reviewed via git diff " + alreadyMergedSha + "^1 " + alreadyMergedSha;
|
|
252
|
+
} else if (branchState === "has-work") {
|
|
253
|
+
reviewBasisNote = "REVIEW_BASIS: branch diff — reviewed via inspect output (task branch vs integration target)";
|
|
254
|
+
} else {
|
|
255
|
+
reviewBasisNote = "REVIEW_BASIS: runtime-state-none — no repo change; plausibility of Build's repo_diff: none claim";
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
// 7. Closeout with the structured verdict record.
|
|
259
|
+
var status = verdictPassed ? "completed" : "rejected";
|
|
260
|
+
var summary = reviewBasisNote + "\n" + summarizeReport(workerText);
|
|
261
|
+
await recordPhase(env, {
|
|
262
|
+
task_id: env.taskId,
|
|
263
|
+
session: {
|
|
264
|
+
id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity,
|
|
265
|
+
step: PHASE.name, status: status, notes: summary,
|
|
266
|
+
},
|
|
267
|
+
event: {
|
|
268
|
+
task_id: env.taskId, type: status, identity: PHASE.identity,
|
|
269
|
+
message: "Review " + status + " by " + PHASE.identity,
|
|
270
|
+
},
|
|
271
|
+
verdict: {
|
|
272
|
+
step: "Review",
|
|
273
|
+
attempt: state.reworkCount,
|
|
274
|
+
reviewer: mechanicalReviewFail ? "workflow" : PHASE.identity,
|
|
275
|
+
verdict: verdictPassed ? "PASS" : "FAIL",
|
|
276
|
+
review_basis: reviewBasis,
|
|
277
|
+
grounds: workerText,
|
|
278
|
+
},
|
|
279
|
+
});
|
|
280
|
+
|
|
281
|
+
if (!verdictPassed) {
|
|
282
|
+
// Shared Review/QA rework budget (spec.reworkPhases): state.reworkCount
|
|
283
|
+
// is the count BEFORE this rejection; the new count would be +1.
|
|
284
|
+
if (state.reworkCount >= MAX_REWORK) {
|
|
285
|
+
log("Shared rework budget exhausted for task " + env.taskId + " — worktree preserved at " + env.worktreePreservedHint + " for manual inspection");
|
|
286
|
+
return await parkTask(env, "Exceeded shared rework budget (" + MAX_REWORK + " total rework attempts across Review and QA) after Review rejection. Worktree preserved.");
|
|
287
|
+
}
|
|
288
|
+
log("Review rejected — bouncing to Build (rework #" + (state.reworkCount + 1) + " of " + MAX_REWORK + ")");
|
|
289
|
+
return { type: "REWIND", next: "Build" };
|
|
290
|
+
}
|
|
291
|
+
return { type: "ADVANCE", next: "Integrate" };
|
|
292
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
// lib/bugfix/phases/triage.js — Triage (sage) phase module (sandbox exit,
|
|
2
|
+
// Phase D). Import-safe: no side effects on import, bare `node` exits 0.
|
|
3
|
+
//
|
|
4
|
+
// Reads: task/project/surface facts, claim/session context.
|
|
5
|
+
// Writes: completed/rejected session + event; the machine-read
|
|
6
|
+
// `experiential: yes|no` marker line, the optional `terminal_targets:`
|
|
7
|
+
// line, and the machine-read `layer: artifact|engine|docs` line. The
|
|
8
|
+
// durable cross-run handoff is the Triage session notes (Reproduce
|
|
9
|
+
// resolves the layer from them; Capture/QA resolve experiential).
|
|
10
|
+
// Transitions: ADVANCE → Capture (a rejected Triage still advances; only
|
|
11
|
+
// Review, Integrate, and Publish have special failure paths).
|
|
12
|
+
|
|
13
|
+
import {
|
|
14
|
+
runWorkBoundary, recordPhase, buildEventPreamble, summarizeReport,
|
|
15
|
+
closeoutPassed, ensureClaimed, log,
|
|
16
|
+
} from "../../workflow-lib.js";
|
|
17
|
+
import { extractExperiential, extractLayer } from "../../extract.js";
|
|
18
|
+
|
|
19
|
+
export const PHASE = { name: "Triage", identity: "sage" };
|
|
20
|
+
|
|
21
|
+
// buildInstructions — pure function of ctx. Verbatim from
|
|
22
|
+
// workflows/bugfix.js (Triage branch): the standard triage prompt plus the
|
|
23
|
+
// bugfix workflow assignment and the LAYER FLAG block.
|
|
24
|
+
export function buildInstructions(ctx) {
|
|
25
|
+
var env = ctx.env;
|
|
26
|
+
return "Validate the task against the project's repo at " + env.repoPath + " — that exact checkout, not any other copy of the project on disk. If you run git commands, cd " + env.repoPath + " first.\nCheck clarity, note dependencies, confirm the bugfix workflow assignment.\nIf the task needs decomposition, note that in your assessment.\nReport back in plain prose — what you found.\nEXPERIENTIAL FLAG: does this task change anything a user can directly observe in the project's user-facing surface? " + env.surfaceTriageDesc + " For an artifact surface that means rendered and visible — pages, components, styles, layout, copy, visual states. For a terminal surface it means the CLI experience — command output, help text, flags, error messages, defaults. The shared UX bar is " + (env.uxDoctrinePath ? env.uxDoctrinePath + " — flag experiential when the task touches anything it covers." : "not classified for this project — flag experiential when the task touches anything user-observable in the surface described above.") + " If yes it is experiential and gets baseline evidence (plus an experiential verdict where the workflow has a QA phase). For terminal-surface tasks also declare the CLI surface to exercise: end your report with a line `terminal_targets: <comma-separated CLI commands/flags>` (machine-read; optional — falls back to the task description). End your report with exactly one line on its own, lowercase, unrephrased: experiential: yes — or experiential: no. This line is machine-read.\nLAYER FLAG: classify the bug's layer — where the reported misbehavior lives. artifact: user-facing behavior of the project's rendered artifact (something a user sees or clicks). engine: the crew's own machinery — workflows, lib scripts, shell scripts, tests, scheduler. docs: a documentation gap or error. End your report with exactly one line on its own, lowercase, unrephrased: layer: artifact — or layer: engine — or layer: docs. This line is machine-read. If the bug genuinely spans layers, pick the layer where reproduction must happen and note the ambiguity in prose.";
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export async function runPhase(ctx) {
|
|
30
|
+
var env = ctx.env, state = ctx.state;
|
|
31
|
+
var claimed = await ensureClaimed(env, state, PHASE);
|
|
32
|
+
if (claimed.type !== "CLAIMED") return claimed;
|
|
33
|
+
|
|
34
|
+
var boundary = await runWorkBoundary(env, state, {
|
|
35
|
+
phase: PHASE.name, identity: PHASE.identity,
|
|
36
|
+
instructions: buildInstructions(ctx),
|
|
37
|
+
eventPreamble: buildEventPreamble(env, PHASE.name),
|
|
38
|
+
crewApiLine: true, verdictStep: false,
|
|
39
|
+
});
|
|
40
|
+
if (boundary.type !== "BOUNDARY_DONE") return boundary;
|
|
41
|
+
|
|
42
|
+
// Triage is not a verdict step: producing output means passed.
|
|
43
|
+
var passed = closeoutPassed(boundary);
|
|
44
|
+
var workerText = boundary.workerText;
|
|
45
|
+
var summary = summarizeReport(workerText);
|
|
46
|
+
var status = passed ? "completed" : "rejected";
|
|
47
|
+
await recordPhase(env, {
|
|
48
|
+
task_id: env.taskId,
|
|
49
|
+
session: { id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity, step: PHASE.name, status: status, notes: summary },
|
|
50
|
+
event: { task_id: env.taskId, type: status, identity: PHASE.identity, message: PHASE.name + " " + status + " by " + PHASE.identity },
|
|
51
|
+
});
|
|
52
|
+
if (!passed) {
|
|
53
|
+
// A rejected Triage still advances; the rejection is recorded.
|
|
54
|
+
log("Triage rejected for task " + env.taskId + " — advancing");
|
|
55
|
+
return { type: "ADVANCE", next: "Capture", experiential: null };
|
|
56
|
+
}
|
|
57
|
+
// Stash the terminal targets and the bug layer run-locally (both are
|
|
58
|
+
// re-derivable from the Triage session notes' marker lines on a resumed
|
|
59
|
+
// run — the layer marker survives in the notes via extractMarkerLines).
|
|
60
|
+
var ttm = /^terminal_targets:\s*(.+)/im.exec(workerText || "");
|
|
61
|
+
state.memo.terminalTargets = ttm ? ttm[1].trim().slice(0, 300) : "";
|
|
62
|
+
var layer = extractLayer(workerText);
|
|
63
|
+
state.memo.layerResolved = layer || "artifact";
|
|
64
|
+
log("Triage complete for task " + env.taskId + " (layer: " + state.memo.layerResolved + ")");
|
|
65
|
+
return {
|
|
66
|
+
type: "ADVANCE", next: "Capture",
|
|
67
|
+
experiential: extractExperiential(workerText),
|
|
68
|
+
};
|
|
69
|
+
}
|