muse-crew 0.14.8 → 0.14.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/decisions/workflow-core.md +14 -7
- package/docs/ooda-report.md +21 -4
- package/lib/AGENTS.md +2 -2
- package/lib/read-ooda-verdict.js +61 -86
- package/lib/see-act.js +616 -6
- package/lib/test-worktree-backend.sh +42 -0
- package/lib/worktree-lifecycle.sh +64 -4
- package/package.json +1 -1
- package/workflows/bugfix.js +66 -17
- package/workflows/chore.js +63 -15
- package/workflows/docs.js +11 -13
- package/workflows/standard.js +66 -17
package/workflows/standard.js
CHANGED
|
@@ -139,7 +139,7 @@ if (!taskId) {
|
|
|
139
139
|
throw new Error("task_id is required in args");
|
|
140
140
|
}
|
|
141
141
|
|
|
142
|
-
// See docs/decisions/workflow-core.md#closeout-envelope: the work
|
|
142
|
+
// See docs/decisions/workflow-core.md#closeout-envelope: the schema-less work courier returns prose, consumed as a plain string; verdict extracted mechanically.
|
|
143
143
|
const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
|
|
144
144
|
function extractVerdict(workerText) {
|
|
145
145
|
// The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
|
|
@@ -229,7 +229,6 @@ function buildVerdictReaskPrompt(stepName, workerText) {
|
|
|
229
229
|
"Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
|
|
230
230
|
"if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
|
|
231
231
|
"Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
|
|
232
|
-
"Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
|
|
233
232
|
"The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
|
|
234
233
|
}
|
|
235
234
|
async function reaskVerdict(stepName, reworkSuffix, workerText) {
|
|
@@ -302,8 +301,8 @@ var TOOL_CHECK_PREAMBLE =
|
|
|
302
301
|
"3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
|
|
303
302
|
"Then do the assignment below.\n\n";
|
|
304
303
|
function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
|
|
305
|
-
// reason: "discarded" (the runtime threw the output away
|
|
306
|
-
//
|
|
304
|
+
// reason: "discarded" (the runtime threw the output away — the JSON-candidate
|
|
305
|
+
// scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
|
|
307
306
|
// nothing usable), "no-tools" (the worker's TOOL CHECK reported
|
|
308
307
|
// artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
|
|
309
308
|
// reported shell_transport: unavailable).
|
|
@@ -314,11 +313,11 @@ function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason)
|
|
|
314
313
|
? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
|
|
315
314
|
: reason === "no-transport"
|
|
316
315
|
? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
|
|
317
|
-
: "your previous attempt's output could not
|
|
316
|
+
: "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
|
|
318
317
|
return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
|
|
319
318
|
"First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
|
|
320
319
|
"if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
|
|
321
|
-
"Then return your report
|
|
320
|
+
"Then return your report in exactly the shape specified above.";
|
|
322
321
|
}
|
|
323
322
|
// parseClassifyFerry — room #26 blocker 42 (2026-09-21). Shape-constrained
|
|
324
323
|
// ferry parser for the classify-branch agent() call. The schema buys SHAPE,
|
|
@@ -791,8 +790,52 @@ if (CAPTURE_INDEX < 0) throw new Error("STEPS missing 'Capture' step");
|
|
|
791
790
|
const REWORK_STEP = STEPS[BUILD_INDEX].name;
|
|
792
791
|
// Shared rework budget: Review and QA rejections draw from the SAME pool of 2.
|
|
793
792
|
// E.g. 2 Review bounces + 1 QA bounce = 3 total > budget -> task blocks.
|
|
793
|
+
// Blocker 46 (2026-09-22): the budget is task-durable, not per-run. A run
|
|
794
|
+
// ends on every operational failure (agent timeout, worktree-confinement
|
|
795
|
+
// failure) and the dispatcher launches a fresh run — the old `let ... = 0`
|
|
796
|
+
// reset the counter on each launch, so a task whose runs kept ending
|
|
797
|
+
// operationally between rejections reworked unboundedly while the designed
|
|
798
|
+
// park stayed unreachable (room #29 J1: 3 rejections across 4 runs, never
|
|
799
|
+
// parked). The counter is seeded from the task's durable event stream at
|
|
800
|
+
// run start (seedReworkCount, after the pin block); in-run increments are
|
|
801
|
+
// unchanged.
|
|
794
802
|
const MAX_TOTAL_REWORK = 2;
|
|
795
803
|
let totalReworkCount = 0;
|
|
804
|
+
// Blocker 46: the first-claim test used totalReworkCount === 0 as a proxy
|
|
805
|
+
// for "this run is entering its first step". The counter is now seeded from
|
|
806
|
+
// task history, so the proxy lies on re-run launches — use an explicit
|
|
807
|
+
// run-local flag instead. The first claim (self-claim with raced-duplicate
|
|
808
|
+
// stand-down) must fire exactly once per run.
|
|
809
|
+
let firstClaimDone = false;
|
|
810
|
+
|
|
811
|
+
// Blocker 46: count verdict-level `rejected` events (Review/QA rejections —
|
|
812
|
+
// record-phase writes the event type equal to the session status) in the
|
|
813
|
+
// task's durable event stream. Operational failure ends are `failed`, never
|
|
814
|
+
// `rejected`, so they must not spend retry headroom. A failed seed must not
|
|
815
|
+
// strand the run: start at 0 and let the in-run budget govern this run; the
|
|
816
|
+
// seed is retried on the next run.
|
|
817
|
+
async function seedReworkCount() {
|
|
818
|
+
try {
|
|
819
|
+
var seed = await agent(
|
|
820
|
+
"Read this task's event history.\n" +
|
|
821
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId, limit: 100 }) + "\n" +
|
|
822
|
+
"Count the events whose type is exactly \"rejected\".\n" +
|
|
823
|
+
"Return JSON { \"rejected_count\": <integer> } and nothing else.",
|
|
824
|
+
{
|
|
825
|
+
key: "rework-seed-" + taskId,
|
|
826
|
+
label: "Seeding shared rework budget from task history",
|
|
827
|
+
schema: { type: "object", properties: { rejected_count: { type: "number" } }, required: ["rejected_count"] }
|
|
828
|
+
}
|
|
829
|
+
);
|
|
830
|
+
var n = Math.floor(Number(seed && seed.rejected_count) || 0);
|
|
831
|
+
if (!(n >= 0)) n = 0;
|
|
832
|
+
if (n > 0) log("Shared rework budget seeded from task history: " + n + " prior rejection(s)");
|
|
833
|
+
return n;
|
|
834
|
+
} catch (e) {
|
|
835
|
+
log("Rework-budget seed unavailable (" + ((e && e.message) || String(e)).slice(0, 120) + ") — starting at 0");
|
|
836
|
+
return 0;
|
|
837
|
+
}
|
|
838
|
+
}
|
|
796
839
|
let rejectionNotes = inputs.rejection_notes || "";
|
|
797
840
|
let mapperSpec = "";
|
|
798
841
|
// Visual verdict state: Sage's experiential flag (true/false/null until the
|
|
@@ -900,6 +943,11 @@ log("Lifecycle scripts pinned to " + RUN_LIB);
|
|
|
900
943
|
CREW_API = CREW_API_PINNED;
|
|
901
944
|
log("Crew API pinned to " + CREW_API);
|
|
902
945
|
|
|
946
|
+
// Blocker 46: seed the task-durable rework budget from the task's event
|
|
947
|
+
// history before the phase loop. Pinned copy, so the seed reads the same
|
|
948
|
+
// release this run executes.
|
|
949
|
+
totalReworkCount = await seedReworkCount();
|
|
950
|
+
|
|
903
951
|
// Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
|
|
904
952
|
// minted at this run's first claim (never a PID — short-lived agent PIDs
|
|
905
953
|
// made every concurrent acquire take the stale path). Set exactly once on
|
|
@@ -908,7 +956,7 @@ let lockHolder = taskId;
|
|
|
908
956
|
|
|
909
957
|
while (i < STEPS.length) {
|
|
910
958
|
const step = STEPS[i];
|
|
911
|
-
const isFirstClaim = (i === startStepIndex &&
|
|
959
|
+
const isFirstClaim = (i === startStepIndex && !firstClaimDone);
|
|
912
960
|
|
|
913
961
|
phase(step.name);
|
|
914
962
|
log(step.name + " step (" + step.identity + ") for task " + taskId);
|
|
@@ -1073,6 +1121,7 @@ while (i < STEPS.length) {
|
|
|
1073
1121
|
return { status: "duplicate", task_id: taskId, reason: "task already claimed by another run" };
|
|
1074
1122
|
}
|
|
1075
1123
|
activeSessionId = claimResult.session_id;
|
|
1124
|
+
firstClaimDone = true; // blocker 46: exactly once per run
|
|
1076
1125
|
} else {
|
|
1077
1126
|
const claimResult = await agent(
|
|
1078
1127
|
"Claim a session for this task step.\n" +
|
|
@@ -1970,14 +2019,15 @@ while (i < STEPS.length) {
|
|
|
1970
2019
|
"Public docs (API.md, README, published action schemas) are NOT source code — read them freely, exactly as a user would.\n\n" +
|
|
1971
2020
|
"STEP 1: Experiential visual inspection — drive the artifact as a user would, one browser step at a time.\n" +
|
|
1972
2021
|
"You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
|
|
2022
|
+
"SESSION PROTOCOL (multi-step flows): one-shot invocations launch a fresh browser each time, so they cannot drive flows where a later step needs an earlier step's living page state (open a deck, then click Study — the second step needs the first step's page state). For those, hold ONE browser alive: node " + crewHome + "/current/lib/see-act.js session-start --name qa --url http://localhost:<N>/ --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — launches Chromium once in a detached daemon (its control server binds 127.0.0.1 only; the daemon exits after 10 min with no act). The --log path is the SAME ooda-log.jsonl the QA guard reads: every session-act appends its own JSON line in the exact step schema (step, attempt, action, args, exit, screenshot, observation), so you do NOT call append-ooda-step.js for session acts — READ each act's JSON result (the screenshot field is the path to READ with your read tool) and decide the next action. Then drive it: node " + crewHome + "/current/lib/see-act.js session-act --name qa --attempt \"1\" <aria|shot|click|scroll|type> [args] — one action inside the living page. Prefix EVERY session-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one). When done: node " + crewHome + "/current/lib/see-act.js session-end --name qa — closes the browser, prints the session summary, and keeps the session dir as evidence. If session-start exits 3 with a \"NOT POSSIBLE: <reason>\" string, the environment cannot drive a browser: report NOT POSSIBLE: <reason> and judge what you can.\n" +
|
|
1973
2023
|
"Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
|
|
1974
2024
|
"a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and continue with the mechanical checks.\n" +
|
|
1975
2025
|
"b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
|
|
1976
2026
|
"c. Bounded see-act loop, at most 8 steps: drive the artifact as a user would, one browser step at a time. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
|
|
1977
|
-
"c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
|
|
2027
|
+
"c2. After EVERY one-shot see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). (Session acts are self-logging — session-act already appended its line; for those, READ the result instead of re-logging.) Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
|
|
1978
2028
|
"c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
|
|
1979
2029
|
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
1980
|
-
"d.
|
|
2030
|
+
"d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the deck, then click Study, then judge the study view. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and judge what you did reach.\n" +
|
|
1981
2031
|
"e. Judge as a user against the task description: does the change render correctly? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
|
|
1982
2032
|
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
1983
2033
|
"Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
|
|
@@ -2061,12 +2111,12 @@ while (i < STEPS.length) {
|
|
|
2061
2111
|
"The returned events are filtered to this task. They contain notes and decisions from prior phases.\n\n";
|
|
2062
2112
|
}
|
|
2063
2113
|
|
|
2064
|
-
// Work agent returns the
|
|
2065
|
-
//
|
|
2066
|
-
//
|
|
2067
|
-
//
|
|
2068
|
-
//
|
|
2069
|
-
// text by extractVerdict below — never by an agent.
|
|
2114
|
+
// Work agent returns prose on the schema-less courier; the workflow
|
|
2115
|
+
// consumes the return as a plain string (blocker 43, 2026-09-22: the
|
|
2116
|
+
// prompt names no JSON envelope — nothing parses one, and the runtime's
|
|
2117
|
+
// JSON-candidate scan discards brace-shaped prose, which is what the
|
|
2118
|
+
// transport retry guards). The verdict is still extracted deterministically
|
|
2119
|
+
// from the report text by extractVerdict below — never by an agent.
|
|
2070
2120
|
var workPromptBase =
|
|
2071
2121
|
TOOL_CHECK_PREAMBLE +
|
|
2072
2122
|
"Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
|
|
@@ -2079,8 +2129,7 @@ while (i < STEPS.length) {
|
|
|
2079
2129
|
"\n## Instructions\n\n" + eventPreamble + instructions + "\n\n" +
|
|
2080
2130
|
"CONSTRAINT: Do NOT call logevent or upsertagentsession — the workflow handles all phase tracking after your step completes.\n\n" +
|
|
2081
2131
|
"Stay in character. Do the work thoroughly.\n\n" +
|
|
2082
|
-
"
|
|
2083
|
-
"The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
2132
|
+
"Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
2084
2133
|
var workKeyBase = "work-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "");
|
|
2085
2134
|
var workerResult = null;
|
|
2086
2135
|
var workAttempts = [];
|