muse-crew 0.14.8 → 0.14.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -139,7 +139,7 @@ if (!taskId) {
139
139
  throw new Error("task_id is required in args");
140
140
  }
141
141
 
142
- // See docs/decisions/workflow-core.md#closeout-envelope: the work agent returns the runtime's native envelope; verdict extracted mechanically.
142
+ // See docs/decisions/workflow-core.md#closeout-envelope: the schema-less work courier returns prose, consumed as a plain string; verdict extracted mechanically.
143
143
  const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
144
144
  function extractVerdict(workerText) {
145
145
  // The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
@@ -229,7 +229,6 @@ function buildVerdictReaskPrompt(stepName, workerText) {
229
229
  "Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
230
230
  "if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
231
231
  "Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
232
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
233
232
  "The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
234
233
  }
235
234
  async function reaskVerdict(stepName, reworkSuffix, workerText) {
@@ -302,8 +301,8 @@ var TOOL_CHECK_PREAMBLE =
302
301
  "3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
303
302
  "Then do the assignment below.\n\n";
304
303
  function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
305
- // reason: "discarded" (the runtime threw the output away - it could not be
306
- // machine-read), "empty" (agent() returned without throwing but produced
304
+ // reason: "discarded" (the runtime threw the output away — the JSON-candidate
305
+ // scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
307
306
  // nothing usable), "no-tools" (the worker's TOOL CHECK reported
308
307
  // artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
309
308
  // reported shell_transport: unavailable).
@@ -314,11 +313,11 @@ function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason)
314
313
  ? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
315
314
  : reason === "no-transport"
316
315
  ? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
317
- : "your previous attempt's output could not be machine-read as JSON and was discarded";
316
+ : "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
318
317
  return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
319
318
  "First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
320
319
  "if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
321
- "Then return your report as JSON in exactly the shape specified above.";
320
+ "Then return your report in exactly the shape specified above.";
322
321
  }
323
322
  // parseClassifyFerry — room #26 blocker 42 (2026-09-21). Shape-constrained
324
323
  // ferry parser for the classify-branch agent() call. The schema buys SHAPE,
@@ -791,8 +790,52 @@ if (CAPTURE_INDEX < 0) throw new Error("STEPS missing 'Capture' step");
791
790
  const REWORK_STEP = STEPS[BUILD_INDEX].name;
792
791
  // Shared rework budget: Review and QA rejections draw from the SAME pool of 2.
793
792
  // E.g. 2 Review bounces + 1 QA bounce = 3 total > budget -> task blocks.
793
+ // Blocker 46 (2026-09-22): the budget is task-durable, not per-run. A run
794
+ // ends on every operational failure (agent timeout, worktree-confinement
795
+ // failure) and the dispatcher launches a fresh run — the old `let ... = 0`
796
+ // reset the counter on each launch, so a task whose runs kept ending
797
+ // operationally between rejections reworked unboundedly while the designed
798
+ // park stayed unreachable (room #29 J1: 3 rejections across 4 runs, never
799
+ // parked). The counter is seeded from the task's durable event stream at
800
+ // run start (seedReworkCount, after the pin block); in-run increments are
801
+ // unchanged.
794
802
  const MAX_TOTAL_REWORK = 2;
795
803
  let totalReworkCount = 0;
804
+ // Blocker 46: the first-claim test used totalReworkCount === 0 as a proxy
805
+ // for "this run is entering its first step". The counter is now seeded from
806
+ // task history, so the proxy lies on re-run launches — use an explicit
807
+ // run-local flag instead. The first claim (self-claim with raced-duplicate
808
+ // stand-down) must fire exactly once per run.
809
+ let firstClaimDone = false;
810
+
811
+ // Blocker 46: count verdict-level `rejected` events (Review/QA rejections —
812
+ // record-phase writes the event type equal to the session status) in the
813
+ // task's durable event stream. Operational failure ends are `failed`, never
814
+ // `rejected`, so they must not spend retry headroom. A failed seed must not
815
+ // strand the run: start at 0 and let the in-run budget govern this run; the
816
+ // seed is retried on the next run.
817
+ async function seedReworkCount() {
818
+ try {
819
+ var seed = await agent(
820
+ "Read this task's event history.\n" +
821
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId, limit: 100 }) + "\n" +
822
+ "Count the events whose type is exactly \"rejected\".\n" +
823
+ "Return JSON { \"rejected_count\": <integer> } and nothing else.",
824
+ {
825
+ key: "rework-seed-" + taskId,
826
+ label: "Seeding shared rework budget from task history",
827
+ schema: { type: "object", properties: { rejected_count: { type: "number" } }, required: ["rejected_count"] }
828
+ }
829
+ );
830
+ var n = Math.floor(Number(seed && seed.rejected_count) || 0);
831
+ if (!(n >= 0)) n = 0;
832
+ if (n > 0) log("Shared rework budget seeded from task history: " + n + " prior rejection(s)");
833
+ return n;
834
+ } catch (e) {
835
+ log("Rework-budget seed unavailable (" + ((e && e.message) || String(e)).slice(0, 120) + ") — starting at 0");
836
+ return 0;
837
+ }
838
+ }
796
839
  let rejectionNotes = inputs.rejection_notes || "";
797
840
  let mapperSpec = "";
798
841
  // Visual verdict state: Sage's experiential flag (true/false/null until the
@@ -900,6 +943,11 @@ log("Lifecycle scripts pinned to " + RUN_LIB);
900
943
  CREW_API = CREW_API_PINNED;
901
944
  log("Crew API pinned to " + CREW_API);
902
945
 
946
+ // Blocker 46: seed the task-durable rework budget from the task's event
947
+ // history before the phase loop. Pinned copy, so the seed reads the same
948
+ // release this run executes.
949
+ totalReworkCount = await seedReworkCount();
950
+
903
951
  // Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
904
952
  // minted at this run's first claim (never a PID — short-lived agent PIDs
905
953
  // made every concurrent acquire take the stale path). Set exactly once on
@@ -908,7 +956,7 @@ let lockHolder = taskId;
908
956
 
909
957
  while (i < STEPS.length) {
910
958
  const step = STEPS[i];
911
- const isFirstClaim = (i === startStepIndex && totalReworkCount === 0);
959
+ const isFirstClaim = (i === startStepIndex && !firstClaimDone);
912
960
 
913
961
  phase(step.name);
914
962
  log(step.name + " step (" + step.identity + ") for task " + taskId);
@@ -1073,6 +1121,7 @@ while (i < STEPS.length) {
1073
1121
  return { status: "duplicate", task_id: taskId, reason: "task already claimed by another run" };
1074
1122
  }
1075
1123
  activeSessionId = claimResult.session_id;
1124
+ firstClaimDone = true; // blocker 46: exactly once per run
1076
1125
  } else {
1077
1126
  const claimResult = await agent(
1078
1127
  "Claim a session for this task step.\n" +
@@ -1970,14 +2019,15 @@ while (i < STEPS.length) {
1970
2019
  "Public docs (API.md, README, published action schemas) are NOT source code — read them freely, exactly as a user would.\n\n" +
1971
2020
  "STEP 1: Experiential visual inspection — drive the artifact as a user would, one browser step at a time.\n" +
1972
2021
  "You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
2022
+ "SESSION PROTOCOL (multi-step flows): one-shot invocations launch a fresh browser each time, so they cannot drive flows where a later step needs an earlier step's living page state (open a deck, then click Study — the second step needs the first step's page state). For those, hold ONE browser alive: node " + crewHome + "/current/lib/see-act.js session-start --name qa --url http://localhost:<N>/ --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — launches Chromium once in a detached daemon (its control server binds 127.0.0.1 only; the daemon exits after 10 min with no act). The --log path is the SAME ooda-log.jsonl the QA guard reads: every session-act appends its own JSON line in the exact step schema (step, attempt, action, args, exit, screenshot, observation), so you do NOT call append-ooda-step.js for session acts — READ each act's JSON result (the screenshot field is the path to READ with your read tool) and decide the next action. Then drive it: node " + crewHome + "/current/lib/see-act.js session-act --name qa --attempt \"1\" <aria|shot|click|scroll|type> [args] — one action inside the living page. Prefix EVERY session-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one). When done: node " + crewHome + "/current/lib/see-act.js session-end --name qa — closes the browser, prints the session summary, and keeps the session dir as evidence. If session-start exits 3 with a \"NOT POSSIBLE: <reason>\" string, the environment cannot drive a browser: report NOT POSSIBLE: <reason> and judge what you can.\n" +
1973
2023
  "Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
1974
2024
  "a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and continue with the mechanical checks.\n" +
1975
2025
  "b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
1976
2026
  "c. Bounded see-act loop, at most 8 steps: drive the artifact as a user would, one browser step at a time. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
1977
- "c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
2027
+ "c2. After EVERY one-shot see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). (Session acts are self-logging — session-act already appended its line; for those, READ the result instead of re-logging.) Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
1978
2028
  "c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
1979
2029
  "c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
1980
- "d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the task's change needs such a sequence, report NOT POSSIBLE for that part and judge what you can.\n" +
2030
+ "d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the deck, then click Study, then judge the study view. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and judge what you did reach.\n" +
1981
2031
  "e. Judge as a user against the task description: does the change render correctly? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
1982
2032
  "f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
1983
2033
  "Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
@@ -2061,12 +2111,12 @@ while (i < STEPS.length) {
2061
2111
  "The returned events are filtered to this task. They contain notes and decisions from prior phases.\n\n";
2062
2112
  }
2063
2113
 
2064
- // Work agent returns the runtime's native envelope {"status": "ok",
2065
- // "result": "<prose>"} with no schema. The runtime requires JSON output;
2066
- // the envelope is its own documented shape, so there is nothing for the
2067
- // agent to improvise. The workflow receives the prose report as a plain
2068
- // string. The verdict is still extracted deterministically from the report
2069
- // text by extractVerdict below — never by an agent.
2114
+ // Work agent returns prose on the schema-less courier; the workflow
2115
+ // consumes the return as a plain string (blocker 43, 2026-09-22: the
2116
+ // prompt names no JSON envelope — nothing parses one, and the runtime's
2117
+ // JSON-candidate scan discards brace-shaped prose, which is what the
2118
+ // transport retry guards). The verdict is still extracted deterministically
2119
+ // from the report text by extractVerdict below — never by an agent.
2070
2120
  var workPromptBase =
2071
2121
  TOOL_CHECK_PREAMBLE +
2072
2122
  "Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
@@ -2079,8 +2129,7 @@ while (i < STEPS.length) {
2079
2129
  "\n## Instructions\n\n" + eventPreamble + instructions + "\n\n" +
2080
2130
  "CONSTRAINT: Do NOT call logevent or upsertagentsession — the workflow handles all phase tracking after your step completes.\n\n" +
2081
2131
  "Stay in character. Do the work thoroughly.\n\n" +
2082
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your report here\"}. " +
2083
- "The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
2132
+ "Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
2084
2133
  var workKeyBase = "work-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "");
2085
2134
  var workerResult = null;
2086
2135
  var workAttempts = [];