muse-crew 0.14.8 → 0.14.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -45,6 +45,48 @@ grep -q "^path=$T/.worktrees/$tid\$" "$T/.worktrees/.registry/$tid" || fail "pre
45
45
  out=$(bash "$LIFECYCLE" prepare "$tid") || fail "prepare (reuse) exited non-zero"
46
46
  echo "$out" | grep -q '^REUSED' || fail "prepare: expected REUSED on second run: $out"
47
47
 
48
+ # Blocker 45 (2026-09-22, room #29 J1): prepare must verify its REUSED
49
+ # invariant. An agent-improvised branch (raw git, bypassing prepare) gets
50
+ # REPAIRED — registry re-registered under the actual branch, never deleted —
51
+ # never silently REUSED.
52
+ tid2="backendtest45"
53
+ out=$(bash "$LIFECYCLE" prepare "$tid2") || fail "prepare (45) exited non-zero: $out"
54
+ echo "$out" | grep -q '^CREATED' || fail "prepare (45): expected CREATED, got: $out"
55
+ # The agent improvises: checks out a branch outside the task/ namespace and
56
+ # commits there, bypassing prepare entirely.
57
+ git -C "$T/.worktrees/$tid2" checkout -qb crew-9eda33ca
58
+ echo improvised >> "$T/.worktrees/$tid2/f.txt"
59
+ (git -C "$T/.worktrees/$tid2" commit -qam improvised) || fail "improvised commit failed"
60
+ out=$(bash "$LIFECYCLE" prepare "$tid2") || fail "prepare (45 repair) exited non-zero: $out"
61
+ echo "$out" | grep -q '^REPAIRED' || fail "prepare: expected REPAIRED on improvised branch, got: $out"
62
+ echo "$out" | grep -q '^REUSED' && fail "prepare: must not print REUSED when the branch invariant fails"
63
+ # The agent's branch survives; the registry attributes the repair honestly.
64
+ git -C "$T" rev-parse --verify "crew-9eda33ca" >/dev/null 2>&1 || fail "repair: agent branch deleted"
65
+ grep -q '^branch=crew-9eda33ca$' "$T/.worktrees/.registry/$tid2" || fail "repair: registry branch not re-registered"
66
+ grep -q '^task_id=backendtest45$' "$T/.worktrees/.registry/$tid2" || fail "repair: registry lost task attribution"
67
+ base_before=$(grep '^base_commit=' "$T/.worktrees/.registry/$tid2")
68
+ # The repair is idempotent: a second prepare repairs again (no spurious
69
+ # REUSED), preserving the registry's other fields.
70
+ out=$(bash "$LIFECYCLE" prepare "$tid2") || fail "prepare (45 second repair) exited non-zero: $out"
71
+ echo "$out" | grep -q '^REPAIRED' || fail "prepare: expected idempotent REPAIRED, got: $out"
72
+ base_after=$(grep '^base_commit=' "$T/.worktrees/.registry/$tid2")
73
+ [ "$base_before" = "$base_after" ] || fail "repair: re-register must preserve base_commit"
74
+ # A worktree back on the canonical branch is REUSED with no repair.
75
+ git -C "$T/.worktrees/$tid2" checkout -q "task/$tid2"
76
+ out=$(bash "$LIFECYCLE" prepare "$tid2") || fail "prepare (45 canonical) exited non-zero: $out"
77
+ echo "$out" | grep -q '^REUSED' || fail "prepare: expected REUSED on canonical branch, got: $out"
78
+ echo "$out" | grep -q '^REPAIRED' && fail "prepare: no spurious REPAIRED on a canonical worktree"
79
+ # resolve-branch falls back to the worktree's actual checked-out branch when
80
+ # the registry is absent and no canonical/prefix ref matches — the
81
+ # classifier's mechanical truth of last resort (not "branch does not exist").
82
+ git -C "$T/.worktrees/$tid2" checkout -qb crew-deadbeef
83
+ rm -f "$T/.worktrees/.registry/$tid2"
84
+ git -C "$T" branch -D -q "task/$tid2"
85
+ resolved=$(bash "$LIFECYCLE" resolve-branch "$tid2") || fail "resolve-branch (45) exited non-zero"
86
+ [ "$resolved" = "crew-deadbeef" ] || fail "resolve-branch: expected the actual branch, got: $resolved"
87
+ bash "$LIFECYCLE" cleanup "$tid2" >/dev/null || fail "cleanup (45) failed"
88
+ [ -d "$T/.worktrees/$tid2" ] && fail "cleanup (45): worktree still exists"
89
+
48
90
  # inspect: diffs the task branch against main
49
91
  echo change >> "$T/.worktrees/$tid/f.txt"
50
92
  (cd "$T/.worktrees/$tid" && git commit -qam work)
@@ -12,11 +12,23 @@
12
12
  # bypassing prepare — task 4e1a1bba, 2026-09-17), resolve_branch() falls
13
13
  # back to enumerating refs: the canonical task/<full-id> ref wins, then a
14
14
  # task/<id-prefix> ref is tolerated (matched by strict prefix, never
15
- # created). Canonical stays task/<full-id>; ambiguous prefixes fail closed.
15
+ # created), then the existing worktree's actual checked-out branch
16
+ # (blocker 45, 2026-09-22 — an agent-improvised branch outside the task/
17
+ # namespace is invisible to the prefix scan; the classifier must see the
18
+ # world, not "branch does not exist"). Canonical stays task/<full-id>;
19
+ # ambiguous prefixes fail closed.
20
+ #
21
+ # prepare's reuse contract (blocker 45, 2026-09-22): an existing worktree
22
+ # prints REUSED only when it is checked out on the canonical task/<id>
23
+ # branch. A worktree checked out on another branch (an agent improvised
24
+ # with raw git) is repaired — the registry is re-registered under the
25
+ # actual branch (the branch is never deleted: its commits may be the
26
+ # task's work) — and prepare prints REPAIRED instead of REUSED, so the
27
+ # repair is attributable in the record.
16
28
  #
17
29
  # Commands:
18
30
  # validate <task_id> — check task ID is safe for branches/paths
19
- # prepare <task_id> — create worktree + branch, register ownership
31
+ # prepare <task_id> — create worktree + branch, register ownership (existing worktree: REUSED, or REPAIRED when re-registered under an improvised branch)
20
32
  # resolve-branch <task_id> — print the resolved task branch name
21
33
  # inspect <task_id> — diff against the integration target for review
22
34
  # classify-branch <task_id> — mechanical branch-state classification for Review:
@@ -136,6 +148,17 @@ resolve_branch() {
136
148
  echo "ERROR: task id '$task_id' matches multiple task branches: ${matches[*]} — ambiguous, refusing to guess" >&2
137
149
  return 1
138
150
  fi
151
+ # Blocker 45 (2026-09-22): the worktree's actual checkout is the mechanical
152
+ # truth of last resort. An agent-improvised branch (outside the task/
153
+ # namespace) is invisible to the prefix scan above — the classifier must
154
+ # see the world, not "branch does not exist".
155
+ if [ -d "$WORKTREE_DIR/$task_id" ]; then
156
+ local actual
157
+ if actual=$(git -C "$WORKTREE_DIR/$task_id" rev-parse --abbrev-ref HEAD 2>/dev/null) && [ -n "$actual" ]; then
158
+ printf '%s' "$actual"
159
+ return 0
160
+ fi
161
+ fi
139
162
  printf '%s' "$canonical"
140
163
  }
141
164
 
@@ -239,6 +262,25 @@ unregister() {
239
262
  rm -f "$REGISTRY_DIR/$1"
240
263
  }
241
264
 
265
+ # Blocker 45 (2026-09-22): repair the registry's branch attribution without
266
+ # touching anything else. An existing entry keeps its task_id, base_commit,
267
+ # and created fields — only branch= is rewritten. A missing entry is
268
+ # registered fresh (base_commit "unknown" — honest, not fabricated). The
269
+ # agent's branch is never deleted: its commits may be the task's work.
270
+ reregister_branch() {
271
+ local task_id="$1" branch="$2"
272
+ local reg="$REGISTRY_DIR/$task_id"
273
+ mkdir -p "$REGISTRY_DIR"
274
+ if [ -f "$reg" ] && grep -q '^branch=' "$reg"; then
275
+ # Escape sed-replacement metacharacters (branch names can carry &).
276
+ local esc=${branch//\\/\\\\}
277
+ esc=${esc//&/\\&}
278
+ sed -i "s|^branch=.*|branch=$esc|" "$reg"
279
+ else
280
+ register "$task_id" "unknown" "$branch" "$WORKTREE_DIR/$task_id"
281
+ fi
282
+ }
283
+
242
284
  # --- commands ---
243
285
 
244
286
  cmd_validate() {
@@ -256,9 +298,27 @@ cmd_prepare() {
256
298
 
257
299
  # Reuse existing worktree
258
300
  if [ -d "$WORKTREE_DIR/$task_id" ]; then
259
- echo "REUSED: worktree exists at .worktrees/$task_id"
260
- local head
301
+ local head actual_branch canonical_branch
261
302
  head=$(cd "$WORKTREE_DIR/$task_id" && git rev-parse HEAD)
303
+ if ! actual_branch=$(cd "$WORKTREE_DIR/$task_id" && git rev-parse --abbrev-ref HEAD 2>/dev/null); then
304
+ echo "ERROR: worktree .worktrees/$task_id exists but its checked-out branch is unreadable — refusing to REUSE an unverifiable worktree"
305
+ return 1
306
+ fi
307
+ canonical_branch="task/$task_id"
308
+ # Blocker 45 (2026-09-22): REUSED asserted an invariant it never checked —
309
+ # the worktree is on the task's canonical branch. Room #29: a Build agent
310
+ # improvised branch crew-9eda33ca; the retry's prepare REUSED it blindly
311
+ # and Review's classify-branch parked on "branch does not exist". Verify
312
+ # the invariant; on mismatch repair the registry (never delete the
313
+ # agent's branch — its commits may be the task's work) and attribute the
314
+ # repair with REPAIRED instead of REUSED.
315
+ if [ "$actual_branch" != "$canonical_branch" ]; then
316
+ reregister_branch "$task_id" "$actual_branch"
317
+ echo "REPAIRED: worktree .worktrees/$task_id checked out on branch $actual_branch (expected $canonical_branch); registry re-registered"
318
+ echo "HEAD: $head"
319
+ return 0
320
+ fi
321
+ echo "REUSED: worktree exists at .worktrees/$task_id"
262
322
  echo "HEAD: $head"
263
323
  return 0
264
324
  fi
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "muse-crew",
3
- "version": "0.14.8",
3
+ "version": "0.14.10",
4
4
  "description": "Opinionated orchestration for Muse — workflows, identities, and tooling for autonomous software development.",
5
5
  "license": "UNLICENSED",
6
6
  "private": false,
@@ -131,7 +131,7 @@ if (!taskId) {
131
131
  throw new Error("task_id is required in args");
132
132
  }
133
133
 
134
- // See docs/decisions/workflow-core.md#closeout-envelope: the work agent returns the runtime's native envelope; verdict extracted mechanically.
134
+ // See docs/decisions/workflow-core.md#closeout-envelope: the schema-less work courier returns prose, consumed as a plain string; verdict extracted mechanically.
135
135
  const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
136
136
  function extractVerdict(workerText) {
137
137
  // The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
@@ -221,7 +221,6 @@ function buildVerdictReaskPrompt(stepName, workerText) {
221
221
  "Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
222
222
  "if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
223
223
  "Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
224
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
225
224
  "The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
226
225
  }
227
226
  async function reaskVerdict(stepName, reworkSuffix, workerText) {
@@ -294,8 +293,8 @@ var TOOL_CHECK_PREAMBLE =
294
293
  "3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
295
294
  "Then do the assignment below.\n\n";
296
295
  function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
297
- // reason: "discarded" (the runtime threw the output away - it could not be
298
- // machine-read), "empty" (agent() returned without throwing but produced
296
+ // reason: "discarded" (the runtime threw the output away — the JSON-candidate
297
+ // scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
299
298
  // nothing usable), "no-tools" (the worker's TOOL CHECK reported
300
299
  // artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
301
300
  // reported shell_transport: unavailable).
@@ -306,11 +305,11 @@ function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason)
306
305
  ? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
307
306
  : reason === "no-transport"
308
307
  ? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
309
- : "your previous attempt's output could not be machine-read as JSON and was discarded";
308
+ : "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
310
309
  return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
311
310
  "First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
312
311
  "if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
313
- "Then return your report as JSON in exactly the shape specified above.";
312
+ "Then return your report in exactly the shape specified above.";
314
313
  }
315
314
  // parseClassifyFerry — room #26 blocker 42 (2026-09-21). Shape-constrained
316
315
  // ferry parser for the classify-branch agent() call. The schema buys SHAPE,
@@ -826,8 +825,52 @@ if (CAPTURE_INDEX < 0) throw new Error("STEPS missing 'Capture' step");
826
825
  const REWORK_STEP = STEPS[BUILD_INDEX].name;
827
826
  // Shared rework budget: Review and QA rejections draw from the SAME pool of 2.
828
827
  // E.g. 2 Review bounces + 1 QA bounce = 3 total > budget -> task blocks.
828
+ // Blocker 46 (2026-09-22): the budget is task-durable, not per-run. A run
829
+ // ends on every operational failure (agent timeout, worktree-confinement
830
+ // failure) and the dispatcher launches a fresh run — the old `let ... = 0`
831
+ // reset the counter on each launch, so a task whose runs kept ending
832
+ // operationally between rejections reworked unboundedly while the designed
833
+ // park stayed unreachable (room #29 J1: 3 rejections across 4 runs, never
834
+ // parked). The counter is seeded from the task's durable event stream at
835
+ // run start (seedReworkCount, after the pin block); in-run increments are
836
+ // unchanged.
829
837
  const MAX_TOTAL_REWORK = 2;
830
838
  let totalReworkCount = 0;
839
+ // Blocker 46: the first-claim test used totalReworkCount === 0 as a proxy
840
+ // for "this run is entering its first step". The counter is now seeded from
841
+ // task history, so the proxy lies on re-run launches — use an explicit
842
+ // run-local flag instead. The first claim (self-claim with raced-duplicate
843
+ // stand-down) must fire exactly once per run.
844
+ let firstClaimDone = false;
845
+
846
+ // Blocker 46: count verdict-level `rejected` events (Review/QA rejections —
847
+ // record-phase writes the event type equal to the session status) in the
848
+ // task's durable event stream. Operational failure ends are `failed`, never
849
+ // `rejected`, so they must not spend retry headroom. A failed seed must not
850
+ // strand the run: start at 0 and let the in-run budget govern this run; the
851
+ // seed is retried on the next run.
852
+ async function seedReworkCount() {
853
+ try {
854
+ var seed = await agent(
855
+ "Read this task's event history.\n" +
856
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId, limit: 100 }) + "\n" +
857
+ "Count the events whose type is exactly \"rejected\".\n" +
858
+ "Return JSON { \"rejected_count\": <integer> } and nothing else.",
859
+ {
860
+ key: "rework-seed-" + taskId,
861
+ label: "Seeding shared rework budget from task history",
862
+ schema: { type: "object", properties: { rejected_count: { type: "number" } }, required: ["rejected_count"] }
863
+ }
864
+ );
865
+ var n = Math.floor(Number(seed && seed.rejected_count) || 0);
866
+ if (!(n >= 0)) n = 0;
867
+ if (n > 0) log("Shared rework budget seeded from task history: " + n + " prior rejection(s)");
868
+ return n;
869
+ } catch (e) {
870
+ log("Rework-budget seed unavailable (" + ((e && e.message) || String(e)).slice(0, 120) + ") — starting at 0");
871
+ return 0;
872
+ }
873
+ }
831
874
  let rejectionNotes = inputs.rejection_notes || "";
832
875
  let mapperSpec = "";
833
876
  // Visual verdict state: Sage's experiential flag (true/false/null until the
@@ -941,6 +984,11 @@ log("Lifecycle scripts pinned to " + RUN_LIB);
941
984
  CREW_API = CREW_API_PINNED;
942
985
  log("Crew API pinned to " + CREW_API);
943
986
 
987
+ // Blocker 46: seed the task-durable rework budget from the task's event
988
+ // history before the phase loop. Pinned copy, so the seed reads the same
989
+ // release this run executes.
990
+ totalReworkCount = await seedReworkCount();
991
+
944
992
  // Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
945
993
  // minted at this run's first claim (never a PID — short-lived agent PIDs
946
994
  // made every concurrent acquire take the stale path). Set exactly once on
@@ -949,7 +997,7 @@ let lockHolder = taskId;
949
997
 
950
998
  while (i < STEPS.length) {
951
999
  const step = STEPS[i];
952
- const isFirstClaim = (i === startStepIndex && totalReworkCount === 0);
1000
+ const isFirstClaim = (i === startStepIndex && !firstClaimDone);
953
1001
 
954
1002
  phase(step.name);
955
1003
  log(step.name + " step (" + step.identity + ") for task " + taskId);
@@ -1114,6 +1162,7 @@ while (i < STEPS.length) {
1114
1162
  return { status: "duplicate", task_id: taskId, reason: "task already claimed by another run" };
1115
1163
  }
1116
1164
  activeSessionId = claimResult.session_id;
1165
+ firstClaimDone = true; // blocker 46: exactly once per run
1117
1166
  } else {
1118
1167
  const claimResult = await agent(
1119
1168
  "Claim a session for this task step.\n" +
@@ -2081,14 +2130,15 @@ while (i < STEPS.length) {
2081
2130
  "You are code-blind QA. You NEVER read source files. Public docs are not source — read them as a user would.\n" +
2082
2131
  "STEP 1: Experiential visual inspection — drive the fixed artifact as a user would, one browser step at a time, and verify the reported bug is actually fixed.\n" +
2083
2132
  "You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
2133
+ "SESSION PROTOCOL (multi-step flows): one-shot invocations launch a fresh browser each time, so they cannot drive flows where a later step needs an earlier step's living page state (open a dialog, then confirm it — the second step needs the first step's page state). For those, hold ONE browser alive: node " + crewHome + "/current/lib/see-act.js session-start --name qa --url http://localhost:<N>/ --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — launches Chromium once in a detached daemon (its control server binds 127.0.0.1 only; the daemon exits after 10 min with no act). The --log path is the SAME ooda-log.jsonl the QA guard reads: every session-act appends its own JSON line in the exact step schema (step, attempt, action, args, exit, screenshot, observation), so you do NOT call append-ooda-step.js for session acts — READ each act's JSON result (the screenshot field is the path to READ with your read tool) and decide the next action. Then drive it: node " + crewHome + "/current/lib/see-act.js session-act --name qa --attempt \"1\" <aria|shot|click|scroll|type> [args] — one action inside the living page. Prefix EVERY session-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one). When done: node " + crewHome + "/current/lib/see-act.js session-end --name qa — closes the browser, prints the session summary, and keeps the session dir as evidence. If session-start exits 3 with a \"NOT POSSIBLE: <reason>\" string, the environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
2084
2134
  "Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
2085
2135
  "a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and continue with the mechanical checks.\n" +
2086
2136
  "b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
2087
2137
  "c. Bounded see-act loop, at most 8 steps: re-run the reproduction steps for the reported bug — does it still occur? Then check the surrounding views for regressions. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
2088
- "c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
2138
+ "c2. After EVERY one-shot see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). (Session acts are self-logging — session-act already appended its line; for those, READ the result instead of re-logging.) Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
2089
2139
  "c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
2090
2140
  "c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
2091
- "d. One action per invocation, fresh browser each time: anything reachable by (navigate, one action) is testable — e.g. clicking any tab or button from the landing page. Sequences needing prior in-page state (open a dialog, then confirm it) are not; if the fix needs such a sequence to verify, report NOT POSSIBLE for that part and judge what you can.\n" +
2141
+ "d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the dialog, then confirm it, then judge the result. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and continue with the mechanical checks.\n" +
2092
2142
  "e. Judge as a user against the task description: is the reported bug fixed AND is nothing else visibly broken? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
2093
2143
  "f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
2094
2144
  "Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
@@ -2172,12 +2222,12 @@ while (i < STEPS.length) {
2172
2222
  "The returned events are filtered to this task. They contain notes and decisions from prior phases.\n\n";
2173
2223
  }
2174
2224
 
2175
- // Work agent returns the runtime's native envelope {"status": "ok",
2176
- // "result": "<prose>"} with no schema. The runtime requires JSON output;
2177
- // the envelope is its own documented shape, so there is nothing for the
2178
- // agent to improvise. The workflow receives the prose report as a plain
2179
- // string. The verdict is still extracted deterministically from the report
2180
- // text by extractVerdict below — never by an agent.
2225
+ // Work agent returns prose on the schema-less courier; the workflow
2226
+ // consumes the return as a plain string (blocker 43, 2026-09-22: the
2227
+ // prompt names no JSON envelope — nothing parses one, and the runtime's
2228
+ // JSON-candidate scan discards brace-shaped prose, which is what the
2229
+ // transport retry guards). The verdict is still extracted deterministically
2230
+ // from the report text by extractVerdict below — never by an agent.
2181
2231
  var workPromptBase =
2182
2232
  TOOL_CHECK_PREAMBLE +
2183
2233
  "Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
@@ -2190,8 +2240,7 @@ while (i < STEPS.length) {
2190
2240
  "\n## Instructions\n\n" + eventPreamble + instructions + "\n\n" +
2191
2241
  "CONSTRAINT: Do NOT call logevent or upsertagentsession — the workflow handles all phase tracking after your step completes.\n\n" +
2192
2242
  "Stay in character. Do the work thoroughly.\n\n" +
2193
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your report here\"}. " +
2194
- "The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
2243
+ "Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
2195
2244
  var workKeyBase = "work-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "");
2196
2245
  // See docs/decisions/qa-reproduce.md#wrong-layer-guard: reproduce must run at the correct layer.
2197
2246
  var reproEvidenceDir = crewHome + "/task-evidence/" + taskId + "/repro";
@@ -197,7 +197,7 @@ if (!taskId) {
197
197
  throw new Error("task_id is required in args");
198
198
  }
199
199
 
200
- // See docs/decisions/workflow-core.md#closeout-envelope: the work agent returns the runtime's native envelope; verdict extracted mechanically.
200
+ // See docs/decisions/workflow-core.md#closeout-envelope: the schema-less work courier returns prose, consumed as a plain string; verdict extracted mechanically.
201
201
  const VERDICT_STEPS = ["Build", "Review", "Integrate", "Publish"];
202
202
  function extractVerdict(workerText) {
203
203
  // The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
@@ -231,7 +231,6 @@ function buildVerdictReaskPrompt(stepName, workerText) {
231
231
  "Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
232
232
  "if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
233
233
  "Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
234
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
235
234
  "The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
236
235
  }
237
236
  async function reaskVerdict(stepName, reworkSuffix, workerText) {
@@ -304,8 +303,8 @@ var TOOL_CHECK_PREAMBLE =
304
303
  "3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
305
304
  "Then do the assignment below.\n\n";
306
305
  function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
307
- // reason: "discarded" (the runtime threw the output away - it could not be
308
- // machine-read), "empty" (agent() returned without throwing but produced
306
+ // reason: "discarded" (the runtime threw the output away — the JSON-candidate
307
+ // scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
309
308
  // nothing usable), "no-tools" (the worker's TOOL CHECK reported
310
309
  // artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
311
310
  // reported shell_transport: unavailable).
@@ -316,11 +315,11 @@ function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason)
316
315
  ? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
317
316
  : reason === "no-transport"
318
317
  ? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
319
- : "your previous attempt's output could not be machine-read as JSON and was discarded";
318
+ : "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
320
319
  return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
321
320
  "First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
322
321
  "if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
323
- "Then return your report as JSON in exactly the shape specified above.";
322
+ "Then return your report in exactly the shape specified above.";
324
323
  }
325
324
  // parseClassifyFerry — room #26 blocker 42 (2026-09-21). Shape-constrained
326
325
  // ferry parser for the classify-branch agent() call. The schema buys SHAPE,
@@ -794,6 +793,50 @@ if (CAPTURE_INDEX < 0) throw new Error("STEPS missing 'Capture' step");
794
793
  const REWORK_STEP = STEPS[BUILD_INDEX].name;
795
794
  const MAX_REWORK = 2;
796
795
  let reworkCount = 0;
796
+ // Blocker 46 (2026-09-22): the budget is task-durable, not per-run. A run
797
+ // ends on every operational failure (agent timeout, worktree-confinement
798
+ // failure) and the dispatcher launches a fresh run — the old `let ... = 0`
799
+ // reset the counter on each launch, so a task whose runs kept ending
800
+ // operationally between rejections reworked unboundedly while the designed
801
+ // park stayed unreachable (room #29 J1: 3 rejections across 4 runs, never
802
+ // parked). The counter is seeded from the task's durable event stream at
803
+ // run start (seedReworkCount, after the pin block); in-run increments are
804
+ // unchanged.
805
+ // Blocker 46: the first-claim test used reworkCount === 0 as a proxy for
806
+ // "this run is entering its first step". The counter is now seeded from
807
+ // task history, so the proxy lies on re-run launches — use an explicit
808
+ // run-local flag instead. The first claim (self-claim with raced-duplicate
809
+ // stand-down) must fire exactly once per run.
810
+ let firstClaimDone = false;
811
+
812
+ // Blocker 46: count verdict-level `rejected` events (Review/QA rejections —
813
+ // record-phase writes the event type equal to the session status) in the
814
+ // task's durable event stream. Operational failure ends are `failed`, never
815
+ // `rejected`, so they must not spend retry headroom. A failed seed must not
816
+ // strand the run: start at 0 and let the in-run budget govern this run; the
817
+ // seed is retried on the next run.
818
+ async function seedReworkCount() {
819
+ try {
820
+ var seed = await agent(
821
+ "Read this task's event history.\n" +
822
+ "Run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId, limit: 100 }) + "\n" +
823
+ "Count the events whose type is exactly \"rejected\".\n" +
824
+ "Return JSON { \"rejected_count\": <integer> } and nothing else.",
825
+ {
826
+ key: "rework-seed-" + taskId,
827
+ label: "Seeding shared rework budget from task history",
828
+ schema: { type: "object", properties: { rejected_count: { type: "number" } }, required: ["rejected_count"] }
829
+ }
830
+ );
831
+ var n = Math.floor(Number(seed && seed.rejected_count) || 0);
832
+ if (!(n >= 0)) n = 0;
833
+ if (n > 0) log("Shared rework budget seeded from task history: " + n + " prior rejection(s)");
834
+ return n;
835
+ } catch (e) {
836
+ log("Rework-budget seed unavailable (" + ((e && e.message) || String(e)).slice(0, 120) + ") — starting at 0");
837
+ return 0;
838
+ }
839
+ }
797
840
  let rejectionNotes = inputs.rejection_notes || "";
798
841
  let mapperSpec = "";
799
842
  // Visual verdict state: Sage's experiential flag (true/false/null until the
@@ -905,6 +948,11 @@ log("Lifecycle scripts pinned to " + RUN_LIB);
905
948
  CREW_API = CREW_API_PINNED;
906
949
  log("Crew API pinned to " + CREW_API);
907
950
 
951
+ // Blocker 46: seed the task-durable rework budget from the task's event
952
+ // history before the phase loop. Pinned copy, so the seed reads the same
953
+ // release this run executes.
954
+ reworkCount = await seedReworkCount();
955
+
908
956
  // Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
909
957
  // minted at this run's first claim (never a PID — short-lived agent PIDs
910
958
  // made every concurrent acquire take the stale path). Set exactly once on
@@ -913,7 +961,7 @@ let lockHolder = taskId;
913
961
 
914
962
  while (i < STEPS.length) {
915
963
  const step = STEPS[i];
916
- const isFirstClaim = (i === startStepIndex && reworkCount === 0);
964
+ const isFirstClaim = (i === startStepIndex && !firstClaimDone);
917
965
 
918
966
  phase(step.name);
919
967
  log(step.name + " step (" + step.identity + ") for task " + taskId);
@@ -1085,6 +1133,7 @@ while (i < STEPS.length) {
1085
1133
  return { status: "duplicate", task_id: taskId, reason: "task already claimed by another run" };
1086
1134
  }
1087
1135
  activeSessionId = claimResult.session_id;
1136
+ firstClaimDone = true; // blocker 46: exactly once per run
1088
1137
  await telemetryEvent("claim", step.name + " claimed");
1089
1138
  // Flush telemetry before proceeding — the claim is a critical point.
1090
1139
  // If a subsequent agent() fails, we want the claim event persisted.
@@ -1900,12 +1949,12 @@ while (i < STEPS.length) {
1900
1949
  "The returned events are filtered to this task. They contain notes and decisions from prior phases.\n\n";
1901
1950
  }
1902
1951
 
1903
- // Work agent returns the runtime's native envelope {"status": "ok",
1904
- // "result": "<prose>"} with no schema. The runtime requires JSON output;
1905
- // the envelope is its own documented shape, so there is nothing for the
1906
- // agent to improvise. The workflow receives the prose report as a plain
1907
- // string. The verdict is still extracted deterministically from the report
1908
- // text by extractVerdict below — never by an agent.
1952
+ // Work agent returns prose on the schema-less courier; the workflow
1953
+ // consumes the return as a plain string (blocker 43, 2026-09-22: the
1954
+ // prompt names no JSON envelope — nothing parses one, and the runtime's
1955
+ // JSON-candidate scan discards brace-shaped prose, which is what the
1956
+ // transport retry guards). The verdict is still extracted deterministically
1957
+ // from the report text by extractVerdict below — never by an agent.
1909
1958
  var workPromptBase =
1910
1959
  TOOL_CHECK_PREAMBLE +
1911
1960
  "Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
@@ -1913,8 +1962,7 @@ while (i < STEPS.length) {
1913
1962
  "Task: " + taskTitle + "\nTask ID: " + taskId + "\nDescription: " + taskDescription + "\nStep: " + step.name + "\n" +
1914
1963
  (step.name !== "Review" ? "Crew API: " + CREW_API + "\n" : "") +
1915
1964
  "\n## Instructions\n\n" + eventPreamble + instructions + "\n\nCONSTRAINT: Do NOT call logevent or upsertagentsession — the workflow handles all phase tracking after your step completes.\n\nStay in character. Do the work thoroughly.\n\n" +
1916
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your report here\"}. " +
1917
- "The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
1965
+ "Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
1918
1966
  var workKeyBase = "work-" + step.name + (reworkCount > 0 ? "-r" + reworkCount : "");
1919
1967
  var workerResult = null;
1920
1968
  var workAttempts = [];
package/workflows/docs.js CHANGED
@@ -269,7 +269,6 @@ while (i < STEPS.length) {
269
269
  "Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
270
270
  "if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
271
271
  "Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
272
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
273
272
  "The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
274
273
  }
275
274
  async function reaskVerdict(stepName, reworkSuffix, workerText) {
@@ -329,8 +328,8 @@ while (i < STEPS.length) {
329
328
  "3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
330
329
  "Then do the assignment below.\n\n";
331
330
  function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
332
- // reason: "discarded" (the runtime threw the output away - it could not be
333
- // machine-read), "empty" (agent() returned without throwing but produced
331
+ // reason: "discarded" (the runtime threw the output away — the JSON-candidate
332
+ // scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
334
333
  // nothing usable), "no-tools" (the worker's TOOL CHECK reported
335
334
  // artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
336
335
  // reported shell_transport: unavailable).
@@ -341,11 +340,11 @@ while (i < STEPS.length) {
341
340
  ? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
342
341
  : reason === "no-transport"
343
342
  ? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
344
- : "your previous attempt's output could not be machine-read as JSON and was discarded";
343
+ : "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
345
344
  return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
346
345
  "First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
347
346
  "if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
348
- "Then return your report as JSON in exactly the shape specified above.";
347
+ "Then return your report in exactly the shape specified above.";
349
348
  }
350
349
  // parseToolSignals - bug 3472bf36. The work agent's TOOL CHECK emits two
351
350
  // exact signal lines: artifact_tools: ok|missing and
@@ -432,20 +431,19 @@ function describeWorkAgentFailure(stepName, identity, attempts) {
432
431
  "\nWrite your review as plain prose — findings, then decision. End your report with exactly one line: VERDICT: PASS if it passes, VERDICT: FAIL if it fails.";
433
432
  }
434
433
 
435
- // Work agent returns the runtime's native envelope {"status": "ok",
436
- // "result": "<prose>"} with no schema. The runtime requires JSON output;
437
- // the envelope is its own documented shape, so there is nothing for the
438
- // agent to improvise. The workflow receives the prose report as a plain
439
- // string. The verdict is still extracted deterministically from the report
440
- // text by extractVerdict below — never by an agent.
434
+ // Work agent returns prose on the schema-less courier; the workflow
435
+ // consumes the return as a plain string (blocker 43, 2026-09-22: the
436
+ // prompt names no JSON envelope — nothing parses one, and the runtime's
437
+ // JSON-candidate scan discards brace-shaped prose, which is what the
438
+ // transport retry guards). The verdict is still extracted deterministically
439
+ // from the report text by extractVerdict below — never by an agent.
441
440
  var workPromptBase =
442
441
  TOOL_CHECK_PREAMBLE +
443
442
  "Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
444
443
  "## Your Assignment\n\n" +
445
444
  "Task: " + taskTitle + "\nTask ID: " + taskId + "\nDescription: " + taskDescription + "\nStep: " + step.name + "\n\n" +
446
445
  "## Instructions\n\n" + instructions + "\n\nCONSTRAINT: Do NOT call the crew API directly — the workflow handles all phase tracking after your step completes.\n\nStay in character. All file work under " + REPO_PATH + "/.\n\n" +
447
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your report here\"}. " +
448
- "The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
446
+ "Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
449
447
  var workKeyBase = "work-" + step.name + (reworkCount > 0 ? "-r" + reworkCount : "");
450
448
  var workerResult = null;
451
449
  var workAttempts = [];