muse-crew 0.14.8 → 0.14.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/decisions/workflow-core.md +14 -7
- package/docs/ooda-report.md +21 -4
- package/lib/AGENTS.md +2 -2
- package/lib/read-ooda-verdict.js +61 -86
- package/lib/see-act.js +616 -6
- package/lib/test-worktree-backend.sh +42 -0
- package/lib/worktree-lifecycle.sh +64 -4
- package/package.json +1 -1
- package/workflows/bugfix.js +66 -17
- package/workflows/chore.js +63 -15
- package/workflows/docs.js +11 -13
- package/workflows/standard.js +66 -17
|
@@ -45,6 +45,48 @@ grep -q "^path=$T/.worktrees/$tid\$" "$T/.worktrees/.registry/$tid" || fail "pre
|
|
|
45
45
|
out=$(bash "$LIFECYCLE" prepare "$tid") || fail "prepare (reuse) exited non-zero"
|
|
46
46
|
echo "$out" | grep -q '^REUSED' || fail "prepare: expected REUSED on second run: $out"
|
|
47
47
|
|
|
48
|
+
# Blocker 45 (2026-09-22, room #29 J1): prepare must verify its REUSED
|
|
49
|
+
# invariant. An agent-improvised branch (raw git, bypassing prepare) gets
|
|
50
|
+
# REPAIRED — registry re-registered under the actual branch, never deleted —
|
|
51
|
+
# never silently REUSED.
|
|
52
|
+
tid2="backendtest45"
|
|
53
|
+
out=$(bash "$LIFECYCLE" prepare "$tid2") || fail "prepare (45) exited non-zero: $out"
|
|
54
|
+
echo "$out" | grep -q '^CREATED' || fail "prepare (45): expected CREATED, got: $out"
|
|
55
|
+
# The agent improvises: checks out a branch outside the task/ namespace and
|
|
56
|
+
# commits there, bypassing prepare entirely.
|
|
57
|
+
git -C "$T/.worktrees/$tid2" checkout -qb crew-9eda33ca
|
|
58
|
+
echo improvised >> "$T/.worktrees/$tid2/f.txt"
|
|
59
|
+
(git -C "$T/.worktrees/$tid2" commit -qam improvised) || fail "improvised commit failed"
|
|
60
|
+
out=$(bash "$LIFECYCLE" prepare "$tid2") || fail "prepare (45 repair) exited non-zero: $out"
|
|
61
|
+
echo "$out" | grep -q '^REPAIRED' || fail "prepare: expected REPAIRED on improvised branch, got: $out"
|
|
62
|
+
echo "$out" | grep -q '^REUSED' && fail "prepare: must not print REUSED when the branch invariant fails"
|
|
63
|
+
# The agent's branch survives; the registry attributes the repair honestly.
|
|
64
|
+
git -C "$T" rev-parse --verify "crew-9eda33ca" >/dev/null 2>&1 || fail "repair: agent branch deleted"
|
|
65
|
+
grep -q '^branch=crew-9eda33ca$' "$T/.worktrees/.registry/$tid2" || fail "repair: registry branch not re-registered"
|
|
66
|
+
grep -q '^task_id=backendtest45$' "$T/.worktrees/.registry/$tid2" || fail "repair: registry lost task attribution"
|
|
67
|
+
base_before=$(grep '^base_commit=' "$T/.worktrees/.registry/$tid2")
|
|
68
|
+
# The repair is idempotent: a second prepare repairs again (no spurious
|
|
69
|
+
# REUSED), preserving the registry's other fields.
|
|
70
|
+
out=$(bash "$LIFECYCLE" prepare "$tid2") || fail "prepare (45 second repair) exited non-zero: $out"
|
|
71
|
+
echo "$out" | grep -q '^REPAIRED' || fail "prepare: expected idempotent REPAIRED, got: $out"
|
|
72
|
+
base_after=$(grep '^base_commit=' "$T/.worktrees/.registry/$tid2")
|
|
73
|
+
[ "$base_before" = "$base_after" ] || fail "repair: re-register must preserve base_commit"
|
|
74
|
+
# A worktree back on the canonical branch is REUSED with no repair.
|
|
75
|
+
git -C "$T/.worktrees/$tid2" checkout -q "task/$tid2"
|
|
76
|
+
out=$(bash "$LIFECYCLE" prepare "$tid2") || fail "prepare (45 canonical) exited non-zero: $out"
|
|
77
|
+
echo "$out" | grep -q '^REUSED' || fail "prepare: expected REUSED on canonical branch, got: $out"
|
|
78
|
+
echo "$out" | grep -q '^REPAIRED' && fail "prepare: no spurious REPAIRED on a canonical worktree"
|
|
79
|
+
# resolve-branch falls back to the worktree's actual checked-out branch when
|
|
80
|
+
# the registry is absent and no canonical/prefix ref matches — the
|
|
81
|
+
# classifier's mechanical truth of last resort (not "branch does not exist").
|
|
82
|
+
git -C "$T/.worktrees/$tid2" checkout -qb crew-deadbeef
|
|
83
|
+
rm -f "$T/.worktrees/.registry/$tid2"
|
|
84
|
+
git -C "$T" branch -D -q "task/$tid2"
|
|
85
|
+
resolved=$(bash "$LIFECYCLE" resolve-branch "$tid2") || fail "resolve-branch (45) exited non-zero"
|
|
86
|
+
[ "$resolved" = "crew-deadbeef" ] || fail "resolve-branch: expected the actual branch, got: $resolved"
|
|
87
|
+
bash "$LIFECYCLE" cleanup "$tid2" >/dev/null || fail "cleanup (45) failed"
|
|
88
|
+
[ -d "$T/.worktrees/$tid2" ] && fail "cleanup (45): worktree still exists"
|
|
89
|
+
|
|
48
90
|
# inspect: diffs the task branch against main
|
|
49
91
|
echo change >> "$T/.worktrees/$tid/f.txt"
|
|
50
92
|
(cd "$T/.worktrees/$tid" && git commit -qam work)
|
|
@@ -12,11 +12,23 @@
|
|
|
12
12
|
# bypassing prepare — task 4e1a1bba, 2026-09-17), resolve_branch() falls
|
|
13
13
|
# back to enumerating refs: the canonical task/<full-id> ref wins, then a
|
|
14
14
|
# task/<id-prefix> ref is tolerated (matched by strict prefix, never
|
|
15
|
-
# created)
|
|
15
|
+
# created), then the existing worktree's actual checked-out branch
|
|
16
|
+
# (blocker 45, 2026-09-22 — an agent-improvised branch outside the task/
|
|
17
|
+
# namespace is invisible to the prefix scan; the classifier must see the
|
|
18
|
+
# world, not "branch does not exist"). Canonical stays task/<full-id>;
|
|
19
|
+
# ambiguous prefixes fail closed.
|
|
20
|
+
#
|
|
21
|
+
# prepare's reuse contract (blocker 45, 2026-09-22): an existing worktree
|
|
22
|
+
# prints REUSED only when it is checked out on the canonical task/<id>
|
|
23
|
+
# branch. A worktree checked out on another branch (an agent improvised
|
|
24
|
+
# with raw git) is repaired — the registry is re-registered under the
|
|
25
|
+
# actual branch (the branch is never deleted: its commits may be the
|
|
26
|
+
# task's work) — and prepare prints REPAIRED instead of REUSED, so the
|
|
27
|
+
# repair is attributable in the record.
|
|
16
28
|
#
|
|
17
29
|
# Commands:
|
|
18
30
|
# validate <task_id> — check task ID is safe for branches/paths
|
|
19
|
-
# prepare <task_id> — create worktree + branch, register ownership
|
|
31
|
+
# prepare <task_id> — create worktree + branch, register ownership (existing worktree: REUSED, or REPAIRED when re-registered under an improvised branch)
|
|
20
32
|
# resolve-branch <task_id> — print the resolved task branch name
|
|
21
33
|
# inspect <task_id> — diff against the integration target for review
|
|
22
34
|
# classify-branch <task_id> — mechanical branch-state classification for Review:
|
|
@@ -136,6 +148,17 @@ resolve_branch() {
|
|
|
136
148
|
echo "ERROR: task id '$task_id' matches multiple task branches: ${matches[*]} — ambiguous, refusing to guess" >&2
|
|
137
149
|
return 1
|
|
138
150
|
fi
|
|
151
|
+
# Blocker 45 (2026-09-22): the worktree's actual checkout is the mechanical
|
|
152
|
+
# truth of last resort. An agent-improvised branch (outside the task/
|
|
153
|
+
# namespace) is invisible to the prefix scan above — the classifier must
|
|
154
|
+
# see the world, not "branch does not exist".
|
|
155
|
+
if [ -d "$WORKTREE_DIR/$task_id" ]; then
|
|
156
|
+
local actual
|
|
157
|
+
if actual=$(git -C "$WORKTREE_DIR/$task_id" rev-parse --abbrev-ref HEAD 2>/dev/null) && [ -n "$actual" ]; then
|
|
158
|
+
printf '%s' "$actual"
|
|
159
|
+
return 0
|
|
160
|
+
fi
|
|
161
|
+
fi
|
|
139
162
|
printf '%s' "$canonical"
|
|
140
163
|
}
|
|
141
164
|
|
|
@@ -239,6 +262,25 @@ unregister() {
|
|
|
239
262
|
rm -f "$REGISTRY_DIR/$1"
|
|
240
263
|
}
|
|
241
264
|
|
|
265
|
+
# Blocker 45 (2026-09-22): repair the registry's branch attribution without
|
|
266
|
+
# touching anything else. An existing entry keeps its task_id, base_commit,
|
|
267
|
+
# and created fields — only branch= is rewritten. A missing entry is
|
|
268
|
+
# registered fresh (base_commit "unknown" — honest, not fabricated). The
|
|
269
|
+
# agent's branch is never deleted: its commits may be the task's work.
|
|
270
|
+
reregister_branch() {
|
|
271
|
+
local task_id="$1" branch="$2"
|
|
272
|
+
local reg="$REGISTRY_DIR/$task_id"
|
|
273
|
+
mkdir -p "$REGISTRY_DIR"
|
|
274
|
+
if [ -f "$reg" ] && grep -q '^branch=' "$reg"; then
|
|
275
|
+
# Escape sed-replacement metacharacters (branch names can carry &).
|
|
276
|
+
local esc=${branch//\\/\\\\}
|
|
277
|
+
esc=${esc//&/\\&}
|
|
278
|
+
sed -i "s|^branch=.*|branch=$esc|" "$reg"
|
|
279
|
+
else
|
|
280
|
+
register "$task_id" "unknown" "$branch" "$WORKTREE_DIR/$task_id"
|
|
281
|
+
fi
|
|
282
|
+
}
|
|
283
|
+
|
|
242
284
|
# --- commands ---
|
|
243
285
|
|
|
244
286
|
cmd_validate() {
|
|
@@ -256,9 +298,27 @@ cmd_prepare() {
|
|
|
256
298
|
|
|
257
299
|
# Reuse existing worktree
|
|
258
300
|
if [ -d "$WORKTREE_DIR/$task_id" ]; then
|
|
259
|
-
|
|
260
|
-
local head
|
|
301
|
+
local head actual_branch canonical_branch
|
|
261
302
|
head=$(cd "$WORKTREE_DIR/$task_id" && git rev-parse HEAD)
|
|
303
|
+
if ! actual_branch=$(cd "$WORKTREE_DIR/$task_id" && git rev-parse --abbrev-ref HEAD 2>/dev/null); then
|
|
304
|
+
echo "ERROR: worktree .worktrees/$task_id exists but its checked-out branch is unreadable — refusing to REUSE an unverifiable worktree"
|
|
305
|
+
return 1
|
|
306
|
+
fi
|
|
307
|
+
canonical_branch="task/$task_id"
|
|
308
|
+
# Blocker 45 (2026-09-22): REUSED asserted an invariant it never checked —
|
|
309
|
+
# the worktree is on the task's canonical branch. Room #29: a Build agent
|
|
310
|
+
# improvised branch crew-9eda33ca; the retry's prepare REUSED it blindly
|
|
311
|
+
# and Review's classify-branch parked on "branch does not exist". Verify
|
|
312
|
+
# the invariant; on mismatch repair the registry (never delete the
|
|
313
|
+
# agent's branch — its commits may be the task's work) and attribute the
|
|
314
|
+
# repair with REPAIRED instead of REUSED.
|
|
315
|
+
if [ "$actual_branch" != "$canonical_branch" ]; then
|
|
316
|
+
reregister_branch "$task_id" "$actual_branch"
|
|
317
|
+
echo "REPAIRED: worktree .worktrees/$task_id checked out on branch $actual_branch (expected $canonical_branch); registry re-registered"
|
|
318
|
+
echo "HEAD: $head"
|
|
319
|
+
return 0
|
|
320
|
+
fi
|
|
321
|
+
echo "REUSED: worktree exists at .worktrees/$task_id"
|
|
262
322
|
echo "HEAD: $head"
|
|
263
323
|
return 0
|
|
264
324
|
fi
|
package/package.json
CHANGED
package/workflows/bugfix.js
CHANGED
|
@@ -131,7 +131,7 @@ if (!taskId) {
|
|
|
131
131
|
throw new Error("task_id is required in args");
|
|
132
132
|
}
|
|
133
133
|
|
|
134
|
-
// See docs/decisions/workflow-core.md#closeout-envelope: the work
|
|
134
|
+
// See docs/decisions/workflow-core.md#closeout-envelope: the schema-less work courier returns prose, consumed as a plain string; verdict extracted mechanically.
|
|
135
135
|
const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
|
|
136
136
|
function extractVerdict(workerText) {
|
|
137
137
|
// The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
|
|
@@ -221,7 +221,6 @@ function buildVerdictReaskPrompt(stepName, workerText) {
|
|
|
221
221
|
"Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
|
|
222
222
|
"if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
|
|
223
223
|
"Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
|
|
224
|
-
"Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
|
|
225
224
|
"The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
|
|
226
225
|
}
|
|
227
226
|
async function reaskVerdict(stepName, reworkSuffix, workerText) {
|
|
@@ -294,8 +293,8 @@ var TOOL_CHECK_PREAMBLE =
|
|
|
294
293
|
"3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
|
|
295
294
|
"Then do the assignment below.\n\n";
|
|
296
295
|
function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
|
|
297
|
-
// reason: "discarded" (the runtime threw the output away
|
|
298
|
-
//
|
|
296
|
+
// reason: "discarded" (the runtime threw the output away — the JSON-candidate
|
|
297
|
+
// scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
|
|
299
298
|
// nothing usable), "no-tools" (the worker's TOOL CHECK reported
|
|
300
299
|
// artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
|
|
301
300
|
// reported shell_transport: unavailable).
|
|
@@ -306,11 +305,11 @@ function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason)
|
|
|
306
305
|
? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
|
|
307
306
|
: reason === "no-transport"
|
|
308
307
|
? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
|
|
309
|
-
: "your previous attempt's output could not
|
|
308
|
+
: "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
|
|
310
309
|
return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
|
|
311
310
|
"First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
|
|
312
311
|
"if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
|
|
313
|
-
"Then return your report
|
|
312
|
+
"Then return your report in exactly the shape specified above.";
|
|
314
313
|
}
|
|
315
314
|
// parseClassifyFerry — room #26 blocker 42 (2026-09-21). Shape-constrained
|
|
316
315
|
// ferry parser for the classify-branch agent() call. The schema buys SHAPE,
|
|
@@ -826,8 +825,52 @@ if (CAPTURE_INDEX < 0) throw new Error("STEPS missing 'Capture' step");
|
|
|
826
825
|
const REWORK_STEP = STEPS[BUILD_INDEX].name;
|
|
827
826
|
// Shared rework budget: Review and QA rejections draw from the SAME pool of 2.
|
|
828
827
|
// E.g. 2 Review bounces + 1 QA bounce = 3 total > budget -> task blocks.
|
|
828
|
+
// Blocker 46 (2026-09-22): the budget is task-durable, not per-run. A run
|
|
829
|
+
// ends on every operational failure (agent timeout, worktree-confinement
|
|
830
|
+
// failure) and the dispatcher launches a fresh run — the old `let ... = 0`
|
|
831
|
+
// reset the counter on each launch, so a task whose runs kept ending
|
|
832
|
+
// operationally between rejections reworked unboundedly while the designed
|
|
833
|
+
// park stayed unreachable (room #29 J1: 3 rejections across 4 runs, never
|
|
834
|
+
// parked). The counter is seeded from the task's durable event stream at
|
|
835
|
+
// run start (seedReworkCount, after the pin block); in-run increments are
|
|
836
|
+
// unchanged.
|
|
829
837
|
const MAX_TOTAL_REWORK = 2;
|
|
830
838
|
let totalReworkCount = 0;
|
|
839
|
+
// Blocker 46: the first-claim test used totalReworkCount === 0 as a proxy
|
|
840
|
+
// for "this run is entering its first step". The counter is now seeded from
|
|
841
|
+
// task history, so the proxy lies on re-run launches — use an explicit
|
|
842
|
+
// run-local flag instead. The first claim (self-claim with raced-duplicate
|
|
843
|
+
// stand-down) must fire exactly once per run.
|
|
844
|
+
let firstClaimDone = false;
|
|
845
|
+
|
|
846
|
+
// Blocker 46: count verdict-level `rejected` events (Review/QA rejections —
|
|
847
|
+
// record-phase writes the event type equal to the session status) in the
|
|
848
|
+
// task's durable event stream. Operational failure ends are `failed`, never
|
|
849
|
+
// `rejected`, so they must not spend retry headroom. A failed seed must not
|
|
850
|
+
// strand the run: start at 0 and let the in-run budget govern this run; the
|
|
851
|
+
// seed is retried on the next run.
|
|
852
|
+
async function seedReworkCount() {
|
|
853
|
+
try {
|
|
854
|
+
var seed = await agent(
|
|
855
|
+
"Read this task's event history.\n" +
|
|
856
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId, limit: 100 }) + "\n" +
|
|
857
|
+
"Count the events whose type is exactly \"rejected\".\n" +
|
|
858
|
+
"Return JSON { \"rejected_count\": <integer> } and nothing else.",
|
|
859
|
+
{
|
|
860
|
+
key: "rework-seed-" + taskId,
|
|
861
|
+
label: "Seeding shared rework budget from task history",
|
|
862
|
+
schema: { type: "object", properties: { rejected_count: { type: "number" } }, required: ["rejected_count"] }
|
|
863
|
+
}
|
|
864
|
+
);
|
|
865
|
+
var n = Math.floor(Number(seed && seed.rejected_count) || 0);
|
|
866
|
+
if (!(n >= 0)) n = 0;
|
|
867
|
+
if (n > 0) log("Shared rework budget seeded from task history: " + n + " prior rejection(s)");
|
|
868
|
+
return n;
|
|
869
|
+
} catch (e) {
|
|
870
|
+
log("Rework-budget seed unavailable (" + ((e && e.message) || String(e)).slice(0, 120) + ") — starting at 0");
|
|
871
|
+
return 0;
|
|
872
|
+
}
|
|
873
|
+
}
|
|
831
874
|
let rejectionNotes = inputs.rejection_notes || "";
|
|
832
875
|
let mapperSpec = "";
|
|
833
876
|
// Visual verdict state: Sage's experiential flag (true/false/null until the
|
|
@@ -941,6 +984,11 @@ log("Lifecycle scripts pinned to " + RUN_LIB);
|
|
|
941
984
|
CREW_API = CREW_API_PINNED;
|
|
942
985
|
log("Crew API pinned to " + CREW_API);
|
|
943
986
|
|
|
987
|
+
// Blocker 46: seed the task-durable rework budget from the task's event
|
|
988
|
+
// history before the phase loop. Pinned copy, so the seed reads the same
|
|
989
|
+
// release this run executes.
|
|
990
|
+
totalReworkCount = await seedReworkCount();
|
|
991
|
+
|
|
944
992
|
// Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
|
|
945
993
|
// minted at this run's first claim (never a PID — short-lived agent PIDs
|
|
946
994
|
// made every concurrent acquire take the stale path). Set exactly once on
|
|
@@ -949,7 +997,7 @@ let lockHolder = taskId;
|
|
|
949
997
|
|
|
950
998
|
while (i < STEPS.length) {
|
|
951
999
|
const step = STEPS[i];
|
|
952
|
-
const isFirstClaim = (i === startStepIndex &&
|
|
1000
|
+
const isFirstClaim = (i === startStepIndex && !firstClaimDone);
|
|
953
1001
|
|
|
954
1002
|
phase(step.name);
|
|
955
1003
|
log(step.name + " step (" + step.identity + ") for task " + taskId);
|
|
@@ -1114,6 +1162,7 @@ while (i < STEPS.length) {
|
|
|
1114
1162
|
return { status: "duplicate", task_id: taskId, reason: "task already claimed by another run" };
|
|
1115
1163
|
}
|
|
1116
1164
|
activeSessionId = claimResult.session_id;
|
|
1165
|
+
firstClaimDone = true; // blocker 46: exactly once per run
|
|
1117
1166
|
} else {
|
|
1118
1167
|
const claimResult = await agent(
|
|
1119
1168
|
"Claim a session for this task step.\n" +
|
|
@@ -2081,14 +2130,15 @@ while (i < STEPS.length) {
|
|
|
2081
2130
|
"You are code-blind QA. You NEVER read source files. Public docs are not source — read them as a user would.\n" +
|
|
2082
2131
|
"STEP 1: Experiential visual inspection — drive the fixed artifact as a user would, one browser step at a time, and verify the reported bug is actually fixed.\n" +
|
|
2083
2132
|
"You have a see-act driver: " + crewHome + "/current/lib/see-act.js (a node script; one browser action per invocation; it prints one JSON line to stdout). It launches its own Chromium through a self-contained loopback proxy — the ONLY url you may give it is the local artifact server you start below. Never point it at any other URL.\n" +
|
|
2133
|
+
"SESSION PROTOCOL (multi-step flows): one-shot invocations launch a fresh browser each time, so they cannot drive flows where a later step needs an earlier step's living page state (open a dialog, then confirm it — the second step needs the first step's page state). For those, hold ONE browser alive: node " + crewHome + "/current/lib/see-act.js session-start --name qa --url http://localhost:<N>/ --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl — launches Chromium once in a detached daemon (its control server binds 127.0.0.1 only; the daemon exits after 10 min with no act). The --log path is the SAME ooda-log.jsonl the QA guard reads: every session-act appends its own JSON line in the exact step schema (step, attempt, action, args, exit, screenshot, observation), so you do NOT call append-ooda-step.js for session acts — READ each act's JSON result (the screenshot field is the path to READ with your read tool) and decide the next action. Then drive it: node " + crewHome + "/current/lib/see-act.js session-act --name qa --attempt \"1\" <aria|shot|click|scroll|type> [args] — one action inside the living page. Prefix EVERY session-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one). When done: node " + crewHome + "/current/lib/see-act.js session-end --name qa — closes the browser, prints the session summary, and keeps the session dir as evidence. If session-start exits 3 with a \"NOT POSSIBLE: <reason>\" string, the environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
|
|
2084
2134
|
"Actions: aria | shot [--out <png>] [--full] | click [--out <png>] --selector <css> | scroll [--out <png>] --y <pixels|bottom> | type [--out <png>] --selector <css> --text <text>. Add --viewport mobile for a 390x844 frame. The JSON reports console_errors — treat any as a defect signal. Exit code 3 with not_possible set (a \"NOT POSSIBLE: <reason>\" string) means this environment cannot drive a browser: report NOT POSSIBLE: <reason> and continue with the mechanical checks — your VERDICT then covers mechanical checks only.\n" +
|
|
2085
2135
|
"a. Verify the built artifact exists: test -d ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/client/dist && test -f ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/server/dist/actions.js — if either is missing, report NOT POSSIBLE: built artifact not present at ~/workspace/ts-spaces/" + PUBLISH_SLUG + " and continue with the mechanical checks.\n" +
|
|
2086
2136
|
"b. Start the local artifact server detached with a log file, then poll the log for readiness (single shell invocation): CREW_HOME=" + crewHome + " setsid nohup node " + crewHome + "/current/lib/serve-artifact.js --space-dir ~/workspace/ts-spaces/" + PUBLISH_SLUG + " --port 0 --tag " + taskId + "-qa > /tmp/qa-server-" + taskId + ".log 2>&1 < /dev/null & for i in $(seq 1 30); do grep -q READY /tmp/qa-server-" + taskId + ".log && break; sleep 1; done; grep -o 'READY port=[0-9]*' /tmp/qa-server-" + taskId + ".log | cut -d= -f2 — the printed number is <N>. If no READY appears, report NOT POSSIBLE: local artifact server did not start (see /tmp/qa-server-" + taskId + ".log) and continue with the mechanical checks.\n" +
|
|
2087
2137
|
"c. Bounded see-act loop, at most 8 steps: re-run the reproduction steps for the reported bug — does it still occur? Then check the surrounding views for regressions. Prefix EVERY see-act invocation with SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ (shell env vars do not persist between your commands, so prefix each one) — every screenshot is then archived automatically as 001-shot-desktop.png, 002-click-desktop.png, ..., so no frame can be lost. Omit --out; the JSON's screenshot field is the path to READ with your read tool (you can see images).\n" +
|
|
2088
|
-
"c2. After EVERY see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
|
|
2138
|
+
"c2. After EVERY one-shot see-act invocation, append it to the OODA report: node " + crewHome + "/current/lib/append-ooda-step.js --log " + crewHome + "/task-evidence/" + taskId + "/postchange/ooda-log.jsonl --attempt \"1\" --step <N> --action <aria|shot|click|scroll|type> --exit <exit-code> --screenshot <screenshot-from-JSON> --observation \"<1-2 sentences: what the frame showed and what you concluded>\" (omit --screenshot for aria). (Session acts are self-logging — session-act already appended its line; for those, READ the result instead of re-logging.) Steps are strictly monotonic: your first step is --step 1, then 2, 3, ... — the script rejects anything else. If you run the loop again (a new idea, after a restart), that run is --attempt \"2\": attempts never overwrite, they accumulate. Your observation is the payload — say what you saw and what it means, not just what you did. A step you ran but did not log is invisible to everyone after you.\n" +
|
|
2089
2139
|
"c2b. IMAGE TOOLS (when a raw screenshot is not enough evidence): python3 " + crewHome + "/current/lib/edit-image.py crop --in <frame.png> --out <crop.png> --region x,y,w,h — pixel-exact crop; zoom --factor <f> --center x,y — pixel-crisp zoom back to the original frame size; label --text \"<caption>\" — caption bar; nup --cols 2 --in <a.png> --in <b.png> --labels \"before|after\" — side-by-side grid. Deterministic, no network. Log each with --action crop|zoom|label|nup, --attempt \"1\", and --screenshot <the derivative you READ>. For rich compositions (before/after with real typography, annotated callouts), write an HTML page and render it: node " + crewHome + "/current/lib/render-html.js --in <page.html> --out <composed.png> [--width 1200] — headless Chromium, hermetic (remote assets are blocked and fail the render), deterministic; log with --action compose and --screenshot <the composition you READ>. Derivatives supplement, never replace: keep the source frame archived and name it in --args. READ every derivative you log — an unread image is not evidence.\n" +
|
|
2090
2140
|
"c3. Start with: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ aria — read the JSON, log the step. Then: SEE_ACT_ARCHIVE_DIR=" + crewHome + "/task-evidence/" + taskId + "/postchange/ node " + crewHome + "/current/lib/see-act.js --url http://localhost:<N>/ shot — READ the screenshot, log the step. Act on what you see: click, scroll, type, then re-observe, logging each step. Prefer aria (cheap text) to find controls; screenshot when the view changes and for your final verdict frames (one desktop, one mobile). If a click exits non-zero, do NOT retry the same ref blindly: re-run aria first (refs go stale between invocations), then click the fresh ref exactly once. If it still fails, log the failure and move on — a flaky control is a finding, not a loop.\n" +
|
|
2091
|
-
"d.
|
|
2141
|
+
"d. Reach: with the session protocol, any flow reachable by N in-page actions is drivable — open the dialog, then confirm it, then judge the result. Without a session (one-shot invocations), anything reachable by (navigate, one action) is testable and sequences needing prior in-page state are not — use a session for those. Report NOT POSSIBLE only when the tooling itself fails (session-start exits 3): a flow you could not reach is not NOT POSSIBLE — name the exact step that stopped you in verdict.json's missing evidence and continue with the mechanical checks.\n" +
|
|
2092
2142
|
"e. Judge as a user against the task description: is the reported bug fixed AND is nothing else visibly broken? Look for broken layout, overflow, missing or wrong content, stale data, and console errors. Compare against the task's expected behavior, never against source code (you are code-blind). Every frame you captured is already archived under " + crewHome + "/task-evidence/" + taskId + "/postchange/ and indexed in ooda-log.jsonl. A frame you did not read is not evidence. Loading, error, or blank frames never pass. If you cannot complete the loop, say exactly which steps are missing — unknown is not PASS.\n" +
|
|
2093
2143
|
"f. Kill ONLY the server you started: pkill -f 'serve-artifact[.]js.*--tag " + taskId + "-qa' — never another task's server. (The [.] keeps pkill from matching its own command line.) Do not leave it running.\n" +
|
|
2094
2144
|
"Then continue with the mechanical checks below. Your VERDICT covers both the visual and the mechanical checks.\n\n" +
|
|
@@ -2172,12 +2222,12 @@ while (i < STEPS.length) {
|
|
|
2172
2222
|
"The returned events are filtered to this task. They contain notes and decisions from prior phases.\n\n";
|
|
2173
2223
|
}
|
|
2174
2224
|
|
|
2175
|
-
// Work agent returns the
|
|
2176
|
-
//
|
|
2177
|
-
//
|
|
2178
|
-
//
|
|
2179
|
-
//
|
|
2180
|
-
// text by extractVerdict below — never by an agent.
|
|
2225
|
+
// Work agent returns prose on the schema-less courier; the workflow
|
|
2226
|
+
// consumes the return as a plain string (blocker 43, 2026-09-22: the
|
|
2227
|
+
// prompt names no JSON envelope — nothing parses one, and the runtime's
|
|
2228
|
+
// JSON-candidate scan discards brace-shaped prose, which is what the
|
|
2229
|
+
// transport retry guards). The verdict is still extracted deterministically
|
|
2230
|
+
// from the report text by extractVerdict below — never by an agent.
|
|
2181
2231
|
var workPromptBase =
|
|
2182
2232
|
TOOL_CHECK_PREAMBLE +
|
|
2183
2233
|
"Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
|
|
@@ -2190,8 +2240,7 @@ while (i < STEPS.length) {
|
|
|
2190
2240
|
"\n## Instructions\n\n" + eventPreamble + instructions + "\n\n" +
|
|
2191
2241
|
"CONSTRAINT: Do NOT call logevent or upsertagentsession — the workflow handles all phase tracking after your step completes.\n\n" +
|
|
2192
2242
|
"Stay in character. Do the work thoroughly.\n\n" +
|
|
2193
|
-
"
|
|
2194
|
-
"The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
2243
|
+
"Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
2195
2244
|
var workKeyBase = "work-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "");
|
|
2196
2245
|
// See docs/decisions/qa-reproduce.md#wrong-layer-guard: reproduce must run at the correct layer.
|
|
2197
2246
|
var reproEvidenceDir = crewHome + "/task-evidence/" + taskId + "/repro";
|
package/workflows/chore.js
CHANGED
|
@@ -197,7 +197,7 @@ if (!taskId) {
|
|
|
197
197
|
throw new Error("task_id is required in args");
|
|
198
198
|
}
|
|
199
199
|
|
|
200
|
-
// See docs/decisions/workflow-core.md#closeout-envelope: the work
|
|
200
|
+
// See docs/decisions/workflow-core.md#closeout-envelope: the schema-less work courier returns prose, consumed as a plain string; verdict extracted mechanically.
|
|
201
201
|
const VERDICT_STEPS = ["Build", "Review", "Integrate", "Publish"];
|
|
202
202
|
function extractVerdict(workerText) {
|
|
203
203
|
// The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
|
|
@@ -231,7 +231,6 @@ function buildVerdictReaskPrompt(stepName, workerText) {
|
|
|
231
231
|
"Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
|
|
232
232
|
"if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
|
|
233
233
|
"Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
|
|
234
|
-
"Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
|
|
235
234
|
"The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
|
|
236
235
|
}
|
|
237
236
|
async function reaskVerdict(stepName, reworkSuffix, workerText) {
|
|
@@ -304,8 +303,8 @@ var TOOL_CHECK_PREAMBLE =
|
|
|
304
303
|
"3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
|
|
305
304
|
"Then do the assignment below.\n\n";
|
|
306
305
|
function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
|
|
307
|
-
// reason: "discarded" (the runtime threw the output away
|
|
308
|
-
//
|
|
306
|
+
// reason: "discarded" (the runtime threw the output away — the JSON-candidate
|
|
307
|
+
// scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
|
|
309
308
|
// nothing usable), "no-tools" (the worker's TOOL CHECK reported
|
|
310
309
|
// artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
|
|
311
310
|
// reported shell_transport: unavailable).
|
|
@@ -316,11 +315,11 @@ function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason)
|
|
|
316
315
|
? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
|
|
317
316
|
: reason === "no-transport"
|
|
318
317
|
? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
|
|
319
|
-
: "your previous attempt's output could not
|
|
318
|
+
: "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
|
|
320
319
|
return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
|
|
321
320
|
"First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
|
|
322
321
|
"if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
|
|
323
|
-
"Then return your report
|
|
322
|
+
"Then return your report in exactly the shape specified above.";
|
|
324
323
|
}
|
|
325
324
|
// parseClassifyFerry — room #26 blocker 42 (2026-09-21). Shape-constrained
|
|
326
325
|
// ferry parser for the classify-branch agent() call. The schema buys SHAPE,
|
|
@@ -794,6 +793,50 @@ if (CAPTURE_INDEX < 0) throw new Error("STEPS missing 'Capture' step");
|
|
|
794
793
|
const REWORK_STEP = STEPS[BUILD_INDEX].name;
|
|
795
794
|
const MAX_REWORK = 2;
|
|
796
795
|
let reworkCount = 0;
|
|
796
|
+
// Blocker 46 (2026-09-22): the budget is task-durable, not per-run. A run
|
|
797
|
+
// ends on every operational failure (agent timeout, worktree-confinement
|
|
798
|
+
// failure) and the dispatcher launches a fresh run — the old `let ... = 0`
|
|
799
|
+
// reset the counter on each launch, so a task whose runs kept ending
|
|
800
|
+
// operationally between rejections reworked unboundedly while the designed
|
|
801
|
+
// park stayed unreachable (room #29 J1: 3 rejections across 4 runs, never
|
|
802
|
+
// parked). The counter is seeded from the task's durable event stream at
|
|
803
|
+
// run start (seedReworkCount, after the pin block); in-run increments are
|
|
804
|
+
// unchanged.
|
|
805
|
+
// Blocker 46: the first-claim test used reworkCount === 0 as a proxy for
|
|
806
|
+
// "this run is entering its first step". The counter is now seeded from
|
|
807
|
+
// task history, so the proxy lies on re-run launches — use an explicit
|
|
808
|
+
// run-local flag instead. The first claim (self-claim with raced-duplicate
|
|
809
|
+
// stand-down) must fire exactly once per run.
|
|
810
|
+
let firstClaimDone = false;
|
|
811
|
+
|
|
812
|
+
// Blocker 46: count verdict-level `rejected` events (Review/QA rejections —
|
|
813
|
+
// record-phase writes the event type equal to the session status) in the
|
|
814
|
+
// task's durable event stream. Operational failure ends are `failed`, never
|
|
815
|
+
// `rejected`, so they must not spend retry headroom. A failed seed must not
|
|
816
|
+
// strand the run: start at 0 and let the in-run budget govern this run; the
|
|
817
|
+
// seed is retried on the next run.
|
|
818
|
+
async function seedReworkCount() {
|
|
819
|
+
try {
|
|
820
|
+
var seed = await agent(
|
|
821
|
+
"Read this task's event history.\n" +
|
|
822
|
+
"Run in shell and return the stdout verbatim:\n" + crewCmd("get-events", { task_id: taskId, limit: 100 }) + "\n" +
|
|
823
|
+
"Count the events whose type is exactly \"rejected\".\n" +
|
|
824
|
+
"Return JSON { \"rejected_count\": <integer> } and nothing else.",
|
|
825
|
+
{
|
|
826
|
+
key: "rework-seed-" + taskId,
|
|
827
|
+
label: "Seeding shared rework budget from task history",
|
|
828
|
+
schema: { type: "object", properties: { rejected_count: { type: "number" } }, required: ["rejected_count"] }
|
|
829
|
+
}
|
|
830
|
+
);
|
|
831
|
+
var n = Math.floor(Number(seed && seed.rejected_count) || 0);
|
|
832
|
+
if (!(n >= 0)) n = 0;
|
|
833
|
+
if (n > 0) log("Shared rework budget seeded from task history: " + n + " prior rejection(s)");
|
|
834
|
+
return n;
|
|
835
|
+
} catch (e) {
|
|
836
|
+
log("Rework-budget seed unavailable (" + ((e && e.message) || String(e)).slice(0, 120) + ") — starting at 0");
|
|
837
|
+
return 0;
|
|
838
|
+
}
|
|
839
|
+
}
|
|
797
840
|
let rejectionNotes = inputs.rejection_notes || "";
|
|
798
841
|
let mapperSpec = "";
|
|
799
842
|
// Visual verdict state: Sage's experiential flag (true/false/null until the
|
|
@@ -905,6 +948,11 @@ log("Lifecycle scripts pinned to " + RUN_LIB);
|
|
|
905
948
|
CREW_API = CREW_API_PINNED;
|
|
906
949
|
log("Crew API pinned to " + CREW_API);
|
|
907
950
|
|
|
951
|
+
// Blocker 46: seed the task-durable rework budget from the task's event
|
|
952
|
+
// history before the phase loop. Pinned copy, so the seed reads the same
|
|
953
|
+
// release this run executes.
|
|
954
|
+
reworkCount = await seedReworkCount();
|
|
955
|
+
|
|
908
956
|
// Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
|
|
909
957
|
// minted at this run's first claim (never a PID — short-lived agent PIDs
|
|
910
958
|
// made every concurrent acquire take the stale path). Set exactly once on
|
|
@@ -913,7 +961,7 @@ let lockHolder = taskId;
|
|
|
913
961
|
|
|
914
962
|
while (i < STEPS.length) {
|
|
915
963
|
const step = STEPS[i];
|
|
916
|
-
const isFirstClaim = (i === startStepIndex &&
|
|
964
|
+
const isFirstClaim = (i === startStepIndex && !firstClaimDone);
|
|
917
965
|
|
|
918
966
|
phase(step.name);
|
|
919
967
|
log(step.name + " step (" + step.identity + ") for task " + taskId);
|
|
@@ -1085,6 +1133,7 @@ while (i < STEPS.length) {
|
|
|
1085
1133
|
return { status: "duplicate", task_id: taskId, reason: "task already claimed by another run" };
|
|
1086
1134
|
}
|
|
1087
1135
|
activeSessionId = claimResult.session_id;
|
|
1136
|
+
firstClaimDone = true; // blocker 46: exactly once per run
|
|
1088
1137
|
await telemetryEvent("claim", step.name + " claimed");
|
|
1089
1138
|
// Flush telemetry before proceeding — the claim is a critical point.
|
|
1090
1139
|
// If a subsequent agent() fails, we want the claim event persisted.
|
|
@@ -1900,12 +1949,12 @@ while (i < STEPS.length) {
|
|
|
1900
1949
|
"The returned events are filtered to this task. They contain notes and decisions from prior phases.\n\n";
|
|
1901
1950
|
}
|
|
1902
1951
|
|
|
1903
|
-
// Work agent returns the
|
|
1904
|
-
//
|
|
1905
|
-
//
|
|
1906
|
-
//
|
|
1907
|
-
//
|
|
1908
|
-
// text by extractVerdict below — never by an agent.
|
|
1952
|
+
// Work agent returns prose on the schema-less courier; the workflow
|
|
1953
|
+
// consumes the return as a plain string (blocker 43, 2026-09-22: the
|
|
1954
|
+
// prompt names no JSON envelope — nothing parses one, and the runtime's
|
|
1955
|
+
// JSON-candidate scan discards brace-shaped prose, which is what the
|
|
1956
|
+
// transport retry guards). The verdict is still extracted deterministically
|
|
1957
|
+
// from the report text by extractVerdict below — never by an agent.
|
|
1909
1958
|
var workPromptBase =
|
|
1910
1959
|
TOOL_CHECK_PREAMBLE +
|
|
1911
1960
|
"Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
|
|
@@ -1913,8 +1962,7 @@ while (i < STEPS.length) {
|
|
|
1913
1962
|
"Task: " + taskTitle + "\nTask ID: " + taskId + "\nDescription: " + taskDescription + "\nStep: " + step.name + "\n" +
|
|
1914
1963
|
(step.name !== "Review" ? "Crew API: " + CREW_API + "\n" : "") +
|
|
1915
1964
|
"\n## Instructions\n\n" + eventPreamble + instructions + "\n\nCONSTRAINT: Do NOT call logevent or upsertagentsession — the workflow handles all phase tracking after your step completes.\n\nStay in character. Do the work thoroughly.\n\n" +
|
|
1916
|
-
"
|
|
1917
|
-
"The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
1965
|
+
"Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
1918
1966
|
var workKeyBase = "work-" + step.name + (reworkCount > 0 ? "-r" + reworkCount : "");
|
|
1919
1967
|
var workerResult = null;
|
|
1920
1968
|
var workAttempts = [];
|
package/workflows/docs.js
CHANGED
|
@@ -269,7 +269,6 @@ while (i < STEPS.length) {
|
|
|
269
269
|
"Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
|
|
270
270
|
"if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
|
|
271
271
|
"Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
|
|
272
|
-
"Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
|
|
273
272
|
"The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
|
|
274
273
|
}
|
|
275
274
|
async function reaskVerdict(stepName, reworkSuffix, workerText) {
|
|
@@ -329,8 +328,8 @@ while (i < STEPS.length) {
|
|
|
329
328
|
"3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
|
|
330
329
|
"Then do the assignment below.\n\n";
|
|
331
330
|
function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
|
|
332
|
-
// reason: "discarded" (the runtime threw the output away
|
|
333
|
-
//
|
|
331
|
+
// reason: "discarded" (the runtime threw the output away — the JSON-candidate
|
|
332
|
+
// scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
|
|
334
333
|
// nothing usable), "no-tools" (the worker's TOOL CHECK reported
|
|
335
334
|
// artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
|
|
336
335
|
// reported shell_transport: unavailable).
|
|
@@ -341,11 +340,11 @@ while (i < STEPS.length) {
|
|
|
341
340
|
? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
|
|
342
341
|
: reason === "no-transport"
|
|
343
342
|
? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
|
|
344
|
-
: "your previous attempt's output could not
|
|
343
|
+
: "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
|
|
345
344
|
return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
|
|
346
345
|
"First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
|
|
347
346
|
"if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
|
|
348
|
-
"Then return your report
|
|
347
|
+
"Then return your report in exactly the shape specified above.";
|
|
349
348
|
}
|
|
350
349
|
// parseToolSignals - bug 3472bf36. The work agent's TOOL CHECK emits two
|
|
351
350
|
// exact signal lines: artifact_tools: ok|missing and
|
|
@@ -432,20 +431,19 @@ function describeWorkAgentFailure(stepName, identity, attempts) {
|
|
|
432
431
|
"\nWrite your review as plain prose — findings, then decision. End your report with exactly one line: VERDICT: PASS if it passes, VERDICT: FAIL if it fails.";
|
|
433
432
|
}
|
|
434
433
|
|
|
435
|
-
// Work agent returns the
|
|
436
|
-
//
|
|
437
|
-
//
|
|
438
|
-
//
|
|
439
|
-
//
|
|
440
|
-
// text by extractVerdict below — never by an agent.
|
|
434
|
+
// Work agent returns prose on the schema-less courier; the workflow
|
|
435
|
+
// consumes the return as a plain string (blocker 43, 2026-09-22: the
|
|
436
|
+
// prompt names no JSON envelope — nothing parses one, and the runtime's
|
|
437
|
+
// JSON-candidate scan discards brace-shaped prose, which is what the
|
|
438
|
+
// transport retry guards). The verdict is still extracted deterministically
|
|
439
|
+
// from the report text by extractVerdict below — never by an agent.
|
|
441
440
|
var workPromptBase =
|
|
442
441
|
TOOL_CHECK_PREAMBLE +
|
|
443
442
|
"Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
|
|
444
443
|
"## Your Assignment\n\n" +
|
|
445
444
|
"Task: " + taskTitle + "\nTask ID: " + taskId + "\nDescription: " + taskDescription + "\nStep: " + step.name + "\n\n" +
|
|
446
445
|
"## Instructions\n\n" + instructions + "\n\nCONSTRAINT: Do NOT call the crew API directly — the workflow handles all phase tracking after your step completes.\n\nStay in character. All file work under " + REPO_PATH + "/.\n\n" +
|
|
447
|
-
"
|
|
448
|
-
"The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
446
|
+
"Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
449
447
|
var workKeyBase = "work-" + step.name + (reworkCount > 0 ? "-r" + reworkCount : "");
|
|
450
448
|
var workerResult = null;
|
|
451
449
|
var workAttempts = [];
|