muse-crew 0.14.7 → 0.14.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -139,7 +139,7 @@ if (!taskId) {
139
139
  throw new Error("task_id is required in args");
140
140
  }
141
141
 
142
- // See docs/decisions/workflow-core.md#closeout-envelope: the work agent returns the runtime's native envelope; verdict extracted mechanically.
142
+ // See docs/decisions/workflow-core.md#closeout-envelope: the schema-less work courier returns prose, consumed as a plain string; verdict extracted mechanically.
143
143
  const VERDICT_STEPS = ["Build", "Review", "QA", "Reproduce", "Integrate", "Publish"];
144
144
  function extractVerdict(workerText) {
145
145
  // The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
@@ -229,7 +229,6 @@ function buildVerdictReaskPrompt(stepName, workerText) {
229
229
  "Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
230
230
  "if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
231
231
  "Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
232
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
233
232
  "The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
234
233
  }
235
234
  async function reaskVerdict(stepName, reworkSuffix, workerText) {
@@ -302,8 +301,8 @@ var TOOL_CHECK_PREAMBLE =
302
301
  "3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
303
302
  "Then do the assignment below.\n\n";
304
303
  function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
305
- // reason: "discarded" (the runtime threw the output away - it could not be
306
- // machine-read), "empty" (agent() returned without throwing but produced
304
+ // reason: "discarded" (the runtime threw the output away — the JSON-candidate
305
+ // scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
307
306
  // nothing usable), "no-tools" (the worker's TOOL CHECK reported
308
307
  // artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
309
308
  // reported shell_transport: unavailable).
@@ -314,12 +313,148 @@ function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason)
314
313
  ? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
315
314
  : reason === "no-transport"
316
315
  ? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
317
- : "your previous attempt's output could not be machine-read as JSON and was discarded";
316
+ : "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
318
317
  return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
319
318
  "First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
320
319
  "if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
321
- "Then return your report as JSON in exactly the shape specified above.";
320
+ "Then return your report in exactly the shape specified above.";
322
321
  }
322
+ // parseClassifyFerry — room #26 blocker 42 (2026-09-21). Shape-constrained
323
+ // ferry parser for the classify-branch agent() call. The schema buys SHAPE,
324
+ // not provenance: the fields are LLM-authored, so the parser stays
325
+ // defensive. Byte-identical across standard.js, bugfix.js, chore.js
326
+ // (pinned by tests/already-merged.test.js). Triplication is structural:
327
+ // each workflow ships as a standalone launch payload (240 KiB budget,
328
+ // individually git-archived) — there is no shared module to hold it.
329
+ // The schema is the guard; the prompt is instruction, never load-bearing.
330
+ // Returns { state, diagField, diagText, failReason }:
331
+ // state: "has-work" | "already-merged:<40-hex>" | "empty-no-work" | null
332
+ // (null = unparseable/ambiguous/channel violation — UNKNOWN)
333
+ // diagField: "stderr" | "stdout" | null (which field sourced the DIAG pin)
334
+ // diagText: the matched DIAG line, or ""
335
+ // failReason: computed reason, for the retry trailer and the park note
336
+ function parseClassifyFerry(bsResult) {
337
+ var bsStdout = (bsResult && typeof bsResult.stdout === "string") ? bsResult.stdout : null;
338
+ var bsStderr = (bsResult && typeof bsResult.stderr === "string") ? bsResult.stderr : null;
339
+ if (bsStdout === null || bsStderr === null) {
340
+ return { state: null, diagField: null, diagText: "",
341
+ failReason: "classifier return fields missing or non-string (stdout and stderr must both be strings)" };
342
+ }
343
+ // State channel: stdout ONLY. A BRANCH_STATE match on stderr is never
344
+ // trusted — workers merge streams constantly, and trusting a channel
345
+ // swap as truth is the wrong shape. Search, never anchor (blocker 11):
346
+ // the ferry labels output ("stdout: BRANCH_STATE: has-work"). Collect
347
+ // ALL matches and dedupe normalized values: exactly one distinct value
348
+ // is truth; zero or two-or-more is UNKNOWN — alternation order never
349
+ // adjudicates.
350
+ var bsRe = /\bBRANCH_STATE:\s*(has-work|already-merged:[0-9a-f]{40}|empty-no-work)(?![\w-])/gi;
351
+ var bsSeen = {};
352
+ var bsM;
353
+ while ((bsM = bsRe.exec(bsStdout)) !== null) {
354
+ bsSeen[bsM[1].toLowerCase()] = true;
355
+ }
356
+ var bsStates = Object.keys(bsSeen);
357
+ if (bsStates.length !== 1) {
358
+ return { state: null, diagField: null, diagText: "",
359
+ failReason: bsStates.length === 0
360
+ ? "no BRANCH_STATE match in the stdout field"
361
+ : "ambiguous BRANCH_STATE values in stdout (" + bsStates.length + " distinct)" };
362
+ }
363
+ var bsState = bsStates[0];
364
+ // DIAG pin: search both fields, stderr first. Keep the DIAG: prefix — a
365
+ // prose mention without it must not satisfy the pin.
366
+ var bsDiagRe = /DIAG:\s*classify-branch\s+records=\d+\s+resolvable=\d+/;
367
+ var bsDiagField = bsDiagRe.test(bsStderr) ? "stderr" : (bsDiagRe.test(bsStdout) ? "stdout" : null);
368
+ var bsDiagText = "";
369
+ if (bsDiagField !== null) {
370
+ bsDiagText = bsDiagRe.exec(bsDiagField === "stderr" ? bsStderr : bsStdout)[0];
371
+ }
372
+ if (bsState.indexOf("already-merged:") === 0 && bsDiagField === null) {
373
+ // already-merged's sha flows into Cass's `git diff <sha>^1 <sha>` — a
374
+ // hallucinated-but-well-formed sha is a false-PASS vector. The script
375
+ // emits exactly one DIAG on every path that emits a BRANCH_STATE, so a
376
+ // missing pin means stderr wasn't faithfully ferried: UNKNOWN.
377
+ // Residual risk, named (blocker-42 fib 4): a worker could fabricate
378
+ // BOTH fields wholesale — the pin is a tripwire, not proof of
379
+ // execution. The workflow has no exec channel for a deterministic
380
+ // cross-check; if the platform ever offers one, verify the sha
381
+ // against the task's durable merge records directly.
382
+ return { state: null, diagField: null, diagText: "",
383
+ failReason: "already-merged without the DIAG pin — stderr not faithfully ferried" };
384
+ }
385
+ return { state: bsState, diagField: bsDiagField, diagText: bsDiagText, failReason: "" };
386
+ }
387
+
388
+ // classifyRetryTrailer — room #26 blocker 42 (2026-09-21). Corrective (not
389
+ // flat) trailer for the in-place classify retries. The failure mode was
390
+ // systematic labeling, not a coin flip: the trailer carries the COMPUTED
391
+ // parse-failure reason, the failed return truncated as a negative example,
392
+ // and the explicit field mapping — feedback from the actual failure,
393
+ // following the buildTransportRetryTrailer idiom. Bound: 2 retries
394
+ // (unmeasured house convention, matches the shared rework bound of 2 —
395
+ // against room retry data). Byte-identical across the three workflows
396
+ // (pinned by tests/already-merged.test.js).
397
+ function classifyRetryTrailer(reason, failedReturn) {
398
+ // Pass-2 simplification: the reason string already carries the truth
399
+ // (throw vs parse failure), so one neutral sentence replaces the
400
+ // throw/parse branch — the stringly-typed prefix contract between the
401
+ // call site and this function is deleted, not moved. Prompt accuracy,
402
+ // not prompt hardening.
403
+ return "\n\nCLASSIFY RETRY: the previous classifier attempt failed (" + reason + "). " +
404
+ "Return ONLY the two named fields: { \"stdout\": \"<the command's exact stdout>\", \"stderr\": \"<the command's exact stderr>\" }. " +
405
+ "Put each stream in its named field — do not add labels like \"stdout:\" in front of the content. " +
406
+ "Previous attempt detail (do not repeat this shape): " + String(failedReturn).slice(0, 200);
407
+ }
408
+
409
+ // parseVersionFerry — review pass 2, blocker-42 class (2026-09-21). The npm
410
+ // version-check ferry had the full pre-42 shape: schema-less agent(),
411
+ // line-anchored parser, and an unparseable fallback asserting a POSITIVE
412
+ // touched claim that spent the shared rework budget. Same treatment as
413
+ // parseClassifyFerry: shape-constrained {stdout} schema, search-and-dedupe
414
+ // parse, UNKNOWN on unparseable with in-place retries, honest park.
415
+ // Byte-identical across standard.js, bugfix.js, chore.js
416
+ // (pinned by tests/already-merged.test.js).
417
+ // Returns { versionState, failReason }:
418
+ // versionState: "touched" | "clean" | null (null = UNKNOWN)
419
+ // failReason: computed reason, for the retry trailer and the park note
420
+ function parseVersionFerry(vtResult) {
421
+ var vtStdout = (vtResult && typeof vtResult.stdout === "string") ? vtResult.stdout : null;
422
+ if (vtStdout === null) {
423
+ return { versionState: null,
424
+ failReason: "version-check return field missing or non-string (stdout must be a string)" };
425
+ }
426
+ // Search, never anchor (blocker 11): the ferry labels output
427
+ // ("stdout: VERSION_CLEAN"). Collect ALL matches and dedupe normalized
428
+ // values: exactly one distinct value is truth; zero or two-or-more is
429
+ // UNKNOWN — alternation order never adjudicates.
430
+ var vtRe = /\bVERSION_(TOUCHED|CLEAN)(?![\w-])/gi;
431
+ var vtSeen = {};
432
+ var vtM;
433
+ while ((vtM = vtRe.exec(vtStdout)) !== null) {
434
+ vtSeen[vtM[1].toUpperCase()] = true;
435
+ }
436
+ var vtStates = Object.keys(vtSeen);
437
+ if (vtStates.length !== 1) {
438
+ return { versionState: null,
439
+ failReason: vtStates.length === 0
440
+ ? "no VERSION_(TOUCHED|CLEAN) match in the stdout field"
441
+ : "ambiguous VERSION values in stdout (" + vtStates.length + " distinct)" };
442
+ }
443
+ return { versionState: vtStates[0] === "CLEAN" ? "clean" : "touched", failReason: "" };
444
+ }
445
+
446
+ // versionRetryTrailer — review pass 2, blocker-42 class (2026-09-21).
447
+ // Neutral corrective trailer for the in-place version-check retries: the
448
+ // computed parse-failure reason plus the failed return as a negative
449
+ // example. Byte-identical across the three workflows
450
+ // (pinned by tests/already-merged.test.js).
451
+ function versionRetryTrailer(reason, failedReturn) {
452
+ return "\n\nVERSION-CHECK RETRY: the previous version-check attempt failed (" + reason + "). " +
453
+ "Return ONLY the named field: { \"stdout\": \"<the command's exact stdout>\" }. " +
454
+ "Put the command's stdout in the named field — do not add labels like \"stdout:\" in front of the content. " +
455
+ "Previous attempt detail (do not repeat this shape): " + String(failedReturn).slice(0, 200);
456
+ }
457
+
323
458
  // parseToolSignals - bug 3472bf36. The work agent's TOOL CHECK emits two
324
459
  // exact signal lines: artifact_tools: ok|missing and
325
460
  // shell_transport: ok|unavailable. The workflow reads ONLY these lines.
@@ -696,7 +831,7 @@ let alreadyMergedSha = null;
696
831
  let branchState = null; // "has-work" | "already-merged:<sha>" | "empty-no-work"
697
832
  let buildClaimedNoDiff = false; // Build declared plain `repo_diff: none` (no repo change — runtime-state deliverable, or believed already-merged). Same-process Build report only; there is no cross-process hydration — an empty branch on a fresh dispatch with no task-attributed merge fails mechanically.
698
833
  let versionTouched = false; // npm: branch changed package.json's `version` (mechanical git fact, blocker-34 follow-up)
699
- let branchStateErr = ""; // classifier failure detail (fail-closed grounds)
834
+ let branchStateErr = ""; // classifier failure detail (retry trailer + park note grounds)
700
835
  // Deterministic publish target — computed by the workflow (registry base +
701
836
  // bumpVersion), never by the Publish agent.
702
837
  let publishTarget = null; // { base, scope, target }
@@ -1176,43 +1311,87 @@ while (i < STEPS.length) {
1176
1311
  // (classify-branch in the lifecycle script): has-work |
1177
1312
  // already-merged:<sha> | empty-no-work. Attribution is identity-first
1178
1313
  // from git, then the task's own durable merge records — there is no
1179
- // agent-authored declared-sha input and no cross-process hydration
1180
- // ferry (the re-entry case it served is covered by the durable record
1181
- // → classifier directly). An unparseable result fails closed as
1182
- // empty-no-work — an empty branch never passes Review silently, and
1183
- // the bounce to Build gives the next round a chance to classify.
1184
- try {
1185
- var bsResult = await agent(
1314
+ // agent-authored declared-sha input.
1315
+ // Blocker 42 (2026-09-21): the classifier output crosses an agent()
1316
+ // ferry — the old comment's "no cross-process hydration ferry" was a
1317
+ // fib. Machine facts travel in structured fields ({stdout, stderr}
1318
+ // schema) now, never in prose to be "returned verbatim". The schema
1319
+ // is the guard; the prompt is instruction, never load-bearing.
1320
+ // Unparseable is UNKNOWN, never a positive empty-no-work claim (room
1321
+ // #27: a labeled ferry — "stdout: BRANCH_STATE: has-work" — defeated
1322
+ // the ^-anchored parsers, and the empty-no-work fallback burned the
1323
+ // whole shared rework budget on a branch with real work). UNKNOWN
1324
+ // retries the classify call IN PLACE (up to 2, fresh -u1/-u2 keys — a
1325
+ // full Build round cannot fix an instrument failure, and the only
1326
+ // in-process bounce path spends the shared budget unconditionally).
1327
+ // Instrumentation failures NEVER touch totalReworkCount: an instrument
1328
+ // failure is not a build-quality failure. Exhaustion parks honestly
1329
+ // as unclassifiable — "unknown keeps the task alive (bounded)"; the
1330
+ // park on exhaustion is the fail-closed part. The UNKNOWN path never
1331
+ // asserts empty-no-work: no log line or park note presents it as the
1332
+ // classification. The string can still appear inside quoted evidence
1333
+ // (the verbatim failed return travels in the trailer and the note).
1334
+ // INVARIANT: branchState leaves this block holding a known value, or
1335
+ // the step has parked — unknown never reaches any consumer below.
1336
+ // Re-classify on EVERY Review entry: rework rewinds this step
1337
+ // in-process, and the branch changed between rounds. The loop's
1338
+ // null-gate governs only the in-place instrument retries within
1339
+ // this pass — without the reset, a second Review pass would trust
1340
+ // the first pass's stale classification and a real fix commit
1341
+ // would never "become has-work".
1342
+ branchState = null;
1343
+ branchStateErr = "";
1344
+ alreadyMergedSha = null;
1345
+ var classifyFailedReturn = "";
1346
+ for (var classifyAttempt = 0; classifyAttempt <= 2 && branchState === null; classifyAttempt++) {
1347
+ var classifyPrompt =
1186
1348
  "Classify the task branch state. This is a mechanical git check, not a judgment call.\n" +
1187
- "Run in shell and return the stdout verbatim, then the single stderr line starting with `DIAG:` verbatim:\n" +
1188
- LIFECYCLE_ENV + LIFECYCLE + " classify-branch " + taskId,
1189
- { key: "classify-branch" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Classifying branch state (mechanical)" }
1190
- );
1191
- var bsStr = (typeof bsResult === "string") ? bsResult : JSON.stringify(bsResult);
1192
- var bsMatch = /^\s*BRANCH_STATE:\s*(has-work|already-merged:[0-9a-f]{40}|empty-no-work)\s*$/im.exec(bsStr);
1193
- if (bsMatch) {
1194
- branchState = bsMatch[1].toLowerCase();
1195
- } else {
1196
- branchStateErr = "unparseable classifier output: " + bsStr.slice(0, 120);
1197
- }
1198
- // DIAG pin (blocker 34 design v2.1): the merge-record scan is
1199
- // load-bearing with no fallback — the classifier's stderr counter is
1200
- // logged verbatim so silent record loss surfaces in the evidence.
1201
- var bsDiagMatch = /^[ \t]*DIAG:\s*classify-branch\s+records=\d+\s+resolvable=\d+[ \t]*$/im.exec(bsStr);
1202
- if (bsDiagMatch) {
1203
- log(bsDiagMatch[0].trim());
1204
- } else {
1205
- log("classify-branch: no DIAG: record-scan counter returned (scan unverifiable)");
1349
+ "Run in shell: " + LIFECYCLE_ENV + LIFECYCLE + " classify-branch " + taskId + "\n" +
1350
+ "Return the two streams as the two named fields and nothing else: { \"stdout\": \"<the command's exact stdout>\", \"stderr\": \"<the command's exact stderr>\" }." +
1351
+ (classifyAttempt === 0 ? "" : classifyRetryTrailer(branchStateErr, classifyFailedReturn));
1352
+ try {
1353
+ var bsResult = await agent(
1354
+ classifyPrompt,
1355
+ { key: "classify-branch" + (classifyAttempt === 0 ? "" : "-u" + classifyAttempt) + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""),
1356
+ label: "Classifying branch state (mechanical)" + (classifyAttempt === 0 ? "" : " (instrument retry " + classifyAttempt + " of 2)"),
1357
+ schema: { type: "object", properties: { stdout: { type: "string" }, stderr: { type: "string" } }, required: ["stdout", "stderr"] } }
1358
+ );
1359
+ classifyFailedReturn = (bsResult && typeof bsResult === "object" ? JSON.stringify(bsResult) : String(bsResult)).slice(0, 500);
1360
+ var classifyParsed = parseClassifyFerry(bsResult);
1361
+ if (classifyParsed.state !== null) {
1362
+ branchState = classifyParsed.state;
1363
+ branchStateErr = "";
1364
+ // DIAG pin: advisory for has-work/empty-no-work (blocker-34 v2.1
1365
+ // semantics — logged with its source field so silent record loss
1366
+ // surfaces in the evidence); already-merged is gated on the pin
1367
+ // inside parseClassifyFerry.
1368
+ if (classifyParsed.diagField !== null) {
1369
+ log("classify-branch DIAG pin (" + classifyParsed.diagField + " field): " + classifyParsed.diagText);
1370
+ } else {
1371
+ log("classify-branch: no DIAG: record-scan counter returned (scan unverifiable)");
1372
+ }
1373
+ } else {
1374
+ branchStateErr = classifyParsed.failReason;
1375
+ if (classifyAttempt < 2) {
1376
+ log("classify-branch unparseable (" + branchStateErr + ") — retrying in place (attempt " + (classifyAttempt + 1) + " of 2)");
1377
+ }
1378
+ }
1379
+ } catch (bsErr) {
1380
+ classifyFailedReturn = String(bsErr && bsErr.message || bsErr).slice(0, 500);
1381
+ branchStateErr = "classifier call failed: " + String(bsErr && bsErr.message || bsErr).slice(0, 120);
1382
+ if (classifyAttempt < 2) {
1383
+ log("classify-branch instrument failure (" + branchStateErr + ") — retrying in place (attempt " + (classifyAttempt + 1) + " of 2)");
1384
+ }
1206
1385
  }
1207
- } catch (bsErr) {
1208
- branchStateErr = "classifier call failed: " + String(bsErr && bsErr.message || bsErr).slice(0, 120);
1209
1386
  }
1210
- if (!branchState) {
1211
- branchState = "empty-no-work";
1212
- log("Branch-state classification failed (" + branchStateErr + ") — failing closed as empty-no-work");
1213
- } else if (branchState.indexOf("already-merged:") === 0) {
1387
+ if (branchState === null) {
1388
+ // Exhaustion: park honestly. An instrument failure is not a
1389
+ // measurement of emptiness — never "empty-no-work".
1390
+ return await parkTask("Branch state unclassifiable after 2 instrument retries (" + branchStateErr + "; last failure: " + classifyFailedReturn + ") — the classifier instrument failed, not the branch. Human attention needed: inspect the task branch directly to determine its state — no trustworthy branch classification was produced.");
1391
+ }
1392
+ if (branchState.indexOf("already-merged:") === 0) {
1214
1393
  alreadyMergedSha = branchState.slice("already-merged:".length);
1215
- log("Branch state already-merged: " + alreadyMergedSha + " (mechanically verified) — Cass reviews the frozen merge diff");
1394
+ log("Branch state already-merged: " + alreadyMergedSha + " (DIAG pin present — tripwire satisfied) — Cass reviews the frozen merge diff");
1216
1395
  } else {
1217
1396
  log("Branch state: " + branchState);
1218
1397
  }
@@ -1224,23 +1403,55 @@ while (i < STEPS.length) {
1224
1403
  // git fact, not a reviewer judgment. For npm projects the workflow checks
1225
1404
  // it here, before Cass is dispatched — a touched `version` fails Review
1226
1405
  // mechanically (versions are assigned at publish time, never in
1227
- // branches). Fail closed: anything but an explicit VERSION_CLEAN counts
1228
- // as touched.
1406
+ // branches). Review pass 2 (blocker-42 class): this ferry had the full
1407
+ // pre-42 shape — schema-less agent(), line-anchored parser, and an
1408
+ // unparseable fallback asserting a POSITIVE touched claim that spent the
1409
+ // shared rework budget. Same treatment as the classifier:
1410
+ // shape-constrained {stdout} schema, search-and-dedupe parse, UNKNOWN
1411
+ // with in-place retries, honest park. An instrument failure is not a
1412
+ // measurement of touched — never assert touched from unparseable output.
1229
1413
  if (PUBLISH_TYPE === "npm") {
1230
- try {
1231
- var vtResult = await agent(
1414
+ var versionState = null; // "touched" | "clean" | null (null = UNKNOWN)
1415
+ var versionStateErr = "";
1416
+ var versionFailedReturn = "";
1417
+ for (var versionAttempt = 0; versionAttempt <= 2 && versionState === null; versionAttempt++) {
1418
+ var versionPrompt =
1232
1419
  "Check whether the task branch changed package.json's `version` field. This is a mechanical git check, not a judgment call.\n" +
1233
- "Run in shell and return the stdout verbatim:\n" +
1234
- LIFECYCLE_ENV + LIFECYCLE + " version-check " + taskId,
1235
- { key: "version-check" + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""), label: "Checking package.json version (mechanical)" }
1236
- );
1237
- var vtStr = (typeof vtResult === "string") ? vtResult : JSON.stringify(vtResult);
1238
- var vtMatch = /^\s*VERSION_(TOUCHED|CLEAN)\s*$/im.exec(vtStr);
1239
- versionTouched = !(vtMatch && vtMatch[1].toUpperCase() === "CLEAN");
1240
- } catch (vtErr) {
1241
- versionTouched = true;
1242
- log("Version check failed (" + String(vtErr && vtErr.message || vtErr).slice(0, 120) + ") — failing closed as touched");
1420
+ "Run in shell: " + LIFECYCLE_ENV + LIFECYCLE + " version-check " + taskId + "\n" +
1421
+ "Return the command's exact stdout as the single named field and nothing else: { \"stdout\": \"<the command's exact stdout>\" }." +
1422
+ (versionAttempt === 0 ? "" : versionRetryTrailer(versionStateErr, versionFailedReturn));
1423
+ try {
1424
+ var vtResult = await agent(
1425
+ versionPrompt,
1426
+ { key: "version-check" + (versionAttempt === 0 ? "" : "-u" + versionAttempt) + (totalReworkCount > 0 ? "-r" + totalReworkCount : ""),
1427
+ label: "Checking package.json version (mechanical)" + (versionAttempt === 0 ? "" : " (instrument retry " + versionAttempt + " of 2)"),
1428
+ schema: { type: "object", properties: { stdout: { type: "string" } }, required: ["stdout"] } }
1429
+ );
1430
+ versionFailedReturn = (vtResult && typeof vtResult === "object" ? JSON.stringify(vtResult) : String(vtResult)).slice(0, 500);
1431
+ var vtParsed = parseVersionFerry(vtResult);
1432
+ if (vtParsed.versionState !== null) {
1433
+ versionState = vtParsed.versionState;
1434
+ versionStateErr = "";
1435
+ } else {
1436
+ versionStateErr = vtParsed.failReason;
1437
+ if (versionAttempt < 2) {
1438
+ log("version-check unparseable (" + versionStateErr + ") — retrying in place (attempt " + (versionAttempt + 1) + " of 2)");
1439
+ }
1440
+ }
1441
+ } catch (vtErr) {
1442
+ versionFailedReturn = String(vtErr && vtErr.message || vtErr).slice(0, 500);
1443
+ versionStateErr = "version-check call failed: " + String(vtErr && vtErr.message || vtErr).slice(0, 120);
1444
+ if (versionAttempt < 2) {
1445
+ log("version-check instrument failure (" + versionStateErr + ") — retrying in place (attempt " + (versionAttempt + 1) + " of 2)");
1446
+ }
1447
+ }
1243
1448
  }
1449
+ if (versionState === null) {
1450
+ // Exhaustion: park honestly. An instrument failure is not a
1451
+ // measurement of touched — never assert touched.
1452
+ return await parkTask("package.json `version` state unclassifiable after 2 instrument retries (" + versionStateErr + "; last failure: " + versionFailedReturn + ") — the version-check instrument failed, not the branch. Human attention needed: inspect the task branch directly to determine its state — no trustworthy version classification was produced.");
1453
+ }
1454
+ versionTouched = (versionState === "touched");
1244
1455
  log("Review: package.json `version` " + (versionTouched ? "touched by the branch — mechanical FAIL, Cass not dispatched" : "untouched — mechanical check clean"));
1245
1456
  }
1246
1457
  // The empty-branch rule Cass used to adjudicate is gone. What she
@@ -1563,13 +1774,26 @@ while (i < STEPS.length) {
1563
1774
  // closeout — the file list is the complete record of what the task
1564
1775
  // changed. The courier is pure hands: it writes the deterministic
1565
1776
  // script's own file list, never interprets it.
1777
+ // Blocker 40 (2026-09-21): this courier carries the same
1778
+ // {output: string} schema as the sibling diff-computation call.
1779
+ // The command is stdout-silent; schema-less, the runtime's
1780
+ // non-empty-result contract throws on the empty string and the
1781
+ // workflow parks fail-closed for a side effect that landed. Empty
1782
+ // stdout is schema-valid, so the contract no longer fires. "The
1783
+ // schema bypasses the contract" is INFERRED, not observed —
1784
+ // falsification: if a schema-carrying courier still parks on
1785
+ // silent stdout, the platform contract is deeper than the schema
1786
+ // and this is wrong; room evidence decides. (The file on disk is
1787
+ // not read as proof until QA closeout — the QA loader fails
1788
+ // closed to "unknown" attribution when it is absent.)
1566
1789
  var publishDiffFilesPath = crewHome + "/.publish-diffs/" + taskId + ".files.json";
1567
1790
  await agent(
1568
1791
  "Write the publish diff file list.\n" +
1569
- "Run exactly this and return the stdout verbatim:\n" +
1792
+ "Run exactly this and return the stdout verbatim as { \"output\": \"<verbatim stdout>\" } and nothing else. Do not interpret it:\n" +
1570
1793
  "mkdir -p " + crewHome + "/.publish-diffs && cat > \"" + publishDiffFilesPath + "\" <<'DIFFFILES_EOF'\n" +
1571
1794
  JSON.stringify(diffSummary.files || []) + "\nDIFFFILES_EOF\n",
1572
- { key: attemptKey("publish-diff-files-" + taskId, totalReworkCount), label: "Persisting publish diff file list" }
1795
+ { key: attemptKey("publish-diff-files-" + taskId, totalReworkCount), label: "Persisting publish diff file list",
1796
+ schema: { type: "object", properties: { output: { type: "string" } }, required: ["output"] } }
1573
1797
  );
1574
1798
  // Budget counts CHANGED lines (added + removed), not raw unified-diff
1575
1799
  // output lines: context lines and file headers inflated the old
@@ -1836,12 +2060,12 @@ while (i < STEPS.length) {
1836
2060
  "The returned events are filtered to this task. They contain notes and decisions from prior phases.\n\n";
1837
2061
  }
1838
2062
 
1839
- // Work agent returns the runtime's native envelope {"status": "ok",
1840
- // "result": "<prose>"} with no schema. The runtime requires JSON output;
1841
- // the envelope is its own documented shape, so there is nothing for the
1842
- // agent to improvise. The workflow receives the prose report as a plain
1843
- // string. The verdict is still extracted deterministically from the report
1844
- // text by extractVerdict below — never by an agent.
2063
+ // Work agent returns prose on the schema-less courier; the workflow
2064
+ // consumes the return as a plain string (blocker 43, 2026-09-22: the
2065
+ // prompt names no JSON envelope — nothing parses one, and the runtime's
2066
+ // JSON-candidate scan discards brace-shaped prose, which is what the
2067
+ // transport retry guards). The verdict is still extracted deterministically
2068
+ // from the report text by extractVerdict below — never by an agent.
1845
2069
  var workPromptBase =
1846
2070
  TOOL_CHECK_PREAMBLE +
1847
2071
  "Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
@@ -1854,8 +2078,7 @@ while (i < STEPS.length) {
1854
2078
  "\n## Instructions\n\n" + eventPreamble + instructions + "\n\n" +
1855
2079
  "CONSTRAINT: Do NOT call logevent or upsertagentsession — the workflow handles all phase tracking after your step completes.\n\n" +
1856
2080
  "Stay in character. Do the work thoroughly.\n\n" +
1857
- "Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your report here\"}. " +
1858
- "The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
2081
+ "Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
1859
2082
  var workKeyBase = "work-" + step.name + (totalReworkCount > 0 ? "-r" + totalReworkCount : "");
1860
2083
  var workerResult = null;
1861
2084
  var workAttempts = [];