muse-crew 0.14.7 → 0.14.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/decisions/workflow-core.md +14 -7
- package/package.json +1 -1
- package/workflows/bugfix.js +287 -64
- package/workflows/chore.js +272 -62
- package/workflows/docs.js +11 -13
- package/workflows/standard.js +287 -64
package/workflows/chore.js
CHANGED
|
@@ -197,7 +197,7 @@ if (!taskId) {
|
|
|
197
197
|
throw new Error("task_id is required in args");
|
|
198
198
|
}
|
|
199
199
|
|
|
200
|
-
// See docs/decisions/workflow-core.md#closeout-envelope: the work
|
|
200
|
+
// See docs/decisions/workflow-core.md#closeout-envelope: the schema-less work courier returns prose, consumed as a plain string; verdict extracted mechanically.
|
|
201
201
|
const VERDICT_STEPS = ["Build", "Review", "Integrate", "Publish"];
|
|
202
202
|
function extractVerdict(workerText) {
|
|
203
203
|
// The verdict is the LAST VERDICT: PASS/FAIL in the report (contract: end
|
|
@@ -231,7 +231,6 @@ function buildVerdictReaskPrompt(stepName, workerText) {
|
|
|
231
231
|
"Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
|
|
232
232
|
"if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
|
|
233
233
|
"Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
|
|
234
|
-
"Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
|
|
235
234
|
"The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
|
|
236
235
|
}
|
|
237
236
|
async function reaskVerdict(stepName, reworkSuffix, workerText) {
|
|
@@ -304,8 +303,8 @@ var TOOL_CHECK_PREAMBLE =
|
|
|
304
303
|
"3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
|
|
305
304
|
"Then do the assignment below.\n\n";
|
|
306
305
|
function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
|
|
307
|
-
// reason: "discarded" (the runtime threw the output away
|
|
308
|
-
//
|
|
306
|
+
// reason: "discarded" (the runtime threw the output away — the JSON-candidate
|
|
307
|
+
// scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
|
|
309
308
|
// nothing usable), "no-tools" (the worker's TOOL CHECK reported
|
|
310
309
|
// artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
|
|
311
310
|
// reported shell_transport: unavailable).
|
|
@@ -316,12 +315,148 @@ function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason)
|
|
|
316
315
|
? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
|
|
317
316
|
: reason === "no-transport"
|
|
318
317
|
? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
|
|
319
|
-
: "your previous attempt's output could not
|
|
318
|
+
: "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
|
|
320
319
|
return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
|
|
321
320
|
"First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
|
|
322
321
|
"if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
|
|
323
|
-
"Then return your report
|
|
322
|
+
"Then return your report in exactly the shape specified above.";
|
|
324
323
|
}
|
|
324
|
+
// parseClassifyFerry — room #26 blocker 42 (2026-09-21). Shape-constrained
|
|
325
|
+
// ferry parser for the classify-branch agent() call. The schema buys SHAPE,
|
|
326
|
+
// not provenance: the fields are LLM-authored, so the parser stays
|
|
327
|
+
// defensive. Byte-identical across standard.js, bugfix.js, chore.js
|
|
328
|
+
// (pinned by tests/already-merged.test.js). Triplication is structural:
|
|
329
|
+
// each workflow ships as a standalone launch payload (240 KiB budget,
|
|
330
|
+
// individually git-archived) — there is no shared module to hold it.
|
|
331
|
+
// The schema is the guard; the prompt is instruction, never load-bearing.
|
|
332
|
+
// Returns { state, diagField, diagText, failReason }:
|
|
333
|
+
// state: "has-work" | "already-merged:<40-hex>" | "empty-no-work" | null
|
|
334
|
+
// (null = unparseable/ambiguous/channel violation — UNKNOWN)
|
|
335
|
+
// diagField: "stderr" | "stdout" | null (which field sourced the DIAG pin)
|
|
336
|
+
// diagText: the matched DIAG line, or ""
|
|
337
|
+
// failReason: computed reason, for the retry trailer and the park note
|
|
338
|
+
function parseClassifyFerry(bsResult) {
|
|
339
|
+
var bsStdout = (bsResult && typeof bsResult.stdout === "string") ? bsResult.stdout : null;
|
|
340
|
+
var bsStderr = (bsResult && typeof bsResult.stderr === "string") ? bsResult.stderr : null;
|
|
341
|
+
if (bsStdout === null || bsStderr === null) {
|
|
342
|
+
return { state: null, diagField: null, diagText: "",
|
|
343
|
+
failReason: "classifier return fields missing or non-string (stdout and stderr must both be strings)" };
|
|
344
|
+
}
|
|
345
|
+
// State channel: stdout ONLY. A BRANCH_STATE match on stderr is never
|
|
346
|
+
// trusted — workers merge streams constantly, and trusting a channel
|
|
347
|
+
// swap as truth is the wrong shape. Search, never anchor (blocker 11):
|
|
348
|
+
// the ferry labels output ("stdout: BRANCH_STATE: has-work"). Collect
|
|
349
|
+
// ALL matches and dedupe normalized values: exactly one distinct value
|
|
350
|
+
// is truth; zero or two-or-more is UNKNOWN — alternation order never
|
|
351
|
+
// adjudicates.
|
|
352
|
+
var bsRe = /\bBRANCH_STATE:\s*(has-work|already-merged:[0-9a-f]{40}|empty-no-work)(?![\w-])/gi;
|
|
353
|
+
var bsSeen = {};
|
|
354
|
+
var bsM;
|
|
355
|
+
while ((bsM = bsRe.exec(bsStdout)) !== null) {
|
|
356
|
+
bsSeen[bsM[1].toLowerCase()] = true;
|
|
357
|
+
}
|
|
358
|
+
var bsStates = Object.keys(bsSeen);
|
|
359
|
+
if (bsStates.length !== 1) {
|
|
360
|
+
return { state: null, diagField: null, diagText: "",
|
|
361
|
+
failReason: bsStates.length === 0
|
|
362
|
+
? "no BRANCH_STATE match in the stdout field"
|
|
363
|
+
: "ambiguous BRANCH_STATE values in stdout (" + bsStates.length + " distinct)" };
|
|
364
|
+
}
|
|
365
|
+
var bsState = bsStates[0];
|
|
366
|
+
// DIAG pin: search both fields, stderr first. Keep the DIAG: prefix — a
|
|
367
|
+
// prose mention without it must not satisfy the pin.
|
|
368
|
+
var bsDiagRe = /DIAG:\s*classify-branch\s+records=\d+\s+resolvable=\d+/;
|
|
369
|
+
var bsDiagField = bsDiagRe.test(bsStderr) ? "stderr" : (bsDiagRe.test(bsStdout) ? "stdout" : null);
|
|
370
|
+
var bsDiagText = "";
|
|
371
|
+
if (bsDiagField !== null) {
|
|
372
|
+
bsDiagText = bsDiagRe.exec(bsDiagField === "stderr" ? bsStderr : bsStdout)[0];
|
|
373
|
+
}
|
|
374
|
+
if (bsState.indexOf("already-merged:") === 0 && bsDiagField === null) {
|
|
375
|
+
// already-merged's sha flows into Cass's `git diff <sha>^1 <sha>` — a
|
|
376
|
+
// hallucinated-but-well-formed sha is a false-PASS vector. The script
|
|
377
|
+
// emits exactly one DIAG on every path that emits a BRANCH_STATE, so a
|
|
378
|
+
// missing pin means stderr wasn't faithfully ferried: UNKNOWN.
|
|
379
|
+
// Residual risk, named (blocker-42 fib 4): a worker could fabricate
|
|
380
|
+
// BOTH fields wholesale — the pin is a tripwire, not proof of
|
|
381
|
+
// execution. The workflow has no exec channel for a deterministic
|
|
382
|
+
// cross-check; if the platform ever offers one, verify the sha
|
|
383
|
+
// against the task's durable merge records directly.
|
|
384
|
+
return { state: null, diagField: null, diagText: "",
|
|
385
|
+
failReason: "already-merged without the DIAG pin — stderr not faithfully ferried" };
|
|
386
|
+
}
|
|
387
|
+
return { state: bsState, diagField: bsDiagField, diagText: bsDiagText, failReason: "" };
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
// classifyRetryTrailer — room #26 blocker 42 (2026-09-21). Corrective (not
|
|
391
|
+
// flat) trailer for the in-place classify retries. The failure mode was
|
|
392
|
+
// systematic labeling, not a coin flip: the trailer carries the COMPUTED
|
|
393
|
+
// parse-failure reason, the failed return truncated as a negative example,
|
|
394
|
+
// and the explicit field mapping — feedback from the actual failure,
|
|
395
|
+
// following the buildTransportRetryTrailer idiom. Bound: 2 retries
|
|
396
|
+
// (unmeasured house convention, matches the shared rework bound of 2 —
|
|
397
|
+
// against room retry data). Byte-identical across the three workflows
|
|
398
|
+
// (pinned by tests/already-merged.test.js).
|
|
399
|
+
function classifyRetryTrailer(reason, failedReturn) {
|
|
400
|
+
// Pass-2 simplification: the reason string already carries the truth
|
|
401
|
+
// (throw vs parse failure), so one neutral sentence replaces the
|
|
402
|
+
// throw/parse branch — the stringly-typed prefix contract between the
|
|
403
|
+
// call site and this function is deleted, not moved. Prompt accuracy,
|
|
404
|
+
// not prompt hardening.
|
|
405
|
+
return "\n\nCLASSIFY RETRY: the previous classifier attempt failed (" + reason + "). " +
|
|
406
|
+
"Return ONLY the two named fields: { \"stdout\": \"<the command's exact stdout>\", \"stderr\": \"<the command's exact stderr>\" }. " +
|
|
407
|
+
"Put each stream in its named field — do not add labels like \"stdout:\" in front of the content. " +
|
|
408
|
+
"Previous attempt detail (do not repeat this shape): " + String(failedReturn).slice(0, 200);
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
// parseVersionFerry — review pass 2, blocker-42 class (2026-09-21). The npm
|
|
412
|
+
// version-check ferry had the full pre-42 shape: schema-less agent(),
|
|
413
|
+
// line-anchored parser, and an unparseable fallback asserting a POSITIVE
|
|
414
|
+
// touched claim that spent the shared rework budget. Same treatment as
|
|
415
|
+
// parseClassifyFerry: shape-constrained {stdout} schema, search-and-dedupe
|
|
416
|
+
// parse, UNKNOWN on unparseable with in-place retries, honest park.
|
|
417
|
+
// Byte-identical across standard.js, bugfix.js, chore.js
|
|
418
|
+
// (pinned by tests/already-merged.test.js).
|
|
419
|
+
// Returns { versionState, failReason }:
|
|
420
|
+
// versionState: "touched" | "clean" | null (null = UNKNOWN)
|
|
421
|
+
// failReason: computed reason, for the retry trailer and the park note
|
|
422
|
+
function parseVersionFerry(vtResult) {
|
|
423
|
+
var vtStdout = (vtResult && typeof vtResult.stdout === "string") ? vtResult.stdout : null;
|
|
424
|
+
if (vtStdout === null) {
|
|
425
|
+
return { versionState: null,
|
|
426
|
+
failReason: "version-check return field missing or non-string (stdout must be a string)" };
|
|
427
|
+
}
|
|
428
|
+
// Search, never anchor (blocker 11): the ferry labels output
|
|
429
|
+
// ("stdout: VERSION_CLEAN"). Collect ALL matches and dedupe normalized
|
|
430
|
+
// values: exactly one distinct value is truth; zero or two-or-more is
|
|
431
|
+
// UNKNOWN — alternation order never adjudicates.
|
|
432
|
+
var vtRe = /\bVERSION_(TOUCHED|CLEAN)(?![\w-])/gi;
|
|
433
|
+
var vtSeen = {};
|
|
434
|
+
var vtM;
|
|
435
|
+
while ((vtM = vtRe.exec(vtStdout)) !== null) {
|
|
436
|
+
vtSeen[vtM[1].toUpperCase()] = true;
|
|
437
|
+
}
|
|
438
|
+
var vtStates = Object.keys(vtSeen);
|
|
439
|
+
if (vtStates.length !== 1) {
|
|
440
|
+
return { versionState: null,
|
|
441
|
+
failReason: vtStates.length === 0
|
|
442
|
+
? "no VERSION_(TOUCHED|CLEAN) match in the stdout field"
|
|
443
|
+
: "ambiguous VERSION values in stdout (" + vtStates.length + " distinct)" };
|
|
444
|
+
}
|
|
445
|
+
return { versionState: vtStates[0] === "CLEAN" ? "clean" : "touched", failReason: "" };
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
// versionRetryTrailer — review pass 2, blocker-42 class (2026-09-21).
|
|
449
|
+
// Neutral corrective trailer for the in-place version-check retries: the
|
|
450
|
+
// computed parse-failure reason plus the failed return as a negative
|
|
451
|
+
// example. Byte-identical across the three workflows
|
|
452
|
+
// (pinned by tests/already-merged.test.js).
|
|
453
|
+
function versionRetryTrailer(reason, failedReturn) {
|
|
454
|
+
return "\n\nVERSION-CHECK RETRY: the previous version-check attempt failed (" + reason + "). " +
|
|
455
|
+
"Return ONLY the named field: { \"stdout\": \"<the command's exact stdout>\" }. " +
|
|
456
|
+
"Put the command's stdout in the named field — do not add labels like \"stdout:\" in front of the content. " +
|
|
457
|
+
"Previous attempt detail (do not repeat this shape): " + String(failedReturn).slice(0, 200);
|
|
458
|
+
}
|
|
459
|
+
|
|
325
460
|
// parseToolSignals - bug 3472bf36. The work agent's TOOL CHECK emits two
|
|
326
461
|
// exact signal lines: artifact_tools: ok|missing and
|
|
327
462
|
// shell_transport: ok|unavailable. The workflow reads ONLY these lines.
|
|
@@ -694,7 +829,7 @@ let alreadyMergedSha = null;
|
|
|
694
829
|
let branchState = null; // "has-work" | "already-merged:<sha>" | "empty-no-work"
|
|
695
830
|
let buildClaimedNoDiff = false; // Build declared plain `repo_diff: none` (no repo change — runtime-state deliverable, or believed already-merged). Same-process Build report only; there is no cross-process hydration — an empty branch on a fresh dispatch with no task-attributed merge fails mechanically.
|
|
696
831
|
let versionTouched = false; // npm: branch changed package.json's `version` (mechanical git fact, blocker-34 follow-up)
|
|
697
|
-
let branchStateErr = ""; // classifier failure detail (
|
|
832
|
+
let branchStateErr = ""; // classifier failure detail (retry trailer + park note grounds)
|
|
698
833
|
// Deterministic publish target — computed by the workflow (registry base +
|
|
699
834
|
// bumpVersion), never by the Publish agent.
|
|
700
835
|
let publishTarget = null; // { base, scope, target }
|
|
@@ -1150,43 +1285,87 @@ while (i < STEPS.length) {
|
|
|
1150
1285
|
// (classify-branch in the lifecycle script): has-work |
|
|
1151
1286
|
// already-merged:<sha> | empty-no-work. Attribution is identity-first
|
|
1152
1287
|
// from git, then the task's own durable merge records — there is no
|
|
1153
|
-
// agent-authored declared-sha input
|
|
1154
|
-
//
|
|
1155
|
-
//
|
|
1156
|
-
//
|
|
1157
|
-
//
|
|
1158
|
-
|
|
1159
|
-
|
|
1288
|
+
// agent-authored declared-sha input.
|
|
1289
|
+
// Blocker 42 (2026-09-21): the classifier output crosses an agent()
|
|
1290
|
+
// ferry — the old comment's "no cross-process hydration ferry" was a
|
|
1291
|
+
// fib. Machine facts travel in structured fields ({stdout, stderr}
|
|
1292
|
+
// schema) now, never in prose to be "returned verbatim". The schema
|
|
1293
|
+
// is the guard; the prompt is instruction, never load-bearing.
|
|
1294
|
+
// Unparseable is UNKNOWN, never a positive empty-no-work claim (room
|
|
1295
|
+
// #27: a labeled ferry — "stdout: BRANCH_STATE: has-work" — defeated
|
|
1296
|
+
// the ^-anchored parsers, and the empty-no-work fallback burned the
|
|
1297
|
+
// whole shared rework budget on a branch with real work). UNKNOWN
|
|
1298
|
+
// retries the classify call IN PLACE (up to 2, fresh -u1/-u2 keys — a
|
|
1299
|
+
// full Build round cannot fix an instrument failure, and the only
|
|
1300
|
+
// in-process bounce path spends the shared budget unconditionally).
|
|
1301
|
+
// Instrumentation failures NEVER touch reworkCount: an instrument
|
|
1302
|
+
// failure is not a build-quality failure. Exhaustion parks honestly
|
|
1303
|
+
// as unclassifiable — "unknown keeps the task alive (bounded)"; the
|
|
1304
|
+
// park on exhaustion is the fail-closed part. The UNKNOWN path never
|
|
1305
|
+
// asserts empty-no-work: no log line or park note presents it as the
|
|
1306
|
+
// classification. The string can still appear inside quoted evidence
|
|
1307
|
+
// (the verbatim failed return travels in the trailer and the note).
|
|
1308
|
+
// INVARIANT: branchState leaves this block holding a known value, or
|
|
1309
|
+
// the step has parked — unknown never reaches any consumer below.
|
|
1310
|
+
// Re-classify on EVERY Review entry: rework rewinds this step
|
|
1311
|
+
// in-process, and the branch changed between rounds. The loop's
|
|
1312
|
+
// null-gate governs only the in-place instrument retries within
|
|
1313
|
+
// this pass — without the reset, a second Review pass would trust
|
|
1314
|
+
// the first pass's stale classification and a real fix commit
|
|
1315
|
+
// would never "become has-work".
|
|
1316
|
+
branchState = null;
|
|
1317
|
+
branchStateErr = "";
|
|
1318
|
+
alreadyMergedSha = null;
|
|
1319
|
+
var classifyFailedReturn = "";
|
|
1320
|
+
for (var classifyAttempt = 0; classifyAttempt <= 2 && branchState === null; classifyAttempt++) {
|
|
1321
|
+
var classifyPrompt =
|
|
1160
1322
|
"Classify the task branch state. This is a mechanical git check, not a judgment call.\n" +
|
|
1161
|
-
"Run in shell
|
|
1162
|
-
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1323
|
+
"Run in shell: " + LIFECYCLE_ENV + LIFECYCLE + " classify-branch " + taskId + "\n" +
|
|
1324
|
+
"Return the two streams as the two named fields and nothing else: { \"stdout\": \"<the command's exact stdout>\", \"stderr\": \"<the command's exact stderr>\" }." +
|
|
1325
|
+
(classifyAttempt === 0 ? "" : classifyRetryTrailer(branchStateErr, classifyFailedReturn));
|
|
1326
|
+
try {
|
|
1327
|
+
var bsResult = await agent(
|
|
1328
|
+
classifyPrompt,
|
|
1329
|
+
{ key: "classify-branch" + (classifyAttempt === 0 ? "" : "-u" + classifyAttempt) + (reworkCount > 0 ? "-r" + reworkCount : ""),
|
|
1330
|
+
label: "Classifying branch state (mechanical)" + (classifyAttempt === 0 ? "" : " (instrument retry " + classifyAttempt + " of 2)"),
|
|
1331
|
+
schema: { type: "object", properties: { stdout: { type: "string" }, stderr: { type: "string" } }, required: ["stdout", "stderr"] } }
|
|
1332
|
+
);
|
|
1333
|
+
classifyFailedReturn = (bsResult && typeof bsResult === "object" ? JSON.stringify(bsResult) : String(bsResult)).slice(0, 500);
|
|
1334
|
+
var classifyParsed = parseClassifyFerry(bsResult);
|
|
1335
|
+
if (classifyParsed.state !== null) {
|
|
1336
|
+
branchState = classifyParsed.state;
|
|
1337
|
+
branchStateErr = "";
|
|
1338
|
+
// DIAG pin: advisory for has-work/empty-no-work (blocker-34 v2.1
|
|
1339
|
+
// semantics — logged with its source field so silent record loss
|
|
1340
|
+
// surfaces in the evidence); already-merged is gated on the pin
|
|
1341
|
+
// inside parseClassifyFerry.
|
|
1342
|
+
if (classifyParsed.diagField !== null) {
|
|
1343
|
+
log("classify-branch DIAG pin (" + classifyParsed.diagField + " field): " + classifyParsed.diagText);
|
|
1344
|
+
} else {
|
|
1345
|
+
log("classify-branch: no DIAG: record-scan counter returned (scan unverifiable)");
|
|
1346
|
+
}
|
|
1347
|
+
} else {
|
|
1348
|
+
branchStateErr = classifyParsed.failReason;
|
|
1349
|
+
if (classifyAttempt < 2) {
|
|
1350
|
+
log("classify-branch unparseable (" + branchStateErr + ") — retrying in place (attempt " + (classifyAttempt + 1) + " of 2)");
|
|
1351
|
+
}
|
|
1352
|
+
}
|
|
1353
|
+
} catch (bsErr) {
|
|
1354
|
+
classifyFailedReturn = String(bsErr && bsErr.message || bsErr).slice(0, 500);
|
|
1355
|
+
branchStateErr = "classifier call failed: " + String(bsErr && bsErr.message || bsErr).slice(0, 120);
|
|
1356
|
+
if (classifyAttempt < 2) {
|
|
1357
|
+
log("classify-branch instrument failure (" + branchStateErr + ") — retrying in place (attempt " + (classifyAttempt + 1) + " of 2)");
|
|
1358
|
+
}
|
|
1180
1359
|
}
|
|
1181
|
-
} catch (bsErr) {
|
|
1182
|
-
branchStateErr = "classifier call failed: " + String(bsErr && bsErr.message || bsErr).slice(0, 120);
|
|
1183
1360
|
}
|
|
1184
|
-
if (
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1361
|
+
if (branchState === null) {
|
|
1362
|
+
// Exhaustion: park honestly. An instrument failure is not a
|
|
1363
|
+
// measurement of emptiness — never "empty-no-work".
|
|
1364
|
+
return await parkTask("Branch state unclassifiable after 2 instrument retries (" + branchStateErr + "; last failure: " + classifyFailedReturn + ") — the classifier instrument failed, not the branch. Human attention needed: inspect the task branch directly to determine its state — no trustworthy branch classification was produced.");
|
|
1365
|
+
}
|
|
1366
|
+
if (branchState.indexOf("already-merged:") === 0) {
|
|
1188
1367
|
alreadyMergedSha = branchState.slice("already-merged:".length);
|
|
1189
|
-
log("Branch state already-merged: " + alreadyMergedSha + " (
|
|
1368
|
+
log("Branch state already-merged: " + alreadyMergedSha + " (DIAG pin present — tripwire satisfied) — Cass reviews the frozen merge diff");
|
|
1190
1369
|
} else {
|
|
1191
1370
|
log("Branch state: " + branchState);
|
|
1192
1371
|
}
|
|
@@ -1198,23 +1377,55 @@ while (i < STEPS.length) {
|
|
|
1198
1377
|
// git fact, not a reviewer judgment. For npm projects the workflow checks
|
|
1199
1378
|
// it here, before Cass is dispatched — a touched `version` fails Review
|
|
1200
1379
|
// mechanically (versions are assigned at publish time, never in
|
|
1201
|
-
// branches).
|
|
1202
|
-
//
|
|
1380
|
+
// branches). Review pass 2 (blocker-42 class): this ferry had the full
|
|
1381
|
+
// pre-42 shape — schema-less agent(), line-anchored parser, and an
|
|
1382
|
+
// unparseable fallback asserting a POSITIVE touched claim that spent the
|
|
1383
|
+
// shared rework budget. Same treatment as the classifier:
|
|
1384
|
+
// shape-constrained {stdout} schema, search-and-dedupe parse, UNKNOWN
|
|
1385
|
+
// with in-place retries, honest park. An instrument failure is not a
|
|
1386
|
+
// measurement of touched — never assert touched from unparseable output.
|
|
1203
1387
|
if (PUBLISH_TYPE === "npm") {
|
|
1204
|
-
|
|
1205
|
-
|
|
1388
|
+
var versionState = null; // "touched" | "clean" | null (null = UNKNOWN)
|
|
1389
|
+
var versionStateErr = "";
|
|
1390
|
+
var versionFailedReturn = "";
|
|
1391
|
+
for (var versionAttempt = 0; versionAttempt <= 2 && versionState === null; versionAttempt++) {
|
|
1392
|
+
var versionPrompt =
|
|
1206
1393
|
"Check whether the task branch changed package.json's `version` field. This is a mechanical git check, not a judgment call.\n" +
|
|
1207
|
-
"Run in shell
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1214
|
-
|
|
1215
|
-
|
|
1216
|
-
|
|
1394
|
+
"Run in shell: " + LIFECYCLE_ENV + LIFECYCLE + " version-check " + taskId + "\n" +
|
|
1395
|
+
"Return the command's exact stdout as the single named field and nothing else: { \"stdout\": \"<the command's exact stdout>\" }." +
|
|
1396
|
+
(versionAttempt === 0 ? "" : versionRetryTrailer(versionStateErr, versionFailedReturn));
|
|
1397
|
+
try {
|
|
1398
|
+
var vtResult = await agent(
|
|
1399
|
+
versionPrompt,
|
|
1400
|
+
{ key: "version-check" + (versionAttempt === 0 ? "" : "-u" + versionAttempt) + (reworkCount > 0 ? "-r" + reworkCount : ""),
|
|
1401
|
+
label: "Checking package.json version (mechanical)" + (versionAttempt === 0 ? "" : " (instrument retry " + versionAttempt + " of 2)"),
|
|
1402
|
+
schema: { type: "object", properties: { stdout: { type: "string" } }, required: ["stdout"] } }
|
|
1403
|
+
);
|
|
1404
|
+
versionFailedReturn = (vtResult && typeof vtResult === "object" ? JSON.stringify(vtResult) : String(vtResult)).slice(0, 500);
|
|
1405
|
+
var vtParsed = parseVersionFerry(vtResult);
|
|
1406
|
+
if (vtParsed.versionState !== null) {
|
|
1407
|
+
versionState = vtParsed.versionState;
|
|
1408
|
+
versionStateErr = "";
|
|
1409
|
+
} else {
|
|
1410
|
+
versionStateErr = vtParsed.failReason;
|
|
1411
|
+
if (versionAttempt < 2) {
|
|
1412
|
+
log("version-check unparseable (" + versionStateErr + ") — retrying in place (attempt " + (versionAttempt + 1) + " of 2)");
|
|
1413
|
+
}
|
|
1414
|
+
}
|
|
1415
|
+
} catch (vtErr) {
|
|
1416
|
+
versionFailedReturn = String(vtErr && vtErr.message || vtErr).slice(0, 500);
|
|
1417
|
+
versionStateErr = "version-check call failed: " + String(vtErr && vtErr.message || vtErr).slice(0, 120);
|
|
1418
|
+
if (versionAttempt < 2) {
|
|
1419
|
+
log("version-check instrument failure (" + versionStateErr + ") — retrying in place (attempt " + (versionAttempt + 1) + " of 2)");
|
|
1420
|
+
}
|
|
1421
|
+
}
|
|
1422
|
+
}
|
|
1423
|
+
if (versionState === null) {
|
|
1424
|
+
// Exhaustion: park honestly. An instrument failure is not a
|
|
1425
|
+
// measurement of touched — never assert touched.
|
|
1426
|
+
return await parkTask("package.json `version` state unclassifiable after 2 instrument retries (" + versionStateErr + "; last failure: " + versionFailedReturn + ") — the version-check instrument failed, not the branch. Human attention needed: inspect the task branch directly to determine its state — no trustworthy version classification was produced.");
|
|
1217
1427
|
}
|
|
1428
|
+
versionTouched = (versionState === "touched");
|
|
1218
1429
|
log("Review: package.json `version` " + (versionTouched ? "touched by the branch — mechanical FAIL, Cass not dispatched" : "untouched — mechanical check clean"));
|
|
1219
1430
|
}
|
|
1220
1431
|
// The empty-branch rule Cass used to adjudicate is gone. What she
|
|
@@ -1688,12 +1899,12 @@ while (i < STEPS.length) {
|
|
|
1688
1899
|
"The returned events are filtered to this task. They contain notes and decisions from prior phases.\n\n";
|
|
1689
1900
|
}
|
|
1690
1901
|
|
|
1691
|
-
// Work agent returns the
|
|
1692
|
-
//
|
|
1693
|
-
//
|
|
1694
|
-
//
|
|
1695
|
-
//
|
|
1696
|
-
// text by extractVerdict below — never by an agent.
|
|
1902
|
+
// Work agent returns prose on the schema-less courier; the workflow
|
|
1903
|
+
// consumes the return as a plain string (blocker 43, 2026-09-22: the
|
|
1904
|
+
// prompt names no JSON envelope — nothing parses one, and the runtime's
|
|
1905
|
+
// JSON-candidate scan discards brace-shaped prose, which is what the
|
|
1906
|
+
// transport retry guards). The verdict is still extracted deterministically
|
|
1907
|
+
// from the report text by extractVerdict below — never by an agent.
|
|
1697
1908
|
var workPromptBase =
|
|
1698
1909
|
TOOL_CHECK_PREAMBLE +
|
|
1699
1910
|
"Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
|
|
@@ -1701,8 +1912,7 @@ while (i < STEPS.length) {
|
|
|
1701
1912
|
"Task: " + taskTitle + "\nTask ID: " + taskId + "\nDescription: " + taskDescription + "\nStep: " + step.name + "\n" +
|
|
1702
1913
|
(step.name !== "Review" ? "Crew API: " + CREW_API + "\n" : "") +
|
|
1703
1914
|
"\n## Instructions\n\n" + eventPreamble + instructions + "\n\nCONSTRAINT: Do NOT call logevent or upsertagentsession — the workflow handles all phase tracking after your step completes.\n\nStay in character. Do the work thoroughly.\n\n" +
|
|
1704
|
-
"
|
|
1705
|
-
"The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
1915
|
+
"Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
1706
1916
|
var workKeyBase = "work-" + step.name + (reworkCount > 0 ? "-r" + reworkCount : "");
|
|
1707
1917
|
var workerResult = null;
|
|
1708
1918
|
var workAttempts = [];
|
package/workflows/docs.js
CHANGED
|
@@ -269,7 +269,6 @@ while (i < STEPS.length) {
|
|
|
269
269
|
"Decide from the report's own content whether the " + stepName + " step clearly describes successful completion: " +
|
|
270
270
|
"if it does, the verdict is PASS; otherwise — failure, error, unfinished work, or unclear — the verdict is FAIL. " +
|
|
271
271
|
"Do NOT copy any VERDICT line from the report — decide from the content.\n\n" +
|
|
272
|
-
"Return your work as JSON in exactly this shape: {\"status\": \"ok\", \"result\": \"your verdict line here\"}. " +
|
|
273
272
|
"The result must be exactly one line and nothing else: VERDICT: PASS or VERDICT: FAIL.";
|
|
274
273
|
}
|
|
275
274
|
async function reaskVerdict(stepName, reworkSuffix, workerText) {
|
|
@@ -329,8 +328,8 @@ while (i < STEPS.length) {
|
|
|
329
328
|
"3. Run the shell command: echo tool-probe-ok - then write exactly one line: shell_transport: ok - or shell_transport: unavailable if you cannot run shell commands.\n" +
|
|
330
329
|
"Then do the assignment below.\n\n";
|
|
331
330
|
function buildTransportRetryTrailer(stepName, repoPath, taskId, attempt, reason) {
|
|
332
|
-
// reason: "discarded" (the runtime threw the output away
|
|
333
|
-
//
|
|
331
|
+
// reason: "discarded" (the runtime threw the output away — the JSON-candidate
|
|
332
|
+
// scan found {...}-shaped fragments it could not parse), "empty" (agent() returned without throwing but produced
|
|
334
333
|
// nothing usable), "no-tools" (the worker's TOOL CHECK reported
|
|
335
334
|
// artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
|
|
336
335
|
// reported shell_transport: unavailable).
|
|
@@ -341,11 +340,11 @@ while (i < STEPS.length) {
|
|
|
341
340
|
? "your previous attempt's TOOL CHECK reported artifact_tools: missing (this is a fresh launch, so run the TOOL CHECK's load step again before the work)"
|
|
342
341
|
: reason === "no-transport"
|
|
343
342
|
? "your previous attempt's TOOL CHECK reported shell_transport: unavailable (this is a fresh launch, so run the TOOL CHECK's shell probe again before the work)"
|
|
344
|
-
: "your previous attempt's output could not
|
|
343
|
+
: "your previous attempt's output was discarded by the transport because it contained {...}-shaped fragments the transport could not parse; write plain prose with no JSON-shaped fragments";
|
|
345
344
|
return "\n\nTRANSPORT RETRY (attempt " + attempt + " of 2): " + why + ". " +
|
|
346
345
|
"First check existing state (worktree/branch at " + repoPath + "/.worktrees/" + taskId + ", the task branch, dashboard sessions for this task) - " +
|
|
347
346
|
"if the " + stepName + " work is already complete, report on what was done rather than duplicating side effects. " +
|
|
348
|
-
"Then return your report
|
|
347
|
+
"Then return your report in exactly the shape specified above.";
|
|
349
348
|
}
|
|
350
349
|
// parseToolSignals - bug 3472bf36. The work agent's TOOL CHECK emits two
|
|
351
350
|
// exact signal lines: artifact_tools: ok|missing and
|
|
@@ -432,20 +431,19 @@ function describeWorkAgentFailure(stepName, identity, attempts) {
|
|
|
432
431
|
"\nWrite your review as plain prose — findings, then decision. End your report with exactly one line: VERDICT: PASS if it passes, VERDICT: FAIL if it fails.";
|
|
433
432
|
}
|
|
434
433
|
|
|
435
|
-
// Work agent returns the
|
|
436
|
-
//
|
|
437
|
-
//
|
|
438
|
-
//
|
|
439
|
-
//
|
|
440
|
-
// text by extractVerdict below — never by an agent.
|
|
434
|
+
// Work agent returns prose on the schema-less courier; the workflow
|
|
435
|
+
// consumes the return as a plain string (blocker 43, 2026-09-22: the
|
|
436
|
+
// prompt names no JSON envelope — nothing parses one, and the runtime's
|
|
437
|
+
// JSON-candidate scan discards brace-shaped prose, which is what the
|
|
438
|
+
// transport retry guards). The verdict is still extracted deterministically
|
|
439
|
+
// from the report text by extractVerdict below — never by an agent.
|
|
441
440
|
var workPromptBase =
|
|
442
441
|
TOOL_CHECK_PREAMBLE +
|
|
443
442
|
"Read the identity file at " + ORCH_PATH + "/identities/" + step.identity + ".md using the read tool, and embody that character fully.\n\n" +
|
|
444
443
|
"## Your Assignment\n\n" +
|
|
445
444
|
"Task: " + taskTitle + "\nTask ID: " + taskId + "\nDescription: " + taskDescription + "\nStep: " + step.name + "\n\n" +
|
|
446
445
|
"## Instructions\n\n" + instructions + "\n\nCONSTRAINT: Do NOT call the crew API directly — the workflow handles all phase tracking after your step completes.\n\nStay in character. All file work under " + REPO_PATH + "/.\n\n" +
|
|
447
|
-
"
|
|
448
|
-
"The result is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
446
|
+
"Your report is plain prose describing what you did and found. For verdict steps, end the report with exactly one line: VERDICT: PASS or VERDICT: FAIL.";
|
|
449
447
|
var workKeyBase = "work-" + step.name + (reworkCount > 0 ? "-r" + reworkCount : "");
|
|
450
448
|
var workerResult = null;
|
|
451
449
|
var workAttempts = [];
|