muse-crew 0.7.10 → 0.7.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +36 -14
- package/docs/guide.md +5 -5
- package/docs/ooda-report.md +150 -0
- package/docs/publish-verification.md +277 -88
- package/docs/visual-verdict.md +81 -67
- package/lib/AGENTS.md +8 -0
- package/lib/append-ooda-step.js +167 -0
- package/lib/build-readback-request.js +130 -0
- package/lib/compose-evidence-caption.js +141 -0
- package/lib/crew-api.js +467 -43
- package/lib/edit-image.py +216 -0
- package/lib/read-ooda-verdict.js +94 -0
- package/lib/readback-disk.js +186 -0
- package/lib/render-html.js +142 -0
- package/lib/see-act.js +327 -0
- package/lib/serve-artifact.js +203 -0
- package/lib/verify-publish.js +323 -0
- package/lib/write-ooda-verdict.js +147 -0
- package/package.json +1 -1
- package/seed/cron-body-template.md +30 -4
- package/seed/crons.json +1 -1
- package/workflows/bugfix.js +512 -220
- package/workflows/chore.js +389 -119
- package/workflows/crew-dispatch.js +72 -13
- package/workflows/docs.js +11 -2
- package/workflows/standard.js +413 -235
|
@@ -103,18 +103,44 @@ phase("dispatch");
|
|
|
103
103
|
// the schema demanded a top-level object with ready_tasks. The LLM resolved
|
|
104
104
|
// the contradiction non-deterministically: some ticks validated, some
|
|
105
105
|
// burned retries and blocked the whole dispatcher.)
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
106
|
+
// Bounded in-tick retry (canary 2026-09-15): the read-board agent
|
|
107
|
+
// occasionally ferries malformed JSON (observed twice: a literal
|
|
108
|
+
// `.replace()` code fragment appended to the dispatch-state string).
|
|
109
|
+
// A malformed ferry used to abort the whole dispatcher tick; retrying
|
|
110
|
+
// the identical dispatcher call recovered both times. Retry the read
|
|
111
|
+
// in-tick instead of burning a tick — two attempts with distinct
|
|
112
|
+
// replay keys, fail closed after that.
|
|
113
|
+
var boardReturn = null;
|
|
114
|
+
var boardReadError = null;
|
|
115
|
+
for (var readAttempt = 1; readAttempt <= 2; readAttempt++) {
|
|
116
|
+
var readBoardKey = readAttempt === 1 ? "read-board" : "read-board-r" + readAttempt;
|
|
117
|
+
try {
|
|
118
|
+
boardReturn = await agent(
|
|
119
|
+
"Read the dispatch state.\n" +
|
|
120
|
+
"Run in shell and return the stdout as a raw string:\n" + crewCmd("get-dispatch-state", {}) + "\n" +
|
|
121
|
+
"Return the command's stdout JSON as a plain string, byte-for-byte, unmodified. " +
|
|
122
|
+
"Do NOT parse the JSON — your return value must be the raw stdout string, never an object. " +
|
|
123
|
+
"Do not select fields, do not rewrite, summarize, paraphrase, or reformat anything. " +
|
|
124
|
+
"The result has keys ready_tasks, projects, config, counts — ferry the string exactly as received.",
|
|
125
|
+
{
|
|
126
|
+
key: readBoardKey,
|
|
127
|
+
label: "Reading board state" + (readAttempt > 1 ? " (attempt " + readAttempt + ")" : "")
|
|
128
|
+
}
|
|
129
|
+
);
|
|
130
|
+
// Parse eagerly inside the attempt so a malformed ferry retries
|
|
131
|
+
// instead of aborting the tick. (function declarations hoist.)
|
|
132
|
+
parseBoardJson(boardReturn);
|
|
133
|
+
boardReadError = null;
|
|
134
|
+
break;
|
|
135
|
+
} catch (readErr) {
|
|
136
|
+
boardReadError = readErr;
|
|
137
|
+
log("WARNING: read-board attempt " + readAttempt + " failed (" + (readErr && readErr.message ? readErr.message : readErr) + ")" + (readAttempt < 2 ? " — retrying" : " — attempts exhausted"));
|
|
138
|
+
boardReturn = null;
|
|
116
139
|
}
|
|
117
|
-
|
|
140
|
+
}
|
|
141
|
+
if (boardReturn === null) {
|
|
142
|
+
throw new Error("read-board: all 2 attempts failed to ferry parseable board JSON: " + (boardReadError && boardReadError.message ? boardReadError.message : boardReadError));
|
|
143
|
+
}
|
|
118
144
|
|
|
119
145
|
// Deterministic board parse — the read-board agent is told to return the
|
|
120
146
|
// CLI stdout as a raw string; the workflow parses it here. Fails closed:
|
|
@@ -210,6 +236,10 @@ function projectTaskRecord(t) {
|
|
|
210
236
|
blocked: t.blocked,
|
|
211
237
|
deps: Array.isArray(t.deps) ? t.deps.slice() : [],
|
|
212
238
|
priority: t.priority,
|
|
239
|
+
// next_phase is the one-shot recovery routing set by recover-task. It is
|
|
240
|
+
// ferried verbatim here and consumed (validated, routed, cleared) by the
|
|
241
|
+
// deterministic eligibility logic below — never interpreted by the LLM.
|
|
242
|
+
next_phase: (typeof t.next_phase === "string" && t.next_phase.length > 0) ? t.next_phase : null,
|
|
213
243
|
latest_session: ls ? { id: ls.id, status: ls.status, step: ls.step, notes: ls.notes } : null,
|
|
214
244
|
retry: {
|
|
215
245
|
consecutive_failures: (typeof cf === "number" && cf >= 0) ? cf : 0,
|
|
@@ -424,7 +454,7 @@ function buildPlaytestDescription(journey, personaName, personaFile, maxFilings)
|
|
|
424
454
|
"\n" +
|
|
425
455
|
"Rules:\n" +
|
|
426
456
|
"- Walk the journey end to end from a user's perspective, wearing the persona.\n" +
|
|
427
|
-
"- Use the normal QA surface for this project (
|
|
457
|
+
"- Use the normal QA surface for this project (the published artifact's own user interface for artifact projects; read-only dashboard API checks otherwise). Note: there is currently no agent-callable visual-inspection tool (artifact_inspect was removed by the platform 2026-09-14; artifact.inspect is malfunction diagnosis, not a substitute) — judge only what you can observe directly.\n" +
|
|
428
458
|
"- There is no code change under test: skip change-verification steps (provenance / publish checks) and judge only what you observe.\n" +
|
|
429
459
|
"- " + filingRule + "\n" +
|
|
430
460
|
"- File via the existing createtask action with filed_by \"hazel\": set workflow to \"standard\" when the fix is clear feature work, or omit workflow (untriaged) when unsure — triage routes untriaged filings to bugfix/feature as usual.\n"
|
|
@@ -497,6 +527,30 @@ for (var t = 0; t < allTasks.length; t++) {
|
|
|
497
527
|
var workflow = task.workflow || "standard";
|
|
498
528
|
var steps = WORKFLOWS[workflow] || WORKFLOWS.standard;
|
|
499
529
|
|
|
530
|
+
// One-shot recovery routing: recover-task sets next_phase to an explicit
|
|
531
|
+
// human-chosen phase. It takes precedence over every session-derived path
|
|
532
|
+
// below — the human's redirect wins over the automatic resume — but never
|
|
533
|
+
// over work in flight, and never on a task in a terminal state.
|
|
534
|
+
// Invalid values fail closed: logged, skipped, and NOT cleared, so a
|
|
535
|
+
// human can fix it with a corrected recover-task. Valid values bypass
|
|
536
|
+
// the retry cap by design: an explicit redirect is not an automatic
|
|
537
|
+
// retry and must not consume retry budget. The launched workflow consumes
|
|
538
|
+
// (clears) next_phase on its successful self-claim, exactly once.
|
|
539
|
+
if (task.next_phase) {
|
|
540
|
+
if (task.state !== "todo" && task.state !== "in_progress") {
|
|
541
|
+
log("Skipped \"" + task.title + "\" — next_phase \"" + task.next_phase + "\" set but state is " + task.state + "; left set for inspection");
|
|
542
|
+
continue;
|
|
543
|
+
}
|
|
544
|
+
if (latest && latest.status === "running") continue; // work in flight
|
|
545
|
+
var npIdx = steps.indexOf(task.next_phase);
|
|
546
|
+
if (npIdx < 0) {
|
|
547
|
+
log("Skipped \"" + task.title + "\" — next_phase \"" + task.next_phase + "\" not in " + workflow + " step registry; left set for a corrected recover-task");
|
|
548
|
+
continue;
|
|
549
|
+
}
|
|
550
|
+
eligible.push({ task: task, startStep: npIdx, reason: "next_phase", workflow: workflow, nextPhase: task.next_phase });
|
|
551
|
+
continue;
|
|
552
|
+
}
|
|
553
|
+
|
|
500
554
|
if (task.state === "todo") {
|
|
501
555
|
// Idle playtests always enter at QA — the filing carries the full
|
|
502
556
|
// assignment, so Triage/Map have nothing to add.
|
|
@@ -844,7 +898,12 @@ for (var p = 0; p < toProcess.length; p++) {
|
|
|
844
898
|
// whether the task record had no workflow (playtest filings carry explicit
|
|
845
899
|
// workflow:'standard', so they correctly get false).
|
|
846
900
|
resolved_workflow: iworkflow,
|
|
847
|
-
workflow_was_null: (itask.workflow == null)
|
|
901
|
+
workflow_was_null: (itask.workflow == null),
|
|
902
|
+
// One-shot recovery routing: the value the dispatcher routed on. The
|
|
903
|
+
// workflow's successful self-claim consumes (clears) it atomically via
|
|
904
|
+
// claim-task's expected_next_phase — only an exact match clears, so a
|
|
905
|
+
// newer recover-task written in the race window survives.
|
|
906
|
+
next_phase: item.nextPhase || null
|
|
848
907
|
};
|
|
849
908
|
|
|
850
909
|
log("Recommended " + iworkflow + " for \"" + itask.title + "\" [" + taskProject + "] at step " + nextStepName);
|
package/workflows/docs.js
CHANGED
|
@@ -22,6 +22,15 @@ const startStepIndex = inputs.start_step_index || 0;
|
|
|
22
22
|
// resolution back via updatetask in the self-claim below.
|
|
23
23
|
const RESOLVED_WORKFLOW = inputs.resolved_workflow || null;
|
|
24
24
|
const WORKFLOW_WAS_NULL = inputs.workflow_was_null === true;
|
|
25
|
+
// One-shot recovery routing: the dispatcher sets inputs.next_phase when it
|
|
26
|
+
// routes this run via an explicit recover-task redirect. The value is
|
|
27
|
+
// consumed (cleared) atomically by the successful self-claim below:
|
|
28
|
+
// claim-task takes expected_next_phase and clears the matching next_phase in
|
|
29
|
+
// the same transaction as the winning session insert, so no platform death
|
|
30
|
+
// can slip between claim and consumption and replay the routing. A stale or
|
|
31
|
+
// superseded routing survives — only an exact match clears.
|
|
32
|
+
// what the dispatcher routed on.
|
|
33
|
+
const NEXT_PHASE_ROUTED = (typeof inputs.next_phase === "string" && inputs.next_phase.length > 0) ? inputs.next_phase : null;
|
|
25
34
|
|
|
26
35
|
// Config from args — backward-compatible fallbacks for manual launches
|
|
27
36
|
const crewHome = inputs.crewHome || "~/workspace/.jarvis";
|
|
@@ -148,7 +157,7 @@ while (i < STEPS.length) {
|
|
|
148
157
|
const claimResult = await agent(
|
|
149
158
|
"Claim this task for the " + step.name + " step.\n" +
|
|
150
159
|
"Run in shell and return the stdout verbatim:\n" + crewCmd("update-task", firstClaimUpdateArgs) + "\n" +
|
|
151
|
-
"Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started" }) + "\n" +
|
|
160
|
+
"Then run in shell and return the stdout verbatim:\n" + crewCmd("claim-task", { task_id: taskId, identity: step.identity, step: step.name, notes: step.name + " step started", ...(NEXT_PHASE_ROUTED ? { expected_next_phase: NEXT_PHASE_ROUTED } : {}) }) + "\n" +
|
|
152
161
|
"Do not interpret the claim response. It already contains an explicit \"claimed\" field — copy it verbatim.\n" +
|
|
153
162
|
"Return { claimed: <verbatim>, session_id: \"<...>\" }. If claimed is false there is no session_id; return { claimed: false, session_id: \"\" }.",
|
|
154
163
|
{
|
|
@@ -393,7 +402,7 @@ function describeWorkAgentFailure(stepName, identity, attempts) {
|
|
|
393
402
|
|
|
394
403
|
var instructions = "";
|
|
395
404
|
if (step.name === "Triage") {
|
|
396
|
-
instructions = "Validate the task,
|
|
405
|
+
instructions = "Validate the task against the project's repo at " + REPO_PATH + " — that exact checkout, not any other copy of the project on disk. If you run git commands, cd " + REPO_PATH + " first.\nCheck clarity, confirm the docs workflow assignment.\nReport back in plain prose — your assessment.";
|
|
397
406
|
} else if (step.name === "Write") {
|
|
398
407
|
instructions = "Write or revise the documentation the task asks for.\nFollow Tate's voice — clear, conversational, no jargon unless it earns its place.\nDo your work in the project repository at " + REPO_PATH + " — all doc files go there, not under the crew home." +
|
|
399
408
|
(rejectionNotes ? "\n\nREWORK after review rejection. Address:\n" + rejectionNotes : "") +
|