muse-crew 0.14.12 → 0.14.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/docs/decisions/AGENTS.md
CHANGED
|
@@ -6,6 +6,12 @@ history lives here. One canonical section per `<a id>` anchor — a critic
|
|
|
6
6
|
re-verified that every `docs/decisions/*.md#anchor` reference in
|
|
7
7
|
`workflows/*.js` resolves (45 unique references).
|
|
8
8
|
|
|
9
|
+
## dispatch-decision-log.md — Decision history: dispatch decision log (hoverboat)
|
|
10
|
+
|
|
11
|
+
- `#checksum-vs-hoverboat` — The ferry-corruption problem and the two candidates
|
|
12
|
+
- `#why-hoverboat-won` — Availability, trust, subtraction, primary-source rationale
|
|
13
|
+
- `#what-was-built` — Declaration shape, skip counters, observer ingestion, retirements
|
|
14
|
+
|
|
9
15
|
## publish-path.md — Decision history: publish path
|
|
10
16
|
|
|
11
17
|
- `#fire-and-forget-trigger` — Fire-and-forget trigger + workflow-owned observation
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# Dispatch decision log (hoverboat) — 2026-09-24
|
|
2
|
+
|
|
3
|
+
<a id="checksum-vs-hoverboat"></a>
|
|
4
|
+
## The problem
|
|
5
|
+
|
|
6
|
+
The Gate 2 observer attributed dispatch decisions by ferrying platform
|
|
7
|
+
`scheduler.job_runs` rows through an LLM paste (tick worker: `muse.db`
|
|
8
|
+
result → heredoc → `record_platform.sh --dispatch` → `dispatch_sample`
|
|
9
|
+
rows → `attribute_dispatch.sh` → `dispatch_decision` rows). The paste
|
|
10
|
+
mutated plausible values: a 2026-09-24 consistency scan found 275
|
|
11
|
+
same-`run_id` contradictory `dispatch_sample` rows and 65 derived
|
|
12
|
+
`dispatch_decision` rows based on corrupt data (5.7% of samples
|
|
13
|
+
quarantined; 63.6% of ticks touched). The older `quarantine_suspects.sh`
|
|
14
|
+
handled near-neighbor UUID mutations, not plausible value mutation.
|
|
15
|
+
|
|
16
|
+
## The two candidates
|
|
17
|
+
|
|
18
|
+
**Checksum ferry** (containment): keep the ferry; the worker checksums the
|
|
19
|
+
raw tool-result JSON before the paste and `record_platform.sh` verifies
|
|
20
|
+
after. Corrupt pastes become loud absences instead of silent corruption.
|
|
21
|
+
|
|
22
|
+
**Hoverboat** (local declaration): the dispatcher writes its decision to
|
|
23
|
+
`$CREW_HOME/.dispatch-decisions.jsonl` in its own bytes (deterministic JS,
|
|
24
|
+
same courier shape as the proven §4b tick-release writer); the observer
|
|
25
|
+
reads the file directly and verifies launch declarations mechanically
|
|
26
|
+
against crew DB effects (`platform_run_tasks`, `dispatch_reservations`).
|
|
27
|
+
|
|
28
|
+
<a id="why-hoverboat-won"></a>
|
|
29
|
+
## Why hoverboat won
|
|
30
|
+
|
|
31
|
+
1. **Availability.** Corruption touched ~63.6% of ticks. Checksum converts
|
|
32
|
+
corruption into absence, but the entry demo needs ten consecutive ticks
|
|
33
|
+
each recording a decision — absence fails the demo as surely as
|
|
34
|
+
corruption. Hoverboat's channel has no LLM re-emission at all.
|
|
35
|
+
2. **Trust.** The checksum would be computed by the same worker agent that
|
|
36
|
+
corrupts the paste — asking the corrupting agent to honestly report its
|
|
37
|
+
own corruption. Hoverboat's declaration is composed by deterministic
|
|
38
|
+
workflow JS and checked against independent DB effects.
|
|
39
|
+
3. **Subtraction.** Checksum keeps the ferry, the rolling 25h re-record
|
|
40
|
+
window, and the quarantine machinery permanently busy. Hoverboat deletes
|
|
41
|
+
the sample→decision attribution entirely; the platform ferry remains
|
|
42
|
+
only as platform-health telemetry.
|
|
43
|
+
4. **The crew already knows.** Having the platform tell the observer what
|
|
44
|
+
the crew decided, through an LLM paste, when the crew can write it
|
|
45
|
+
directly, is the ferry. The declaration is the primary source; the DB
|
|
46
|
+
effects are the independent verification.
|
|
47
|
+
|
|
48
|
+
Rejected alternative considered: a deterministic Crew API action for the
|
|
49
|
+
write. It would not remove the agent from the path (workflows reach the
|
|
50
|
+
CLI only through `agent()` shell calls), so it adds API surface for no
|
|
51
|
+
integrity gain. The embedded courier is the proven seam.
|
|
52
|
+
|
|
53
|
+
<a id="what-was-built"></a>
|
|
54
|
+
## What was built
|
|
55
|
+
|
|
56
|
+
- `workflows/crew-dispatch.js` §4c: per-tick declaration
|
|
57
|
+
`{seq, tick_seq, release, decision, launched[], completed[], parked[],
|
|
58
|
+
board{seen,eligible}, skipped{reason:count sparse}, partial}`.
|
|
59
|
+
`tick_seq` couples each line to the `.tick-releases.jsonl` line for the
|
|
60
|
+
same poll — a tick-release line with no decision line means the tick died
|
|
61
|
+
after poll-ack, visibly.
|
|
62
|
+
- Per-reason ineligibility counters (`countSkipped`) at every skip site in
|
|
63
|
+
the eligibility loop + the simultaneity-limit skip, so stand-downs are
|
|
64
|
+
auditable without re-deriving eligibility.
|
|
65
|
+
- Observer: `ingest_decisions.sh` replaces `attribute_dispatch.sh`;
|
|
66
|
+
`snapshot.sh` dispatch section rewritten around decision rows.
|
|
67
|
+
- Retired: sample→decision attribution, `state/attributed.json` watermark.
|
|
68
|
+
Kept: `record_platform.sh --dispatch` as platform-health telemetry,
|
|
69
|
+
`quarantine_suspects.sh` for the health samples' ID mutations.
|
package/package.json
CHANGED
|
@@ -561,6 +561,12 @@ const eligible = [];
|
|
|
561
561
|
const retryCandidates = [];
|
|
562
562
|
const rejectionCandidates = [];
|
|
563
563
|
|
|
564
|
+
// Dispatch-decision log (2026-09-24): per-reason ineligibility counters for
|
|
565
|
+
// the .dispatch-decisions.jsonl declaration. Sparse — only non-zero keys are
|
|
566
|
+
// emitted. Additive only; the eligibility decisions below are unchanged.
|
|
567
|
+
var decisionSkipped = {};
|
|
568
|
+
function countSkipped(reason) { decisionSkipped[reason] = (decisionSkipped[reason] || 0) + 1; }
|
|
569
|
+
|
|
564
570
|
// ── Retry cap ────────────────────────────────────────────────────────
|
|
565
571
|
// Symphony owns phase-redispatch policy (coordination layer): the spec
|
|
566
572
|
// defines backoff but no attempt cap ("implementation-defined"), so the
|
|
@@ -591,12 +597,13 @@ var MAX_CONSECUTIVE_REJECTIONS = configInt(config, "maxConsecutiveRejections", 2
|
|
|
591
597
|
|
|
592
598
|
for (var t = 0; t < allTasks.length; t++) {
|
|
593
599
|
var task = allTasks[t];
|
|
594
|
-
if (task.blocked) continue;
|
|
600
|
+
if (task.blocked) { countSkipped("blocked"); continue; }
|
|
595
601
|
|
|
596
602
|
// Skip tasks with an active dispatch reservation: the worker launched a
|
|
597
603
|
// workflow for this task on a previous tick, but the workflow has not yet
|
|
598
604
|
// self-claimed (claims can take 15+ minutes for cron-launched runs).
|
|
599
605
|
if (reservedTaskIds.has(task.id)) {
|
|
606
|
+
countSkipped("reserved");
|
|
600
607
|
log("Skipped \"" + task.title + "\" — active dispatch reservation (workflow launched, claim pending)");
|
|
601
608
|
continue;
|
|
602
609
|
}
|
|
@@ -604,6 +611,7 @@ for (var t = 0; t < allTasks.length; t++) {
|
|
|
604
611
|
// Skip tasks from quiesced projects
|
|
605
612
|
var taskProject = task.project || DEFAULT_PROJECT;
|
|
606
613
|
if (quiescedProjects[taskProject]) {
|
|
614
|
+
countSkipped("quiesced");
|
|
607
615
|
log("Skipped \"" + task.title + "\" — project " + taskProject + " is quiesced");
|
|
608
616
|
continue;
|
|
609
617
|
}
|
|
@@ -614,6 +622,7 @@ for (var t = 0; t < allTasks.length; t++) {
|
|
|
614
622
|
// task never moves. Set repo_path via updateproject to re-enable.
|
|
615
623
|
var taskProjCfg = PROJECTS[taskProject];
|
|
616
624
|
if (!taskProjCfg || !taskProjCfg.repo_path) {
|
|
625
|
+
countSkipped("no_repo");
|
|
617
626
|
log("Skipped \"" + task.title + "\" — project " + taskProject + " has no repo_path configured");
|
|
618
627
|
continue;
|
|
619
628
|
}
|
|
@@ -633,12 +642,14 @@ for (var t = 0; t < allTasks.length; t++) {
|
|
|
633
642
|
// (clears) next_phase on its successful self-claim, exactly once.
|
|
634
643
|
if (task.next_phase) {
|
|
635
644
|
if (task.state !== "todo" && task.state !== "in_progress") {
|
|
645
|
+
countSkipped("next_phase_state");
|
|
636
646
|
log("Skipped \"" + task.title + "\" — next_phase \"" + task.next_phase + "\" set but state is " + task.state + "; left set for inspection");
|
|
637
647
|
continue;
|
|
638
648
|
}
|
|
639
|
-
if (latest && latest.status === "running") continue; // work in flight
|
|
649
|
+
if (latest && latest.status === "running") { countSkipped("in_flight"); continue; } // work in flight
|
|
640
650
|
var npIdx = steps.indexOf(task.next_phase);
|
|
641
651
|
if (npIdx < 0) {
|
|
652
|
+
countSkipped("next_phase_bad_step");
|
|
642
653
|
log("Skipped \"" + task.title + "\" — next_phase \"" + task.next_phase + "\" not in " + workflow + " step registry; left set for a corrected recover-task");
|
|
643
654
|
continue;
|
|
644
655
|
}
|
|
@@ -654,19 +665,20 @@ for (var t = 0; t < allTasks.length; t++) {
|
|
|
654
665
|
continue;
|
|
655
666
|
}
|
|
656
667
|
|
|
657
|
-
if (task.state !== "in_progress") continue;
|
|
668
|
+
if (task.state !== "in_progress") { countSkipped("terminal_state"); continue; }
|
|
658
669
|
|
|
659
670
|
if (!latest) {
|
|
660
671
|
eligible.push({ task: task, startStep: 0, reason: "no_session", workflow: workflow });
|
|
661
672
|
continue;
|
|
662
673
|
}
|
|
663
674
|
|
|
664
|
-
if (latest.status === "running") continue; // work in flight
|
|
675
|
+
if (latest.status === "running") { countSkipped("in_flight"); continue; } // work in flight
|
|
665
676
|
|
|
666
677
|
if (latest.status === "completed") {
|
|
667
678
|
var stepName = latest.step || "";
|
|
668
679
|
var stepIndex = steps.indexOf(stepName);
|
|
669
680
|
if (stepIndex < 0) {
|
|
681
|
+
countSkipped("unknown_step");
|
|
670
682
|
log("Skipped \"" + task.title + "\" — step \"" + stepName + "\" not in " + workflow + " step registry");
|
|
671
683
|
continue;
|
|
672
684
|
}
|
|
@@ -695,6 +707,7 @@ for (var t = 0; t < allTasks.length; t++) {
|
|
|
695
707
|
var failedStep = latest.step || "";
|
|
696
708
|
var retryIdx = steps.indexOf(failedStep);
|
|
697
709
|
if (retryIdx < 0) {
|
|
710
|
+
countSkipped("unknown_step");
|
|
698
711
|
log("Skipped \"" + task.title + "\" — step \"" + failedStep + "\" not in " + workflow + " step registry");
|
|
699
712
|
continue;
|
|
700
713
|
}
|
|
@@ -756,6 +769,7 @@ if (rejectionCandidates.length > 0) {
|
|
|
756
769
|
// the task stays retryable and is re-parked next tick, so this is
|
|
757
770
|
// fail-safe. The dashboard stamps retry_reset_at on parked→todo, which
|
|
758
771
|
// restarts both counters mechanically.
|
|
772
|
+
var parkedTaskIds = [];
|
|
759
773
|
if (parkJobs.length > 0) {
|
|
760
774
|
var parkSteps = [];
|
|
761
775
|
for (var pji = 0; pji < parkJobs.length; pji++) {
|
|
@@ -775,6 +789,7 @@ if (parkJobs.length > 0) {
|
|
|
775
789
|
{ key: "park-batch", label: "Parking " + parkJobs.length + " task(s) at retry cap" }
|
|
776
790
|
);
|
|
777
791
|
log("Parked " + parkJobs.length + " task(s)");
|
|
792
|
+
parkedTaskIds = parkJobs.map(function(pj) { return pj.task.id; });
|
|
778
793
|
} catch (parkErr) {
|
|
779
794
|
log("WARNING: park batch failed (" + (parkErr.message || String(parkErr)).slice(0, 200) + ") — tasks remain retryable");
|
|
780
795
|
}
|
|
@@ -959,6 +974,7 @@ for (var ei = 0; ei < eligible.length; ei++) {
|
|
|
959
974
|
toProcess.push(eitem);
|
|
960
975
|
inFlightByProject[ep] = current + 1;
|
|
961
976
|
} else {
|
|
977
|
+
countSkipped("at_limit");
|
|
962
978
|
log("Skipped \"" + eitem.task.title + "\" — project " + ep + " at simultaneity limit (" + limit + ")");
|
|
963
979
|
}
|
|
964
980
|
}
|
|
@@ -1164,6 +1180,63 @@ if (recommended.length > 0) {
|
|
|
1164
1180
|
recommended = acquired;
|
|
1165
1181
|
}
|
|
1166
1182
|
|
|
1183
|
+
// ── 4c. Dispatch-decision log ──────────────────────────────────
|
|
1184
|
+
// Hoverboat (2026-09-24): the dispatcher declares its own decision, in its
|
|
1185
|
+
// own bytes, on crew-home disk — one append-only line per completed tick in
|
|
1186
|
+
// $CREW_HOME/.dispatch-decisions.jsonl. The observer reads the file
|
|
1187
|
+
// directly (no LLM re-emission, no platform ferry), so the plausible-value
|
|
1188
|
+
// mutation that poisoned the old dispatch_sample attribution cannot occur.
|
|
1189
|
+
// Launch declarations are verified mechanically against crew DB effects
|
|
1190
|
+
// (platform_run_tasks / dispatch_reservations); a stand-down is the
|
|
1191
|
+
// dispatcher's explicit declaration with board context, auditable without
|
|
1192
|
+
// re-deriving eligibility. tick_seq couples each line to the .tick-releases
|
|
1193
|
+
// line for the same poll — a tick-release line with no decision line means
|
|
1194
|
+
// the tick died after poll-ack and is visible as such.
|
|
1195
|
+
// seq is the non-empty line count + 1 of this file (file order is the
|
|
1196
|
+
// proof — no wall-clock calls, see tests/determinism.test.js).
|
|
1197
|
+
// Fire-and-forget: the script swallows its own errors and exits 0, the
|
|
1198
|
+
// agent call carries no schema, and the whole call is wrapped in
|
|
1199
|
+
// try/catch — a failed write can never fail the tick. The script echoes
|
|
1200
|
+
// the appended line (same rooms #12–#14 empty-result contract as §4b).
|
|
1201
|
+
var dispatchDecisionScript = [
|
|
1202
|
+
"var crewHome=process.argv[1];",
|
|
1203
|
+
"var decisionJson=process.argv[2];",
|
|
1204
|
+
"try{",
|
|
1205
|
+
"var fs=require(\"fs\"),path=require(\"path\");",
|
|
1206
|
+
"var decision=JSON.parse(decisionJson);",
|
|
1207
|
+
"if(!decision||typeof decision.decision!==\"string\"||!Array.isArray(decision.launched)){throw new Error(\"bad decision payload\");}",
|
|
1208
|
+
"var release=path.basename(path.resolve(crewHome,fs.readlinkSync(path.join(crewHome,\"current\"))));",
|
|
1209
|
+
"var file=path.join(crewHome,\".dispatch-decisions.jsonl\");",
|
|
1210
|
+
"var count=0;",
|
|
1211
|
+
"try{var lines=fs.readFileSync(file,\"utf8\").split(\"\\n\");for(var i=0;i<lines.length;i++){if(lines[i].trim()!==\"\"){count++;}}}catch(e){}",
|
|
1212
|
+
"var tickSeq=0;",
|
|
1213
|
+
"try{var tlines=fs.readFileSync(path.join(crewHome,\".tick-releases.jsonl\"),\"utf8\").split(\"\\n\");for(var j=0;j<tlines.length;j++){if(tlines[j].trim()!==\"\"){tickSeq++;}}}catch(e){}",
|
|
1214
|
+
"var line=JSON.stringify({seq:count+1,tick_seq:tickSeq,release:release,decision:decision.decision,launched:decision.launched,completed:decision.completed||[],parked:decision.parked||[],board:decision.board||null,skipped:decision.skipped||{},partial:!!decision.partial});",
|
|
1215
|
+
"fs.appendFileSync(file,line+\"\\n\");",
|
|
1216
|
+
"process.stdout.write(line+\"\\n\");",
|
|
1217
|
+
"}catch(e){}",
|
|
1218
|
+
"process.exit(0);"
|
|
1219
|
+
].join("");
|
|
1220
|
+
var launchedRecs = recommended.map(function(r) { return { task_id: r.task_id, workflow: r.workflow, step: r.step }; });
|
|
1221
|
+
var decisionObj = {
|
|
1222
|
+
decision: launchedRecs.length > 0 ? "launch" : "stand_down",
|
|
1223
|
+
launched: launchedRecs,
|
|
1224
|
+
completed: completed.map(function(r) { return r.task_id; }),
|
|
1225
|
+
parked: parkedTaskIds,
|
|
1226
|
+
board: { seen: allTasks.length, eligible: eligible.length },
|
|
1227
|
+
skipped: decisionSkipped,
|
|
1228
|
+
partial: partial
|
|
1229
|
+
};
|
|
1230
|
+
var dispatchDecisionCmd = "node -e '" + dispatchDecisionScript + "' '" + crewHome.replace(/'/g, "'\\''") + "' '" + JSON.stringify(decisionObj).replace(/'/g, "'\\''") + "'";
|
|
1231
|
+
try {
|
|
1232
|
+
await agent(
|
|
1233
|
+
"Record this tick's dispatch decision.\nRun in shell and return the stdout verbatim:\n" + dispatchDecisionCmd,
|
|
1234
|
+
{ key: "dispatch-decision", label: "Recording dispatch decision" }
|
|
1235
|
+
);
|
|
1236
|
+
} catch (e) {
|
|
1237
|
+
log("WARNING: dispatch-decision log write failed (tick continues): " + e.message);
|
|
1238
|
+
}
|
|
1239
|
+
|
|
1167
1240
|
var msg = "Dispatch complete.";
|
|
1168
1241
|
if (recommended.length > 0) {
|
|
1169
1242
|
msg += " Recommended: " + recommended.map(function(r) { return r.workflow + "/" + r.step + " for " + r.task_id; }).join(", ") + ".";
|