muse-crew 0.17.1 → 0.17.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -21,11 +21,20 @@
|
|
|
21
21
|
// evidence, not a false alarm.
|
|
22
22
|
//
|
|
23
23
|
// Usage: node compare-dispatch-shadow.js --crew-home <path>
|
|
24
|
-
// Prints
|
|
25
|
-
//
|
|
26
|
-
//
|
|
27
|
-
//
|
|
28
|
-
//
|
|
24
|
+
// Prints per-shadow SHADOW_MATCH / SHADOW_DIVERGE / SHADOW_UNPAIRED /
|
|
25
|
+
// SHADOW_PENDING lines plus a SHADOW_SUMMARY; SHADOW_NO_EVIDENCE when no
|
|
26
|
+
// shadow lines exist yet, SHADOW_CURRENT when every shadow already has a
|
|
27
|
+
// verdict. Appends verdicts to $CREW_HOME/.dispatch-shadow-verdicts.jsonl.
|
|
28
|
+
// Exit 0 on a verdict (even a diverge — divergence is evidence, not a
|
|
29
|
+
// failure); exit 2 on usage or IO errors.
|
|
30
|
+
//
|
|
31
|
+
// Retryability: a shadow with no matching decision is left PENDING (no
|
|
32
|
+
// verdict written) unless the authoritative dispatcher has provably moved
|
|
33
|
+
// past its tick — i.e. a decision exists with tick_seq greater than the
|
|
34
|
+
// wanted one. Each tick's dispatcher appends exactly one tick-release line
|
|
35
|
+
// and then writes its decision synchronously, so a later decision proves
|
|
36
|
+
// the missing one will never be written. Pending shadows are retried on
|
|
37
|
+
// every later run; only provably-dead ticks become terminal UNPAIRED.
|
|
29
38
|
|
|
30
39
|
import { readFileSync, appendFileSync } from "node:fs";
|
|
31
40
|
import { join } from "node:path";
|
|
@@ -104,15 +113,31 @@ if (pending.length === 0) {
|
|
|
104
113
|
process.exit(0);
|
|
105
114
|
}
|
|
106
115
|
|
|
107
|
-
let matched = 0, diverged = 0, unpaired = 0;
|
|
116
|
+
let matched = 0, diverged = 0, unpaired = 0, pendingCount = 0;
|
|
117
|
+
// The authoritative dispatcher appends one tick-release line and then writes
|
|
118
|
+
// its decision synchronously in the same step, so decision tick_seqs advance
|
|
119
|
+
// monotonically and each tick gets exactly one decision write. A decision
|
|
120
|
+
// with tick_seq past a shadow's wanted tick therefore proves the missing
|
|
121
|
+
// decision will never be written: the owning tick already spent its single
|
|
122
|
+
// write. (The release-line append and the decision write are adjacent in one
|
|
123
|
+
// process; a >15min stall between them — the only out-of-order path — would
|
|
124
|
+
// cost one sample, not correctness.)
|
|
125
|
+
const maxDecisionTickSeq = decisions.reduce((m, d) => Math.max(m, d.tick_seq || 0), 0);
|
|
108
126
|
for (const shadow of pending) {
|
|
109
127
|
const wantTickSeq = (shadow.tick_seq_before || 0) + 1;
|
|
110
128
|
const decision = decisions.find((d) => d.tick_seq === wantTickSeq);
|
|
111
129
|
|
|
112
130
|
if (!decision) {
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
131
|
+
if (maxDecisionTickSeq > wantTickSeq) {
|
|
132
|
+
emit({ verdict: "unpaired", shadow_seq: shadow.seq, want_tick_seq: wantTickSeq });
|
|
133
|
+
console.log(`SHADOW_UNPAIRED shadow_seq=${shadow.seq} want_tick_seq=${wantTickSeq} (authoritative dispatcher advanced to tick_seq=${maxDecisionTickSeq} without writing this decision)`);
|
|
134
|
+
unpaired++;
|
|
135
|
+
} else {
|
|
136
|
+
// The authoritative decision simply hasn't been written yet — no
|
|
137
|
+
// verdict, the shadow stays pending and a later run retries.
|
|
138
|
+
console.log(`SHADOW_PENDING shadow_seq=${shadow.seq} want_tick_seq=${wantTickSeq} (authoritative decision not yet written; retrying next run)`);
|
|
139
|
+
pendingCount++;
|
|
140
|
+
}
|
|
116
141
|
continue;
|
|
117
142
|
}
|
|
118
143
|
|
|
@@ -146,5 +171,5 @@ for (const shadow of pending) {
|
|
|
146
171
|
diverged++;
|
|
147
172
|
}
|
|
148
173
|
}
|
|
149
|
-
console.log(`SHADOW_SUMMARY matched=${matched} diverged=${diverged} unpaired=${unpaired}`);
|
|
174
|
+
console.log(`SHADOW_SUMMARY matched=${matched} diverged=${diverged} unpaired=${unpaired} pending=${pendingCount}`);
|
|
150
175
|
process.exit(0);
|
package/package.json
CHANGED
|
@@ -113,7 +113,7 @@ The failures below are settled and recorded in the Gate 1 OODA state. The author
|
|
|
113
113
|
|
|
114
114
|
4.6. **Shadow verdict (Piece 1, 2026-09-26):** after the Step-3 dispatcher's decision is recorded, pair its decision with the Step-2.5 shadow evidence. Run in shell:
|
|
115
115
|
`node {crewHome}/lib/compare-dispatch-shadow.js --crew-home {crewHome}`
|
|
116
|
-
Log every `SHADOW_*` line verbatim. `SHADOW_MATCH` means the worker layer agreed with the sandbox on this tick's claims. `SHADOW_DIVERGE` names the differing claim triples (`task_id|workflow|step`) — log them; divergence is evidence for the cutover review, not a tick failure, so continue the tick. `SHADOW_UNPAIRED` means the authoritative decision line is missing (the Step-3 dispatcher died after the shadow ran) — log it loudly and continue. `SHADOW_NO_EVIDENCE` on the first shadowed tick is normal. Verdicts append to `{crewHome}/.dispatch-shadow-verdicts.jsonl`; the cutover decision after ~100-200 ticks is made from that file, never from a single tick.
|
|
116
|
+
Log every `SHADOW_*` line verbatim. `SHADOW_MATCH` means the worker layer agreed with the sandbox on this tick's claims. `SHADOW_DIVERGE` names the differing claim triples (`task_id|workflow|step`) — log them; divergence is evidence for the cutover review, not a tick failure, so continue the tick. `SHADOW_UNPAIRED` means the authoritative decision line is missing (the Step-3 dispatcher died after the shadow ran) — log it loudly and continue. `SHADOW_PENDING` means the decision hasn't been written yet; the shadow stays pending and a later tick retries the pairing — log and continue. `SHADOW_NO_EVIDENCE` on the first shadowed tick is normal. Verdicts append to `{crewHome}/.dispatch-shadow-verdicts.jsonl`; the cutover decision after ~100-200 ticks is made from that file, never from a single tick.
|
|
117
117
|
- Never stamp provenance from prose. Never infer a verdict from an inspector's summary text. The exact version string is the sole positive signal.
|
|
118
118
|
|
|
119
119
|
5. **Monitor launched workflows until terminal (stay-alive — 2026-09-13):** The platform ties async workflow `agent()` authorization to the launcher's lifetime: if THIS tick ends while a workflow is still running, the workflow's next `agent()` call fails with "subagent bootstrap is no longer authorized" / "subagent reservation owner is terminal". Prevention beats recovery here, so this tick is configured with a 90-minute execution timeout (`timeout_secs: 5400` in seed/crons.json) and you MUST stay alive until every launched run reaches a terminal state. Do not exit early while a launched run is still `running` — your death is what kills it.
|