muse-crew 0.16.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +14 -0
- package/lib/AGENTS.md +3 -0
- package/lib/compare-dispatch-shadow.js +150 -0
- package/lib/crew-api.js +31 -0
- package/lib/crew-dispatch-worker.js +1059 -0
- package/lib/schema.sql +19 -0
- package/lib/spawn-boundary.js +101 -0
- package/package.json +1 -1
- package/seed/cron-body-template.md +8 -0
- package/workflows/crew-dispatch.js +5 -1
package/lib/schema.sql
CHANGED
|
@@ -281,3 +281,22 @@ CREATE TABLE IF NOT EXISTS builder_reports (
|
|
|
281
281
|
);
|
|
282
282
|
CREATE INDEX IF NOT EXISTS builder_reports_task_id_idx ON builder_reports(task_id);
|
|
283
283
|
CREATE INDEX IF NOT EXISTS builder_reports_version_idx ON builder_reports(version);
|
|
284
|
+
|
|
285
|
+
-- Worker-layer run ledger (Piece 1, 2026-09-26): every worker-layer
|
|
286
|
+
-- orchestration run (dispatcher, and later workflow phases) records its
|
|
287
|
+
-- start/finish here. The sandboxed workflow runtime keeps writing
|
|
288
|
+
-- workflow_runs; this table is the worker-layer counterpart. Timestamps
|
|
289
|
+
-- are written by SQLite (strftime), never by worker JS — the determinism
|
|
290
|
+
-- guard forbids wall-clock in orchestration code (G3).
|
|
291
|
+
CREATE TABLE IF NOT EXISTS worker_runs (
|
|
292
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
293
|
+
task_id TEXT REFERENCES tasks(id) ON DELETE CASCADE,
|
|
294
|
+
phase TEXT NOT NULL,
|
|
295
|
+
executor TEXT NOT NULL DEFAULT 'worker' CHECK (executor = 'worker'),
|
|
296
|
+
status TEXT NOT NULL CHECK (status IN ('running', 'completed', 'failed')),
|
|
297
|
+
error TEXT,
|
|
298
|
+
started_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ', 'now')),
|
|
299
|
+
ended_at TEXT
|
|
300
|
+
);
|
|
301
|
+
CREATE INDEX IF NOT EXISTS worker_runs_task_id_idx ON worker_runs(task_id);
|
|
302
|
+
CREATE INDEX IF NOT EXISTS worker_runs_phase_idx ON worker_runs(phase);
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
// lib/spawn-boundary.js — fail-closed spawn bounds validator (G1a).
|
|
2
|
+
//
|
|
3
|
+
// Piece 1 of the sandbox exit. Every creative-agent spawn on the worker
|
|
4
|
+
// layer goes through this module BEFORE the orchestrator emits NEED_SPAWN.
|
|
5
|
+
// It does not perform the spawn — the worker agent does that — it validates
|
|
6
|
+
// the bounds and FAILS CLOSED (throws, no spawn) unless every rule holds.
|
|
7
|
+
//
|
|
8
|
+
// Rules (DESIGN-v2 G1a):
|
|
9
|
+
// - prompt_file, schema_file, cwd, env, workdir, crewHome must ALL be declared.
|
|
10
|
+
// No spawn without declared bounds. Ever.
|
|
11
|
+
// - prompt_file and schema_file must be co-located (same directory).
|
|
12
|
+
// - cwd must resolve inside workdir (the task's own worktree or a designated
|
|
13
|
+
// scratch dir). `..` escapes fail closed.
|
|
14
|
+
// - cwd must NOT contain denied fragments (defense in depth): crew-state
|
|
15
|
+
// databases, crew homes, observer state, dispatch logs.
|
|
16
|
+
// - env must be an object; no key may match a denied sensitive pattern.
|
|
17
|
+
//
|
|
18
|
+
// This module is import-safe: no side effects on import, bare `node` exits 0.
|
|
19
|
+
|
|
20
|
+
import { resolve, dirname, relative } from "node:path";
|
|
21
|
+
|
|
22
|
+
const DENIED_PATH_FRAGMENTS = [
|
|
23
|
+
"crew-state.db",
|
|
24
|
+
".crew-soak-gate2",
|
|
25
|
+
".observer",
|
|
26
|
+
".tick-releases.jsonl",
|
|
27
|
+
".dispatch-decisions.jsonl",
|
|
28
|
+
];
|
|
29
|
+
|
|
30
|
+
const DENIED_ENV_PATTERNS = [
|
|
31
|
+
"KEY",
|
|
32
|
+
"TOKEN",
|
|
33
|
+
"SECRET",
|
|
34
|
+
"PASSWORD",
|
|
35
|
+
"PRIVATE",
|
|
36
|
+
"CREDENTIAL",
|
|
37
|
+
"AUTH",
|
|
38
|
+
];
|
|
39
|
+
|
|
40
|
+
function fail(violation) {
|
|
41
|
+
const err = new Error("spawn-boundary: BOUNDS_REJECTED: " + violation);
|
|
42
|
+
err.code = "BOUNDS_REJECTED";
|
|
43
|
+
throw err;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// Validate spawn bounds. Returns { cwd, env } (resolved/normalized) on
|
|
47
|
+
// success. Throws with code BOUNDS_REJECTED on any violation.
|
|
48
|
+
export function validateSpawnBounds(bounds) {
|
|
49
|
+
if (!bounds || typeof bounds !== "object") fail("bounds must be an object");
|
|
50
|
+
|
|
51
|
+
const { prompt_file, schema_file, cwd, env, workdir, crewHome } = bounds;
|
|
52
|
+
|
|
53
|
+
if (typeof prompt_file !== "string" || prompt_file.length === 0)
|
|
54
|
+
fail("prompt_file must be a declared non-empty string");
|
|
55
|
+
if (typeof schema_file !== "string" || schema_file.length === 0)
|
|
56
|
+
fail("schema_file must be a declared non-empty string");
|
|
57
|
+
if (typeof cwd !== "string" || cwd.length === 0)
|
|
58
|
+
fail("cwd must be a declared non-empty string");
|
|
59
|
+
if (!env || typeof env !== "object" || Array.isArray(env))
|
|
60
|
+
fail("env must be a declared object");
|
|
61
|
+
if (typeof workdir !== "string" || workdir.length === 0)
|
|
62
|
+
fail("workdir must be a declared non-empty string");
|
|
63
|
+
if (typeof crewHome !== "string" || crewHome.length === 0)
|
|
64
|
+
fail("crewHome must be a declared non-empty string");
|
|
65
|
+
|
|
66
|
+
// Schema co-location: prompt and schema travel together.
|
|
67
|
+
if (dirname(resolve(prompt_file)) !== dirname(resolve(schema_file)))
|
|
68
|
+
fail("prompt_file and schema_file must be co-located (same directory)");
|
|
69
|
+
|
|
70
|
+
// cwd must resolve inside workdir — no `..` escapes.
|
|
71
|
+
const resolvedCwd = resolve(cwd);
|
|
72
|
+
const resolvedWorkdir = resolve(workdir);
|
|
73
|
+
const rel = relative(resolvedWorkdir, resolvedCwd);
|
|
74
|
+
if (rel === ".." || rel.startsWith("../") || resolve(resolvedWorkdir, rel) !== resolvedCwd)
|
|
75
|
+
fail("cwd must resolve inside workdir (got " + cwd + ")");
|
|
76
|
+
|
|
77
|
+
// Defense in depth: denied fragments anywhere in the resolved cwd.
|
|
78
|
+
for (const frag of DENIED_PATH_FRAGMENTS) {
|
|
79
|
+
if (resolvedCwd.includes(frag)) fail("cwd inside denied path: " + frag);
|
|
80
|
+
}
|
|
81
|
+
// The crew home itself is never a spawn cwd, even without a fragment hit.
|
|
82
|
+
const crewRel = relative(resolve(crewHome), resolvedCwd);
|
|
83
|
+
if (crewRel === "" || (!crewRel.startsWith("../") && crewRel !== ".."))
|
|
84
|
+
fail("cwd must not be inside crewHome");
|
|
85
|
+
|
|
86
|
+
// Sensitive environment denied by default.
|
|
87
|
+
for (const key of Object.keys(env)) {
|
|
88
|
+
const upper = String(key).toUpperCase();
|
|
89
|
+
for (const pat of DENIED_ENV_PATTERNS) {
|
|
90
|
+
if (upper.includes(pat)) fail("env key denied: " + key);
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
return { cwd: resolvedCwd, env: { ...env } };
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// Declared constants for tests and for orchestrators that build bounds.
|
|
98
|
+
export const BOUNDARY = {
|
|
99
|
+
DENIED_PATH_FRAGMENTS: [...DENIED_PATH_FRAGMENTS],
|
|
100
|
+
DENIED_ENV_PATTERNS: [...DENIED_ENV_PATTERNS],
|
|
101
|
+
};
|
package/package.json
CHANGED
|
@@ -43,6 +43,10 @@ The failures below are settled and recorded in the Gate 1 OODA state. The author
|
|
|
43
43
|
|
|
44
44
|
2. **Load the workflow registry:** Read the file "{crewHome}/workflows/registry.json" with the read tool and keep its RAW TEXT — do NOT parse it, do NOT re-emit it as a JSON object. Pass the text verbatim: never reformat, summarize, or hand-transcribe it (room #26: 15+ launch rejections 09-19→09-21 from garbled nested-object args, and one hallucinated registry that passed validation). If the file does not exist (the live release predates the registry), proceed without it — omit the `registryText` arg and the dispatcher will load the registry the slow way and log a warning. If the read FAILS on a file that exists (transient read error — observed 2026-09-13; the file itself was healthy and later ticks read it fine), retry the read once; if it still fails, write the error text into your run summary (observability — never silently swallow a failed read) and proceed without the registry the same way.
|
|
45
45
|
|
|
46
|
+
2.5. **Shadow dispatch (Piece 1, 2026-09-26):** the worker-layer dispatcher (`lib/crew-dispatch-worker.js`) is the sandbox exit's replacement for the sandboxed dispatcher in Step 3. It is NOT authoritative yet — Step 3's sandboxed dispatcher still owns every launch decision. Run the worker dispatcher in read-only shadow mode BEFORE Step 3 (both must read the same board state; the shadow never mutates it). Run in shell:
|
|
47
|
+
`node {crewHome}/lib/crew-dispatch-worker.js --crew-home {crewHome} --read-only`
|
|
48
|
+
The last stdout line is a JSON result (`{status, claims, ...}`); lines before it are the decision log. The script appends its own shadow evidence to `{crewHome}/.dispatch-shadow.jsonl` — do NOT transcribe or re-emit the claims yourself. If the script exits non-zero, log `SHADOW_FAIL <stderr tail>` and continue to Step 3 (the shadow is evidence, never a gate). A zero exit with claims is normal — the claims are recommendations only; NOTHING is launched from them. Every claim carries `executor: "worker"`; the shadow comparison ignores it.
|
|
49
|
+
|
|
46
50
|
3. **Run the dispatcher:** Call workflow_launch with scriptPath "{crewHome}/workflows/crew-dispatch.js" and args {"crewHome": "{crewHome}", "registryText": <raw registry file text, or omit the key when the file was missing>}. The dispatcher parses the text itself — a nested-object `registry` arg is not accepted. If the launch is rejected, retry once with only {"crewHome": "{crewHome}"} (the sanctioned fallback — the dispatcher loads the registry the slow way); a second rejection is a platform problem, not an args problem — write it in your run summary and move on.
|
|
47
51
|
|
|
48
52
|
Wait for it to complete. It reads the crew's task state, determines eligibility, claims tasks, acknowledges the poll, and returns structured results.
|
|
@@ -106,6 +110,10 @@ The failures below are settled and recorded in the Gate 1 OODA state. The author
|
|
|
106
110
|
- `timeouts` = fully handled by code — take NO action, just log. The 2-hour total budget from first issuance is exhausted with no acknowledgement; code writes the terminal `publish: version-timeout` note — parked for human attention. `timeouts` is terminal-budget exhaustion only — a window expiry is a `reissued`, never a timeout.
|
|
107
111
|
- `reissued` = fresh intents for the next attempt — a 30-minute window expired with no acknowledgement, so code re-issued with a fresh version (by derivation), immediately claimable (no backoff). This tick does NOT issue them; the NEXT tick's `scan-publish-intent` claims them. Log only.
|
|
108
112
|
- `skipped` = window still open or evidence absent — log only, take no action.
|
|
113
|
+
|
|
114
|
+
4.6. **Shadow verdict (Piece 1, 2026-09-26):** after the Step-3 dispatcher's decision is recorded, pair its decision with the Step-2.5 shadow evidence. Run in shell:
|
|
115
|
+
`node {crewHome}/lib/compare-dispatch-shadow.js --crew-home {crewHome}`
|
|
116
|
+
Log every `SHADOW_*` line verbatim. `SHADOW_MATCH` means the worker layer agreed with the sandbox on this tick's claims. `SHADOW_DIVERGE` names the differing claim triples (`task_id|workflow|step`) — log them; divergence is evidence for the cutover review, not a tick failure, so continue the tick. `SHADOW_UNPAIRED` means the authoritative decision line is missing (the Step-3 dispatcher died after the shadow ran) — log it loudly and continue. `SHADOW_NO_EVIDENCE` on the first shadowed tick is normal. Verdicts append to `{crewHome}/.dispatch-shadow-verdicts.jsonl`; the cutover decision after ~100-200 ticks is made from that file, never from a single tick.
|
|
109
117
|
- Never stamp provenance from prose. Never infer a verdict from an inspector's summary text. The exact version string is the sole positive signal.
|
|
110
118
|
|
|
111
119
|
5. **Monitor launched workflows until terminal (stay-alive — 2026-09-13):** The platform ties async workflow `agent()` authorization to the launcher's lifetime: if THIS tick ends while a workflow is still running, the workflow's next `agent()` call fails with "subagent bootstrap is no longer authorized" / "subagent reservation owner is terminal". Prevention beats recovery here, so this tick is configured with a 90-minute execution timeout (`timeout_secs: 5400` in seed/crons.json) and you MUST stay alive until every launched run reaches a terminal state. Do not exit early while a launched run is still `running` — your death is what kills it.
|
|
@@ -1054,7 +1054,11 @@ for (var p = 0; p < toProcess.length; p++) {
|
|
|
1054
1054
|
};
|
|
1055
1055
|
|
|
1056
1056
|
log("Recommended " + iworkflow + " for \"" + itask.title + "\" [" + taskProject + "] at step " + nextStepName);
|
|
1057
|
-
|
|
1057
|
+
// executor tag (Piece 1, 2026-09-26): the sandboxed dispatcher claims
|
|
1058
|
+
// "sandbox"; the worker-layer dispatcher (lib/crew-dispatch-worker.js)
|
|
1059
|
+
// claims "worker". Piece 2 routes launches on this tag. The shadow
|
|
1060
|
+
// comparison ignores it.
|
|
1061
|
+
results.push({ task_id: itask.id, workflow: iworkflow, step: nextStepName, action: "recommended", scriptPath: scriptPath, args: launchArgs, executor: "sandbox" });
|
|
1058
1062
|
|
|
1059
1063
|
} // end for (per-task loop)
|
|
1060
1064
|
} // end processing block
|