muse-crew 0.7.19 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/guide.md +179 -7
- package/docs/publish-verification.md +4 -2
- package/lib/AGENTS.md +6 -2
- package/lib/build-registry.js +3 -2
- package/lib/compose-evidence-caption.js +57 -14
- package/lib/publish-npm.sh +33 -0
- package/lib/test-publish-preflight.sh +210 -0
- package/lib/test-worktree-backend.sh +51 -1
- package/lib/update-watch.js +633 -0
- package/lib/verify-publish.js +51 -13
- package/lib/worktree-lifecycle.sh +68 -10
- package/package.json +1 -1
- package/seed/AGENTS.md +1 -0
- package/seed/cron-body-update-watch.md +13 -0
- package/seed/crons.json +14 -1
- package/seed/workflows/upgrade.md +21 -0
- package/workflows/AGENTS.md +1 -1
- package/workflows/bugfix.js +416 -74
- package/workflows/chore.js +223 -72
- package/workflows/crew-dispatch.js +43 -6
- package/workflows/crew-init.js +173 -10
- package/workflows/standard.js +231 -79
- package/workflows/upgrade.js +794 -0
package/workflows/standard.js
CHANGED
|
@@ -57,7 +57,10 @@ if (!inputs.crewHome) throw new Error("crewHome is required — pass the crew ho
|
|
|
57
57
|
const crewHome = inputs.crewHome;
|
|
58
58
|
// Crew API: the workflow calls the crew-owned CLI, not the dashboard.
|
|
59
59
|
// The CLI implements the API.md contract against $CREW_HOME/crew-state.db.
|
|
60
|
-
const
|
|
60
|
+
const CREW_API_SRC = crewHome + "/current/lib/crew-api.js";
|
|
61
|
+
// Pinned at pinLifecycle: after the pin, CREW_API points into RUN_LIB so a
|
|
62
|
+
// mid-flight release swap cannot change the CLI under a running workflow.
|
|
63
|
+
let CREW_API = CREW_API_SRC;
|
|
61
64
|
// Build a shell command invoking the CLI. Args are JSON-encoded and
|
|
62
65
|
// single-quote-wrapped for safe shell passing. The agent runs this and
|
|
63
66
|
// returns the stdout verbatim (the CLI emits JSON on stdout).
|
|
@@ -83,9 +86,12 @@ const LIFECYCLE = RUN_LIB + "/worktree-lifecycle.sh";
|
|
|
83
86
|
const MERGE_LOCK = RUN_LIB + "/merge-lock.sh";
|
|
84
87
|
const PUBLISH_NPM_SRC = crewHome + "/lib/publish-npm.sh";
|
|
85
88
|
const PUBLISH_NPM = RUN_LIB + "/publish-npm.sh";
|
|
86
|
-
|
|
89
|
+
const CREW_API_PINNED = RUN_LIB + "/crew-api.js";
|
|
90
|
+
const SCHEMA_SQL_SRC = crewHome + "/lib/schema.sql";
|
|
91
|
+
const SCHEMA_SQL_PINNED = RUN_LIB + "/schema.sql";
|
|
92
|
+
// The five basenames the pin step must materialize — asserted mechanically
|
|
87
93
|
// by workflow code from the verbatim listing, never from agent prose.
|
|
88
|
-
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM].map(function (p) { return p.split("/").pop(); });
|
|
94
|
+
const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED].map(function (p) { return p.split("/").pop(); });
|
|
89
95
|
|
|
90
96
|
// Project config — passed by dispatcher, falls back to dashboard defaults
|
|
91
97
|
const projectConfig = inputs.project_config || {};
|
|
@@ -231,7 +237,7 @@ function attemptKey(base, reworkCount) {
|
|
|
231
237
|
return base + (reworkCount > 0 ? "-r" + reworkCount : "");
|
|
232
238
|
}
|
|
233
239
|
// pinLifecycle(key) — snapshot the lifecycle scripts into RUN_LIB and return
|
|
234
|
-
// the verbatim `ls -1` listing so WORKFLOW CODE asserts the
|
|
240
|
+
// the verbatim `ls -1` listing so WORKFLOW CODE asserts the five pinned
|
|
235
241
|
// basenames; the agent cannot self-certify. (The pin step was the one place
|
|
236
242
|
// the workflows trusted agent prose: task 24be1cd6 walked to Publish on an
|
|
237
243
|
// empty pin dir.) Byte-identical across standard/bugfix/chore — pinned by
|
|
@@ -239,7 +245,7 @@ function attemptKey(base, reworkCount) {
|
|
|
239
245
|
function pinLifecycle(key) {
|
|
240
246
|
return agent(
|
|
241
247
|
"Snapshot lifecycle scripts for version pinning.\n" +
|
|
242
|
-
"Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
|
|
248
|
+
"Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
|
|
243
249
|
"Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
|
|
244
250
|
{ key: key, label: "Pinning lifecycle scripts",
|
|
245
251
|
schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
|
|
@@ -451,13 +457,16 @@ function parseUnifiedDiff(diffText) {
|
|
|
451
457
|
}
|
|
452
458
|
|
|
453
459
|
|
|
454
|
-
// Publish read-back request (currently unavailable): the
|
|
455
|
-
// the parent protocol (docs/publish-verification.md) would
|
|
456
|
-
// independent read-back tool after the artifact build lands.
|
|
457
|
-
// was removed by the platform (2026-09-14);
|
|
458
|
-
// diagnosis, not a substitute — so no
|
|
459
|
-
//
|
|
460
|
-
//
|
|
460
|
+
// Publish read-back request (agent path currently unavailable): the
|
|
461
|
+
// verbatim_request the parent protocol (docs/publish-verification.md) would
|
|
462
|
+
// hand to an independent read-back tool after the artifact build lands.
|
|
463
|
+
// artifact_inspect was removed by the platform (2026-09-14);
|
|
464
|
+
// artifact.inspect is malfunction diagnosis, not a substitute — so no
|
|
465
|
+
// agent-callable read-back tool exists and this LLM-inspector request
|
|
466
|
+
// cannot currently be issued. The primary sensor is now the deterministic
|
|
467
|
+
// lib/readback-disk.js (reads the on-disk tree the artifact is served
|
|
468
|
+
// from); this request builder is retained only as the manual fallback.
|
|
469
|
+
// Pure function — no I/O, no clock. The request carries the merged diff as the expected change and asks
|
|
461
470
|
// for an independent read of the artifact's actual source: for each file, the
|
|
462
471
|
// exact current text of the changed regions plus a per-line present/absent
|
|
463
472
|
// finding. Until a read-back path exists, the parent cannot independently
|
|
@@ -531,7 +540,7 @@ function extractMarkerLines(workerText) {
|
|
|
531
540
|
var markers = [];
|
|
532
541
|
for (var i = 0; i < lines.length; i++) {
|
|
533
542
|
var line = lines[i].trim();
|
|
534
|
-
if (/^(repo_diff:|release:|version_bump:|VERDICT:|TARGET_VERSION=|published:|experiential:|capture_targets:|worktree:)/i.test(line)) {
|
|
543
|
+
if (/^(repo_diff:|release:|version_bump:|VERDICT:|TARGET_VERSION=|published:|experiential:|layer:|capture_targets:|worktree:)/i.test(line)) {
|
|
535
544
|
markers.push(line);
|
|
536
545
|
}
|
|
537
546
|
}
|
|
@@ -802,7 +811,7 @@ let i = startStepIndex;
|
|
|
802
811
|
// ── Pin lifecycle scripts ────────────────────────────────────────────
|
|
803
812
|
// Copy lifecycle scripts into a per-task temp dir so this run is immune
|
|
804
813
|
// to upgrades that land while it's in flight. Verified mechanically:
|
|
805
|
-
// workflow code asserts the
|
|
814
|
+
// workflow code asserts the five basenames from the verbatim listing —
|
|
806
815
|
// the agent cannot self-certify. Any miss parks the task before Triage.
|
|
807
816
|
const initialPins = parsePinListing(await pinLifecycle("pin-lifecycle"));
|
|
808
817
|
const missingInitialPins = PIN_BASENAMES.filter(function (b) { return initialPins.indexOf(b) === -1; });
|
|
@@ -810,6 +819,10 @@ if (missingInitialPins.length > 0) {
|
|
|
810
819
|
return await parkTask("Lifecycle pin incomplete before Triage — missing " + missingInitialPins.join(", ") + " in " + RUN_LIB + ".");
|
|
811
820
|
}
|
|
812
821
|
log("Lifecycle scripts pinned to " + RUN_LIB);
|
|
822
|
+
// From here on, every crew-api.js invocation uses the pinned copy: immune
|
|
823
|
+
// to a release swap landing mid-flight.
|
|
824
|
+
CREW_API = CREW_API_PINNED;
|
|
825
|
+
log("Crew API pinned to " + CREW_API);
|
|
813
826
|
|
|
814
827
|
// Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
|
|
815
828
|
// minted at this run's first claim (never a PID — short-lived agent PIDs
|
|
@@ -1112,7 +1125,9 @@ while (i < STEPS.length) {
|
|
|
1112
1125
|
var mapBaselineRefs = "";
|
|
1113
1126
|
var mapBaselineNone = false;
|
|
1114
1127
|
if (step.name === "Map") {
|
|
1115
|
-
|
|
1128
|
+
// Must match Capture's run condition (experiential + artifact publish):
|
|
1129
|
+
// when Capture skips, no baseline notes exist, so the gate must not apply.
|
|
1130
|
+
if ((await resolveExperiential()) === "yes" && PUBLISH_TYPE === "artifact") {
|
|
1116
1131
|
var gateStatus = await baselineStatus();
|
|
1117
1132
|
if (!gateStatus.baseline_found) {
|
|
1118
1133
|
log("Map gate: no baseline evidence for experiential task " + taskId + " — bouncing to Capture");
|
|
@@ -1280,6 +1295,10 @@ while (i < STEPS.length) {
|
|
|
1280
1295
|
"TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
|
|
1281
1296
|
"skipped: no-lock-held (empty-diff Integrate — nothing merged, nothing to ship)\n" +
|
|
1282
1297
|
"VERDICT: PASS\n\n" +
|
|
1298
|
+
"If the script's output contains PUBLISH_SKIPPED=no-npm-publish, the publish was skipped gracefully: npm publish is not configured on this machine (helper or credential absent) — the merge stands, the version was not cut, nothing was shipped. Paste the marker block verbatim into your report, then end your report with exactly these three lines, in this order — lowercase, no trailing period, do not rephrase:\n" +
|
|
1299
|
+
"TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
|
|
1300
|
+
"skipped: no-npm-publish (npm publish not configured — helper or credential missing; nothing versioned or published)\n" +
|
|
1301
|
+
"VERDICT: PASS\n\n" +
|
|
1283
1302
|
"If it exits zero, paste the script's COMPLETE marker block verbatim into your report, then end your report with exactly these three lines, in this order — lowercase, no trailing period, do not rephrase:\n" +
|
|
1284
1303
|
"TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
|
|
1285
1304
|
"published: muse-crew@" + publishTarget.target + "\n" +
|
|
@@ -1299,9 +1318,10 @@ while (i < STEPS.length) {
|
|
|
1299
1318
|
// 2026-09-11), so the stamp moved to the parent — after the build
|
|
1300
1319
|
// lands, the workflow records the session completed and parks with
|
|
1301
1320
|
// "publish: verification-requested". The parent owns verification
|
|
1302
|
-
// (docs/publish-verification.md); the
|
|
1303
|
-
//
|
|
1304
|
-
// artifact_inspect was removed by the platform
|
|
1321
|
+
// (docs/publish-verification.md); the primary sensor is the
|
|
1322
|
+
// deterministic lib/readback-disk.js (the agent-callable read-back
|
|
1323
|
+
// tool is unavailable — artifact_inspect was removed by the platform
|
|
1324
|
+
// 2026-09-14 — so the LLM-inspector path is manual-fallback only).
|
|
1305
1325
|
// QA's provenance check enforces the stamp mechanically.
|
|
1306
1326
|
var artifactPublish = null;
|
|
1307
1327
|
var publishLockRefreshed = false;
|
|
@@ -1373,9 +1393,17 @@ while (i < STEPS.length) {
|
|
|
1373
1393
|
} else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
|
|
1374
1394
|
return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
|
|
1375
1395
|
}
|
|
1396
|
+
// The empty tree is not a commit: git merge-base --is-ancestor fails on it.
|
|
1397
|
+
// The workflow knows publishBase == EMPTY_TREE_SHA (set above), so it
|
|
1398
|
+
// hardcodes ANCESTOR=yes for a first publish instead of asking the agent
|
|
1399
|
+
// to execute the conditional (clean-room 2026-09-16: the agent ran
|
|
1400
|
+
// merge-base on the empty tree directly and parked).
|
|
1401
|
+
var ancestorShell = (publishBase === EMPTY_TREE_SHA)
|
|
1402
|
+
? "ANCESTOR=yes && "
|
|
1403
|
+
: "git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no && ";
|
|
1376
1404
|
var diffResult = await agent(
|
|
1377
1405
|
"Run: cd " + REPO_PATH + " && BASE='" + publishBase + "' && HEAD=$(git rev-parse HEAD) && " +
|
|
1378
|
-
|
|
1406
|
+
ancestorShell +
|
|
1379
1407
|
"echo '---COMMIT---' && echo \"$HEAD\" && echo '---BASE---' && echo \"$BASE\" && echo '---ANCESTOR---' && echo \"$ANCESTOR\" && " +
|
|
1380
1408
|
"if [ \"$ANCESTOR\" = yes ]; then echo '---DIFF---' && git diff \"$BASE\" \"$HEAD\" && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r \"$BASE\" \"$HEAD\"; fi\n" +
|
|
1381
1409
|
"Return JSON { \"commit\": \"<HEAD trimmed>\", \"base\": \"<BASE trimmed>\", \"ancestor\": \"<yes|no>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
|
|
@@ -1448,8 +1476,10 @@ while (i < STEPS.length) {
|
|
|
1448
1476
|
// The builder's applied report is gone (2026-09-16): it rode on the
|
|
1449
1477
|
// trigger's JSON closeout contract, which is removed below. The
|
|
1450
1478
|
// parent's independent read-back (docs/publish-verification.md) is
|
|
1451
|
-
// the verification — this field stays "missing-report" on
|
|
1452
|
-
//
|
|
1479
|
+
// the verification — this field stays "missing-report" on ledger
|
|
1480
|
+
// lines for issued triggers; pre-trigger parks (toolcheck
|
|
1481
|
+
// rejected/inconclusive) and unattributed-unknown parks write null
|
|
1482
|
+
// (no trigger was observed, so there is nothing to report).
|
|
1453
1483
|
var publishAppliedObservation = "missing-report";
|
|
1454
1484
|
// Durable-evidence snapshot (2026-09-14): the observation below only
|
|
1455
1485
|
// detects IN-FLIGHT builds. A build that finished before the
|
|
@@ -1460,10 +1490,13 @@ while (i < STEPS.length) {
|
|
|
1460
1490
|
// fallback can diff before/after: a directory appearing during the
|
|
1461
1491
|
// trigger window is positive evidence the edit went through and
|
|
1462
1492
|
// the build completed. Best-effort and non-gating: if the snapshot
|
|
1463
|
-
// fails,
|
|
1464
|
-
//
|
|
1493
|
+
// fails, auditBeforeOk stays false and BOTH fallback comparisons
|
|
1494
|
+
// are disabled (2026-09-16, critic finding 4) — without a baseline,
|
|
1495
|
+
// an empty before-list would make every historical audit dir look
|
|
1496
|
+
// "new". No wall-clock in-script (deterministic replay) — the
|
|
1465
1497
|
// comparison is a pure before/after set diff.
|
|
1466
1498
|
var auditDirsBeforeTrigger = [];
|
|
1499
|
+
var auditBeforeOk = false;
|
|
1467
1500
|
try {
|
|
1468
1501
|
var auditBefore = await agent(
|
|
1469
1502
|
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
|
|
@@ -1473,9 +1506,10 @@ while (i < STEPS.length) {
|
|
|
1473
1506
|
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1474
1507
|
);
|
|
1475
1508
|
auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1509
|
+
auditBeforeOk = true;
|
|
1476
1510
|
log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
|
|
1477
1511
|
} catch (auditBeforeErr) {
|
|
1478
|
-
log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal,
|
|
1512
|
+
log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
|
|
1479
1513
|
}
|
|
1480
1514
|
// Fire-and-forget trigger + workflow-owned observation (2026-09-16,
|
|
1481
1515
|
// clean-room task e2a8d9f8): the trigger's JSON closeout contract
|
|
@@ -1499,10 +1533,16 @@ while (i < STEPS.length) {
|
|
|
1499
1533
|
// Pre-trigger toolcheck (tiny, schema'd): the artifact namespace is
|
|
1500
1534
|
// deferred for workflow children — the child self-loads it and emits
|
|
1501
1535
|
// one exact signal line, read mechanically (never English prose).
|
|
1502
|
-
//
|
|
1503
|
-
//
|
|
1504
|
-
//
|
|
1536
|
+
// Only a parsed ARTIFACT_TOOLS: missing signal is explicit negative
|
|
1537
|
+
// evidence: it gets one bounded retry with a fresh key, then parks
|
|
1538
|
+
// rejected — without the tools the edit provably did NOT go through,
|
|
1539
|
+
// so this is the one safe retry on the publish path. A throw (or an
|
|
1540
|
+
// unparseable signal) is INCONCLUSIVE transport noise, never
|
|
1541
|
+
// evidence of missing tools (2026-09-16, critic finding 3): it is
|
|
1542
|
+
// recorded, it retries once in case the flake clears, but it can
|
|
1543
|
+
// never take the rejected path.
|
|
1505
1544
|
var publishToolsOk = false;
|
|
1545
|
+
var publishToolsMissing = false;
|
|
1506
1546
|
for (var toolcheckAttempt = 1; toolcheckAttempt <= 2 && !publishToolsOk; toolcheckAttempt++) {
|
|
1507
1547
|
try {
|
|
1508
1548
|
var toolcheckResult = await agent(
|
|
@@ -1514,12 +1554,29 @@ while (i < STEPS.length) {
|
|
|
1514
1554
|
label: "Checking artifact tool availability" + (toolcheckAttempt === 1 ? "" : " (retry)"),
|
|
1515
1555
|
schema: { type: "object", properties: { signal: { type: "string" } }, required: ["signal"] } }
|
|
1516
1556
|
);
|
|
1517
|
-
|
|
1518
|
-
|
|
1557
|
+
var toolSignal = String((toolcheckResult && toolcheckResult.signal) || "");
|
|
1558
|
+
if (/ARTIFACT_TOOLS:\s*ok/.test(toolSignal)) {
|
|
1559
|
+
publishToolsOk = true;
|
|
1560
|
+
} else if (/ARTIFACT_TOOLS:\s*missing/.test(toolSignal)) {
|
|
1561
|
+
publishToolsMissing = true;
|
|
1562
|
+
}
|
|
1563
|
+
log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2): " +
|
|
1564
|
+
(publishToolsOk ? "tools ok" : publishToolsMissing ? "tools missing (explicit parsed signal)" : "inconclusive (no ARTIFACT_TOOLS signal parsed)"));
|
|
1519
1565
|
} catch (toolcheckErr) {
|
|
1520
|
-
log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2)
|
|
1566
|
+
log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2) threw (" + (toolcheckErr && toolcheckErr.message ? toolcheckErr.message : toolcheckErr) + ") — inconclusive: a throw proves nothing about tool availability, never counted as missing");
|
|
1521
1567
|
}
|
|
1522
1568
|
}
|
|
1569
|
+
if (!publishToolsOk && !publishToolsMissing) {
|
|
1570
|
+
await recordPublishLedger({
|
|
1571
|
+
commit: mergeCommitForPublish,
|
|
1572
|
+
attempt: rebuildAttemptKey,
|
|
1573
|
+
agent_id: null,
|
|
1574
|
+
applied_report: null,
|
|
1575
|
+
outcome: "unknown",
|
|
1576
|
+
detail: "artifact toolcheck inconclusive after two attempts (throws or unparseable signals — never an explicit ARTIFACT_TOOLS: missing): tool availability unproven, so the trigger was NOT issued; unknown parks fail closed with no blind retry"
|
|
1577
|
+
}, totalReworkCount);
|
|
1578
|
+
return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact toolcheck was inconclusive after two attempts (no explicit ARTIFACT_TOOLS signal parsed — a throw is transport noise, not evidence). Tool availability is unproven, so no edit was issued and nothing was retried blindly. Human attention needed.");
|
|
1579
|
+
}
|
|
1523
1580
|
if (!publishToolsOk) {
|
|
1524
1581
|
await recordPublishLedger({
|
|
1525
1582
|
commit: mergeCommitForPublish,
|
|
@@ -1527,9 +1584,9 @@ while (i < STEPS.length) {
|
|
|
1527
1584
|
agent_id: null,
|
|
1528
1585
|
applied_report: null,
|
|
1529
1586
|
outcome: "rejected",
|
|
1530
|
-
detail: "artifact tool namespace missing
|
|
1587
|
+
detail: "artifact tool namespace explicitly missing (parsed ARTIFACT_TOOLS: missing signal, one bounded retry spent): the edit provably did not go through — no trigger issued, no blind retry"
|
|
1531
1588
|
}, totalReworkCount);
|
|
1532
|
-
return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was missing
|
|
1589
|
+
return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was explicitly missing (parsed signal — the edit provably did not go through, so no trigger was issued and nothing was retried blindly). Human attention needed.");
|
|
1533
1590
|
}
|
|
1534
1591
|
// Pre-trigger build-state baseline (tiny, schema'd): one read of
|
|
1535
1592
|
// artifact_status. The post-trigger observation diffs against this
|
|
@@ -1555,19 +1612,26 @@ while (i < STEPS.length) {
|
|
|
1555
1612
|
baselineFailed = true;
|
|
1556
1613
|
log("Publish pre-trigger baseline read failed for task " + taskId + " (" + (baselineErr && baselineErr.message ? baselineErr.message : baselineErr) + ") — receipt attribution skipped; durable audit-dir evidence is the only positive signal");
|
|
1557
1614
|
}
|
|
1558
|
-
// The trigger itself:
|
|
1559
|
-
//
|
|
1560
|
-
//
|
|
1561
|
-
// call
|
|
1562
|
-
//
|
|
1563
|
-
//
|
|
1564
|
-
//
|
|
1565
|
-
//
|
|
1615
|
+
// The trigger itself: the artifact_edit call is AWAITED (the workflow
|
|
1616
|
+
// waits for it to complete) but its return value is intentionally
|
|
1617
|
+
// UNCONSUMED — NO schema, so no schema validation can fail this
|
|
1618
|
+
// call: a schema-less call resolves to the child's raw response as
|
|
1619
|
+
// a plain string (probed live 2026-09-16 — never parsed, never
|
|
1620
|
+
// throws on content). One caveat, also probed: the runtime still
|
|
1621
|
+
// scans the response for a JSON candidate, and an unparseable
|
|
1622
|
+
// {...}-looking substring in the child's prose throws ("response
|
|
1623
|
+
// JSON candidate", probe P6). The prompt tells the child to end its
|
|
1624
|
+
// turn with no prose at all, which keeps the common case clean —
|
|
1625
|
+
// but the channel is stochastic, so any throw is possible and
|
|
1626
|
+
// inconclusive: the edit may still have gone through, so the
|
|
1627
|
+
// outcome stays unknown until the observation below confirms it —
|
|
1628
|
+
// never inferred from the throw, and never blind-retried (a blind
|
|
1629
|
+
// re-trigger duplicated the edit on 2026-09-12).
|
|
1566
1630
|
var rebuildTrigger = null;
|
|
1567
1631
|
try {
|
|
1568
1632
|
var triggerResultLength = String(await agent(rebuildPrompt,
|
|
1569
1633
|
{ key: rebuildAttemptKey, label: "Triggering artifact rebuild" }) || "").length;
|
|
1570
|
-
log("Publish rebuild trigger for task " + taskId + " returned (" + triggerResultLength + " chars
|
|
1634
|
+
log("Publish rebuild trigger for task " + taskId + " returned (" + triggerResultLength + " chars; awaited but return intentionally unconsumed)");
|
|
1571
1635
|
} catch (triggerErr) {
|
|
1572
1636
|
log("Publish rebuild trigger for task " + taskId + " threw (" + (triggerErr && triggerErr.message ? triggerErr.message : triggerErr) + ") — outcome unknown until observation confirms it; the edit may have gone through");
|
|
1573
1637
|
}
|
|
@@ -1602,6 +1666,16 @@ while (i < STEPS.length) {
|
|
|
1602
1666
|
log("Publish post-trigger build-state check failed for task " + taskId + " (" + (buildCheckErr && buildCheckErr.message ? buildCheckErr.message : buildCheckErr) + ") — this signal is unknown, not negative");
|
|
1603
1667
|
}
|
|
1604
1668
|
var observedAgentId = (buildState && buildState.build && typeof buildState.build.agent_id === "string" && buildState.build.agent_id) || null;
|
|
1669
|
+
// Known limitation (failure-mode audit 2026-09-16): attribution
|
|
1670
|
+
// is timing-based — any agent_id new relative to the baseline is
|
|
1671
|
+
// treated as this edit's receipt. A stranger's build starting inside
|
|
1672
|
+
// the trigger window is indistinguishable by timing and would be
|
|
1673
|
+
// misattributed here. The consequence is bounded: the completion
|
|
1674
|
+
// poll below tracks the recorded id, and the parent's mechanical
|
|
1675
|
+
// content read-back (docs/publish-verification.md) certifies the
|
|
1676
|
+
// exact commit's content — a wrong build's content fails closed as
|
|
1677
|
+
// verification-failed, never stamped. Timing narrows the candidate;
|
|
1678
|
+
// content decides.
|
|
1605
1679
|
var receiptAgentId = (!buildStateFailed && !baselineFailed && observedAgentId && observedAgentId !== baselineAgentId) ? observedAgentId : null;
|
|
1606
1680
|
if (receiptAgentId) {
|
|
1607
1681
|
// The edit went through — a build with a new agent_id appeared
|
|
@@ -1621,6 +1695,31 @@ while (i < STEPS.length) {
|
|
|
1621
1695
|
}, totalReworkCount);
|
|
1622
1696
|
} else {
|
|
1623
1697
|
var newAuditDirs = [];
|
|
1698
|
+
// auditReportOk: pure tri-state read of a report.json body —
|
|
1699
|
+
// true (build ok), false (build failed), null (missing or
|
|
1700
|
+
// unreadable — not evidence either way). The child returns the
|
|
1701
|
+
// raw body verbatim; interpretation lives here, never in prose.
|
|
1702
|
+
// Defined here so both the immediate and post-poll audit
|
|
1703
|
+
// fallbacks share it.
|
|
1704
|
+
var auditReportOk = function (raw) {
|
|
1705
|
+
if (typeof raw !== "string") return null;
|
|
1706
|
+
var trimmed = raw.trim();
|
|
1707
|
+
if (trimmed === "" || trimmed === "MISSING") return null;
|
|
1708
|
+
var parsed;
|
|
1709
|
+
try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
|
|
1710
|
+
if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
|
|
1711
|
+
return null;
|
|
1712
|
+
};
|
|
1713
|
+
// (2026-09-16, critic finding 2) When durable audit evidence
|
|
1714
|
+
// confirms (or refutes) the build, there is no receipt agent_id
|
|
1715
|
+
// to chain the completion poll to — skipReceiptPoll bypasses the
|
|
1716
|
+
// poll below, which with a null receipt could only observe
|
|
1717
|
+
// strangers or nothing.
|
|
1718
|
+
var skipReceiptPoll = false;
|
|
1719
|
+
// publishFailure is declared here (moved up from below) so the
|
|
1720
|
+
// immediate audit fallback can record an explicit build failure
|
|
1721
|
+
// without the later declaration resetting it.
|
|
1722
|
+
var publishFailure = null;
|
|
1624
1723
|
try {
|
|
1625
1724
|
var auditAfter = await agent(
|
|
1626
1725
|
"List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
|
|
@@ -1631,25 +1730,77 @@ while (i < STEPS.length) {
|
|
|
1631
1730
|
);
|
|
1632
1731
|
var auditDirsAfterTrigger = String((auditAfter && auditAfter.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1633
1732
|
// Only timestamped build dirs count — the "latest" symlink
|
|
1634
|
-
// and anything else are not builds.
|
|
1635
|
-
|
|
1733
|
+
// and anything else are not builds. Gated on auditBeforeOk:
|
|
1734
|
+
// without a baseline every historical dir would look new.
|
|
1735
|
+
newAuditDirs = auditBeforeOk ? auditDirsAfterTrigger.filter(function (d) {
|
|
1636
1736
|
return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
|
|
1637
|
-
});
|
|
1737
|
+
}) : [];
|
|
1638
1738
|
} catch (auditAfterErr) {
|
|
1639
1739
|
log("Publish audit-dir re-list after trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
|
|
1640
1740
|
}
|
|
1641
1741
|
if (newAuditDirs.length > 0) {
|
|
1642
1742
|
rebuildTrigger = { edit_started: true };
|
|
1643
1743
|
rebuildAgentId = null;
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1744
|
+
newAuditDirs.sort();
|
|
1745
|
+
var newestImmediateDir = newAuditDirs[newAuditDirs.length - 1];
|
|
1746
|
+
log("Publish rebuild trigger for task " + taskId + ": new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and a build completed; no in-flight receipt was observed.");
|
|
1747
|
+
// (2026-09-16, critic finding 2) Durable audit evidence exists,
|
|
1748
|
+
// but there is no receipt agent_id to chain the completion poll
|
|
1749
|
+
// to — polling with a null receipt can only observe strangers
|
|
1750
|
+
// (any running build differs from "null") or nothing, burning
|
|
1751
|
+
// 10.5 minutes to park unknown. Read the build report now
|
|
1752
|
+
// instead of polling: ok=true confirms completion and routes
|
|
1753
|
+
// directly to parent verification (the poll is skipped);
|
|
1754
|
+
// ok=false is explicit failure; unreadable is unknown.
|
|
1755
|
+
var immediateReportOk = null;
|
|
1756
|
+
try {
|
|
1757
|
+
var immediateOkRead = await agent(
|
|
1758
|
+
"Read the artifact build report for slug \"" + PUBLISH_SLUG + "\".\n" +
|
|
1759
|
+
"Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestImmediateDir + "/report.json 2>/dev/null || echo MISSING\n" +
|
|
1760
|
+
"Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
|
|
1761
|
+
{ key: attemptKey("publish-audit-ok-immediate-" + taskId, totalReworkCount), label: "Reading build report for audit-confirmed build",
|
|
1762
|
+
schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
|
|
1763
|
+
);
|
|
1764
|
+
immediateReportOk = auditReportOk(immediateOkRead && immediateOkRead.raw);
|
|
1765
|
+
} catch (immediateOkErr) {
|
|
1766
|
+
log("Publish build-report read for audit-confirmed dir failed for task " + taskId + " (treated as unknown): " + (immediateOkErr && immediateOkErr.message ? immediateOkErr.message : immediateOkErr));
|
|
1767
|
+
immediateReportOk = null;
|
|
1768
|
+
}
|
|
1769
|
+
if (immediateReportOk === true) {
|
|
1770
|
+
publishBuildLanded = true;
|
|
1771
|
+
artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
|
|
1772
|
+
skipReceiptPoll = true;
|
|
1773
|
+
log("Publish build landed for task " + taskId + " via immediate durable audit evidence (audit dir " + newestImmediateDir + ", report ok=true) — receipt poll skipped (no receipt to chain to), routing directly to parent verification");
|
|
1774
|
+
await recordPublishLedger({
|
|
1775
|
+
commit: mergeCommitForPublish,
|
|
1776
|
+
attempt: rebuildAttemptKey,
|
|
1777
|
+
agent_id: null,
|
|
1778
|
+
applied_report: publishAppliedObservation,
|
|
1779
|
+
outcome: "submitted",
|
|
1780
|
+
detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestImmediateDir + ", report ok=true); receipt poll skipped (no receipt agent_id), routed to parent verification"
|
|
1781
|
+
}, totalReworkCount);
|
|
1782
|
+
} else if (immediateReportOk === false) {
|
|
1783
|
+
skipReceiptPoll = true;
|
|
1784
|
+
publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestImmediateDir + ", report ok=false — immediate audit evidence, no receipt observed). Explicit negative evidence: a build ran and failed. The publish did not land — provenance was not stamped. Fail-closed.";
|
|
1785
|
+
await recordPublishLedger({
|
|
1786
|
+
commit: mergeCommitForPublish,
|
|
1787
|
+
attempt: rebuildAttemptKey,
|
|
1788
|
+
agent_id: null,
|
|
1789
|
+
applied_report: publishAppliedObservation,
|
|
1790
|
+
outcome: "failed",
|
|
1791
|
+
detail: "a build ran and failed: audit dir " + newestImmediateDir + " report ok=false (immediate audit evidence, no receipt)"
|
|
1792
|
+
}, totalReworkCount);
|
|
1793
|
+
} else {
|
|
1794
|
+
await recordPublishLedger({
|
|
1795
|
+
commit: mergeCommitForPublish,
|
|
1796
|
+
attempt: rebuildAttemptKey,
|
|
1797
|
+
agent_id: null,
|
|
1798
|
+
applied_report: null,
|
|
1799
|
+
outcome: "unknown",
|
|
1800
|
+
detail: "new audit dir " + newestImmediateDir + " appeared during the trigger window but its build report is unreadable/missing; no receipt agent_id to poll — outcome unknown, fail-closed with no blind retry"
|
|
1801
|
+
}, totalReworkCount);
|
|
1802
|
+
return await parkTask("Publish outcome unknown for task " + taskId + ": a new audit dir (" + newestImmediateDir + ") appeared during the trigger window but its build report is unreadable, and no in-flight receipt was observed to poll. The edit may have completed. Correlate the accepted edit via the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl — do NOT reissue the edit blindly: if the trigger was accepted, a retry duplicates it (2026-09-12). Verify independently whether the build completed (audit dir + report, or the parent's content read-back) before deciding the next step. Fail-closed.");
|
|
1803
|
+
}
|
|
1653
1804
|
} else {
|
|
1654
1805
|
// No attributable build and no durable evidence — but that
|
|
1655
1806
|
// proves nothing (a fast-completing build can finish between
|
|
@@ -1666,7 +1817,7 @@ while (i < STEPS.length) {
|
|
|
1666
1817
|
outcome: "unknown",
|
|
1667
1818
|
detail: "fire-and-forget trigger; post-trigger build-state poll saw no attributable build (or the check failed) and the audit-dir diff found no new dir; the edit may have been accepted as pending_init"
|
|
1668
1819
|
}, totalReworkCount);
|
|
1669
|
-
return await parkTask("Publish outcome unknown for task " + taskId + ": the rebuild trigger was issued fire-and-forget (no
|
|
1820
|
+
return await parkTask("Publish outcome unknown for task " + taskId + ": the rebuild trigger was issued fire-and-forget (no schema, so no validation failure mode; a candidate-parse throw stays possible and is inconclusive), and the follow-up observation could not attribute a build to the edit for slug " + PUBLISH_SLUG + " — no in-flight build with a new agent_id appeared in the poll window and no new audit dir landed. The edit may have been accepted as pending_init, so no retry was issued: a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion — do NOT reissue the edit blindly. Verify independently whether the build completed before deciding the next step. Fail-closed.");
|
|
1670
1821
|
}
|
|
1671
1822
|
}
|
|
1672
1823
|
|
|
@@ -1679,8 +1830,9 @@ while (i < STEPS.length) {
|
|
|
1679
1830
|
// already recorded the ledger's submitted line on both positive paths
|
|
1680
1831
|
// and parked on unknown — there is no applied report to observe and
|
|
1681
1832
|
// no rejection signal to record.
|
|
1682
|
-
|
|
1683
|
-
|
|
1833
|
+
// (publishFailure is declared with the immediate audit fallback
|
|
1834
|
+
// above so an explicit build failure there survives to here.)
|
|
1835
|
+
if (rebuildTrigger.edit_started && !skipReceiptPoll) {
|
|
1684
1836
|
// (2026-09-16) There is no builder report: the fire-and-forget
|
|
1685
1837
|
// trigger carries no JSON contract, so there is nothing to
|
|
1686
1838
|
// compare and no pre-hash diagnostic. The builder's old
|
|
@@ -1781,10 +1933,11 @@ while (i < STEPS.length) {
|
|
|
1781
1933
|
// the old report check was circular — a fabricated report
|
|
1782
1934
|
// passed by construction, and every phase went green on a hollow
|
|
1783
1935
|
// build. The stamp moves to the parent (docs/publish-verification.md);
|
|
1784
|
-
// the
|
|
1785
|
-
// agent-callable read-back tool
|
|
1786
|
-
// removed by the platform 2026-09-14
|
|
1787
|
-
//
|
|
1936
|
+
// the deterministic lib/readback-disk.js is the primary sensor
|
|
1937
|
+
// (the agent-callable read-back tool is unavailable —
|
|
1938
|
+
// artifact_inspect was removed by the platform 2026-09-14 — so
|
|
1939
|
+
// the LLM-inspector path is manual-fallback only), and the task
|
|
1940
|
+
// parks for parent verification.
|
|
1788
1941
|
// QA's provenance check enforces the stamp mechanically.
|
|
1789
1942
|
// An unverified publish fails loudly in QA instead of passing
|
|
1790
1943
|
// silently here.
|
|
@@ -1827,26 +1980,17 @@ while (i < STEPS.length) {
|
|
|
1827
1980
|
schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
|
|
1828
1981
|
);
|
|
1829
1982
|
var auditDirsAfterPollList = String((auditAfterPoll && auditAfterPoll.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
|
|
1830
|
-
|
|
1983
|
+
// Gated on auditBeforeOk (critic finding 4): without a baseline
|
|
1984
|
+
// every historical dir would look new.
|
|
1985
|
+
newAuditDirsAfterPoll = auditBeforeOk ? auditDirsAfterPollList.filter(function (d) {
|
|
1831
1986
|
return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
|
|
1832
|
-
});
|
|
1987
|
+
}) : [];
|
|
1833
1988
|
log("Publish audit-dir re-list after build poll for task " + taskId + ": " + newAuditDirsAfterPoll.length + " new timestamped dir(s)");
|
|
1834
1989
|
} catch (auditAfterPollErr) {
|
|
1835
1990
|
log("Publish audit-dir re-list after build poll failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterPollErr && auditAfterPollErr.message ? auditAfterPollErr.message : auditAfterPollErr));
|
|
1836
1991
|
}
|
|
1837
|
-
//
|
|
1838
|
-
//
|
|
1839
|
-
// unreadable — not evidence either way). The child returns the
|
|
1840
|
-
// raw body verbatim; interpretation lives here, never in prose.
|
|
1841
|
-
var auditReportOk = function (raw) {
|
|
1842
|
-
if (typeof raw !== "string") return null;
|
|
1843
|
-
var trimmed = raw.trim();
|
|
1844
|
-
if (trimmed === "" || trimmed === "MISSING") return null;
|
|
1845
|
-
var parsed;
|
|
1846
|
-
try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
|
|
1847
|
-
if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
|
|
1848
|
-
return null;
|
|
1849
|
-
};
|
|
1992
|
+
// The shared auditReportOk (defined with the immediate fallback
|
|
1993
|
+
// above) interprets the raw body here too.
|
|
1850
1994
|
var auditOkAfterPoll = null;
|
|
1851
1995
|
var newestAuditDirAfterPoll = null;
|
|
1852
1996
|
if (newAuditDirsAfterPoll.length > 0 && !strangerObserved) {
|
|
@@ -1907,7 +2051,7 @@ while (i < STEPS.length) {
|
|
|
1907
2051
|
} else {
|
|
1908
2052
|
// Unreachable: the observation above either attributes the edit
|
|
1909
2053
|
// (edit_started) or parks. Defensive only — never a silent pass.
|
|
1910
|
-
publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish
|
|
2054
|
+
publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish is unattributed (not proven landed, not proven failed) — provenance was not stamped. Fail-closed.";
|
|
1911
2055
|
}
|
|
1912
2056
|
} // end: publishSkippedNoLock — no rebuild, no stamp, nothing to ship
|
|
1913
2057
|
// STEP 2 (mechanical, always — skip path included): post-deploy
|
|
@@ -2406,16 +2550,24 @@ while (i < STEPS.length) {
|
|
|
2406
2550
|
// Skip-aware (park 2026-09-11): when the deterministic publish script found
|
|
2407
2551
|
// no merge lock held (empty-diff Integrate), it skips the publish path
|
|
2408
2552
|
// gracefully and emits the machine-readable PUBLISH_SKIPPED=no-lock-held
|
|
2409
|
-
// marker.
|
|
2553
|
+
// marker. The preflight (bugfix 2026-09-17) emits
|
|
2554
|
+
// PUBLISH_SKIPPED=no-npm-publish when npm publish is not configured on
|
|
2555
|
+
// this machine (helper or credential absent) — also before any mutation.
|
|
2556
|
+
// Verification is then vacuous — nothing was shipped, and the
|
|
2410
2557
|
// registry must NOT have moved. The marker is script-emitted explicit state
|
|
2411
2558
|
// (pasted verbatim per the Publish agent instructions), not agent prose; a
|
|
2412
2559
|
// report without the marker still runs the full verification fail-closed.
|
|
2413
2560
|
var publishVerified = false;
|
|
2414
2561
|
var npmPublishSkipped = false;
|
|
2415
2562
|
if (step.name === "Publish" && PUBLISH_TYPE === "npm" && publishTarget) {
|
|
2416
|
-
|
|
2563
|
+
var publishSkipMatch = /^PUBLISH_SKIPPED=(no-lock-held|no-npm-publish)$/m.exec(workerText || "");
|
|
2564
|
+
if (publishSkipMatch) {
|
|
2417
2565
|
npmPublishSkipped = true;
|
|
2418
|
-
|
|
2566
|
+
if (publishSkipMatch[1] === "no-npm-publish") {
|
|
2567
|
+
log("Publish skipped for task " + taskId + " (no-npm-publish — npm publish not configured): publish verification vacuous, nothing was shipped");
|
|
2568
|
+
} else {
|
|
2569
|
+
log("Publish skipped for task " + taskId + " (no merge lock held — empty-diff Integrate): publish verification vacuous, nothing was shipped");
|
|
2570
|
+
}
|
|
2419
2571
|
}
|
|
2420
2572
|
if (!npmPublishSkipped) {
|
|
2421
2573
|
try {
|