muse-crew 0.7.19 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -57,7 +57,10 @@ if (!inputs.crewHome) throw new Error("crewHome is required — pass the crew ho
57
57
  const crewHome = inputs.crewHome;
58
58
  // Crew API: the workflow calls the crew-owned CLI, not the dashboard.
59
59
  // The CLI implements the API.md contract against $CREW_HOME/crew-state.db.
60
- const CREW_API = crewHome + "/current/lib/crew-api.js";
60
+ const CREW_API_SRC = crewHome + "/current/lib/crew-api.js";
61
+ // Pinned at pinLifecycle: after the pin, CREW_API points into RUN_LIB so a
62
+ // mid-flight release swap cannot change the CLI under a running workflow.
63
+ let CREW_API = CREW_API_SRC;
61
64
  // Build a shell command invoking the CLI. Args are JSON-encoded and
62
65
  // single-quote-wrapped for safe shell passing. The agent runs this and
63
66
  // returns the stdout verbatim (the CLI emits JSON on stdout).
@@ -83,9 +86,12 @@ const LIFECYCLE = RUN_LIB + "/worktree-lifecycle.sh";
83
86
  const MERGE_LOCK = RUN_LIB + "/merge-lock.sh";
84
87
  const PUBLISH_NPM_SRC = crewHome + "/lib/publish-npm.sh";
85
88
  const PUBLISH_NPM = RUN_LIB + "/publish-npm.sh";
86
- // The three basenames the pin step must materialize — asserted mechanically
89
+ const CREW_API_PINNED = RUN_LIB + "/crew-api.js";
90
+ const SCHEMA_SQL_SRC = crewHome + "/lib/schema.sql";
91
+ const SCHEMA_SQL_PINNED = RUN_LIB + "/schema.sql";
92
+ // The five basenames the pin step must materialize — asserted mechanically
87
93
  // by workflow code from the verbatim listing, never from agent prose.
88
- const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM].map(function (p) { return p.split("/").pop(); });
94
+ const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED].map(function (p) { return p.split("/").pop(); });
89
95
 
90
96
  // Project config — passed by dispatcher, falls back to dashboard defaults
91
97
  const projectConfig = inputs.project_config || {};
@@ -231,7 +237,7 @@ function attemptKey(base, reworkCount) {
231
237
  return base + (reworkCount > 0 ? "-r" + reworkCount : "");
232
238
  }
233
239
  // pinLifecycle(key) — snapshot the lifecycle scripts into RUN_LIB and return
234
- // the verbatim `ls -1` listing so WORKFLOW CODE asserts the four pinned
240
+ // the verbatim `ls -1` listing so WORKFLOW CODE asserts the five pinned
235
241
  // basenames; the agent cannot self-certify. (The pin step was the one place
236
242
  // the workflows trusted agent prose: task 24be1cd6 walked to Publish on an
237
243
  // empty pin dir.) Byte-identical across standard/bugfix/chore — pinned by
@@ -239,7 +245,7 @@ function attemptKey(base, reworkCount) {
239
245
  function pinLifecycle(key) {
240
246
  return agent(
241
247
  "Snapshot lifecycle scripts for version pinning.\n" +
242
- "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
248
+ "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
243
249
  "Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
244
250
  { key: key, label: "Pinning lifecycle scripts",
245
251
  schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
@@ -451,13 +457,16 @@ function parseUnifiedDiff(diffText) {
451
457
  }
452
458
 
453
459
 
454
- // Publish read-back request (currently unavailable): the verbatim_request
455
- // the parent protocol (docs/publish-verification.md) would hand to an
456
- // independent read-back tool after the artifact build lands. artifact_inspect
457
- // was removed by the platform (2026-09-14); artifact.inspect is malfunction
458
- // diagnosis, not a substitute — so no agent-callable read-back tool exists
459
- // and this request cannot currently be issued. Pure function — no I/O, no
460
- // clock. The request carries the merged diff as the expected change and asks
460
+ // Publish read-back request (agent path currently unavailable): the
461
+ // verbatim_request the parent protocol (docs/publish-verification.md) would
462
+ // hand to an independent read-back tool after the artifact build lands.
463
+ // artifact_inspect was removed by the platform (2026-09-14);
464
+ // artifact.inspect is malfunction diagnosis, not a substitute — so no
465
+ // agent-callable read-back tool exists and this LLM-inspector request
466
+ // cannot currently be issued. The primary sensor is now the deterministic
467
+ // lib/readback-disk.js (reads the on-disk tree the artifact is served
468
+ // from); this request builder is retained only as the manual fallback.
469
+ // Pure function — no I/O, no clock. The request carries the merged diff as the expected change and asks
461
470
  // for an independent read of the artifact's actual source: for each file, the
462
471
  // exact current text of the changed regions plus a per-line present/absent
463
472
  // finding. Until a read-back path exists, the parent cannot independently
@@ -531,7 +540,7 @@ function extractMarkerLines(workerText) {
531
540
  var markers = [];
532
541
  for (var i = 0; i < lines.length; i++) {
533
542
  var line = lines[i].trim();
534
- if (/^(repo_diff:|release:|version_bump:|VERDICT:|TARGET_VERSION=|published:|experiential:|capture_targets:|worktree:)/i.test(line)) {
543
+ if (/^(repo_diff:|release:|version_bump:|VERDICT:|TARGET_VERSION=|published:|experiential:|layer:|capture_targets:|worktree:)/i.test(line)) {
535
544
  markers.push(line);
536
545
  }
537
546
  }
@@ -802,7 +811,7 @@ let i = startStepIndex;
802
811
  // ── Pin lifecycle scripts ────────────────────────────────────────────
803
812
  // Copy lifecycle scripts into a per-task temp dir so this run is immune
804
813
  // to upgrades that land while it's in flight. Verified mechanically:
805
- // workflow code asserts the four basenames from the verbatim listing —
814
+ // workflow code asserts the five basenames from the verbatim listing —
806
815
  // the agent cannot self-certify. Any miss parks the task before Triage.
807
816
  const initialPins = parsePinListing(await pinLifecycle("pin-lifecycle"));
808
817
  const missingInitialPins = PIN_BASENAMES.filter(function (b) { return initialPins.indexOf(b) === -1; });
@@ -810,6 +819,10 @@ if (missingInitialPins.length > 0) {
810
819
  return await parkTask("Lifecycle pin incomplete before Triage — missing " + missingInitialPins.join(", ") + " in " + RUN_LIB + ".");
811
820
  }
812
821
  log("Lifecycle scripts pinned to " + RUN_LIB);
822
+ // From here on, every crew-api.js invocation uses the pinned copy: immune
823
+ // to a release swap landing mid-flight.
824
+ CREW_API = CREW_API_PINNED;
825
+ log("Crew API pinned to " + CREW_API);
813
826
 
814
827
  // Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
815
828
  // minted at this run's first claim (never a PID — short-lived agent PIDs
@@ -1112,7 +1125,9 @@ while (i < STEPS.length) {
1112
1125
  var mapBaselineRefs = "";
1113
1126
  var mapBaselineNone = false;
1114
1127
  if (step.name === "Map") {
1115
- if ((await resolveExperiential()) === "yes") {
1128
+ // Must match Capture's run condition (experiential + artifact publish):
1129
+ // when Capture skips, no baseline notes exist, so the gate must not apply.
1130
+ if ((await resolveExperiential()) === "yes" && PUBLISH_TYPE === "artifact") {
1116
1131
  var gateStatus = await baselineStatus();
1117
1132
  if (!gateStatus.baseline_found) {
1118
1133
  log("Map gate: no baseline evidence for experiential task " + taskId + " — bouncing to Capture");
@@ -1280,6 +1295,10 @@ while (i < STEPS.length) {
1280
1295
  "TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
1281
1296
  "skipped: no-lock-held (empty-diff Integrate — nothing merged, nothing to ship)\n" +
1282
1297
  "VERDICT: PASS\n\n" +
1298
+ "If the script's output contains PUBLISH_SKIPPED=no-npm-publish, the publish was skipped gracefully: npm publish is not configured on this machine (helper or credential absent) — the merge stands, the version was not cut, nothing was shipped. Paste the marker block verbatim into your report, then end your report with exactly these three lines, in this order — lowercase, no trailing period, do not rephrase:\n" +
1299
+ "TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
1300
+ "skipped: no-npm-publish (npm publish not configured — helper or credential missing; nothing versioned or published)\n" +
1301
+ "VERDICT: PASS\n\n" +
1283
1302
  "If it exits zero, paste the script's COMPLETE marker block verbatim into your report, then end your report with exactly these three lines, in this order — lowercase, no trailing period, do not rephrase:\n" +
1284
1303
  "TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
1285
1304
  "published: muse-crew@" + publishTarget.target + "\n" +
@@ -1299,9 +1318,10 @@ while (i < STEPS.length) {
1299
1318
  // 2026-09-11), so the stamp moved to the parent — after the build
1300
1319
  // lands, the workflow records the session completed and parks with
1301
1320
  // "publish: verification-requested". The parent owns verification
1302
- // (docs/publish-verification.md); the independent read-back step is
1303
- // currently unavailable (no agent-callable read-back tool exists —
1304
- // artifact_inspect was removed by the platform 2026-09-14).
1321
+ // (docs/publish-verification.md); the primary sensor is the
1322
+ // deterministic lib/readback-disk.js (the agent-callable read-back
1323
+ // tool is unavailable — artifact_inspect was removed by the platform
1324
+ // 2026-09-14 — so the LLM-inspector path is manual-fallback only).
1305
1325
  // QA's provenance check enforces the stamp mechanically.
1306
1326
  var artifactPublish = null;
1307
1327
  var publishLockRefreshed = false;
@@ -1373,9 +1393,17 @@ while (i < STEPS.length) {
1373
1393
  } else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
1374
1394
  return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
1375
1395
  }
1396
+ // The empty tree is not a commit: git merge-base --is-ancestor fails on it.
1397
+ // The workflow knows publishBase == EMPTY_TREE_SHA (set above), so it
1398
+ // hardcodes ANCESTOR=yes for a first publish instead of asking the agent
1399
+ // to execute the conditional (clean-room 2026-09-16: the agent ran
1400
+ // merge-base on the empty tree directly and parked).
1401
+ var ancestorShell = (publishBase === EMPTY_TREE_SHA)
1402
+ ? "ANCESTOR=yes && "
1403
+ : "git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no && ";
1376
1404
  var diffResult = await agent(
1377
1405
  "Run: cd " + REPO_PATH + " && BASE='" + publishBase + "' && HEAD=$(git rev-parse HEAD) && " +
1378
- "if [ \"$BASE\" = '" + EMPTY_TREE_SHA + "' ]; then ANCESTOR=yes; else git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no; fi && " +
1406
+ ancestorShell +
1379
1407
  "echo '---COMMIT---' && echo \"$HEAD\" && echo '---BASE---' && echo \"$BASE\" && echo '---ANCESTOR---' && echo \"$ANCESTOR\" && " +
1380
1408
  "if [ \"$ANCESTOR\" = yes ]; then echo '---DIFF---' && git diff \"$BASE\" \"$HEAD\" && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r \"$BASE\" \"$HEAD\"; fi\n" +
1381
1409
  "Return JSON { \"commit\": \"<HEAD trimmed>\", \"base\": \"<BASE trimmed>\", \"ancestor\": \"<yes|no>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
@@ -1448,8 +1476,10 @@ while (i < STEPS.length) {
1448
1476
  // The builder's applied report is gone (2026-09-16): it rode on the
1449
1477
  // trigger's JSON closeout contract, which is removed below. The
1450
1478
  // parent's independent read-back (docs/publish-verification.md) is
1451
- // the verification — this field stays "missing-report" on every
1452
- // ledger line the workflow writes.
1479
+ // the verification — this field stays "missing-report" on ledger
1480
+ // lines for issued triggers; pre-trigger parks (toolcheck
1481
+ // rejected/inconclusive) and unattributed-unknown parks write null
1482
+ // (no trigger was observed, so there is nothing to report).
1453
1483
  var publishAppliedObservation = "missing-report";
1454
1484
  // Durable-evidence snapshot (2026-09-14): the observation below only
1455
1485
  // detects IN-FLIGHT builds. A build that finished before the
@@ -1460,10 +1490,13 @@ while (i < STEPS.length) {
1460
1490
  // fallback can diff before/after: a directory appearing during the
1461
1491
  // trigger window is positive evidence the edit went through and
1462
1492
  // the build completed. Best-effort and non-gating: if the snapshot
1463
- // fails, the durable check is skipped and the fallback behaves as
1464
- // before. No wall-clock in-script (deterministic replay) — the
1493
+ // fails, auditBeforeOk stays false and BOTH fallback comparisons
1494
+ // are disabled (2026-09-16, critic finding 4) — without a baseline,
1495
+ // an empty before-list would make every historical audit dir look
1496
+ // "new". No wall-clock in-script (deterministic replay) — the
1465
1497
  // comparison is a pure before/after set diff.
1466
1498
  var auditDirsBeforeTrigger = [];
1499
+ var auditBeforeOk = false;
1467
1500
  try {
1468
1501
  var auditBefore = await agent(
1469
1502
  "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
@@ -1473,9 +1506,10 @@ while (i < STEPS.length) {
1473
1506
  schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1474
1507
  );
1475
1508
  auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1509
+ auditBeforeOk = true;
1476
1510
  log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
1477
1511
  } catch (auditBeforeErr) {
1478
- log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1512
+ log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1479
1513
  }
1480
1514
  // Fire-and-forget trigger + workflow-owned observation (2026-09-16,
1481
1515
  // clean-room task e2a8d9f8): the trigger's JSON closeout contract
@@ -1499,10 +1533,16 @@ while (i < STEPS.length) {
1499
1533
  // Pre-trigger toolcheck (tiny, schema'd): the artifact namespace is
1500
1534
  // deferred for workflow children — the child self-loads it and emits
1501
1535
  // one exact signal line, read mechanically (never English prose).
1502
- // Explicit negative evidence (missing) gets one bounded retry with a
1503
- // fresh key, then parks: without the tools the edit provably did NOT
1504
- // go through, so this is the one safe retry on the publish path.
1536
+ // Only a parsed ARTIFACT_TOOLS: missing signal is explicit negative
1537
+ // evidence: it gets one bounded retry with a fresh key, then parks
1538
+ // rejected — without the tools the edit provably did NOT go through,
1539
+ // so this is the one safe retry on the publish path. A throw (or an
1540
+ // unparseable signal) is INCONCLUSIVE transport noise, never
1541
+ // evidence of missing tools (2026-09-16, critic finding 3): it is
1542
+ // recorded, it retries once in case the flake clears, but it can
1543
+ // never take the rejected path.
1505
1544
  var publishToolsOk = false;
1545
+ var publishToolsMissing = false;
1506
1546
  for (var toolcheckAttempt = 1; toolcheckAttempt <= 2 && !publishToolsOk; toolcheckAttempt++) {
1507
1547
  try {
1508
1548
  var toolcheckResult = await agent(
@@ -1514,12 +1554,29 @@ while (i < STEPS.length) {
1514
1554
  label: "Checking artifact tool availability" + (toolcheckAttempt === 1 ? "" : " (retry)"),
1515
1555
  schema: { type: "object", properties: { signal: { type: "string" } }, required: ["signal"] } }
1516
1556
  );
1517
- publishToolsOk = /ARTIFACT_TOOLS:\s*ok/.test(String((toolcheckResult && toolcheckResult.signal) || ""));
1518
- log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2): " + (publishToolsOk ? "tools ok" : "tools missing"));
1557
+ var toolSignal = String((toolcheckResult && toolcheckResult.signal) || "");
1558
+ if (/ARTIFACT_TOOLS:\s*ok/.test(toolSignal)) {
1559
+ publishToolsOk = true;
1560
+ } else if (/ARTIFACT_TOOLS:\s*missing/.test(toolSignal)) {
1561
+ publishToolsMissing = true;
1562
+ }
1563
+ log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2): " +
1564
+ (publishToolsOk ? "tools ok" : publishToolsMissing ? "tools missing (explicit parsed signal)" : "inconclusive (no ARTIFACT_TOOLS signal parsed)"));
1519
1565
  } catch (toolcheckErr) {
1520
- log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2) failed (" + (toolcheckErr && toolcheckErr.message ? toolcheckErr.message : toolcheckErr) + ") — counted as missing for this attempt");
1566
+ log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2) threw (" + (toolcheckErr && toolcheckErr.message ? toolcheckErr.message : toolcheckErr) + ") — inconclusive: a throw proves nothing about tool availability, never counted as missing");
1521
1567
  }
1522
1568
  }
1569
+ if (!publishToolsOk && !publishToolsMissing) {
1570
+ await recordPublishLedger({
1571
+ commit: mergeCommitForPublish,
1572
+ attempt: rebuildAttemptKey,
1573
+ agent_id: null,
1574
+ applied_report: null,
1575
+ outcome: "unknown",
1576
+ detail: "artifact toolcheck inconclusive after two attempts (throws or unparseable signals — never an explicit ARTIFACT_TOOLS: missing): tool availability unproven, so the trigger was NOT issued; unknown parks fail closed with no blind retry"
1577
+ }, totalReworkCount);
1578
+ return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact toolcheck was inconclusive after two attempts (no explicit ARTIFACT_TOOLS signal parsed — a throw is transport noise, not evidence). Tool availability is unproven, so no edit was issued and nothing was retried blindly. Human attention needed.");
1579
+ }
1523
1580
  if (!publishToolsOk) {
1524
1581
  await recordPublishLedger({
1525
1582
  commit: mergeCommitForPublish,
@@ -1527,9 +1584,9 @@ while (i < STEPS.length) {
1527
1584
  agent_id: null,
1528
1585
  applied_report: null,
1529
1586
  outcome: "rejected",
1530
- detail: "artifact tool namespace missing in two toolcheck attempts (explicit negative evidence): the edit provably did not go through — no trigger issued, no blind retry"
1587
+ detail: "artifact tool namespace explicitly missing (parsed ARTIFACT_TOOLS: missing signal, one bounded retry spent): the edit provably did not go through — no trigger issued, no blind retry"
1531
1588
  }, totalReworkCount);
1532
- return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was missing in two toolcheck attempts (explicit negative evidence — the edit provably did not go through, so no trigger was issued and nothing was retried blindly). Human attention needed.");
1589
+ return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was explicitly missing (parsed signal — the edit provably did not go through, so no trigger was issued and nothing was retried blindly). Human attention needed.");
1533
1590
  }
1534
1591
  // Pre-trigger build-state baseline (tiny, schema'd): one read of
1535
1592
  // artifact_status. The post-trigger observation diffs against this
@@ -1555,19 +1612,26 @@ while (i < STEPS.length) {
1555
1612
  baselineFailed = true;
1556
1613
  log("Publish pre-trigger baseline read failed for task " + taskId + " (" + (baselineErr && baselineErr.message ? baselineErr.message : baselineErr) + ") — receipt attribution skipped; durable audit-dir evidence is the only positive signal");
1557
1614
  }
1558
- // The trigger itself: fire-and-forget transport for the
1559
- // artifact_edit call. NO schema — the return value is not consumed,
1560
- // so the runtime's JSON-candidate heuristic never runs on this
1561
- // call. A transport throw is possible and inconclusive: the edit
1562
- // may still have gone through, so the outcome stays unknown until
1563
- // the observation below confirms it — never inferred from the
1564
- // throw, and never blind-retried (a blind re-trigger duplicated the
1565
- // edit on 2026-09-12).
1615
+ // The trigger itself: the artifact_edit call is AWAITED (the workflow
1616
+ // waits for it to complete) but its return value is intentionally
1617
+ // UNCONSUMED — NO schema, so no schema validation can fail this
1618
+ // call: a schema-less call resolves to the child's raw response as
1619
+ // a plain string (probed live 2026-09-16 — never parsed, never
1620
+ // throws on content). One caveat, also probed: the runtime still
1621
+ // scans the response for a JSON candidate, and an unparseable
1622
+ // {...}-looking substring in the child's prose throws ("response
1623
+ // JSON candidate", probe P6). The prompt tells the child to end its
1624
+ // turn with no prose at all, which keeps the common case clean —
1625
+ // but the channel is stochastic, so any throw is possible and
1626
+ // inconclusive: the edit may still have gone through, so the
1627
+ // outcome stays unknown until the observation below confirms it —
1628
+ // never inferred from the throw, and never blind-retried (a blind
1629
+ // re-trigger duplicated the edit on 2026-09-12).
1566
1630
  var rebuildTrigger = null;
1567
1631
  try {
1568
1632
  var triggerResultLength = String(await agent(rebuildPrompt,
1569
1633
  { key: rebuildAttemptKey, label: "Triggering artifact rebuild" }) || "").length;
1570
- log("Publish rebuild trigger for task " + taskId + " returned (" + triggerResultLength + " chars, fire-and-forget: not consumed)");
1634
+ log("Publish rebuild trigger for task " + taskId + " returned (" + triggerResultLength + " chars; awaited but return intentionally unconsumed)");
1571
1635
  } catch (triggerErr) {
1572
1636
  log("Publish rebuild trigger for task " + taskId + " threw (" + (triggerErr && triggerErr.message ? triggerErr.message : triggerErr) + ") — outcome unknown until observation confirms it; the edit may have gone through");
1573
1637
  }
@@ -1602,6 +1666,16 @@ while (i < STEPS.length) {
1602
1666
  log("Publish post-trigger build-state check failed for task " + taskId + " (" + (buildCheckErr && buildCheckErr.message ? buildCheckErr.message : buildCheckErr) + ") — this signal is unknown, not negative");
1603
1667
  }
1604
1668
  var observedAgentId = (buildState && buildState.build && typeof buildState.build.agent_id === "string" && buildState.build.agent_id) || null;
1669
+ // Known limitation (failure-mode audit 2026-09-16): attribution
1670
+ // is timing-based — any agent_id new relative to the baseline is
1671
+ // treated as this edit's receipt. A stranger's build starting inside
1672
+ // the trigger window is indistinguishable by timing and would be
1673
+ // misattributed here. The consequence is bounded: the completion
1674
+ // poll below tracks the recorded id, and the parent's mechanical
1675
+ // content read-back (docs/publish-verification.md) certifies the
1676
+ // exact commit's content — a wrong build's content fails closed as
1677
+ // verification-failed, never stamped. Timing narrows the candidate;
1678
+ // content decides.
1605
1679
  var receiptAgentId = (!buildStateFailed && !baselineFailed && observedAgentId && observedAgentId !== baselineAgentId) ? observedAgentId : null;
1606
1680
  if (receiptAgentId) {
1607
1681
  // The edit went through — a build with a new agent_id appeared
@@ -1621,6 +1695,31 @@ while (i < STEPS.length) {
1621
1695
  }, totalReworkCount);
1622
1696
  } else {
1623
1697
  var newAuditDirs = [];
1698
+ // auditReportOk: pure tri-state read of a report.json body —
1699
+ // true (build ok), false (build failed), null (missing or
1700
+ // unreadable — not evidence either way). The child returns the
1701
+ // raw body verbatim; interpretation lives here, never in prose.
1702
+ // Defined here so both the immediate and post-poll audit
1703
+ // fallbacks share it.
1704
+ var auditReportOk = function (raw) {
1705
+ if (typeof raw !== "string") return null;
1706
+ var trimmed = raw.trim();
1707
+ if (trimmed === "" || trimmed === "MISSING") return null;
1708
+ var parsed;
1709
+ try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
1710
+ if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
1711
+ return null;
1712
+ };
1713
+ // (2026-09-16, critic finding 2) When durable audit evidence
1714
+ // confirms (or refutes) the build, there is no receipt agent_id
1715
+ // to chain the completion poll to — skipReceiptPoll bypasses the
1716
+ // poll below, which with a null receipt could only observe
1717
+ // strangers or nothing.
1718
+ var skipReceiptPoll = false;
1719
+ // publishFailure is declared here (moved up from below) so the
1720
+ // immediate audit fallback can record an explicit build failure
1721
+ // without the later declaration resetting it.
1722
+ var publishFailure = null;
1624
1723
  try {
1625
1724
  var auditAfter = await agent(
1626
1725
  "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
@@ -1631,25 +1730,77 @@ while (i < STEPS.length) {
1631
1730
  );
1632
1731
  var auditDirsAfterTrigger = String((auditAfter && auditAfter.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1633
1732
  // Only timestamped build dirs count — the "latest" symlink
1634
- // and anything else are not builds.
1635
- newAuditDirs = auditDirsAfterTrigger.filter(function (d) {
1733
+ // and anything else are not builds. Gated on auditBeforeOk:
1734
+ // without a baseline every historical dir would look new.
1735
+ newAuditDirs = auditBeforeOk ? auditDirsAfterTrigger.filter(function (d) {
1636
1736
  return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
1637
- });
1737
+ }) : [];
1638
1738
  } catch (auditAfterErr) {
1639
1739
  log("Publish audit-dir re-list after trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
1640
1740
  }
1641
1741
  if (newAuditDirs.length > 0) {
1642
1742
  rebuildTrigger = { edit_started: true };
1643
1743
  rebuildAgentId = null;
1644
- log("Publish rebuild trigger for task " + taskId + ": new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and the build completed; no in-flight receipt was observed.");
1645
- await recordPublishLedger({
1646
- commit: mergeCommitForPublish,
1647
- attempt: rebuildAttemptKey,
1648
- agent_id: null,
1649
- applied_report: publishAppliedObservation,
1650
- outcome: "submitted",
1651
- detail: "fire-and-forget trigger; edit confirmed via durable audit evidence (new audit dir " + newAuditDirs[0] + "); no in-flight receipt observed"
1652
- }, totalReworkCount);
1744
+ newAuditDirs.sort();
1745
+ var newestImmediateDir = newAuditDirs[newAuditDirs.length - 1];
1746
+ log("Publish rebuild trigger for task " + taskId + ": new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and a build completed; no in-flight receipt was observed.");
1747
+ // (2026-09-16, critic finding 2) Durable audit evidence exists,
1748
+ // but there is no receipt agent_id to chain the completion poll
1749
+ // to — polling with a null receipt can only observe strangers
1750
+ // (any running build differs from "null") or nothing, burning
1751
+ // 10.5 minutes to park unknown. Read the build report now
1752
+ // instead of polling: ok=true confirms completion and routes
1753
+ // directly to parent verification (the poll is skipped);
1754
+ // ok=false is explicit failure; unreadable is unknown.
1755
+ var immediateReportOk = null;
1756
+ try {
1757
+ var immediateOkRead = await agent(
1758
+ "Read the artifact build report for slug \"" + PUBLISH_SLUG + "\".\n" +
1759
+ "Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestImmediateDir + "/report.json 2>/dev/null || echo MISSING\n" +
1760
+ "Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
1761
+ { key: attemptKey("publish-audit-ok-immediate-" + taskId, totalReworkCount), label: "Reading build report for audit-confirmed build",
1762
+ schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
1763
+ );
1764
+ immediateReportOk = auditReportOk(immediateOkRead && immediateOkRead.raw);
1765
+ } catch (immediateOkErr) {
1766
+ log("Publish build-report read for audit-confirmed dir failed for task " + taskId + " (treated as unknown): " + (immediateOkErr && immediateOkErr.message ? immediateOkErr.message : immediateOkErr));
1767
+ immediateReportOk = null;
1768
+ }
1769
+ if (immediateReportOk === true) {
1770
+ publishBuildLanded = true;
1771
+ artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
1772
+ skipReceiptPoll = true;
1773
+ log("Publish build landed for task " + taskId + " via immediate durable audit evidence (audit dir " + newestImmediateDir + ", report ok=true) — receipt poll skipped (no receipt to chain to), routing directly to parent verification");
1774
+ await recordPublishLedger({
1775
+ commit: mergeCommitForPublish,
1776
+ attempt: rebuildAttemptKey,
1777
+ agent_id: null,
1778
+ applied_report: publishAppliedObservation,
1779
+ outcome: "submitted",
1780
+ detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestImmediateDir + ", report ok=true); receipt poll skipped (no receipt agent_id), routed to parent verification"
1781
+ }, totalReworkCount);
1782
+ } else if (immediateReportOk === false) {
1783
+ skipReceiptPoll = true;
1784
+ publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestImmediateDir + ", report ok=false — immediate audit evidence, no receipt observed). Explicit negative evidence: a build ran and failed. The publish did not land — provenance was not stamped. Fail-closed.";
1785
+ await recordPublishLedger({
1786
+ commit: mergeCommitForPublish,
1787
+ attempt: rebuildAttemptKey,
1788
+ agent_id: null,
1789
+ applied_report: publishAppliedObservation,
1790
+ outcome: "failed",
1791
+ detail: "a build ran and failed: audit dir " + newestImmediateDir + " report ok=false (immediate audit evidence, no receipt)"
1792
+ }, totalReworkCount);
1793
+ } else {
1794
+ await recordPublishLedger({
1795
+ commit: mergeCommitForPublish,
1796
+ attempt: rebuildAttemptKey,
1797
+ agent_id: null,
1798
+ applied_report: null,
1799
+ outcome: "unknown",
1800
+ detail: "new audit dir " + newestImmediateDir + " appeared during the trigger window but its build report is unreadable/missing; no receipt agent_id to poll — outcome unknown, fail-closed with no blind retry"
1801
+ }, totalReworkCount);
1802
+ return await parkTask("Publish outcome unknown for task " + taskId + ": a new audit dir (" + newestImmediateDir + ") appeared during the trigger window but its build report is unreadable, and no in-flight receipt was observed to poll. The edit may have completed. Correlate the accepted edit via the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl — do NOT reissue the edit blindly: if the trigger was accepted, a retry duplicates it (2026-09-12). Verify independently whether the build completed (audit dir + report, or the parent's content read-back) before deciding the next step. Fail-closed.");
1803
+ }
1653
1804
  } else {
1654
1805
  // No attributable build and no durable evidence — but that
1655
1806
  // proves nothing (a fast-completing build can finish between
@@ -1666,7 +1817,7 @@ while (i < STEPS.length) {
1666
1817
  outcome: "unknown",
1667
1818
  detail: "fire-and-forget trigger; post-trigger build-state poll saw no attributable build (or the check failed) and the audit-dir diff found no new dir; the edit may have been accepted as pending_init"
1668
1819
  }, totalReworkCount);
1669
- return await parkTask("Publish outcome unknown for task " + taskId + ": the rebuild trigger was issued fire-and-forget (no JSON closeout for the runtime heuristic to misfire on), and the follow-up observation could not attribute a build to the edit for slug " + PUBLISH_SLUG + " — no in-flight build with a new agent_id appeared in the poll window and no new audit dir landed. The edit may have been accepted as pending_init, so no retry was issued: a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion before re-driving Publish. Fail-closed.");
1820
+ return await parkTask("Publish outcome unknown for task " + taskId + ": the rebuild trigger was issued fire-and-forget (no schema, so no validation failure mode; a candidate-parse throw stays possible and is inconclusive), and the follow-up observation could not attribute a build to the edit for slug " + PUBLISH_SLUG + " — no in-flight build with a new agent_id appeared in the poll window and no new audit dir landed. The edit may have been accepted as pending_init, so no retry was issued: a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion — do NOT reissue the edit blindly. Verify independently whether the build completed before deciding the next step. Fail-closed.");
1670
1821
  }
1671
1822
  }
1672
1823
 
@@ -1679,8 +1830,9 @@ while (i < STEPS.length) {
1679
1830
  // already recorded the ledger's submitted line on both positive paths
1680
1831
  // and parked on unknown — there is no applied report to observe and
1681
1832
  // no rejection signal to record.
1682
- var publishFailure = null;
1683
- if (rebuildTrigger.edit_started) {
1833
+ // (publishFailure is declared with the immediate audit fallback
1834
+ // above so an explicit build failure there survives to here.)
1835
+ if (rebuildTrigger.edit_started && !skipReceiptPoll) {
1684
1836
  // (2026-09-16) There is no builder report: the fire-and-forget
1685
1837
  // trigger carries no JSON contract, so there is nothing to
1686
1838
  // compare and no pre-hash diagnostic. The builder's old
@@ -1781,10 +1933,11 @@ while (i < STEPS.length) {
1781
1933
  // the old report check was circular — a fabricated report
1782
1934
  // passed by construction, and every phase went green on a hollow
1783
1935
  // build. The stamp moves to the parent (docs/publish-verification.md);
1784
- // the independent read-back step is currently unavailable (no
1785
- // agent-callable read-back tool exists — artifact_inspect was
1786
- // removed by the platform 2026-09-14), so the parent cannot
1787
- // confirm content and the task parks for verification.
1936
+ // the deterministic lib/readback-disk.js is the primary sensor
1937
+ // (the agent-callable read-back tool is unavailable —
1938
+ // artifact_inspect was removed by the platform 2026-09-14 — so
1939
+ // the LLM-inspector path is manual-fallback only), and the task
1940
+ // parks for parent verification.
1788
1941
  // QA's provenance check enforces the stamp mechanically.
1789
1942
  // An unverified publish fails loudly in QA instead of passing
1790
1943
  // silently here.
@@ -1827,26 +1980,17 @@ while (i < STEPS.length) {
1827
1980
  schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1828
1981
  );
1829
1982
  var auditDirsAfterPollList = String((auditAfterPoll && auditAfterPoll.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1830
- newAuditDirsAfterPoll = auditDirsAfterPollList.filter(function (d) {
1983
+ // Gated on auditBeforeOk (critic finding 4): without a baseline
1984
+ // every historical dir would look new.
1985
+ newAuditDirsAfterPoll = auditBeforeOk ? auditDirsAfterPollList.filter(function (d) {
1831
1986
  return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
1832
- });
1987
+ }) : [];
1833
1988
  log("Publish audit-dir re-list after build poll for task " + taskId + ": " + newAuditDirsAfterPoll.length + " new timestamped dir(s)");
1834
1989
  } catch (auditAfterPollErr) {
1835
1990
  log("Publish audit-dir re-list after build poll failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterPollErr && auditAfterPollErr.message ? auditAfterPollErr.message : auditAfterPollErr));
1836
1991
  }
1837
- // auditReportOk: pure tri-state read of a report.json body —
1838
- // true (build ok), false (build failed), null (missing or
1839
- // unreadable — not evidence either way). The child returns the
1840
- // raw body verbatim; interpretation lives here, never in prose.
1841
- var auditReportOk = function (raw) {
1842
- if (typeof raw !== "string") return null;
1843
- var trimmed = raw.trim();
1844
- if (trimmed === "" || trimmed === "MISSING") return null;
1845
- var parsed;
1846
- try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
1847
- if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
1848
- return null;
1849
- };
1992
+ // The shared auditReportOk (defined with the immediate fallback
1993
+ // above) interprets the raw body here too.
1850
1994
  var auditOkAfterPoll = null;
1851
1995
  var newestAuditDirAfterPoll = null;
1852
1996
  if (newAuditDirsAfterPoll.length > 0 && !strangerObserved) {
@@ -1907,7 +2051,7 @@ while (i < STEPS.length) {
1907
2051
  } else {
1908
2052
  // Unreachable: the observation above either attributes the edit
1909
2053
  // (edit_started) or parks. Defensive only — never a silent pass.
1910
- publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish did not land.";
2054
+ publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish is unattributed (not proven landed, not proven failed) — provenance was not stamped. Fail-closed.";
1911
2055
  }
1912
2056
  } // end: publishSkippedNoLock — no rebuild, no stamp, nothing to ship
1913
2057
  // STEP 2 (mechanical, always — skip path included): post-deploy
@@ -2406,16 +2550,24 @@ while (i < STEPS.length) {
2406
2550
  // Skip-aware (park 2026-09-11): when the deterministic publish script found
2407
2551
  // no merge lock held (empty-diff Integrate), it skips the publish path
2408
2552
  // gracefully and emits the machine-readable PUBLISH_SKIPPED=no-lock-held
2409
- // marker. Verification is then vacuous — nothing was shipped, and the
2553
+ // marker. The preflight (bugfix 2026-09-17) emits
2554
+ // PUBLISH_SKIPPED=no-npm-publish when npm publish is not configured on
2555
+ // this machine (helper or credential absent) — also before any mutation.
2556
+ // Verification is then vacuous — nothing was shipped, and the
2410
2557
  // registry must NOT have moved. The marker is script-emitted explicit state
2411
2558
  // (pasted verbatim per the Publish agent instructions), not agent prose; a
2412
2559
  // report without the marker still runs the full verification fail-closed.
2413
2560
  var publishVerified = false;
2414
2561
  var npmPublishSkipped = false;
2415
2562
  if (step.name === "Publish" && PUBLISH_TYPE === "npm" && publishTarget) {
2416
- if (/^PUBLISH_SKIPPED=no-lock-held$/m.test(workerText || "")) {
2563
+ var publishSkipMatch = /^PUBLISH_SKIPPED=(no-lock-held|no-npm-publish)$/m.exec(workerText || "");
2564
+ if (publishSkipMatch) {
2417
2565
  npmPublishSkipped = true;
2418
- log("Publish skipped for task " + taskId + " (no merge lock held — empty-diff Integrate): publish verification vacuous, nothing was shipped");
2566
+ if (publishSkipMatch[1] === "no-npm-publish") {
2567
+ log("Publish skipped for task " + taskId + " (no-npm-publish — npm publish not configured): publish verification vacuous, nothing was shipped");
2568
+ } else {
2569
+ log("Publish skipped for task " + taskId + " (no merge lock held — empty-diff Integrate): publish verification vacuous, nothing was shipped");
2570
+ }
2419
2571
  }
2420
2572
  if (!npmPublishSkipped) {
2421
2573
  try {