muse-crew 0.7.19 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -55,7 +55,10 @@ if (!inputs.crewHome) throw new Error("crewHome is required — pass the crew ho
55
55
  const crewHome = inputs.crewHome;
56
56
  // Crew API: the workflow calls the crew-owned CLI, not the dashboard.
57
57
  // The CLI implements the API.md contract against $CREW_HOME/crew-state.db.
58
- const CREW_API = crewHome + "/current/lib/crew-api.js";
58
+ const CREW_API_SRC = crewHome + "/current/lib/crew-api.js";
59
+ // Pinned at pinLifecycle: after the pin, CREW_API points into RUN_LIB so a
60
+ // mid-flight release swap cannot change the CLI under a running workflow.
61
+ let CREW_API = CREW_API_SRC;
59
62
  // Build a shell command invoking the CLI. Args are JSON-encoded and
60
63
  // single-quote-wrapped for safe shell passing. The agent runs this and
61
64
  // returns the stdout verbatim (the CLI emits JSON on stdout).
@@ -132,9 +135,12 @@ const LIFECYCLE = RUN_LIB + "/worktree-lifecycle.sh";
132
135
  const MERGE_LOCK = RUN_LIB + "/merge-lock.sh";
133
136
  const PUBLISH_NPM_SRC = crewHome + "/lib/publish-npm.sh";
134
137
  const PUBLISH_NPM = RUN_LIB + "/publish-npm.sh";
135
- // The four basenames the pin step must materialize — asserted mechanically
138
+ const CREW_API_PINNED = RUN_LIB + "/crew-api.js";
139
+ const SCHEMA_SQL_SRC = crewHome + "/lib/schema.sql";
140
+ const SCHEMA_SQL_PINNED = RUN_LIB + "/schema.sql";
141
+ // The five basenames the pin step must materialize — asserted mechanically
136
142
  // by workflow code from the verbatim listing, never from agent prose.
137
- const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM].map(function (p) { return p.split("/").pop(); });
143
+ const PIN_BASENAMES = [LIFECYCLE, MERGE_LOCK, PUBLISH_NPM, CREW_API_PINNED, SCHEMA_SQL_PINNED].map(function (p) { return p.split("/").pop(); });
138
144
 
139
145
  // Project config — passed by dispatcher, falls back to dashboard defaults
140
146
  const projectConfig = inputs.project_config || {};
@@ -280,7 +286,7 @@ function attemptKey(base, reworkCount) {
280
286
  return base + (reworkCount > 0 ? "-r" + reworkCount : "");
281
287
  }
282
288
  // pinLifecycle(key) — snapshot the lifecycle scripts into RUN_LIB and return
283
- // the verbatim `ls -1` listing so WORKFLOW CODE asserts the four pinned
289
+ // the verbatim `ls -1` listing so WORKFLOW CODE asserts the five pinned
284
290
  // basenames; the agent cannot self-certify. (The pin step was the one place
285
291
  // the workflows trusted agent prose: task 24be1cd6 walked to Publish on an
286
292
  // empty pin dir.) Byte-identical across standard/bugfix/chore — pinned by
@@ -288,7 +294,7 @@ function attemptKey(base, reworkCount) {
288
294
  function pinLifecycle(key) {
289
295
  return agent(
290
296
  "Snapshot lifecycle scripts for version pinning.\n" +
291
- "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
297
+ "Run: mkdir -p " + RUN_LIB + " && cp " + LIFECYCLE_SRC + " " + LIFECYCLE + " && cp " + MERGE_LOCK_SRC + " " + MERGE_LOCK + " && cp " + PUBLISH_NPM_SRC + " " + PUBLISH_NPM + " && cp " + CREW_API_SRC + " " + CREW_API_PINNED + " && cp " + SCHEMA_SQL_SRC + " " + SCHEMA_SQL_PINNED + " && chmod +x " + LIFECYCLE + " " + MERGE_LOCK + " " + PUBLISH_NPM + " && ls -1 " + RUN_LIB + "\n" +
292
298
  "Return the verbatim output of the ls -1 command as { \"listing\": \"<verbatim output>\" } and nothing else.",
293
299
  { key: key, label: "Pinning lifecycle scripts",
294
300
  schema: { type: "object", properties: { listing: { type: "string" } }, required: ["listing"] } }
@@ -579,7 +585,7 @@ function extractMarkerLines(workerText) {
579
585
  var markers = [];
580
586
  for (var i = 0; i < lines.length; i++) {
581
587
  var line = lines[i].trim();
582
- if (/^(repo_diff:|release:|version_bump:|VERDICT:|TARGET_VERSION=|published:|experiential:|capture_targets:|worktree:)/i.test(line)) {
588
+ if (/^(repo_diff:|release:|version_bump:|VERDICT:|TARGET_VERSION=|published:|experiential:|layer:|capture_targets:|worktree:)/i.test(line)) {
583
589
  markers.push(line);
584
590
  }
585
591
  }
@@ -854,7 +860,7 @@ await telemetryStart("chore");
854
860
  // ── Pin lifecycle scripts ────────────────────────────────────────────
855
861
  // Copy lifecycle scripts into a per-task temp dir so this run is immune
856
862
  // to upgrades that land while it's in flight. Verified mechanically:
857
- // workflow code asserts the four basenames from the verbatim listing —
863
+ // workflow code asserts the five basenames from the verbatim listing —
858
864
  // the agent cannot self-certify. Any miss parks the task before Triage.
859
865
  await telemetryEvent("pin_start");
860
866
  const initialPins = parsePinListing(await pinLifecycle("pin-lifecycle"));
@@ -864,6 +870,10 @@ if (missingInitialPins.length > 0) {
864
870
  return await parkTask("Lifecycle pin incomplete before Triage — missing " + missingInitialPins.join(", ") + " in " + RUN_LIB + ".");
865
871
  }
866
872
  log("Lifecycle scripts pinned to " + RUN_LIB);
873
+ // From here on, every crew-api.js invocation uses the pinned copy: immune
874
+ // to a release swap landing mid-flight.
875
+ CREW_API = CREW_API_PINNED;
876
+ log("Crew API pinned to " + CREW_API);
867
877
 
868
878
  // Merge-lock holder identity (bug 2fc8f52f): the opaque task+run identity
869
879
  // minted at this run's first claim (never a PID — short-lived agent PIDs
@@ -1170,7 +1180,9 @@ while (i < STEPS.length) {
1170
1180
  var mapBaselineRefs = "";
1171
1181
  var mapBaselineNone = false;
1172
1182
  if (step.name === "Map") {
1173
- if ((await resolveExperiential()) === "yes") {
1183
+ // Must match Capture's run condition (experiential + artifact publish):
1184
+ // when Capture skips, no baseline notes exist, so the gate must not apply.
1185
+ if ((await resolveExperiential()) === "yes" && PUBLISH_TYPE === "artifact") {
1174
1186
  var gateStatus = await baselineStatus();
1175
1187
  if (!gateStatus.baseline_found) {
1176
1188
  log("Map gate: no baseline evidence for experiential task " + taskId + " — bouncing to Capture");
@@ -1328,6 +1340,10 @@ while (i < STEPS.length) {
1328
1340
  "TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
1329
1341
  "skipped: no-lock-held (empty-diff Integrate — nothing merged, nothing to ship)\n" +
1330
1342
  "VERDICT: PASS\n\n" +
1343
+ "If the script's output contains PUBLISH_SKIPPED=no-npm-publish, the publish was skipped gracefully: npm publish is not configured on this machine (helper or credential absent) — the merge stands, the version was not cut, nothing was shipped. Paste the marker block verbatim into your report, then end your report with exactly these three lines, in this order — lowercase, no trailing period, do not rephrase:\n" +
1344
+ "TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
1345
+ "skipped: no-npm-publish (npm publish not configured — helper or credential missing; nothing versioned or published)\n" +
1346
+ "VERDICT: PASS\n\n" +
1331
1347
  "If it exits zero, paste the script's COMPLETE marker block verbatim into your report, then end your report with exactly these three lines, in this order — lowercase, no trailing period, do not rephrase:\n" +
1332
1348
  "TARGET_VERSION=" + publishTarget.target + " computed as " + publishTarget.base + " + " + publishTarget.scope + " → " + publishTarget.target + "\n" +
1333
1349
  "published: muse-crew@" + publishTarget.target + "\n" +
@@ -1347,9 +1363,10 @@ while (i < STEPS.length) {
1347
1363
  // 2026-09-11), so the stamp moved to the parent — after the build
1348
1364
  // lands, the workflow records the session completed and parks with
1349
1365
  // "publish: verification-requested". The parent owns verification
1350
- // (docs/publish-verification.md); the independent read-back step is
1351
- // currently unavailable (no agent-callable read-back tool exists —
1352
- // artifact_inspect was removed by the platform 2026-09-14).
1366
+ // (docs/publish-verification.md); the primary sensor is the
1367
+ // deterministic lib/readback-disk.js (the agent-callable read-back
1368
+ // tool is unavailable — artifact_inspect was removed by the platform
1369
+ // 2026-09-14 — so the LLM-inspector path is manual-fallback only).
1353
1370
  // Chore has no QA: the parent's verification is the final gate.
1354
1371
  var artifactPublish = null;
1355
1372
  var publishLockRefreshed = false;
@@ -1421,9 +1438,17 @@ while (i < STEPS.length) {
1421
1438
  } else if (!/^[0-9a-f]{40}$/.test(publishBase)) {
1422
1439
  return await parkTask("Publish base '" + publishBase + "' is not a valid commit SHA — cannot compute the publish diff. Human attention needed.");
1423
1440
  }
1441
+ // The empty tree is not a commit: git merge-base --is-ancestor fails on it.
1442
+ // The workflow knows publishBase == EMPTY_TREE_SHA (set above), so it
1443
+ // hardcodes ANCESTOR=yes for a first publish instead of asking the agent
1444
+ // to execute the conditional (clean-room 2026-09-16: the agent ran
1445
+ // merge-base on the empty tree directly and parked).
1446
+ var ancestorShell = (publishBase === EMPTY_TREE_SHA)
1447
+ ? "ANCESTOR=yes && "
1448
+ : "git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no && ";
1424
1449
  var diffResult = await agent(
1425
1450
  "Run: cd " + REPO_PATH + " && BASE='" + publishBase + "' && HEAD=$(git rev-parse HEAD) && " +
1426
- "if [ \"$BASE\" = '" + EMPTY_TREE_SHA + "' ]; then ANCESTOR=yes; else git merge-base --is-ancestor \"$BASE\" \"$HEAD\" && ANCESTOR=yes || ANCESTOR=no; fi && " +
1451
+ ancestorShell +
1427
1452
  "echo '---COMMIT---' && echo \"$HEAD\" && echo '---BASE---' && echo \"$BASE\" && echo '---ANCESTOR---' && echo \"$ANCESTOR\" && " +
1428
1453
  "if [ \"$ANCESTOR\" = yes ]; then echo '---DIFF---' && git diff \"$BASE\" \"$HEAD\" && echo '---NAMES---' && git diff-tree --no-commit-id --name-only -r \"$BASE\" \"$HEAD\"; fi\n" +
1429
1454
  "Return JSON { \"commit\": \"<HEAD trimmed>\", \"base\": \"<BASE trimmed>\", \"ancestor\": \"<yes|no>\", \"diff\": \"<raw unified diff, may be multi-line>\", \"files\": \"<newline-separated paths>\" } and nothing else.",
@@ -1496,8 +1521,10 @@ while (i < STEPS.length) {
1496
1521
  // The builder's applied report is gone (2026-09-16): it rode on the
1497
1522
  // trigger's JSON closeout contract, which is removed below. The
1498
1523
  // parent's independent read-back (docs/publish-verification.md) is
1499
- // the verification — this field stays "missing-report" on every
1500
- // ledger line the workflow writes.
1524
+ // the verification — this field stays "missing-report" on ledger
1525
+ // lines for issued triggers; pre-trigger parks (toolcheck
1526
+ // rejected/inconclusive) and unattributed-unknown parks write null
1527
+ // (no trigger was observed, so there is nothing to report).
1501
1528
  var publishAppliedObservation = "missing-report";
1502
1529
  // Durable-evidence snapshot (2026-09-14): the observation below only
1503
1530
  // detects IN-FLIGHT builds. A build that finished before the
@@ -1508,10 +1535,13 @@ while (i < STEPS.length) {
1508
1535
  // fallback can diff before/after: a directory appearing during the
1509
1536
  // trigger window is positive evidence the edit went through and
1510
1537
  // the build completed. Best-effort and non-gating: if the snapshot
1511
- // fails, the durable check is skipped and the fallback behaves as
1512
- // before. No wall-clock in-script (deterministic replay) — the
1538
+ // fails, auditBeforeOk stays false and BOTH fallback comparisons
1539
+ // are disabled (2026-09-16, critic finding 4) — without a baseline,
1540
+ // an empty before-list would make every historical audit dir look
1541
+ // "new". No wall-clock in-script (deterministic replay) — the
1513
1542
  // comparison is a pure before/after set diff.
1514
1543
  var auditDirsBeforeTrigger = [];
1544
+ var auditBeforeOk = false;
1515
1545
  try {
1516
1546
  var auditBefore = await agent(
1517
1547
  "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort snapshot, never a gate).\n" +
@@ -1521,9 +1551,10 @@ while (i < STEPS.length) {
1521
1551
  schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1522
1552
  );
1523
1553
  auditDirsBeforeTrigger = String((auditBefore && auditBefore.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1554
+ auditBeforeOk = true;
1524
1555
  log("Publish audit-dir snapshot before trigger for task " + taskId + ": " + auditDirsBeforeTrigger.length + " entries");
1525
1556
  } catch (auditBeforeErr) {
1526
- log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1557
+ log("Publish audit-dir snapshot before trigger failed for task " + taskId + " (non-fatal): audit fallback DISABLED for this attempt — without a baseline, historical dirs would look new: " + (auditBeforeErr && auditBeforeErr.message ? auditBeforeErr.message : auditBeforeErr));
1527
1558
  }
1528
1559
  // Fire-and-forget trigger + workflow-owned observation (2026-09-16,
1529
1560
  // clean-room task e2a8d9f8): the trigger's JSON closeout contract
@@ -1547,10 +1578,16 @@ while (i < STEPS.length) {
1547
1578
  // Pre-trigger toolcheck (tiny, schema'd): the artifact namespace is
1548
1579
  // deferred for workflow children — the child self-loads it and emits
1549
1580
  // one exact signal line, read mechanically (never English prose).
1550
- // Explicit negative evidence (missing) gets one bounded retry with a
1551
- // fresh key, then parks: without the tools the edit provably did NOT
1552
- // go through, so this is the one safe retry on the publish path.
1581
+ // Only a parsed ARTIFACT_TOOLS: missing signal is explicit negative
1582
+ // evidence: it gets one bounded retry with a fresh key, then parks
1583
+ // rejected — without the tools the edit provably did NOT go through,
1584
+ // so this is the one safe retry on the publish path. A throw (or an
1585
+ // unparseable signal) is INCONCLUSIVE transport noise, never
1586
+ // evidence of missing tools (2026-09-16, critic finding 3): it is
1587
+ // recorded, it retries once in case the flake clears, but it can
1588
+ // never take the rejected path.
1553
1589
  var publishToolsOk = false;
1590
+ var publishToolsMissing = false;
1554
1591
  for (var toolcheckAttempt = 1; toolcheckAttempt <= 2 && !publishToolsOk; toolcheckAttempt++) {
1555
1592
  try {
1556
1593
  var toolcheckResult = await agent(
@@ -1562,12 +1599,29 @@ while (i < STEPS.length) {
1562
1599
  label: "Checking artifact tool availability" + (toolcheckAttempt === 1 ? "" : " (retry)"),
1563
1600
  schema: { type: "object", properties: { signal: { type: "string" } }, required: ["signal"] } }
1564
1601
  );
1565
- publishToolsOk = /ARTIFACT_TOOLS:\s*ok/.test(String((toolcheckResult && toolcheckResult.signal) || ""));
1566
- log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2): " + (publishToolsOk ? "tools ok" : "tools missing"));
1602
+ var toolSignal = String((toolcheckResult && toolcheckResult.signal) || "");
1603
+ if (/ARTIFACT_TOOLS:\s*ok/.test(toolSignal)) {
1604
+ publishToolsOk = true;
1605
+ } else if (/ARTIFACT_TOOLS:\s*missing/.test(toolSignal)) {
1606
+ publishToolsMissing = true;
1607
+ }
1608
+ log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2): " +
1609
+ (publishToolsOk ? "tools ok" : publishToolsMissing ? "tools missing (explicit parsed signal)" : "inconclusive (no ARTIFACT_TOOLS signal parsed)"));
1567
1610
  } catch (toolcheckErr) {
1568
- log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2) failed (" + (toolcheckErr && toolcheckErr.message ? toolcheckErr.message : toolcheckErr) + ") — counted as missing for this attempt");
1611
+ log("Publish artifact toolcheck for task " + taskId + " (attempt " + toolcheckAttempt + " of 2) threw (" + (toolcheckErr && toolcheckErr.message ? toolcheckErr.message : toolcheckErr) + ") — inconclusive: a throw proves nothing about tool availability, never counted as missing");
1569
1612
  }
1570
1613
  }
1614
+ if (!publishToolsOk && !publishToolsMissing) {
1615
+ await recordPublishLedger({
1616
+ commit: mergeCommitForPublish,
1617
+ attempt: rebuildAttemptKey,
1618
+ agent_id: null,
1619
+ applied_report: null,
1620
+ outcome: "unknown",
1621
+ detail: "artifact toolcheck inconclusive after two attempts (throws or unparseable signals — never an explicit ARTIFACT_TOOLS: missing): tool availability unproven, so the trigger was NOT issued; unknown parks fail closed with no blind retry"
1622
+ }, reworkCount);
1623
+ return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact toolcheck was inconclusive after two attempts (no explicit ARTIFACT_TOOLS signal parsed — a throw is transport noise, not evidence). Tool availability is unproven, so no edit was issued and nothing was retried blindly. Human attention needed.");
1624
+ }
1571
1625
  if (!publishToolsOk) {
1572
1626
  await recordPublishLedger({
1573
1627
  commit: mergeCommitForPublish,
@@ -1575,9 +1629,9 @@ while (i < STEPS.length) {
1575
1629
  agent_id: null,
1576
1630
  applied_report: null,
1577
1631
  outcome: "rejected",
1578
- detail: "artifact tool namespace missing in two toolcheck attempts (explicit negative evidence): the edit provably did not go through — no trigger issued, no blind retry"
1632
+ detail: "artifact tool namespace explicitly missing (parsed ARTIFACT_TOOLS: missing signal, one bounded retry spent): the edit provably did not go through — no trigger issued, no blind retry"
1579
1633
  }, reworkCount);
1580
- return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was missing in two toolcheck attempts (explicit negative evidence — the edit provably did not go through, so no trigger was issued and nothing was retried blindly). Human attention needed.");
1634
+ return await parkTask("Publish cannot proceed for task " + taskId + ": the artifact tool namespace was explicitly missing (parsed signal — the edit provably did not go through, so no trigger was issued and nothing was retried blindly). Human attention needed.");
1581
1635
  }
1582
1636
  // Pre-trigger build-state baseline (tiny, schema'd): one read of
1583
1637
  // artifact_status. The post-trigger observation diffs against this
@@ -1603,19 +1657,26 @@ while (i < STEPS.length) {
1603
1657
  baselineFailed = true;
1604
1658
  log("Publish pre-trigger baseline read failed for task " + taskId + " (" + (baselineErr && baselineErr.message ? baselineErr.message : baselineErr) + ") — receipt attribution skipped; durable audit-dir evidence is the only positive signal");
1605
1659
  }
1606
- // The trigger itself: fire-and-forget transport for the
1607
- // artifact_edit call. NO schema — the return value is not consumed,
1608
- // so the runtime's JSON-candidate heuristic never runs on this
1609
- // call. A transport throw is possible and inconclusive: the edit
1610
- // may still have gone through, so the outcome stays unknown until
1611
- // the observation below confirms it — never inferred from the
1612
- // throw, and never blind-retried (a blind re-trigger duplicated the
1613
- // edit on 2026-09-12).
1660
+ // The trigger itself: the artifact_edit call is AWAITED (the workflow
1661
+ // waits for it to complete) but its return value is intentionally
1662
+ // UNCONSUMED — NO schema, so no schema validation can fail this
1663
+ // call: a schema-less call resolves to the child's raw response as
1664
+ // a plain string (probed live 2026-09-16 — never parsed, never
1665
+ // throws on content). One caveat, also probed: the runtime still
1666
+ // scans the response for a JSON candidate, and an unparseable
1667
+ // {...}-looking substring in the child's prose throws ("response
1668
+ // JSON candidate", probe P6). The prompt tells the child to end its
1669
+ // turn with no prose at all, which keeps the common case clean —
1670
+ // but the channel is stochastic, so any throw is possible and
1671
+ // inconclusive: the edit may still have gone through, so the
1672
+ // outcome stays unknown until the observation below confirms it —
1673
+ // never inferred from the throw, and never blind-retried (a blind
1674
+ // re-trigger duplicated the edit on 2026-09-12).
1614
1675
  var rebuildTrigger = null;
1615
1676
  try {
1616
1677
  var triggerResultLength = String(await agent(rebuildPrompt,
1617
1678
  { key: rebuildAttemptKey, label: "Triggering artifact rebuild" }) || "").length;
1618
- log("Publish rebuild trigger for task " + taskId + " returned (" + triggerResultLength + " chars, fire-and-forget: not consumed)");
1679
+ log("Publish rebuild trigger for task " + taskId + " returned (" + triggerResultLength + " chars; awaited but return intentionally unconsumed)");
1619
1680
  } catch (triggerErr) {
1620
1681
  log("Publish rebuild trigger for task " + taskId + " threw (" + (triggerErr && triggerErr.message ? triggerErr.message : triggerErr) + ") — outcome unknown until observation confirms it; the edit may have gone through");
1621
1682
  }
@@ -1650,6 +1711,16 @@ while (i < STEPS.length) {
1650
1711
  log("Publish post-trigger build-state check failed for task " + taskId + " (" + (buildCheckErr && buildCheckErr.message ? buildCheckErr.message : buildCheckErr) + ") — this signal is unknown, not negative");
1651
1712
  }
1652
1713
  var observedAgentId = (buildState && buildState.build && typeof buildState.build.agent_id === "string" && buildState.build.agent_id) || null;
1714
+ // Known limitation (failure-mode audit 2026-09-16): attribution
1715
+ // is timing-based — any agent_id new relative to the baseline is
1716
+ // treated as this edit's receipt. A stranger's build starting inside
1717
+ // the trigger window is indistinguishable by timing and would be
1718
+ // misattributed here. The consequence is bounded: the completion
1719
+ // poll below tracks the recorded id, and the parent's mechanical
1720
+ // content read-back (docs/publish-verification.md) certifies the
1721
+ // exact commit's content — a wrong build's content fails closed as
1722
+ // verification-failed, never stamped. Timing narrows the candidate;
1723
+ // content decides.
1653
1724
  var receiptAgentId = (!buildStateFailed && !baselineFailed && observedAgentId && observedAgentId !== baselineAgentId) ? observedAgentId : null;
1654
1725
  if (receiptAgentId) {
1655
1726
  // The edit went through — a build with a new agent_id appeared
@@ -1669,6 +1740,31 @@ while (i < STEPS.length) {
1669
1740
  }, reworkCount);
1670
1741
  } else {
1671
1742
  var newAuditDirs = [];
1743
+ // auditReportOk: pure tri-state read of a report.json body —
1744
+ // true (build ok), false (build failed), null (missing or
1745
+ // unreadable — not evidence either way). The child returns the
1746
+ // raw body verbatim; interpretation lives here, never in prose.
1747
+ // Defined here so both the immediate and post-poll audit
1748
+ // fallbacks share it.
1749
+ var auditReportOk = function (raw) {
1750
+ if (typeof raw !== "string") return null;
1751
+ var trimmed = raw.trim();
1752
+ if (trimmed === "" || trimmed === "MISSING") return null;
1753
+ var parsed;
1754
+ try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
1755
+ if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
1756
+ return null;
1757
+ };
1758
+ // (2026-09-16, critic finding 2) When durable audit evidence
1759
+ // confirms (or refutes) the build, there is no receipt agent_id
1760
+ // to chain the completion poll to — skipReceiptPoll bypasses the
1761
+ // poll below, which with a null receipt could only observe
1762
+ // strangers or nothing.
1763
+ var skipReceiptPoll = false;
1764
+ // publishFailure is declared here (moved up from below) so the
1765
+ // immediate audit fallback can record an explicit build failure
1766
+ // without the later declaration resetting it.
1767
+ var publishFailure = null;
1672
1768
  try {
1673
1769
  var auditAfter = await agent(
1674
1770
  "List the artifact audit directories for slug \"" + PUBLISH_SLUG + "\" (best-effort, never a gate).\n" +
@@ -1679,25 +1775,77 @@ while (i < STEPS.length) {
1679
1775
  );
1680
1776
  var auditDirsAfterTrigger = String((auditAfter && auditAfter.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1681
1777
  // Only timestamped build dirs count — the "latest" symlink
1682
- // and anything else are not builds.
1683
- newAuditDirs = auditDirsAfterTrigger.filter(function (d) {
1778
+ // and anything else are not builds. Gated on auditBeforeOk:
1779
+ // without a baseline every historical dir would look new.
1780
+ newAuditDirs = auditBeforeOk ? auditDirsAfterTrigger.filter(function (d) {
1684
1781
  return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
1685
- });
1782
+ }) : [];
1686
1783
  } catch (auditAfterErr) {
1687
1784
  log("Publish audit-dir re-list after trigger failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterErr && auditAfterErr.message ? auditAfterErr.message : auditAfterErr));
1688
1785
  }
1689
1786
  if (newAuditDirs.length > 0) {
1690
1787
  rebuildTrigger = { edit_started: true };
1691
1788
  rebuildAgentId = null;
1692
- log("Publish rebuild trigger for task " + taskId + ": new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and the build completed; no in-flight receipt was observed.");
1693
- await recordPublishLedger({
1694
- commit: mergeCommitForPublish,
1695
- attempt: rebuildAttemptKey,
1696
- agent_id: null,
1697
- applied_report: publishAppliedObservation,
1698
- outcome: "submitted",
1699
- detail: "fire-and-forget trigger; edit confirmed via durable audit evidence (new audit dir " + newAuditDirs[0] + "); no in-flight receipt observed"
1700
- }, reworkCount);
1789
+ newAuditDirs.sort();
1790
+ var newestImmediateDir = newAuditDirs[newAuditDirs.length - 1];
1791
+ log("Publish rebuild trigger for task " + taskId + ": new audit dir(s) during the trigger window (" + newAuditDirs.join(", ") + ") — the edit went through and a build completed; no in-flight receipt was observed.");
1792
+ // (2026-09-16, critic finding 2) Durable audit evidence exists,
1793
+ // but there is no receipt agent_id to chain the completion poll
1794
+ // to — polling with a null receipt can only observe strangers
1795
+ // (any running build differs from "null") or nothing, burning
1796
+ // 10.5 minutes to park unknown. Read the build report now
1797
+ // instead of polling: ok=true confirms completion and routes
1798
+ // directly to parent verification (the poll is skipped);
1799
+ // ok=false is explicit failure; unreadable is unknown.
1800
+ var immediateReportOk = null;
1801
+ try {
1802
+ var immediateOkRead = await agent(
1803
+ "Read the artifact build report for slug \"" + PUBLISH_SLUG + "\".\n" +
1804
+ "Run: cat ~/workspace/ts-spaces/" + PUBLISH_SLUG + "/audits/" + newestImmediateDir + "/report.json 2>/dev/null || echo MISSING\n" +
1805
+ "Return JSON { \"raw\": \"<verbatim file contents, or the literal string MISSING when the file does not exist>\" } and nothing else.",
1806
+ { key: attemptKey("publish-audit-ok-immediate-" + taskId, reworkCount), label: "Reading build report for audit-confirmed build",
1807
+ schema: { type: "object", properties: { raw: { type: "string" } }, required: ["raw"] } }
1808
+ );
1809
+ immediateReportOk = auditReportOk(immediateOkRead && immediateOkRead.raw);
1810
+ } catch (immediateOkErr) {
1811
+ log("Publish build-report read for audit-confirmed dir failed for task " + taskId + " (treated as unknown): " + (immediateOkErr && immediateOkErr.message ? immediateOkErr.message : immediateOkErr));
1812
+ immediateReportOk = null;
1813
+ }
1814
+ if (immediateReportOk === true) {
1815
+ publishBuildLanded = true;
1816
+ artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
1817
+ skipReceiptPoll = true;
1818
+ log("Publish build landed for task " + taskId + " via immediate durable audit evidence (audit dir " + newestImmediateDir + ", report ok=true) — receipt poll skipped (no receipt to chain to), routing directly to parent verification");
1819
+ await recordPublishLedger({
1820
+ commit: mergeCommitForPublish,
1821
+ attempt: rebuildAttemptKey,
1822
+ agent_id: null,
1823
+ applied_report: publishAppliedObservation,
1824
+ outcome: "submitted",
1825
+ detail: "durable audit evidence shows a build completed during the attempt window (audit dir " + newestImmediateDir + ", report ok=true); receipt poll skipped (no receipt agent_id), routed to parent verification"
1826
+ }, reworkCount);
1827
+ } else if (immediateReportOk === false) {
1828
+ skipReceiptPoll = true;
1829
+ publishFailure = "Artifact build FAILED for slug " + PUBLISH_SLUG + " (audit dir " + newestImmediateDir + ", report ok=false — immediate audit evidence, no receipt observed). Explicit negative evidence: a build ran and failed. The publish did not land — provenance was not stamped. Fail-closed.";
1830
+ await recordPublishLedger({
1831
+ commit: mergeCommitForPublish,
1832
+ attempt: rebuildAttemptKey,
1833
+ agent_id: null,
1834
+ applied_report: publishAppliedObservation,
1835
+ outcome: "failed",
1836
+ detail: "a build ran and failed: audit dir " + newestImmediateDir + " report ok=false (immediate audit evidence, no receipt)"
1837
+ }, reworkCount);
1838
+ } else {
1839
+ await recordPublishLedger({
1840
+ commit: mergeCommitForPublish,
1841
+ attempt: rebuildAttemptKey,
1842
+ agent_id: null,
1843
+ applied_report: null,
1844
+ outcome: "unknown",
1845
+ detail: "new audit dir " + newestImmediateDir + " appeared during the trigger window but its build report is unreadable/missing; no receipt agent_id to poll — outcome unknown, fail-closed with no blind retry"
1846
+ }, reworkCount);
1847
+ return await parkTask("Publish outcome unknown for task " + taskId + ": a new audit dir (" + newestImmediateDir + ") appeared during the trigger window but its build report is unreadable, and no in-flight receipt was observed to poll. The edit may have completed. Correlate the accepted edit via the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl — do NOT reissue the edit blindly: if the trigger was accepted, a retry duplicates it (2026-09-12). Verify independently whether the build completed (audit dir + report, or the parent's content read-back) before deciding the next step. Fail-closed.");
1848
+ }
1701
1849
  } else {
1702
1850
  // No attributable build and no durable evidence — but that
1703
1851
  // proves nothing (a fast-completing build can finish between
@@ -1714,7 +1862,7 @@ while (i < STEPS.length) {
1714
1862
  outcome: "unknown",
1715
1863
  detail: "fire-and-forget trigger; post-trigger build-state poll saw no attributable build (or the check failed) and the audit-dir diff found no new dir; the edit may have been accepted as pending_init"
1716
1864
  }, reworkCount);
1717
- return await parkTask("Publish outcome unknown for task " + taskId + ": the rebuild trigger was issued fire-and-forget (no JSON closeout for the runtime heuristic to misfire on), and the follow-up observation could not attribute a build to the edit for slug " + PUBLISH_SLUG + " — no in-flight build with a new agent_id appeared in the poll window and no new audit dir landed. The edit may have been accepted as pending_init, so no retry was issued: a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion before re-driving Publish. Fail-closed.");
1865
+ return await parkTask("Publish outcome unknown for task " + taskId + ": the rebuild trigger was issued fire-and-forget (no schema, so no validation failure mode; a candidate-parse throw stays possible and is inconclusive), and the follow-up observation could not attribute a build to the edit for slug " + PUBLISH_SLUG + " — no in-flight build with a new agent_id appeared in the poll window and no new audit dir landed. The edit may have been accepted as pending_init, so no retry was issued: a blind retry duplicated the edit on 2026-09-12. The attempt is recorded in the publish ledger at " + crewHome + "/.publish-ledger/" + PUBLISH_SLUG + ".jsonl (commit " + String(mergeCommitForPublish || "unknown").slice(0, 12) + "). Correlate the accepted edit via the ledger and the builder's eventual completion — do NOT reissue the edit blindly. Verify independently whether the build completed before deciding the next step. Fail-closed.");
1718
1866
  }
1719
1867
  }
1720
1868
 
@@ -1727,8 +1875,9 @@ while (i < STEPS.length) {
1727
1875
  // already recorded the ledger's submitted line on both positive paths
1728
1876
  // and parked on unknown — there is no applied report to observe and
1729
1877
  // no rejection signal to record.
1730
- var publishFailure = null;
1731
- if (rebuildTrigger.edit_started) {
1878
+ // (publishFailure is declared with the immediate audit fallback
1879
+ // above so an explicit build failure there survives to here.)
1880
+ if (rebuildTrigger.edit_started && !skipReceiptPoll) {
1732
1881
  // (2026-09-16) There is no builder report: the fire-and-forget
1733
1882
  // trigger carries no JSON contract, so there is nothing to
1734
1883
  // compare and no pre-hash diagnostic. The builder's old
@@ -1829,11 +1978,14 @@ while (i < STEPS.length) {
1829
1978
  // the old report check was circular — a fabricated report
1830
1979
  // passed by construction, and every phase went green on a hollow
1831
1980
  // build. The stamp moves to the parent (docs/publish-verification.md);
1832
- // the independent read-back step is currently unavailable (no
1833
- // agent-callable read-back tool exists — artifact_inspect was
1834
- // removed by the platform 2026-09-14), so the parent cannot
1835
- // confirm content and the task parks for verification.
1981
+ // the deterministic lib/readback-disk.js is the primary sensor
1982
+ // (the agent-callable read-back tool is unavailable —
1983
+ // artifact_inspect was removed by the platform 2026-09-14 — so
1984
+ // the LLM-inspector path is manual-fallback only), and the task
1985
+ // parks for parent verification.
1836
1986
  // Chore has no QA: the parent's verification is the final gate.
1987
+ // An unverified publish fails loudly in QA instead of passing
1988
+ // silently here.
1837
1989
  publishBuildLanded = true;
1838
1990
  artifactPublish = { source_commit: mergeCommitForPublish, pending_parent_verification: true };
1839
1991
  log("Publish build landed for task " + taskId + " — provenance stamp deferred to parent content verification");
@@ -1873,26 +2025,17 @@ while (i < STEPS.length) {
1873
2025
  schema: { type: "object", properties: { dirs: { type: "string" } }, required: ["dirs"] } }
1874
2026
  );
1875
2027
  var auditDirsAfterPollList = String((auditAfterPoll && auditAfterPoll.dirs) || "").split("\n").map(function (s) { return s.trim(); }).filter(function (s) { return s.length > 0; });
1876
- newAuditDirsAfterPoll = auditDirsAfterPollList.filter(function (d) {
2028
+ // Gated on auditBeforeOk (critic finding 4): without a baseline
2029
+ // every historical dir would look new.
2030
+ newAuditDirsAfterPoll = auditBeforeOk ? auditDirsAfterPollList.filter(function (d) {
1877
2031
  return auditDirsBeforeTrigger.indexOf(d) === -1 && /^20\d\d-\d\d-\d\dT\d\d-\d\d-\d\dZ-/.test(d);
1878
- });
2032
+ }) : [];
1879
2033
  log("Publish audit-dir re-list after build poll for task " + taskId + ": " + newAuditDirsAfterPoll.length + " new timestamped dir(s)");
1880
2034
  } catch (auditAfterPollErr) {
1881
2035
  log("Publish audit-dir re-list after build poll failed for task " + taskId + " (non-fatal, durable-evidence check degraded): " + (auditAfterPollErr && auditAfterPollErr.message ? auditAfterPollErr.message : auditAfterPollErr));
1882
2036
  }
1883
- // auditReportOk: pure tri-state read of a report.json body —
1884
- // true (build ok), false (build failed), null (missing or
1885
- // unreadable — not evidence either way). The child returns the
1886
- // raw body verbatim; interpretation lives here, never in prose.
1887
- var auditReportOk = function (raw) {
1888
- if (typeof raw !== "string") return null;
1889
- var trimmed = raw.trim();
1890
- if (trimmed === "" || trimmed === "MISSING") return null;
1891
- var parsed;
1892
- try { parsed = JSON.parse(trimmed); } catch (e) { return null; }
1893
- if (parsed && typeof parsed.ok === "boolean") return parsed.ok;
1894
- return null;
1895
- };
2037
+ // The shared auditReportOk (defined with the immediate fallback
2038
+ // above) interprets the raw body here too.
1896
2039
  var auditOkAfterPoll = null;
1897
2040
  var newestAuditDirAfterPoll = null;
1898
2041
  if (newAuditDirsAfterPoll.length > 0 && !strangerObserved) {
@@ -1953,7 +2096,7 @@ while (i < STEPS.length) {
1953
2096
  } else {
1954
2097
  // Unreachable: the observation above either attributes the edit
1955
2098
  // (edit_started) or parks. Defensive only — never a silent pass.
1956
- publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish did not land.";
2099
+ publishFailure = "Artifact rebuild trigger failed: the edit was not attributed to any observed build. The publish is unattributed (not proven landed, not proven failed) — provenance was not stamped. Fail-closed.";
1957
2100
  }
1958
2101
  } // end: publishSkippedNoLock — no rebuild, no stamp, nothing to ship
1959
2102
  // STEP 2 (mechanical, always — skip path included): post-deploy
@@ -2307,16 +2450,24 @@ while (i < STEPS.length) {
2307
2450
  // Skip-aware (park 2026-09-11): when the deterministic publish script found
2308
2451
  // no merge lock held (empty-diff Integrate), it skips the publish path
2309
2452
  // gracefully and emits the machine-readable PUBLISH_SKIPPED=no-lock-held
2310
- // marker. Verification is then vacuous — nothing was shipped, and the
2453
+ // marker. The preflight (bugfix 2026-09-17) emits
2454
+ // PUBLISH_SKIPPED=no-npm-publish when npm publish is not configured on
2455
+ // this machine (helper or credential absent) — also before any mutation.
2456
+ // Verification is then vacuous — nothing was shipped, and the
2311
2457
  // registry must NOT have moved. The marker is script-emitted explicit state
2312
2458
  // (pasted verbatim per the Publish agent instructions), not agent prose; a
2313
2459
  // report without the marker still runs the full verification fail-closed.
2314
2460
  var publishVerified = false;
2315
2461
  var npmPublishSkipped = false;
2316
2462
  if (step.name === "Publish" && PUBLISH_TYPE === "npm" && publishTarget) {
2317
- if (/^PUBLISH_SKIPPED=no-lock-held$/m.test(workerText || "")) {
2463
+ var publishSkipMatch = /^PUBLISH_SKIPPED=(no-lock-held|no-npm-publish)$/m.exec(workerText || "");
2464
+ if (publishSkipMatch) {
2318
2465
  npmPublishSkipped = true;
2319
- log("Publish skipped for task " + taskId + " (no merge lock held — empty-diff Integrate): publish verification vacuous, nothing was shipped");
2466
+ if (publishSkipMatch[1] === "no-npm-publish") {
2467
+ log("Publish skipped for task " + taskId + " (no-npm-publish — npm publish not configured): publish verification vacuous, nothing was shipped");
2468
+ } else {
2469
+ log("Publish skipped for task " + taskId + " (no merge lock held — empty-diff Integrate): publish verification vacuous, nothing was shipped");
2470
+ }
2320
2471
  }
2321
2472
  if (!npmPublishSkipped) {
2322
2473
  try {
@@ -53,13 +53,14 @@ var registryResult = inputs.registry;
53
53
  if (!registryResult) {
54
54
  log("WARNING: no registry in args — falling back to slow registry load");
55
55
  registryResult = await agent(
56
- "Read the following 4 workflow files using the read tool and extract the `steps` array and `reworkTarget` string from each file's `export const meta` block at the top of the file.\n\n" +
56
+ "Read the following 5 workflow files using the read tool and extract the `steps` array and `reworkTarget` string from each file's `export const meta` block at the top of the file.\n\n" +
57
57
  "Files:\n" +
58
58
  "1. " + WORKFLOW_DIR + "/standard.js\n" +
59
59
  "2. " + WORKFLOW_DIR + "/bugfix.js\n" +
60
60
  "3. " + WORKFLOW_DIR + "/chore.js\n" +
61
- "4. " + WORKFLOW_DIR + "/docs.js\n\n" +
62
- "Return an object with keys: standard, bugfix, chore, docs. Each value has { steps: [{name, identity}], reworkTarget: string }.",
61
+ "4. " + WORKFLOW_DIR + "/docs.js\n" +
62
+ "5. " + WORKFLOW_DIR + "/upgrade.js\n\n" +
63
+ "Return an object with keys: standard, bugfix, chore, docs, upgrade. Each value has { steps: [{name, identity}], reworkTarget: string }.",
63
64
  {
64
65
  key: "load-registry",
65
66
  label: "Loading workflow step registry",
@@ -69,9 +70,10 @@ if (!registryResult) {
69
70
  standard: { type: "object" },
70
71
  bugfix: { type: "object" },
71
72
  chore: { type: "object" },
72
- docs: { type: "object" }
73
+ docs: { type: "object" },
74
+ upgrade: { type: "object" }
73
75
  },
74
- required: ["standard", "bugfix", "chore", "docs"]
76
+ required: ["standard", "bugfix", "chore", "docs", "upgrade"]
75
77
  }
76
78
  }
77
79
  );
@@ -80,7 +82,7 @@ if (!registryResult) {
80
82
  // Derive lookup tables from canonical definitions
81
83
  const WORKFLOWS = {};
82
84
  const BUILD_STEPS = {};
83
- var wfNames = ["standard", "bugfix", "chore", "docs"];
85
+ var wfNames = ["standard", "bugfix", "chore", "docs", "upgrade"];
84
86
  for (var wi = 0; wi < wfNames.length; wi++) {
85
87
  var wfName = wfNames[wi];
86
88
  var wfData = registryResult[wfName];
@@ -927,6 +929,41 @@ await agent(
927
929
  { key: "ack", label: "Acknowledging poll" }
928
930
  );
929
931
 
932
+ // ── 4b. Tick release-identity log ──────────────────────────────────
933
+ // Observable proof of which release each scheduler tick ran on, for the
934
+ // self-upgrade proof: one append-only line per dispatcher run in
935
+ // $CREW_HOME/.tick-releases.jsonl: {"seq": <n>, "release": "<active release id>"}.
936
+ // seq is the non-empty line count + 1. The file's ordering is the proof,
937
+ // not timestamps — no wall-clock calls (see tests/determinism.test.js).
938
+ // Fire-and-forget: the script swallows its own errors and exits 0, the
939
+ // agent call carries no schema, and the whole call is wrapped in
940
+ // try/catch — a failed write can never fail the tick. Placement: every run
941
+ // that completes the poll (dispatching, NO_DISPATCH, partial-cap) reaches
942
+ // this point. Concurrency edge: overlapping ticks could duplicate seq
943
+ // (count+append is not atomic) but appends are atomic, so lines never
944
+ // interleave — seq is advisory, file order is the proof.
945
+ var tickReleaseScript = [
946
+ "var crewHome=process.argv[1];",
947
+ "try{",
948
+ "var fs=require(\"fs\"),path=require(\"path\");",
949
+ "var release=path.basename(path.resolve(crewHome,fs.readlinkSync(path.join(crewHome,\"current\"))));",
950
+ "var file=path.join(crewHome,\".tick-releases.jsonl\");",
951
+ "var count=0;",
952
+ "try{var lines=fs.readFileSync(file,\"utf8\").split(\"\\n\");for(var i=0;i<lines.length;i++){if(lines[i].trim()!==\"\"){count++;}}}catch(e){}",
953
+ "fs.appendFileSync(file,JSON.stringify({seq:count+1,release:release})+\"\\n\");",
954
+ "}catch(e){}",
955
+ "process.exit(0);"
956
+ ].join("");
957
+ var tickReleaseCmd = "node -e '" + tickReleaseScript + "' '" + crewHome.replace(/'/g, "'\\''") + "'";
958
+ try {
959
+ await agent(
960
+ "Record this tick's release identity.\nRun in shell and return the stdout verbatim:\n" + tickReleaseCmd,
961
+ { key: "tick-release", label: "Recording tick release identity" }
962
+ );
963
+ } catch (e) {
964
+ log("WARNING: tick release-identity log write failed (tick continues): " + e.message);
965
+ }
966
+
930
967
  var recommended = [];
931
968
  var seenClaims = {};
932
969
  results.forEach(function(r) {