ruvnet-brain 4.4.0 → 4.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/README.md +3 -3
  2. package/bin/install.mjs +679 -121
  3. package/console/app.js +178 -81
  4. package/console/index.html +1 -1
  5. package/console/install-architecture.html +1 -0
  6. package/console/scope.css +4 -1
  7. package/console/style.css +13 -0
  8. package/console/tips.html +4 -4
  9. package/kb/brain-profile.mjs +1 -0
  10. package/kb/corpus-release-identity.mjs +1 -1
  11. package/kb/forge-update.mjs +41 -17
  12. package/kb/model-requirements.mjs +4 -1
  13. package/kb/update-storage-transaction.mjs +79 -0
  14. package/kb/zip-extract.mjs +22 -0
  15. package/package.json +1 -1
  16. package/plugin/.claude-plugin/plugin.json +1 -1
  17. package/plugin/.codex-plugin/plugin.json +1 -1
  18. package/plugin/commands/brain-console.md +5 -4
  19. package/plugin/commands/configure.md +5 -4
  20. package/plugin/commands/rnb-brief.md +41 -0
  21. package/plugin/commands/rnb.md +80 -0
  22. package/plugin/commands/rnbc.md +80 -0
  23. package/plugin/commands/rvbc.md +5 -4
  24. package/plugin/commands/rvcb.md +5 -4
  25. package/plugin/commands/whats-new.md +4 -4
  26. package/plugin/mcp/server.mjs +10 -2
  27. package/plugin/scripts/advocacy-route.mjs +59 -23
  28. package/plugin/scripts/anticipate.sh +4 -0
  29. package/plugin/scripts/brain-confirmation.mjs +258 -0
  30. package/plugin/scripts/brain-footprint.mjs +494 -0
  31. package/plugin/scripts/brain-location.mjs +47 -0
  32. package/plugin/scripts/capability-registry.mjs +11 -1
  33. package/plugin/scripts/continuity-brief.mjs +324 -0
  34. package/plugin/scripts/continuity-events.mjs +327 -0
  35. package/plugin/scripts/continuity-journal.mjs +500 -0
  36. package/plugin/scripts/decision-gate.mjs +56 -3
  37. package/plugin/scripts/footprint-io.mjs +186 -0
  38. package/plugin/scripts/ground-before-write.sh +8 -1
  39. package/plugin/scripts/ground-ruvnet.sh +103 -12
  40. package/plugin/scripts/grounding-answer.mjs +2 -1
  41. package/plugin/scripts/grounding-stamp.sh +3 -0
  42. package/plugin/scripts/grounding-substance.mjs +1 -1
  43. package/plugin/scripts/grounding-turn-evidence.mjs +68 -6
  44. package/plugin/scripts/hook-input.mjs +78 -4
  45. package/plugin/scripts/kb-copy-proof.mjs +148 -0
  46. package/plugin/scripts/lesson-bridge.mjs +6 -2
  47. package/plugin/scripts/nightly-controller.mjs +8 -1
  48. package/plugin/scripts/node-sqlite.mjs +41 -0
  49. package/plugin/scripts/package-cards.json +797 -0
  50. package/plugin/scripts/package-cards.rvf +0 -0
  51. package/plugin/scripts/package-cards.rvf.idmap.json +1 -0
  52. package/plugin/scripts/package-cards.rvf.meta.json +1 -0
  53. package/plugin/scripts/package-recommender-client.mjs +138 -0
  54. package/plugin/scripts/package-recommender-flag.mjs +30 -0
  55. package/plugin/scripts/package-recommender.mjs +391 -0
  56. package/plugin/scripts/project-progression-outbox.mjs +26 -8
  57. package/plugin/scripts/project-progression-reader.mjs +14 -2
  58. package/plugin/scripts/project-progression-store.mjs +153 -6
  59. package/plugin/scripts/protect-brain-state.sh +4 -1
  60. package/plugin/scripts/session-snapshot-hook.mjs +225 -41
  61. package/plugin/scripts/session-start-budget.mjs +1 -0
  62. package/plugin/scripts/session-start-core.mjs +34 -4
  63. package/plugin/scripts/session-start-health.mjs +7 -1
  64. package/plugin/scripts/session-start-update-plane.mjs +35 -0
  65. package/plugin/scripts/turn-outcome-capture.mjs +12 -1
  66. package/plugin/scripts/unprompted-runtime.mjs +2 -2
  67. package/plugin/skills/brain-console/SKILL.md +3 -3
  68. package/plugin/skills/rnbc/SKILL.md +24 -0
  69. package/plugin/skills/rvbc/SKILL.md +2 -2
  70. package/scripts/approved-runtime.mjs +2 -2
  71. package/scripts/ci/warm-brain-models.mjs +28 -0
  72. package/scripts/codex-hook-trust.mjs +94 -0
  73. package/scripts/console-instances.mjs +70 -12
  74. package/scripts/console-runtime-identity.mjs +5 -0
  75. package/scripts/corpus-canary.mjs +46 -6
  76. package/scripts/corpus-dispatch-decision.mjs +2 -2
  77. package/scripts/corpus-promotion.mjs +1 -1
  78. package/scripts/full-suite-gate.mjs +9 -2
  79. package/scripts/hook-qualify-hosts.mjs +15 -3
  80. package/scripts/host-install-matrix.mjs +63 -2
  81. package/scripts/human-approval-phrases.mjs +46 -0
  82. package/scripts/installed-brain-health.mjs +53 -0
  83. package/scripts/move-brain.mjs +310 -0
  84. package/scripts/onboarding-console.mjs +93 -10
  85. package/scripts/oracle/abstain-threshold-sweep.mjs +62 -0
  86. package/scripts/oracle/abstain-trace.mjs +139 -0
  87. package/scripts/oracle/doc2query-generate.mjs +162 -0
  88. package/scripts/oracle/doc2query-reach.mjs +110 -0
  89. package/scripts/oracle/judge-train.mjs +158 -0
  90. package/scripts/oracle/need-set-split.mjs +48 -0
  91. package/scripts/oracle/sona-query-adapter-eval.mjs +139 -0
  92. package/scripts/package-cards.mjs +374 -0
  93. package/scripts/publication-receipt.mjs +37 -9
  94. package/scripts/recommendation-e2e.mjs +110 -0
  95. package/scripts/recommendation-eval.mjs +105 -0
  96. package/scripts/recommendation-floor.mjs +56 -0
  97. package/scripts/recommendation-judge-score.mjs +74 -0
  98. package/scripts/recommendation-latency.mjs +95 -0
  99. package/scripts/recommendation-real-host-score.mjs +76 -0
  100. package/scripts/recommendation-real-host.mjs +137 -0
  101. package/scripts/release-channel-kind.mjs +1 -1
  102. package/scripts/release-environment-policy.mjs +33 -0
  103. package/scripts/single-source-check.mjs +15 -10
  104. package/scripts/sync-commands.mjs +5 -2
  105. package/scripts/wired-check.mjs +17 -2
@@ -180,6 +180,145 @@ export function rufloCwdFor(storePath, { root = rufloScratchRoot() } = {}) {
180
180
  return ensurePrivateDir(path.join(root, projectKey));
181
181
  }
182
182
 
183
+ /**
184
+ * 4.3.40 ran ruflo with cwd `<project>/.swarm`, so upgraded projects still hold ruflo's cwd artifacts
185
+ * INSIDE the store directory — including `.swarm/.swarm/hnsw.metadata.json`, a copy of snapshot content.
186
+ * Removes exactly those, and only when every path in them is one ruflo is known to write there; anything
187
+ * unexpected (or any symlink) leaves that artifact untouched and is REPORTED. The store itself
188
+ * (memory.db, -wal/-shm, schema.sql), the outbox and queue files are never candidates.
189
+ *
190
+ * The allowlist is measured, not guessed: real ruflo 3.49.0 run with cwd=<dir> and --path elsewhere
191
+ * writes .claude/{memory.db,.proven-config-version,proven-config.json} (init/store),
192
+ * .claude-flow/harness-active-policy.json and .swarm/{hnsw.index,hnsw.metadata.json} and ruvector.db
193
+ * (store), .claude-flow/policy/state.json (retrieve). The owner's real 4.3.40 projects also hold
194
+ * .swarm/.swarm/agentdb-memory.db(-wal,-shm): ruflo's AgentDB bridge opens <cwd>/.swarm/agentdb-memory.db
195
+ * (ruflo/v3/@claude-flow/cli/src/memory/memory-bridge.ts getAgentDbPath), and the native bindings drop
196
+ * ruvector.db into whatever cwd they run in (ruflo/scripts/smoke-memory-no-stray-db.mjs, ADR-125 Phase 7).
197
+ */
198
+ const SQLITE = (name) => [name, `${name}-wal`, `${name}-shm`, `${name}-journal`];
199
+ const LEGACY_CWD_ARTIFACTS = Object.freeze({
200
+ '.swarm': new Set(['hnsw.index', 'hnsw.metadata.json', ...SQLITE('agentdb-memory.db')]),
201
+ '.claude': new Set(['.proven-config-version', 'proven-config.json', ...SQLITE('memory.db')]),
202
+ '.claude-flow': new Set(['harness-active-policy.json', 'policy/', 'policy/state.json']),
203
+ 'ruvector.db': null, // a regular file
204
+ });
205
+
206
+ /** Every path under `dir`, relative, directories with a trailing slash; symlinks reported, never followed. */
207
+ function relativeEntries(dir) {
208
+ const out = [];
209
+ const walk = (current, prefix) => {
210
+ for (const name of fs.readdirSync(current)) {
211
+ const full = path.join(current, name);
212
+ const stat = fs.lstatSync(full);
213
+ const rel = prefix + name;
214
+ if (stat.isSymbolicLink()) out.push({ rel, link: true });
215
+ else if (stat.isDirectory()) { out.push({ rel: `${rel}/` }); walk(full, `${rel}/`); }
216
+ else out.push({ rel, file: stat.isFile() });
217
+ }
218
+ };
219
+ walk(dir, '');
220
+ return out;
221
+ }
222
+
223
+ /**
224
+ * With `dryRun`, reports what WOULD be removed and touches nothing (--doctor). Returns
225
+ * { removed: [paths], refused: [{ path, reason }] }; a refusal must be shown to the user, never dropped.
226
+ */
227
+ const IN_USE_MS = 10 * 60_000;
228
+ // The files that carry DATA: the database, its WAL and its rollback journal. Never -shm: it is the WAL
229
+ // index, and every reader — including this proof's own read-only open — rewrites it. Counting it made
230
+ // every real 4.3.40 store look "changed while it was being checked" (and "in use" on the next run).
231
+ const dataFiles = (dbPath) => {
232
+ const base = path.basename(dbPath);
233
+ return [base, `${base}-wal`, `${base}-journal`];
234
+ };
235
+ const sqliteFingerprint = (dbPath) => dataFiles(dbPath).map((name) => {
236
+ try { const st = fs.statSync(path.join(path.dirname(dbPath), name)); return `${name}:${st.size}:${st.mtimeMs}`; }
237
+ catch { return `${name}:absent`; }
238
+ });
239
+ const sameFiles = (a, b) => a.join('|') === b.join('|');
240
+
241
+ /**
242
+ * Is every active (namespace, key, content) row of a stray nested AgentDB present, byte-identical, in
243
+ * the canonical store? Read-only through the progression reader (node:sqlite, readOnly — no write, no
244
+ * checkpoint). Anything it cannot prove — recently written (in use), unreadable, a row missing or
245
+ * different — keeps the file and says why.
246
+ */
247
+ export function proveMirrored(nestedDb, canonicalDb, { now = Date.now(), maxRows = 100_000 } = {}) {
248
+ const fingerprint = sqliteFingerprint(nestedDb);
249
+ const newest = Math.max(...dataFiles(nestedDb).map((name) => {
250
+ try { return fs.statSync(path.join(path.dirname(nestedDb), name)).mtimeMs; } catch { return 0; }
251
+ }));
252
+ if (now - newest < IN_USE_MS) return { ok: false, reason: `kept: agentdb-memory.db was written ${Math.round((now - newest) / 1000)}s ago (in use)` };
253
+ const nested = withProgressionReader(nestedDb, (reader) => reader.allRows({ maxRows }));
254
+ if (!nested.ok) return { ok: false, reason: `kept: agentdb-memory.db could not be read to prove it is mirrored (${nested.reason})` };
255
+ const canonical = withProgressionReader(canonicalDb, (reader) => reader.allRows({ maxRows }));
256
+ if (!canonical.ok) return { ok: false, reason: `kept: memory.db could not be read to prove agentdb-memory.db is mirrored (${canonical.reason})` };
257
+ const have = new Map(canonical.value.map((row) => [`${row.namespace}\u0000${row.key}`, row.content]));
258
+ const missing = nested.value.filter((row) => row.content === null || have.get(`${row.namespace}\u0000${row.key}`) !== row.content);
259
+ if (missing.length) return { ok: false, reason: `kept: ${missing.length} of ${nested.value.length} rows not in memory.db`, missing: missing.length };
260
+ return { ok: true, rows: nested.value.length, fingerprint };
261
+ }
262
+
263
+ export function cleanLegacyRufloDebris(storeDir, { dryRun = false, now = Date.now() } = {}) {
264
+ const removed = [];
265
+ const refused = [];
266
+ for (const [name, allowed] of Object.entries(LEGACY_CWD_ARTIFACTS)) {
267
+ const entry = path.join(storeDir, name);
268
+ let stat;
269
+ try { stat = fs.lstatSync(entry); } catch { continue; } // absent: nothing to do
270
+ if (stat.isSymbolicLink()) { refused.push({ path: entry, reason: 'symbolic link' }); continue; }
271
+ if (allowed === null) {
272
+ if (!stat.isFile()) { refused.push({ path: entry, reason: 'not a regular file' }); continue; }
273
+ } else {
274
+ if (!stat.isDirectory()) { refused.push({ path: entry, reason: 'not a directory' }); continue; }
275
+ const unknown = relativeEntries(entry).filter((item) => item.link || item.file === false || !allowed.has(item.rel))
276
+ .map((item) => (item.link ? `${item.rel} (symbolic link)` : item.rel));
277
+ if (unknown.length) { refused.push({ path: entry, reason: `unexpected entries: ${unknown.join(', ')}` }); continue; }
278
+ }
279
+ // A nested AgentDB is a real store (owner's projects: 1-41 rows, still written by open 4.3.40
280
+ // sessions). It goes only when every row is PROVEN present, identically, in the canonical memory.db.
281
+ const nestedDb = path.join(entry, 'agentdb-memory.db');
282
+ const mirror = name === '.swarm' && fs.existsSync(nestedDb) ? proveMirrored(nestedDb, path.join(storeDir, 'memory.db'), { now }) : null;
283
+ if (mirror && !mirror.ok) { refused.push({ path: entry, reason: mirror.reason, kept: true }); continue; }
284
+ if (!dryRun) {
285
+ // The proof is only valid for the bytes it read: anything written since means the file is in use.
286
+ if (mirror && !sameFiles(mirror.fingerprint, sqliteFingerprint(nestedDb))) {
287
+ refused.push({ path: entry, reason: 'kept: agentdb-memory.db changed while it was being checked (in use)', kept: true });
288
+ continue;
289
+ }
290
+ fs.rmSync(entry, { recursive: true, force: true });
291
+ }
292
+ removed.push(entry);
293
+ }
294
+ return { removed, refused };
295
+ }
296
+
297
+ // What ruflo leaves in a cwd (measured, ruflo 3.49.0): `.swarm/` (hnsw.index, hnsw.metadata.json — a
298
+ // copy of every stored value), `.claude/`, `.claude-flow/`, `ruvector.db`. None of it is read back by the
299
+ // product: every call carries --path, and the store of record is that file.
300
+ const RUFLO_CWD_ARTIFACTS = Object.freeze(['.swarm', '.claude', '.claude-flow', 'ruvector.db']);
301
+ const STALE_RUN_MS = 3_600_000;
302
+
303
+ /**
304
+ * One ruflo invocation's working directory: a fresh private `run-*` directory inside the project's
305
+ * scratch dir, removed again by the caller after the call. ruflo copies every snapshot it touches into
306
+ * `<cwd>/.swarm/hnsw.metadata.json`; with one cwd per call that copy never accumulates (4.4.0 kept one
307
+ * per project that grew without bound and outlived the project). Also clears what 4.4.0 left directly
308
+ * in the project scratch dir and any `run-*` older than an hour (a call killed before its cleanup).
309
+ */
310
+ export function rufloRunDir(storePath, { root = rufloScratchRoot(), now = Date.now() } = {}) {
311
+ const projectScratch = rufloCwdFor(storePath, { root });
312
+ for (const name of fs.readdirSync(projectScratch)) {
313
+ const entry = path.join(projectScratch, name);
314
+ let stat;
315
+ try { stat = fs.lstatSync(entry); } catch { continue; }
316
+ const stale = name.startsWith('run-') && stat.isDirectory() && now - stat.mtimeMs > STALE_RUN_MS;
317
+ if (RUFLO_CWD_ARTIFACTS.includes(name) || stale) fs.rmSync(entry, { recursive: true, force: true });
318
+ }
319
+ return fs.mkdtempSync(path.join(projectScratch, 'run-'));
320
+ }
321
+
183
322
  export class ProjectProgressionStore {
184
323
  constructor({
185
324
  projectDir,
@@ -194,6 +333,9 @@ export class ProjectProgressionStore {
194
333
  } = {}) {
195
334
  if (!rufloBinary) throw new Error(RUFLO_MISSING);
196
335
  this.resolution = resolveProjectStore({ projectDir, requestedStorePath });
336
+ // Best effort: a cleanup that cannot run must never stop a capture or a restore.
337
+ try { this.legacyDebris = cleanLegacyRufloDebris(path.dirname(this.resolution.canonicalAgentDbPath)); }
338
+ catch (error) { this.legacyDebris = { removed: [], refused: [{ path: null, reason: error.message }] }; }
197
339
  this.rufloBinary = rufloBinary;
198
340
  this.runner = runner;
199
341
  this.clock = clock;
@@ -212,12 +354,17 @@ export class ProjectProgressionStore {
212
354
  }
213
355
 
214
356
  run(args) {
215
- return this.runner(this.rufloBinary, args, {
216
- cwd: rufloCwdFor(this.resolution.canonicalAgentDbPath),
217
- encoding: 'utf8',
218
- timeout: 120_000,
219
- env: { ...process.env, RUFLO_DAEMON_AUTOSTART: '0' },
220
- });
357
+ const cwd = rufloRunDir(this.resolution.canonicalAgentDbPath);
358
+ try {
359
+ return this.runner(this.rufloBinary, args, {
360
+ cwd,
361
+ encoding: 'utf8',
362
+ timeout: 120_000,
363
+ env: { ...process.env, RUFLO_DAEMON_AUTOSTART: '0' },
364
+ });
365
+ } finally {
366
+ fs.rmSync(cwd, { recursive: true, force: true });
367
+ }
221
368
  }
222
369
 
223
370
  validateSnapshot(snapshot) {
@@ -52,7 +52,10 @@ INPUT="${INPUT:0:65536}"
52
52
 
53
53
  field() { local re="\"$1\"[[:space:]]*:[[:space:]]*\"([^\"]*)\""; [[ $INPUT =~ $re ]] && printf '%s' "${BASH_REMATCH[1]}"; }
54
54
 
55
- case "$(field tool_name)" in Write|Edit|MultiEdit|NotebookEdit) ;; *) exit 0 ;; esac
55
+ # Case-insensitive (4.5): Grok sends its own spelling (`write`, `search_replace`) — a case-sensitive
56
+ # match let a Grok write past this guard silently. nocasematch is scoped to this one test.
57
+ shopt -s nocasematch
58
+ case "$(field tool_name)" in Write|Edit|MultiEdit|NotebookEdit|search_replace|multi_edit) shopt -u nocasematch ;; *) exit 0 ;; esac
56
59
 
57
60
  FILE_PATH=$(field file_path)
58
61
  [ -n "$FILE_PATH" ] || exit 0
@@ -13,6 +13,7 @@ import { buildProjectProgression } from './project-progression-producer.mjs';
13
13
  import { ProjectProgressionStore } from './project-progression-store.mjs';
14
14
  import { resolveProjectStore } from './project-store-resolver.mjs';
15
15
  import { captureTurnOutcome } from './turn-outcome-capture.mjs';
16
+ import { captureContinuityEvents, stopNotice } from './continuity-journal.mjs';
16
17
 
17
18
  /**
18
19
  * The capture boundary's whole budget. hooks.json declares 10s; this keeps the internal work well
@@ -122,6 +123,7 @@ export function runSessionSnapshotHook(projectDir, event, {
122
123
  makeStoreFactory = boundedStoreFactory,
123
124
  spawnReplay = replayOutboxDetached,
124
125
  ordered = null,
126
+ captureEvents = captureContinuityEvents,
125
127
  } = {}) {
126
128
  // The detached worker re-runs a QUEUED boundary; its session receipt was already written then.
127
129
  const metadataWritten = writeMetadata ? writeSessionSnapshot(projectDir, event) : false;
@@ -134,7 +136,14 @@ export function runSessionSnapshotHook(projectDir, event, {
134
136
  try { turn = captureTurn({ projectDir, event, payload, host }); } catch (error) {
135
137
  turn = { recorded: false, skipped: `turn capture failed: ${error.message}` };
136
138
  }
137
- const idle = { metadataWritten, progressionCaptured: false, receipt: null, turn };
139
+ // MATERIAL EVENTS (continuity-journal.mjs): commits, releases, gates, findings, decisions, lessons —
140
+ // journalled to the durable outbox with one fsync and committed by a detached drainer. Like turn
141
+ // capture it is independent of the progression lock below and costs this boundary only a read.
142
+ let continuity;
143
+ try { continuity = captureEvents({ projectDir, event, payload, host }); } catch (error) {
144
+ continuity = { recorded: 0, skipped: `continuity capture failed: ${error.message}` };
145
+ }
146
+ const idle = { metadataWritten, progressionCaptured: false, receipt: null, turn, continuity };
138
147
 
139
148
  if (hasProjectProgression(payload)) {
140
149
  if (payload.hook_event_name !== event) {
@@ -183,12 +192,12 @@ export function runSessionSnapshotHook(projectDir, event, {
183
192
  if (!handed && token && token !== ordered) releaseReplayLock(root, token);
184
193
  return { ...idle, replayed: 0, progressionCaptured: false, deferredToReplayer: Boolean(queued),
185
194
  replaySkipped: `${why}; this capture ${queued ? 'queued behind it' : 'NOT queued (queue unwritable)'}`
186
- + `${queued ? (handed ? ', handed to a detached worker' : ' (a live worker holds the lock and drains the queue)') : ''}` };
195
+ + `${queued ? (handed ? ', handed to a detached worker' : ' (the current lock holder hands the queue to a worker when it releases)') : ''}` };
187
196
  };
188
197
  if (!ordered) {
189
198
  token = takeReplayLock(root);
190
199
  if (!token) return handOff('a worker is committing older work');
191
- const queuedAhead = queuedCaptures(root).length;
200
+ const queuedAhead = queuedWork(root);
192
201
  if (queuedAhead) return handOff(`${queuedAhead} older capture(s) queued`);
193
202
  if (budgetMs < REPLAY_MIN_BUDGET_MS) {
194
203
  const pending = pendingCount();
@@ -207,6 +216,14 @@ export function runSessionSnapshotHook(projectDir, event, {
207
216
  } catch { /* the debt stays durable in the outbox; this capture is still worth attempting */ }
208
217
  }
209
218
 
219
+ // Before COMMITTING anything, re-check the lock is still ours: with three racers, a stale-lock
220
+ // put-back can leave a holder that no longer owns it. One that lost it queues itself instead.
221
+ if (!ordered && !refreshReplayLock(root, token)) {
222
+ token = null;
223
+ handedLock = true; // nothing of ours to release
224
+ return handOff('the lock was taken over before this capture committed');
225
+ }
226
+
210
227
  let produced;
211
228
  try {
212
229
  produced = produce({ resolution, payload, host, trigger: event });
@@ -235,12 +252,22 @@ export function runSessionSnapshotHook(projectDir, event, {
235
252
  metadataWritten,
236
253
  progressionCaptured: true,
237
254
  turn,
255
+ continuity,
238
256
  replayed,
239
257
  receipt: result.receipt,
240
258
  provenance: produced.provenance,
241
259
  };
242
260
  } finally {
243
- if (!ordered && !handedLock) releaseReplayLock(root, token);
261
+ // A boundary that fired while this one held the lock queued itself and could not start a worker
262
+ // (this lock was in the way). Releasing without looking stranded it until the next boundary — two
263
+ // simultaneous SessionEnds lost the second one's final state (4.4.1). So: queued work → hand THIS
264
+ // lock to a worker; and re-check after releasing, for a boundary that queued in between.
265
+ if (!ordered && !handedLock) {
266
+ if (!(queuedWork(root) && spawnReplay({ projectDir: root, token }))) {
267
+ releaseReplayLock(root, token);
268
+ if (queuedWork(root)) spawnReplay({ projectDir: root });
269
+ }
270
+ }
244
271
  }
245
272
  }
246
273
 
@@ -253,44 +280,171 @@ export const REPLAY_LOCK_STALE_MS = 120_000;
253
280
  const REPLAY_LOCK = '.progression-replay.lock';
254
281
  const QUEUE_PREFIX = '.progression-capture-queue-';
255
282
  const lockPath = (projectDir) => path.join(projectDir, '.swarm', REPLAY_LOCK);
256
- const readLock = (projectDir) => { try { return fs.readFileSync(lockPath(projectDir), 'utf8').trim(); } catch { return null; } };
257
- let queueSeq = 0;
283
+ // The lock's FIRST line is the owner token; a second `pid <n>` line names the process holding it.
284
+ const readLock = (projectDir) => { try { return fs.readFileSync(lockPath(projectDir), 'utf8').split('\n')[0].trim(); } catch { return null; } };
285
+ const CLAIM_PREFIX = '.progression-capture-claimed-';
286
+ /** After this long a stale lock is taken over even if its holder pid looks alive (pid reuse, a wedged process). */
287
+ export const REPLAY_LOCK_ABANDON_MS = 30 * 60_000;
258
288
 
259
- /** Queue one boundary's capture for the worker (0600, inside the project's own .swarm), in arrival order. */
260
- export function queueCapture({ projectDir, event, host, payload, now = Date.now() }) {
261
- try {
262
- queueSeq += 1;
263
- const name = `${QUEUE_PREFIX}${String(now).padStart(15, '0')}-${String(process.hrtime.bigint() % 1_000_000_000n).padStart(9, '0')}-${process.pid}-${queueSeq}.json`;
264
- const file = path.join(projectDir, '.swarm', name);
265
- fs.writeFileSync(file, JSON.stringify({ event, host, payload }), { flag: 'wx', mode: 0o600 });
266
- return file;
267
- } catch { return null; }
289
+ /** Is a process with this pid alive? EPERM means alive but not ours. Never throws. */
290
+ export function pidAlive(pid) {
291
+ if (!Number.isSafeInteger(pid) || pid <= 0) return false;
292
+ try { process.kill(pid, 0); return true; } catch (error) { return error?.code === 'EPERM'; }
268
293
  }
294
+ const seqOf = (name) => Number((/(\d{12})\.json$/.exec(name) || [])[1] ?? 0);
295
+ const swarmEntries = (projectDir) => { try { return fs.readdirSync(path.join(projectDir, '.swarm')); } catch { return []; } };
296
+ // 4.4.0 named queue files by wall clock: `<prefix><15-digit ms>-<hrtime>-<pid>-<n>.json`. Open 4.4.0
297
+ // sessions keep queuing in that format after the update, so the two formats coexist for a while.
298
+ const LEGACY_QUEUE = /^\d{15}-/;
299
+ const queueTail = (name) => (name.startsWith(QUEUE_PREFIX) ? name.slice(QUEUE_PREFIX.length) : name.slice(CLAIM_PREFIX.length).replace(/^\d+-[A-Za-z0-9]*-\d+-/, '')); // <pid>-<start>-<queuedAt>-
300
+ const mtimeOf = (projectDir, name) => { try { return fs.statSync(path.join(projectDir, '.swarm', name)).mtimeMs; } catch { return Infinity; } };
269
301
 
302
+ /**
303
+ * Queue one boundary's capture for the worker (0600, inside the project's own .swarm). ORDER IS THE
304
+ * ORDER OF EXCLUSIVE CREATION: the name is the next sequence number after every queued or claimed one,
305
+ * created with O_EXCL and retried on collision — never a clock, which can step backwards or wrap.
306
+ */
307
+ export function queueCapture({ projectDir, event, host, payload }) {
308
+ const body = JSON.stringify({ event, host, payload });
309
+ for (let attempt = 0; attempt < 64; attempt += 1) {
310
+ const seq = Math.max(0, ...swarmEntries(projectDir).filter((n) => n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)).map(seqOf)) + 1;
311
+ const file = path.join(projectDir, '.swarm', `${QUEUE_PREFIX}${String(seq).padStart(12, '0')}.json`);
312
+ try { fs.writeFileSync(file, body, { flag: 'wx', mode: 0o600 }); return file; } catch (error) {
313
+ if (error?.code !== 'EEXIST') return null;
314
+ }
315
+ }
316
+ return null;
317
+ }
318
+
319
+ /**
320
+ * Unclaimed queued captures, in queue order. While any 4.4.0 (timestamp-named) entry is present the
321
+ * order is creation time (mtime, then name) — by NAME every 4.4.1 sequence file would sort before every
322
+ * 4.4.0 one, replaying newer captures before older ones across the upgrade window. Once the legacy
323
+ * entries are gone the order is the sequence alone, independent of any clock.
324
+ */
270
325
  export function queuedCaptures(projectDir) {
326
+ const all = swarmEntries(projectDir).filter((n) => (n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)) && n.endsWith('.json'));
327
+ const queued = all.filter((n) => n.startsWith(QUEUE_PREFIX));
328
+ const mixed = all.some((n) => LEGACY_QUEUE.test(queueTail(n)));
329
+ const ordered = mixed
330
+ ? queued.map((n) => [mtimeOf(projectDir, n), n]).sort((a, b) => a[0] - b[0] || (a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0)).map(([, n]) => n)
331
+ : queued.sort();
332
+ return ordered.map((n) => path.join(projectDir, '.swarm', n));
333
+ }
334
+
335
+ /** All older work a boundary must wait behind: unclaimed captures plus captures a worker has claimed. */
336
+ export function queuedWork(projectDir) {
337
+ return swarmEntries(projectDir).filter((n) => (n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)) && n.endsWith('.json')).length;
338
+ }
339
+
340
+ /**
341
+ * A process's START TIME, as a filename-safe token, or null where it cannot be read (no `ps`, e.g.
342
+ * Windows). With the pid it identifies the process: a reused pid has a different start time.
343
+ */
344
+ export function processStart(pid) {
271
345
  try {
272
- return fs.readdirSync(path.join(projectDir, '.swarm')).filter((n) => n.startsWith(QUEUE_PREFIX) && n.endsWith('.json'))
273
- .sort().map((n) => path.join(projectDir, '.swarm', n));
274
- } catch { return []; }
346
+ // TZ and locale PINNED: `lstart` prints local time in the locale's format, so two workers with
347
+ // different settings would record the same live process differently and read it as pid reuse.
348
+ const r = spawnSync('ps', ['-o', 'lstart=', '-p', String(pid)], { encoding: 'utf8', timeout: 2000, windowsHide: true,
349
+ env: { ...process.env, TZ: 'UTC', LC_ALL: 'C', LANG: 'C' } });
350
+ const s = String(r.stdout || '').replace(/[^A-Za-z0-9]/g, '');
351
+ return r.status === 0 && s ? s : null;
352
+ } catch { return null; }
353
+ }
354
+ let selfStart;
355
+ const ownStart = () => (selfStart === undefined ? (selfStart = processStart(process.pid)) : selfStart);
356
+
357
+ /**
358
+ * Claim a queued capture by atomic rename; null if taken. The claim's name records pid, start time and
359
+ * the queue file's ORIGINAL mtime (its creation order, which the mixed upgrade window sorts by); the
360
+ * claim file's own mtime is then set to the claim time, from which the orphan ceiling counts.
361
+ */
362
+ function claimQueued(file) {
363
+ let queuedAt = 0;
364
+ try { queuedAt = Math.floor(fs.statSync(file).mtimeMs); } catch { return null; }
365
+ const claimed = path.join(path.dirname(file), `${CLAIM_PREFIX}${process.pid}-${ownStart() || 'na'}-${queuedAt}-${path.basename(file).slice(QUEUE_PREFIX.length)}`);
366
+ try { fs.renameSync(file, claimed); } catch { return null; }
367
+ try { const t = new Date(); fs.utimesSync(claimed, t, t); } catch { /* the ceiling then counts from queue time: earlier, never later */ }
368
+ return claimed;
369
+ }
370
+ const unclaimedName = (claimed) => path.join(path.dirname(claimed), `${QUEUE_PREFIX}${queueTail(path.basename(claimed))}`);
371
+ /** Put a claim back in the queue: rename (atomic, needs no hard links), restoring its creation order. */
372
+ const returnClaim = (claimed) => {
373
+ const queuedAt = Number(path.basename(claimed).slice(CLAIM_PREFIX.length).split('-')[2]);
374
+ const back = unclaimedName(claimed);
375
+ try { fs.renameSync(claimed, back); } catch { return false; }
376
+ if (Number.isFinite(queuedAt) && queuedAt > 0) { try { const t = new Date(queuedAt); fs.utimesSync(back, t, t); } catch { /* best effort */ } }
377
+ return true;
378
+ };
379
+
380
+ /**
381
+ * Return to the queue every claim whose worker is gone: its pid is dead, OR the pid now belongs to a
382
+ * different process (start time differs — pid reuse), OR the claim is older than REPLAY_LOCK_ABANDON_MS
383
+ * whatever the pid says (a reused pid where no start time can be read, a wedged worker). Without the
384
+ * last two a reused pid stranded a claim forever and every Stop spawned a worker that could not run it.
385
+ */
386
+ export function reclaimOrphans(projectDir, { isAlive = pidAlive, startOf = processStart, now = Date.now() } = {}) {
387
+ let n = 0;
388
+ for (const name of swarmEntries(projectDir).filter((x) => x.startsWith(CLAIM_PREFIX))) {
389
+ const [pidText, start] = name.slice(CLAIM_PREFIX.length).split('-');
390
+ const pid = Number(pidText);
391
+ const claimed = path.join(projectDir, '.swarm', name);
392
+ const abandoned = now - mtimeOf(projectDir, name) > REPLAY_LOCK_ABANDON_MS;
393
+ let gone = abandoned || !isAlive(pid);
394
+ if (!gone && start && start !== 'na') {
395
+ const current = startOf(pid);
396
+ gone = Boolean(current) && current !== start;
397
+ }
398
+ if (gone && returnClaim(claimed)) n += 1;
399
+ }
400
+ return n;
275
401
  }
276
402
 
403
+ const lockFacts = (file) => { const st = fs.statSync(file); return { content: fs.readFileSync(file, 'utf8'), mtimeMs: st.mtimeMs, ino: st.ino }; };
404
+ const sameFacts = (a, b) => a.content === b.content && a.mtimeMs === b.mtimeMs && a.ino === b.ino;
405
+ /** The pid ACTUALLY holding a lock: the `pid <n>` line a worker writes for itself, else the token's pid. */
406
+ const holderPid = (content) => {
407
+ const line = /^pid (\d+)$/m.exec(String(content));
408
+ return Number(line ? line[1] : String(content).trim().split('-')[0]);
409
+ };
410
+
277
411
  /**
278
- * Take the lock. Returns this holder's TOKEN, or null when a live holder has it. A lock not refreshed
279
- * for REPLAY_LOCK_STALE_MS belongs to a dead worker and is taken over: renamed aside first, so two
280
- * would-be successors cannot both win the exclusive create.
412
+ * Take the lock. Returns this holder's TOKEN (`<pid>-<time>-<random>`), or null.
413
+ * • Free → exclusive create.
414
+ * • Fresh (refreshed within REPLAY_LOCK_STALE_MS) → null.
415
+ * • Stale but its holder pid is ALIVE → null until REPLAY_LOCK_ABANDON_MS: a laptop asleep mid-step,
416
+ * or a long step, is not a dead worker, and taking over would put two workers on one job. The holder
417
+ * pid is the WORKER's own (it rewrites the lock on start), not the hook that spawned it and exited.
418
+ * • Otherwise taken over: the stale file is renamed aside and VERIFIED to be the very file judged
419
+ * stale (content, mtime, inode). If a successor's fresh lock was moved instead (it took over between
420
+ * our check and our rename), it is put back — never over a third lock — and we back off. One winner.
281
421
  */
282
- export function takeReplayLock(projectDir, now = Date.now()) {
422
+ export function takeReplayLock(projectDir, now = Date.now(), { isAlive = pidAlive, beforeRename = null } = {}) {
283
423
  const lock = lockPath(projectDir);
284
424
  const token = `${process.pid}-${now}-${Math.random().toString(36).slice(2, 10)}`;
285
- const create = () => { fs.writeFileSync(lock, `${token}\n`, { flag: 'wx', mode: 0o600 }); return token; };
425
+ const create = () => { fs.writeFileSync(lock, `${token}\npid ${process.pid}\n`, { flag: 'wx', mode: 0o600 }); return token; };
286
426
  try { return create(); } catch { /* held, or stale */ }
287
- try {
288
- if (now - fs.statSync(lock).mtimeMs <= REPLAY_LOCK_STALE_MS) return null;
289
- const aside = `${lock}.stale-${process.pid}-${now}`;
290
- fs.renameSync(lock, aside);
291
- fs.rmSync(aside, { force: true });
292
- return create();
293
- } catch { return null; }
427
+ let seen;
428
+ try { seen = lockFacts(lock); } catch { try { return create(); } catch { return null; } }
429
+ const age = now - seen.mtimeMs;
430
+ if (age <= REPLAY_LOCK_STALE_MS) return null;
431
+ if (age <= REPLAY_LOCK_ABANDON_MS && isAlive(holderPid(seen.content))) return null;
432
+ beforeRename?.();
433
+ const aside = `${lock}.stale-${token}`;
434
+ try { fs.renameSync(lock, aside); } catch { return null; }
435
+ let moved = null;
436
+ try { moved = lockFacts(aside); } catch { /* vanished */ }
437
+ if (!moved || !sameFacts(moved, seen)) {
438
+ // Put the successor's lock back without ever overwriting a third holder's: a hard link where the
439
+ // filesystem has them, else an exclusive copy.
440
+ try { fs.linkSync(aside, lock); } catch {
441
+ try { fs.copyFileSync(aside, lock, fs.constants.COPYFILE_EXCL); } catch { /* a third holder exists; the successor sees it lost the lock and stops */ }
442
+ }
443
+ try { fs.rmSync(aside, { force: true }); } catch { /* best effort */ }
444
+ return null;
445
+ }
446
+ try { fs.rmSync(aside, { force: true }); } catch { /* best effort */ }
447
+ try { return create(); } catch { return null; }
294
448
  }
295
449
 
296
450
  /** Heartbeat: refresh the lock's mtime if (and only if) this holder still owns it. */
@@ -299,6 +453,12 @@ export function refreshReplayLock(projectDir, token) {
299
453
  try { const t = new Date(); fs.utimesSync(lockPath(projectDir), t, t); return true; } catch { return false; }
300
454
  }
301
455
 
456
+ /** A worker that inherited the lock records ITS OWN pid on it, keeping the owner token. */
457
+ export function adoptReplayLock(projectDir, token) {
458
+ if (!token || readLock(projectDir) !== token) return false;
459
+ try { fs.writeFileSync(lockPath(projectDir), `${token}\npid ${process.pid}\n`, { mode: 0o600 }); return true; } catch { return false; }
460
+ }
461
+
302
462
  /** Release ONLY a lock this holder owns; a successor's lock is never deleted. */
303
463
  export function releaseReplayLock(projectDir, token) {
304
464
  if (!token || readLock(projectDir) !== token) return false;
@@ -323,19 +483,22 @@ export function replayOutboxDetached({ projectDir, token = null, spawnFn = spawn
323
483
  }
324
484
 
325
485
  /**
326
- * The detached worker's body, holding the lock `token`: replay the outbox, then run every queued capture
327
- * IN ORDER — each re-entering the boundary as `ordered`, so it replays before it produces — refreshing
328
- * the lock between steps and STOPPING the moment the lock is no longer its own (a successor took it
329
- * over: running on would duplicate its work). Releases only its own lock, then re-checks for captures
486
+ * The detached worker's body, holding the lock `token`: record its own pid on the lock, return orphaned
487
+ * claims to the queue, replay the outbox, then run every queued capture IN ORDER — each CLAIMED by
488
+ * atomic rename first, so no other worker can run it too, and each re-entering the boundary as
489
+ * `ordered`, so it replays before it produces. Ownership is re-checked before every step and right
490
+ * after each claim; a worker that lost the lock puts an unstarted claim back and stops. A finished
491
+ * claim is the claimer's own and is deleted. Releases only its own lock, then re-checks for captures
330
492
  * queued while it held it.
331
493
  */
332
494
  export function runOutboxReplay({ projectDir, token = process.env.RUVNET_REPLAY_LOCK_TOKEN || null, budgetMs = DETACHED_REPLAY_BUDGET_MS,
333
- makeStoreFactory = boundedStoreFactory, now = Date.now, runCapture = runSessionSnapshotHook } = {}) {
495
+ makeStoreFactory = boundedStoreFactory, now = Date.now, runCapture = runSessionSnapshotHook, onClaim = null } = {}) {
334
496
  let held = token || takeReplayLock(projectDir);
335
497
  let replayed = 0;
336
498
  for (let round = 0; held && round < 8; round += 1) {
337
499
  try {
338
- if (!refreshReplayLock(projectDir, held)) return replayed;
500
+ if (!adoptReplayLock(projectDir, held)) return replayed;
501
+ reclaimOrphans(projectDir);
339
502
  const resolution = resolveProjectStore({ projectDir });
340
503
  const store = makeStoreFactory(now() + budgetMs)({ projectDir, requestedStorePath: resolution.canonicalAgentDbPath });
341
504
  for (const snapshot of store.outbox.pendingSnapshots()) {
@@ -345,19 +508,27 @@ export function runOutboxReplay({ projectDir, token = process.env.RUVNET_REPLAY_
345
508
  }
346
509
  for (const file of queuedCaptures(projectDir)) {
347
510
  if (!refreshReplayLock(projectDir, held)) return replayed;
511
+ const claimed = claimQueued(file);
512
+ if (!claimed) continue;
513
+ onClaim?.(claimed);
514
+ if (!refreshReplayLock(projectDir, held)) {
515
+ returnClaim(claimed);
516
+ return replayed;
517
+ }
348
518
  let job = null;
349
- try { job = JSON.parse(fs.readFileSync(file, 'utf8')); } catch { /* torn: dropped below */ }
519
+ try { job = JSON.parse(fs.readFileSync(claimed, 'utf8')); } catch { /* torn: dropped below */ }
350
520
  try {
351
521
  if (job) runCapture(projectDir, job.event, { rawInput: JSON.stringify(job.payload), host: job.host,
352
522
  budgetMs, makeStoreFactory, now, ordered: held, writeMetadata: false,
353
- captureTurn: () => ({ recorded: false, skipped: 'detached replay' }) });
523
+ captureTurn: () => ({ recorded: false, skipped: 'detached replay' }),
524
+ captureEvents: () => ({ recorded: 0, skipped: 'detached replay' }) });
354
525
  } catch { /* a failed capture leaves its own snapshot durable in the outbox */ }
355
- try { fs.rmSync(file, { force: true }); } catch { /* best effort */ }
526
+ try { fs.rmSync(claimed, { force: true }); } catch { /* best effort */ }
356
527
  }
357
528
  } catch { /* the debt stays durable; the next boundary hands it on again */ } finally {
358
529
  releaseReplayLock(projectDir, held);
359
530
  }
360
- held = queuedCaptures(projectDir).length ? takeReplayLock(projectDir) : null;
531
+ held = queuedWork(projectDir) ? takeReplayLock(projectDir) : null;
361
532
  }
362
533
  return replayed;
363
534
  }
@@ -369,7 +540,20 @@ if (process.argv[1] && path.resolve(process.argv[1]).endsWith('session-snapshot-
369
540
  // independently is what let this hook write a receipt the Console then reported as missing (#85).
370
541
  const rawInput = fs.readFileSync(0, 'utf8');
371
542
  try {
372
- runSessionSnapshotHook(projectDirectory(), process.argv[2] || 'SessionEnd', { rawInput });
543
+ const result = runSessionSnapshotHook(projectDirectory(), process.argv[2] || 'SessionEnd', { rawInput });
544
+ // FAIL LOUDLY, NEVER SILENTLY — AND ONCE. When recording is stuck (events pending past STUCK_AFTER_MS,
545
+ // a quarantined conflict, a corrupt outbox line, a cap drop) Claude Code shows this systemMessage, at
546
+ // most once per session per condition (stopNotice; it used to repeat at every turn). "Not applicable"
547
+ // (no store, no ruflo) is never stuck. Codex is excluded on purpose: its Stop schema turns any `reason`
548
+ // into a BLOCK (codex-hook-adapter.mjs), and a recording problem must never hold a turn open.
549
+ const host = process.env.RUVNET_HOOK_HOST || 'claude';
550
+ const { status, journal } = result?.continuity || {};
551
+ if (host === 'claude' && status?.stuck && journal) {
552
+ let session = null;
553
+ try { session = JSON.parse(rawInput || '{}').session_id || null; } catch { /* no session: still once per 'unknown' */ }
554
+ const message = stopNotice({ journal, status, session });
555
+ if (message) process.stdout.write(JSON.stringify({ systemMessage: message }));
556
+ }
373
557
  } catch (error) {
374
558
  // ADVISORY, ALWAYS. A capture boundary fires at Stop, PreCompact and SessionEnd; one that can
375
559
  // return a non-zero status can interrupt a turn, a compaction, or a clean exit. Report and exit 0.
@@ -43,6 +43,7 @@ export const STAGE_BUDGETS_MS = {
43
43
  'signal-surface': 400, // bounded CI-signal transition poll (see session-start-signals.mjs)
44
44
  'router-nudge': 50, // one fs.existsSync + at-most-one-time write
45
45
  'knowledge-currency': 100, // refresh receipts + registration + SOURCE.json reads; no spawn, no network
46
+ footprint: 150, // name-only classification (measured ~20ms on a 1.4 GB brain) + at-most-one detach launch
46
47
  'stable-spine': 300, // seed-dispatch decision + a single detach launch
47
48
  heartbeat: 300, // update-check dispatch launch
48
49
  'ascii-drift': 300, // optional ascii->svg drift advisory, already spawnSync-timeout bounded