ruvnet-brain 4.3.40 → 4.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +2 -2
  2. package/bin/install.mjs +179 -36
  3. package/kb/brain-profile.mjs +17 -2
  4. package/kb/forge-update.mjs +6 -2
  5. package/kb/lifecycle-evidence-retention.mjs +12 -9
  6. package/kb/refresh-run.mjs +17 -1
  7. package/kb/update-storage-transaction.mjs +33 -5
  8. package/package.json +1 -1
  9. package/plugin/.claude-plugin/plugin.json +1 -1
  10. package/plugin/.codex-plugin/plugin.json +1 -1
  11. package/plugin/hooks/codex-hooks.json +2 -2
  12. package/plugin/hooks/hooks.json +1 -1
  13. package/plugin/scripts/capability-registry.mjs +3 -3
  14. package/plugin/scripts/codex-hook-wrapper.mjs +7 -2
  15. package/plugin/scripts/design-wall.sh +1 -0
  16. package/plugin/scripts/ground-before-write.sh +1 -0
  17. package/plugin/scripts/ground-ruvnet.sh +3 -3
  18. package/plugin/scripts/grounding-answer.mjs +129 -0
  19. package/plugin/scripts/grounding-stamp.sh +32 -31
  20. package/plugin/scripts/grounding-turn-evidence.mjs +146 -5
  21. package/plugin/scripts/grounding-turn-gate.mjs +25 -6
  22. package/plugin/scripts/hook-shim.mjs +3 -0
  23. package/plugin/scripts/kling-preflight.sh +1 -0
  24. package/plugin/scripts/learn-capture.sh +1 -0
  25. package/plugin/scripts/project-progression-reader.mjs +10 -0
  26. package/plugin/scripts/project-progression-sources.mjs +16 -4
  27. package/plugin/scripts/project-progression-store.mjs +201 -6
  28. package/plugin/scripts/protect-brain-state.sh +1 -0
  29. package/plugin/scripts/route-dispatch.sh +1 -0
  30. package/plugin/scripts/session-snapshot-hook.mjs +383 -37
  31. package/plugin/scripts/session-start-health.mjs +24 -3
  32. package/plugin/scripts/session-start-update-plane.mjs +1 -1
  33. package/plugin/scripts/update-apply.mjs +22 -2
  34. package/scripts/console-instances.mjs +203 -0
  35. package/scripts/console-runtime-identity.mjs +2 -0
  36. package/scripts/corpus-canary.mjs +130 -18
  37. package/scripts/customer-seams.mjs +84 -0
  38. package/scripts/customer-state-matrix.mjs +363 -0
  39. package/scripts/full-suite-gate.mjs +162 -0
  40. package/scripts/grounding-turn-replay.mjs +11 -3
  41. package/scripts/hook-qualify-core.mjs +346 -0
  42. package/scripts/hook-qualify-hosts.mjs +115 -0
  43. package/scripts/hook-qualify.mjs +101 -0
  44. package/scripts/host-cli.mjs +115 -0
  45. package/scripts/qe/agentic-qe-4.3.mjs +0 -1
  46. package/scripts/route-gold-rank.mjs +156 -0
  47. package/scripts/route-index-memory.mjs +51 -0
  48. package/scripts/route-latency-warm.mjs +123 -0
  49. package/scripts/wired-check.mjs +17 -3
@@ -1,6 +1,8 @@
1
1
  import fs from 'node:fs';
2
2
  import path from 'node:path';
3
- import { spawnSync } from 'node:child_process';
3
+ import { spawn, spawnSync } from 'node:child_process';
4
+ import { fileURLToPath } from 'node:url';
5
+ import { ProgressionOutbox } from './project-progression-outbox.mjs';
4
6
  import {
5
7
  captureProjectTransition,
6
8
  hasProjectProgression,
@@ -19,6 +21,23 @@ import { captureTurnOutcome } from './turn-outcome-capture.mjs';
19
21
  */
20
22
  export const CAPTURE_BUDGET_MS = 8_000;
21
23
 
24
+ /**
25
+ * Replaying an interrupted session's outbox costs one `ruflo` write per pending snapshot, each ~3s
26
+ * cold (project-progression-store.mjs). Under this budget there is room for the NEW snapshot or the
27
+ * old ones, not both — and the new one is the one nothing else will ever write.
28
+ */
29
+ export const REPLAY_MIN_BUDGET_MS = 4_000;
30
+
31
+ /**
32
+ * The budget this invocation really has. The Codex wrapper hands its own kill deadline down as
33
+ * RUVNET_CODEX_BUDGET_MS (2200ms at SessionEnd, which Codex caps at 3s); planning for 8s there meant
34
+ * being SIGKILLed mid-write with nothing reported. 300ms is left for the adapter → shim → body spawns.
35
+ */
36
+ export function effectiveBudgetMs(env = process.env) {
37
+ const handed = Number(env.RUVNET_CODEX_BUDGET_MS);
38
+ return Number.isFinite(handed) && handed > 0 ? Math.max(0, Math.min(CAPTURE_BUDGET_MS, handed - 300)) : CAPTURE_BUDGET_MS;
39
+ }
40
+
22
41
  function regularOrAbsent(file) {
23
42
  try {
24
43
  const stat = fs.lstatSync(file);
@@ -96,11 +115,16 @@ export function runSessionSnapshotHook(projectDir, event, {
96
115
  host = process.env.RUVNET_HOOK_HOST || 'claude',
97
116
  captureProgression = captureProjectTransition,
98
117
  produce = buildProjectProgression,
99
- budgetMs = CAPTURE_BUDGET_MS,
118
+ budgetMs = effectiveBudgetMs(),
100
119
  now = Date.now,
101
120
  captureTurn = captureTurnOutcome,
121
+ writeMetadata = true,
122
+ makeStoreFactory = boundedStoreFactory,
123
+ spawnReplay = replayOutboxDetached,
124
+ ordered = null,
102
125
  } = {}) {
103
- const metadataWritten = writeSessionSnapshot(projectDir, event);
126
+ // The detached worker re-runs a QUEUED boundary; its session receipt was already written then.
127
+ const metadataWritten = writeMetadata ? writeSessionSnapshot(projectDir, event) : false;
104
128
  let payload;
105
129
  try { payload = rawInput ? JSON.parse(rawInput) : {}; } catch { payload = {}; }
106
130
  // TURN OUTCOMES FIRST, and independent of `.swarm`: every turn in every repository is recorded
@@ -134,51 +158,373 @@ export function runSessionSnapshotHook(projectDir, event, {
134
158
  return { ...idle, skipped: 'project has not adopted the canonical store' };
135
159
  }
136
160
 
137
- const deadlineAt = now() + budgetMs;
138
- const storeFactory = boundedStoreFactory(deadlineAt);
161
+ const root = resolution.projectRoot;
162
+ const pendingCount = () => {
163
+ try { return new ProgressionOutbox({ projectRoot: root }).pendingSnapshots().length; } catch { return 0; }
164
+ };
139
165
 
140
- // Commit anything a previously interrupted session left durable-but-uncommitted. SessionStart is
141
- // forbidden from doing this (ADR-073 §5) because replay is a write; a capture boundary already
142
- // owns a write budget, so this is where that debt is settled.
143
- let replayed = 0;
166
+ // CAUSAL ORDER. The producer links a new snapshot to the COMMITTED heads
167
+ // (project-progression-producer.mjs), so a snapshot produced while older work is uncommitted would
168
+ // not descend from it and the project would end with two unrelated heads
169
+ // (tests/acceptance/cross-host-project-resume.test.mjs). "Older work" is BOTH the outbox (captures
170
+ // interrupted after their fsync) AND the capture queue (whole boundaries waiting for a worker).
171
+ //
172
+ // THE REPLAY LOCK IS THE RIGHT TO COMMIT IN ORDER (4.4.0 re-review S-A). Every boundary takes it
173
+ // before doing anything that commits. If it cannot — a live worker holds it — or older work is
174
+ // queued, or (on a short budget) outbox debt cannot be replayed here, this boundary QUEUES itself
175
+ // behind that work and the lock passes to a DETACHED, bounded worker that drains everything in
176
+ // order. On Codex no boundary has the replay budget (Stop 3700ms effective, SessionEnd 1900ms, no
177
+ // PreCompact). Any boundary, of any budget, therefore also drains a queue a dead worker stranded.
178
+ // `ordered` is the worker's own re-entry: it already holds the lock and is draining in order.
179
+ let token = ordered;
180
+ const handOff = (why) => {
181
+ const queued = queueCapture({ projectDir: root, event, host, payload });
182
+ const handed = queued ? spawnReplay({ projectDir: root, token }) : false;
183
+ if (!handed && token && token !== ordered) releaseReplayLock(root, token);
184
+ return { ...idle, replayed: 0, progressionCaptured: false, deferredToReplayer: Boolean(queued),
185
+ replaySkipped: `${why}; this capture ${queued ? 'queued behind it' : 'NOT queued (queue unwritable)'}`
186
+ + `${queued ? (handed ? ', handed to a detached worker' : ' (the current lock holder hands the queue to a worker when it releases)') : ''}` };
187
+ };
188
+ if (!ordered) {
189
+ token = takeReplayLock(root);
190
+ if (!token) return handOff('a worker is committing older work');
191
+ const queuedAhead = queuedWork(root);
192
+ if (queuedAhead) return handOff(`${queuedAhead} older capture(s) queued`);
193
+ if (budgetMs < REPLAY_MIN_BUDGET_MS) {
194
+ const pending = pendingCount();
195
+ if (pending) return handOff(`outbox replay deferred: budget ${budgetMs}ms < ${REPLAY_MIN_BUDGET_MS}ms; ${pending} pending`);
196
+ }
197
+ }
198
+
199
+ let handedLock = false;
144
200
  try {
145
- replayed = storeFactory({ projectDir, requestedStorePath: resolution.canonicalAgentDbPath }).replay().length;
146
- } catch { /* the new capture below is still worth attempting */ }
201
+ const deadlineAt = now() + budgetMs;
202
+ const storeFactory = makeStoreFactory(deadlineAt);
203
+ let replayed = 0;
204
+ if (budgetMs >= REPLAY_MIN_BUDGET_MS) {
205
+ try {
206
+ replayed = storeFactory({ projectDir, requestedStorePath: resolution.canonicalAgentDbPath }).replay().length;
207
+ } catch { /* the debt stays durable in the outbox; this capture is still worth attempting */ }
208
+ }
209
+
210
+ // Before COMMITTING anything, re-check the lock is still ours: with three racers, a stale-lock
211
+ // put-back can leave a holder that no longer owns it. One that lost it queues itself instead.
212
+ if (!ordered && !refreshReplayLock(root, token)) {
213
+ token = null;
214
+ handedLock = true; // nothing of ours to release
215
+ return handOff('the lock was taken over before this capture committed');
216
+ }
217
+
218
+ let produced;
219
+ try {
220
+ produced = produce({ resolution, payload, host, trigger: event });
221
+ } catch (error) {
222
+ return { ...idle, replayed, skipped: `producer failed: ${error.message}` };
223
+ }
224
+ if (produced.skipped) return { ...idle, replayed, skipped: produced.skipped.reason };
225
+
226
+ let result;
227
+ try {
228
+ result = captureProgression({
229
+ host,
230
+ payload: { ...payload, hook_event_name: event, projectProgression: produced.projectProgression },
231
+ projectDir,
232
+ storeFactory,
233
+ });
234
+ } catch (error) {
235
+ // NOT LOST — DEFERRED. capture() fsyncs the snapshot to the durable outbox BEFORE it writes to
236
+ // the store, so a budget overrun leaves the evidence on disk. On a short budget nothing later in
237
+ // this process can settle it, so the lock goes straight to a detached worker.
238
+ handedLock = !ordered && budgetMs < REPLAY_MIN_BUDGET_MS && pendingCount() > 0 && spawnReplay({ projectDir: root, token });
239
+ return { ...idle, replayed, skipped: `capture deferred: ${error.message}`,
240
+ ...(handedLock ? { replaySkipped: 'deferred capture handed to a detached worker' } : {}) };
241
+ }
242
+ return {
243
+ metadataWritten,
244
+ progressionCaptured: true,
245
+ turn,
246
+ replayed,
247
+ receipt: result.receipt,
248
+ provenance: produced.provenance,
249
+ };
250
+ } finally {
251
+ // A boundary that fired while this one held the lock queued itself and could not start a worker
252
+ // (this lock was in the way). Releasing without looking stranded it until the next boundary — two
253
+ // simultaneous SessionEnds lost the second one's final state (4.4.1). So: queued work → hand THIS
254
+ // lock to a worker; and re-check after releasing, for a boundary that queued in between.
255
+ if (!ordered && !handedLock) {
256
+ if (!(queuedWork(root) && spawnReplay({ projectDir: root, token }))) {
257
+ releaseReplayLock(root, token);
258
+ if (queuedWork(root)) spawnReplay({ projectDir: root });
259
+ }
260
+ }
261
+ }
262
+ }
263
+
264
+ /**
265
+ * How long the detached worker may spend per step. The lock is refreshed between steps and goes stale
266
+ * after REPLAY_LOCK_STALE_MS, which is more than twice a step, so a live worker never looks dead.
267
+ */
268
+ export const DETACHED_REPLAY_BUDGET_MS = 45_000;
269
+ export const REPLAY_LOCK_STALE_MS = 120_000;
270
+ const REPLAY_LOCK = '.progression-replay.lock';
271
+ const QUEUE_PREFIX = '.progression-capture-queue-';
272
+ const lockPath = (projectDir) => path.join(projectDir, '.swarm', REPLAY_LOCK);
273
+ // The lock's FIRST line is the owner token; a second `pid <n>` line names the process holding it.
274
+ const readLock = (projectDir) => { try { return fs.readFileSync(lockPath(projectDir), 'utf8').split('\n')[0].trim(); } catch { return null; } };
275
+ const CLAIM_PREFIX = '.progression-capture-claimed-';
276
+ /** After this long a stale lock is taken over even if its holder pid looks alive (pid reuse, a wedged process). */
277
+ export const REPLAY_LOCK_ABANDON_MS = 30 * 60_000;
278
+
279
+ /** Is a process with this pid alive? EPERM means alive but not ours. Never throws. */
280
+ export function pidAlive(pid) {
281
+ if (!Number.isSafeInteger(pid) || pid <= 0) return false;
282
+ try { process.kill(pid, 0); return true; } catch (error) { return error?.code === 'EPERM'; }
283
+ }
284
+ const seqOf = (name) => Number((/(\d{12})\.json$/.exec(name) || [])[1] ?? 0);
285
+ const swarmEntries = (projectDir) => { try { return fs.readdirSync(path.join(projectDir, '.swarm')); } catch { return []; } };
286
+ // 4.4.0 named queue files by wall clock: `<prefix><15-digit ms>-<hrtime>-<pid>-<n>.json`. Open 4.4.0
287
+ // sessions keep queuing in that format after the update, so the two formats coexist for a while.
288
+ const LEGACY_QUEUE = /^\d{15}-/;
289
+ const queueTail = (name) => (name.startsWith(QUEUE_PREFIX) ? name.slice(QUEUE_PREFIX.length) : name.slice(CLAIM_PREFIX.length).replace(/^\d+-[A-Za-z0-9]*-\d+-/, '')); // <pid>-<start>-<queuedAt>-
290
+ const mtimeOf = (projectDir, name) => { try { return fs.statSync(path.join(projectDir, '.swarm', name)).mtimeMs; } catch { return Infinity; } };
291
+
292
+ /**
293
+ * Queue one boundary's capture for the worker (0600, inside the project's own .swarm). ORDER IS THE
294
+ * ORDER OF EXCLUSIVE CREATION: the name is the next sequence number after every queued or claimed one,
295
+ * created with O_EXCL and retried on collision — never a clock, which can step backwards or wrap.
296
+ */
297
+ export function queueCapture({ projectDir, event, host, payload }) {
298
+ const body = JSON.stringify({ event, host, payload });
299
+ for (let attempt = 0; attempt < 64; attempt += 1) {
300
+ const seq = Math.max(0, ...swarmEntries(projectDir).filter((n) => n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)).map(seqOf)) + 1;
301
+ const file = path.join(projectDir, '.swarm', `${QUEUE_PREFIX}${String(seq).padStart(12, '0')}.json`);
302
+ try { fs.writeFileSync(file, body, { flag: 'wx', mode: 0o600 }); return file; } catch (error) {
303
+ if (error?.code !== 'EEXIST') return null;
304
+ }
305
+ }
306
+ return null;
307
+ }
147
308
 
148
- let produced;
309
+ /**
310
+ * Unclaimed queued captures, in queue order. While any 4.4.0 (timestamp-named) entry is present the
311
+ * order is creation time (mtime, then name) — by NAME every 4.4.1 sequence file would sort before every
312
+ * 4.4.0 one, replaying newer captures before older ones across the upgrade window. Once the legacy
313
+ * entries are gone the order is the sequence alone, independent of any clock.
314
+ */
315
+ export function queuedCaptures(projectDir) {
316
+ const all = swarmEntries(projectDir).filter((n) => (n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)) && n.endsWith('.json'));
317
+ const queued = all.filter((n) => n.startsWith(QUEUE_PREFIX));
318
+ const mixed = all.some((n) => LEGACY_QUEUE.test(queueTail(n)));
319
+ const ordered = mixed
320
+ ? queued.map((n) => [mtimeOf(projectDir, n), n]).sort((a, b) => a[0] - b[0] || (a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0)).map(([, n]) => n)
321
+ : queued.sort();
322
+ return ordered.map((n) => path.join(projectDir, '.swarm', n));
323
+ }
324
+
325
+ /** All older work a boundary must wait behind: unclaimed captures plus captures a worker has claimed. */
326
+ export function queuedWork(projectDir) {
327
+ return swarmEntries(projectDir).filter((n) => (n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)) && n.endsWith('.json')).length;
328
+ }
329
+
330
+ /**
331
+ * A process's START TIME, as a filename-safe token, or null where it cannot be read (no `ps`, e.g.
332
+ * Windows). With the pid it identifies the process: a reused pid has a different start time.
333
+ */
334
+ export function processStart(pid) {
149
335
  try {
150
- produced = produce({ resolution, payload, host, trigger: event });
151
- } catch (error) {
152
- return { ...idle, replayed, skipped: `producer failed: ${error.message}` };
336
+ // TZ and locale PINNED: `lstart` prints local time in the locale's format, so two workers with
337
+ // different settings would record the same live process differently and read it as pid reuse.
338
+ const r = spawnSync('ps', ['-o', 'lstart=', '-p', String(pid)], { encoding: 'utf8', timeout: 2000, windowsHide: true,
339
+ env: { ...process.env, TZ: 'UTC', LC_ALL: 'C', LANG: 'C' } });
340
+ const s = String(r.stdout || '').replace(/[^A-Za-z0-9]/g, '');
341
+ return r.status === 0 && s ? s : null;
342
+ } catch { return null; }
343
+ }
344
+ let selfStart;
345
+ const ownStart = () => (selfStart === undefined ? (selfStart = processStart(process.pid)) : selfStart);
346
+
347
+ /**
348
+ * Claim a queued capture by atomic rename; null if taken. The claim's name records pid, start time and
349
+ * the queue file's ORIGINAL mtime (its creation order, which the mixed upgrade window sorts by); the
350
+ * claim file's own mtime is then set to the claim time, from which the orphan ceiling counts.
351
+ */
352
+ function claimQueued(file) {
353
+ let queuedAt = 0;
354
+ try { queuedAt = Math.floor(fs.statSync(file).mtimeMs); } catch { return null; }
355
+ const claimed = path.join(path.dirname(file), `${CLAIM_PREFIX}${process.pid}-${ownStart() || 'na'}-${queuedAt}-${path.basename(file).slice(QUEUE_PREFIX.length)}`);
356
+ try { fs.renameSync(file, claimed); } catch { return null; }
357
+ try { const t = new Date(); fs.utimesSync(claimed, t, t); } catch { /* the ceiling then counts from queue time: earlier, never later */ }
358
+ return claimed;
359
+ }
360
+ const unclaimedName = (claimed) => path.join(path.dirname(claimed), `${QUEUE_PREFIX}${queueTail(path.basename(claimed))}`);
361
+ /** Put a claim back in the queue: rename (atomic, needs no hard links), restoring its creation order. */
362
+ const returnClaim = (claimed) => {
363
+ const queuedAt = Number(path.basename(claimed).slice(CLAIM_PREFIX.length).split('-')[2]);
364
+ const back = unclaimedName(claimed);
365
+ try { fs.renameSync(claimed, back); } catch { return false; }
366
+ if (Number.isFinite(queuedAt) && queuedAt > 0) { try { const t = new Date(queuedAt); fs.utimesSync(back, t, t); } catch { /* best effort */ } }
367
+ return true;
368
+ };
369
+
370
+ /**
371
+ * Return to the queue every claim whose worker is gone: its pid is dead, OR the pid now belongs to a
372
+ * different process (start time differs — pid reuse), OR the claim is older than REPLAY_LOCK_ABANDON_MS
373
+ * whatever the pid says (a reused pid where no start time can be read, a wedged worker). Without the
374
+ * last two a reused pid stranded a claim forever and every Stop spawned a worker that could not run it.
375
+ */
376
+ export function reclaimOrphans(projectDir, { isAlive = pidAlive, startOf = processStart, now = Date.now() } = {}) {
377
+ let n = 0;
378
+ for (const name of swarmEntries(projectDir).filter((x) => x.startsWith(CLAIM_PREFIX))) {
379
+ const [pidText, start] = name.slice(CLAIM_PREFIX.length).split('-');
380
+ const pid = Number(pidText);
381
+ const claimed = path.join(projectDir, '.swarm', name);
382
+ const abandoned = now - mtimeOf(projectDir, name) > REPLAY_LOCK_ABANDON_MS;
383
+ let gone = abandoned || !isAlive(pid);
384
+ if (!gone && start && start !== 'na') {
385
+ const current = startOf(pid);
386
+ gone = Boolean(current) && current !== start;
387
+ }
388
+ if (gone && returnClaim(claimed)) n += 1;
153
389
  }
154
- if (produced.skipped) return { ...idle, replayed, skipped: produced.skipped.reason };
390
+ return n;
391
+ }
392
+
393
+ const lockFacts = (file) => { const st = fs.statSync(file); return { content: fs.readFileSync(file, 'utf8'), mtimeMs: st.mtimeMs, ino: st.ino }; };
394
+ const sameFacts = (a, b) => a.content === b.content && a.mtimeMs === b.mtimeMs && a.ino === b.ino;
395
+ /** The pid ACTUALLY holding a lock: the `pid <n>` line a worker writes for itself, else the token's pid. */
396
+ const holderPid = (content) => {
397
+ const line = /^pid (\d+)$/m.exec(String(content));
398
+ return Number(line ? line[1] : String(content).trim().split('-')[0]);
399
+ };
155
400
 
156
- let result;
401
+ /**
402
+ * Take the lock. Returns this holder's TOKEN (`<pid>-<time>-<random>`), or null.
403
+ * • Free → exclusive create.
404
+ * • Fresh (refreshed within REPLAY_LOCK_STALE_MS) → null.
405
+ * • Stale but its holder pid is ALIVE → null until REPLAY_LOCK_ABANDON_MS: a laptop asleep mid-step,
406
+ * or a long step, is not a dead worker, and taking over would put two workers on one job. The holder
407
+ * pid is the WORKER's own (it rewrites the lock on start), not the hook that spawned it and exited.
408
+ * • Otherwise taken over: the stale file is renamed aside and VERIFIED to be the very file judged
409
+ * stale (content, mtime, inode). If a successor's fresh lock was moved instead (it took over between
410
+ * our check and our rename), it is put back — never over a third lock — and we back off. One winner.
411
+ */
412
+ export function takeReplayLock(projectDir, now = Date.now(), { isAlive = pidAlive, beforeRename = null } = {}) {
413
+ const lock = lockPath(projectDir);
414
+ const token = `${process.pid}-${now}-${Math.random().toString(36).slice(2, 10)}`;
415
+ const create = () => { fs.writeFileSync(lock, `${token}\npid ${process.pid}\n`, { flag: 'wx', mode: 0o600 }); return token; };
416
+ try { return create(); } catch { /* held, or stale */ }
417
+ let seen;
418
+ try { seen = lockFacts(lock); } catch { try { return create(); } catch { return null; } }
419
+ const age = now - seen.mtimeMs;
420
+ if (age <= REPLAY_LOCK_STALE_MS) return null;
421
+ if (age <= REPLAY_LOCK_ABANDON_MS && isAlive(holderPid(seen.content))) return null;
422
+ beforeRename?.();
423
+ const aside = `${lock}.stale-${token}`;
424
+ try { fs.renameSync(lock, aside); } catch { return null; }
425
+ let moved = null;
426
+ try { moved = lockFacts(aside); } catch { /* vanished */ }
427
+ if (!moved || !sameFacts(moved, seen)) {
428
+ // Put the successor's lock back without ever overwriting a third holder's: a hard link where the
429
+ // filesystem has them, else an exclusive copy.
430
+ try { fs.linkSync(aside, lock); } catch {
431
+ try { fs.copyFileSync(aside, lock, fs.constants.COPYFILE_EXCL); } catch { /* a third holder exists; the successor sees it lost the lock and stops */ }
432
+ }
433
+ try { fs.rmSync(aside, { force: true }); } catch { /* best effort */ }
434
+ return null;
435
+ }
436
+ try { fs.rmSync(aside, { force: true }); } catch { /* best effort */ }
437
+ try { return create(); } catch { return null; }
438
+ }
439
+
440
+ /** Heartbeat: refresh the lock's mtime if (and only if) this holder still owns it. */
441
+ export function refreshReplayLock(projectDir, token) {
442
+ if (!token || readLock(projectDir) !== token) return false;
443
+ try { const t = new Date(); fs.utimesSync(lockPath(projectDir), t, t); return true; } catch { return false; }
444
+ }
445
+
446
+ /** A worker that inherited the lock records ITS OWN pid on it, keeping the owner token. */
447
+ export function adoptReplayLock(projectDir, token) {
448
+ if (!token || readLock(projectDir) !== token) return false;
449
+ try { fs.writeFileSync(lockPath(projectDir), `${token}\npid ${process.pid}\n`, { mode: 0o600 }); return true; } catch { return false; }
450
+ }
451
+
452
+ /** Release ONLY a lock this holder owns; a successor's lock is never deleted. */
453
+ export function releaseReplayLock(projectDir, token) {
454
+ if (!token || readLock(projectDir) !== token) return false;
455
+ try { fs.rmSync(lockPath(projectDir), { force: true }); return true; } catch { return false; }
456
+ }
457
+
458
+ /** Hand the lock (or take it, if free) to a detached worker. Returns whether one was started. Never throws. */
459
+ export function replayOutboxDetached({ projectDir, token = null, spawnFn = spawn } = {}) {
460
+ const held = token || takeReplayLock(projectDir);
461
+ if (!held) return false;
157
462
  try {
158
- result = captureProgression({
159
- host,
160
- payload: { ...payload, hook_event_name: event, projectProgression: produced.projectProgression },
161
- projectDir,
162
- storeFactory,
463
+ const child = spawnFn(process.execPath, [fileURLToPath(import.meta.url), '--replay-outbox'], {
464
+ cwd: projectDir, detached: true, stdio: 'ignore', windowsHide: true,
465
+ env: { ...process.env, RUVNET_REPLAY_LOCK_TOKEN: held },
163
466
  });
164
- } catch (error) {
165
- // NOT LOST — DEFERRED. capture() fsyncs the snapshot to the durable outbox BEFORE it writes to
166
- // the store, so a budget overrun here leaves the evidence on disk and the next capture boundary
167
- // (or /checkpoint) commits it. Reporting that plainly is the whole difference between a bounded
168
- // hook and a lossy one, so the reason is returned rather than thrown at a lifecycle boundary.
169
- return { ...idle, replayed, skipped: `capture deferred: ${error.message}` };
467
+ child.unref?.();
468
+ return true;
469
+ } catch {
470
+ releaseReplayLock(projectDir, held);
471
+ return false;
170
472
  }
171
- return {
172
- metadataWritten,
173
- progressionCaptured: true,
174
- turn,
175
- replayed,
176
- receipt: result.receipt,
177
- provenance: produced.provenance,
178
- };
179
473
  }
180
474
 
181
- if (process.argv[1] && path.resolve(process.argv[1]).endsWith('session-snapshot-hook.mjs')) {
475
+ /**
476
+ * The detached worker's body, holding the lock `token`: record its own pid on the lock, return orphaned
477
+ * claims to the queue, replay the outbox, then run every queued capture IN ORDER — each CLAIMED by
478
+ * atomic rename first, so no other worker can run it too, and each re-entering the boundary as
479
+ * `ordered`, so it replays before it produces. Ownership is re-checked before every step and right
480
+ * after each claim; a worker that lost the lock puts an unstarted claim back and stops. A finished
481
+ * claim is the claimer's own and is deleted. Releases only its own lock, then re-checks for captures
482
+ * queued while it held it.
483
+ */
484
+ export function runOutboxReplay({ projectDir, token = process.env.RUVNET_REPLAY_LOCK_TOKEN || null, budgetMs = DETACHED_REPLAY_BUDGET_MS,
485
+ makeStoreFactory = boundedStoreFactory, now = Date.now, runCapture = runSessionSnapshotHook, onClaim = null } = {}) {
486
+ let held = token || takeReplayLock(projectDir);
487
+ let replayed = 0;
488
+ for (let round = 0; held && round < 8; round += 1) {
489
+ try {
490
+ if (!adoptReplayLock(projectDir, held)) return replayed;
491
+ reclaimOrphans(projectDir);
492
+ const resolution = resolveProjectStore({ projectDir });
493
+ const store = makeStoreFactory(now() + budgetMs)({ projectDir, requestedStorePath: resolution.canonicalAgentDbPath });
494
+ for (const snapshot of store.outbox.pendingSnapshots()) {
495
+ if (!refreshReplayLock(projectDir, held)) return replayed;
496
+ store.outbox.markCommitted(store.appendExact(snapshot));
497
+ replayed += 1;
498
+ }
499
+ for (const file of queuedCaptures(projectDir)) {
500
+ if (!refreshReplayLock(projectDir, held)) return replayed;
501
+ const claimed = claimQueued(file);
502
+ if (!claimed) continue;
503
+ onClaim?.(claimed);
504
+ if (!refreshReplayLock(projectDir, held)) {
505
+ returnClaim(claimed);
506
+ return replayed;
507
+ }
508
+ let job = null;
509
+ try { job = JSON.parse(fs.readFileSync(claimed, 'utf8')); } catch { /* torn: dropped below */ }
510
+ try {
511
+ if (job) runCapture(projectDir, job.event, { rawInput: JSON.stringify(job.payload), host: job.host,
512
+ budgetMs, makeStoreFactory, now, ordered: held, writeMetadata: false,
513
+ captureTurn: () => ({ recorded: false, skipped: 'detached replay' }) });
514
+ } catch { /* a failed capture leaves its own snapshot durable in the outbox */ }
515
+ try { fs.rmSync(claimed, { force: true }); } catch { /* best effort */ }
516
+ }
517
+ } catch { /* the debt stays durable; the next boundary hands it on again */ } finally {
518
+ releaseReplayLock(projectDir, held);
519
+ }
520
+ held = queuedWork(projectDir) ? takeReplayLock(projectDir) : null;
521
+ }
522
+ return replayed;
523
+ }
524
+
525
+ if (process.argv[1] && path.resolve(process.argv[1]).endsWith('session-snapshot-hook.mjs') && process.argv[2] === '--replay-outbox') {
526
+ try { runOutboxReplay({ projectDir: process.cwd() }); } catch { /* the debt stays durable in the outbox */ }
527
+ } else if (process.argv[1] && path.resolve(process.argv[1]).endsWith('session-snapshot-hook.mjs')) {
182
528
  // projectDirectory() is the SAME derivation the Console's detector uses. Deriving it here
183
529
  // independently is what let this hook write a receipt the Console then reported as missing (#85).
184
530
  const rawInput = fs.readFileSync(0, 'utf8');
@@ -93,13 +93,29 @@ export const knowledgeFacts = ({ env = process.env, home, now = Date.now() } = {
93
93
  lockMs: Date.parse(json(auto.lockFile)?.at || '') || mtimeMs(auto.lockFile) };
94
94
  };
95
95
 
96
+ /**
97
+ * agentic-kit ownership is a CLAIM in kit.json, not a delivery: on the owner's Mac (2026-09-30)
98
+ * kit.json said ruvnetBrain:true while agentic-kit scheduled nothing, so the self-heal stood down
99
+ * forever and the knowledge base aged by hand only. Ownership is honoured only while an update is
100
+ * PROVEN inside this window (a successful refresh receipt or a CURRENT --check verdict); 36h leaves
101
+ * the self-heal 12h to land one before the 48h invariant breaks.
102
+ */
103
+ export const AGENTIC_KIT_PROOF_HOURS = 36;
104
+ /** 'none' | 'delivering' (kit.json claims it AND an update is proven) | 'not-delivering'. */
105
+ export const agenticKitUpdates = ({ home, facts }) => {
106
+ if (!updateOwnedByAgenticKit(home)) return 'none';
107
+ return facts.provenWithin(AGENTIC_KIT_PROOF_HOURS) ? 'delivering' : 'not-delivering';
108
+ };
109
+
96
110
  /** Why the SessionStart knowledge auto-update may NEVER run on this machine ('' = it may). */
97
111
  export const autoUpdateOptOut = ({ env = process.env, home, facts }) => {
98
112
  const flag = String(env.RUVNET_AUTO_UPDATE || '').toLowerCase();
99
113
  if (flag === 'off') return 'RUVNET_AUTO_UPDATE=off';
100
114
  if (env.RUVNET_BRAIN_TEST === '1' && flag !== 'on') return 'test mode';
101
115
  if (read(path.join(facts.brainHome, '.auto-update-pref')).trim() === 'no') return 'you answered no to background auto-update';
102
- if (updateOwnedByAgenticKit(home)) return 'agentic-kit owns updates: ak sync';
116
+ if (agenticKitUpdates({ home, facts }) === 'delivering') {
117
+ return `agentic-kit owns updates and one is proven within ${AGENTIC_KIT_PROOF_HOURS}h: ak sync`;
118
+ }
103
119
  if (!exists(path.join(facts.kbDir, 'forge-update.mjs'))) return 'this install predates the self-updater';
104
120
  return '';
105
121
  };
@@ -122,8 +138,10 @@ export const knowledgeCurrency = ({ env = process.env, home, now = Date.now(), w
122
138
  const ageKnown = Number.isFinite(builtMs);
123
139
  if (!failing && proven) return '';
124
140
  if (!failing && ageKnown && hours(builtMs) <= windowHours) return '';
125
- const agentKit = updateOwnedByAgenticKit(home);
126
- const scheduled = agentKit || readNightlyRegistration({ brainHome }).ok;
141
+ const kit = agenticKitUpdates({ home, facts });
142
+ const agentKit = kit === 'delivering';
143
+ // An agentic-kit machine must never be told to also --enable-nightly (one owner per machine).
144
+ const scheduled = kit !== 'none' || readNightlyRegistration({ brainHome }).ok;
127
145
  const parts = [ageKnown ? `knowledge base built ${day(builtMs)} (${age(builtMs)})`
128
146
  : 'knowledge base age UNKNOWN (SOURCE.json missing or unreadable)'];
129
147
  if (autoFailed) {
@@ -138,6 +156,9 @@ export const knowledgeCurrency = ({ env = process.env, home, now = Date.now(), w
138
156
  parts.push(history.receipts
139
157
  ? `${history.failuresSinceSuccess} failed run(s) since the last success (${history.lastSuccess ? day(history.lastSuccess.at) : 'none recorded'})`
140
158
  : 'no refresh has ever run on this machine');
159
+ if (kit === 'not-delivering') {
160
+ parts.push(`agentic-kit claims updates (kit.json ruvnetBrain:true) but no update is proven in ${AGENTIC_KIT_PROOF_HOURS}h, so the Brain's own self-heal runs instead`);
161
+ }
141
162
  if (!scheduled) parts.push('no nightly refresh is scheduled');
142
163
  if (history.unreadable) parts.push(`${history.unreadable} unreadable receipt(s)`);
143
164
  const fix = agentKit ? 'ak sync' : scheduled ? 'npx ruvnet-brain@latest --update'
@@ -145,7 +145,7 @@ export const heartbeat = ({ env, hookDir, stateDir, home, running, seedDispatche
145
145
  if (pref === 'yes' && exists(path.join(kbDir, 'forge-update.mjs'))) {
146
146
  const kbLog = path.join(stateDir, '.last-kb-check.log');
147
147
  if (/\bBEHIND\b/.test(read(kbLog))) {
148
- emit('[RuvNet Brain — a newer knowledge bundle is available. It is signed (Ed25519) and the updater verifies that signature before extracting anything. We do NOT auto-apply it: applying replaces executable tool files, which is your call. To update: cd ~/.cache/ruvnet-brain/kb && node forge-update.mjs --apply]');
148
+ emit('[RuvNet Brain — a newer knowledge bundle is available. It is signed (Ed25519) and the updater verifies that signature before extracting anything. We do NOT auto-apply it: applying replaces executable tool files, which is your call. To update: npx ruvnet-brain@latest --update]');
149
149
  }
150
150
  // S2 (ONE CURRENCY VERDICT): --result-file records the SAME structured verdict --check/--apply
151
151
  // and bin/install.mjs already read (forge-update.mjs's currencyVerdict()), at the well-known path
@@ -53,10 +53,19 @@ const RECEIPTS = path.join(BRAIN_HOME, 'update-receipts.jsonl');
53
53
  const LEASES = path.join(BRAIN_HOME, 'leases');
54
54
  const DEV = path.join(BRAIN_HOME, 'dev.json');
55
55
  const SEEDED = path.join(BRAIN_HOME, '.spine-seeded');
56
+ // Claude Code honours CLAUDE_CONFIG_DIR for its whole config tree, plugins included; Codex honours CODEX_HOME.
57
+ const CLAUDE_CONFIG = process.env.CLAUDE_CONFIG_DIR || path.join(os.homedir(), '.claude');
58
+ const CODEX_CONFIG = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
56
59
  const PLUGIN_CACHES = [
57
- path.join(os.homedir(), '.claude', 'plugins', 'cache', 'ruvnet-brain', 'ruvnet-brain'),
58
- path.join(process.env.CODEX_HOME || path.join(os.homedir(), '.codex'), 'plugins', 'cache', 'ruvnet-brain', 'ruvnet-brain'),
60
+ path.join(CLAUDE_CONFIG, 'plugins', 'cache', 'ruvnet-brain', 'ruvnet-brain'),
61
+ path.join(CODEX_CONFIG, 'plugins', 'cache', 'ruvnet-brain', 'ruvnet-brain'),
59
62
  ];
63
+ // Evidence that a host has been pointed at the Brain's plugin at all (marketplace registered or plugin
64
+ // cache created) — even when `plugin install` then failed and staged nothing.
65
+ const HOST_PLUGIN_EVIDENCE = [CLAUDE_CONFIG, CODEX_CONFIG].flatMap((root) => [
66
+ path.join(root, 'plugins', 'marketplaces', 'ruvnet-brain'),
67
+ path.join(root, 'plugins', 'cache', 'ruvnet-brain'),
68
+ ]);
60
69
  const LEASE_FRESH_MS = 6 * 3600_000; // a lease older than 6h is stale (its process is long gone)
61
70
 
62
71
  const argv = process.argv.slice(2);
@@ -425,6 +434,17 @@ function main() {
425
434
  console.log(`already on ${activeNow.version}, at or above requested ${expectedVersion} — nothing to apply.`);
426
435
  return 0;
427
436
  }
437
+ // NO HOST AT ALL is not a stale spine. With no active spine and no payload of ANY version in
438
+ // any host cache, nothing was ever seeded, so nothing can be behind (a desktop-app/IDE-extension
439
+ // customer whose shell has no host CLI). A host cache holding some OTHER version still fails
440
+ // closed below — issue #64's exact-selection guard is untouched.
441
+ // Existence, not a parse: a corrupt active.json is a damaged spine and still fails closed below.
442
+ // A host that registered the Brain's plugin but staged nothing (its `plugin install` failed) is
443
+ // NOT "no host": that is a failed install and must keep failing.
444
+ if (!fs.existsSync(ACTIVE) && !newestStagedCC() && !HOST_PLUGIN_EVIDENCE.some((dir) => fs.existsSync(dir))) {
445
+ console.log(`no host has staged a payload and no spine is active — nothing to converge for ${expectedVersion}.`);
446
+ return 0;
447
+ }
428
448
  console.error(`✗ no staged host payload exactly matches expected version ${expectedVersion} — spine unchanged`);
429
449
  return 1;
430
450
  }