ruvnet-brain 4.3.40 → 4.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/bin/install.mjs +179 -36
- package/kb/brain-profile.mjs +17 -2
- package/kb/forge-update.mjs +6 -2
- package/kb/lifecycle-evidence-retention.mjs +12 -9
- package/kb/refresh-run.mjs +17 -1
- package/kb/update-storage-transaction.mjs +33 -5
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/hooks/codex-hooks.json +2 -2
- package/plugin/hooks/hooks.json +1 -1
- package/plugin/scripts/capability-registry.mjs +3 -3
- package/plugin/scripts/codex-hook-wrapper.mjs +7 -2
- package/plugin/scripts/design-wall.sh +1 -0
- package/plugin/scripts/ground-before-write.sh +1 -0
- package/plugin/scripts/ground-ruvnet.sh +3 -3
- package/plugin/scripts/grounding-answer.mjs +129 -0
- package/plugin/scripts/grounding-stamp.sh +32 -31
- package/plugin/scripts/grounding-turn-evidence.mjs +146 -5
- package/plugin/scripts/grounding-turn-gate.mjs +25 -6
- package/plugin/scripts/hook-shim.mjs +3 -0
- package/plugin/scripts/kling-preflight.sh +1 -0
- package/plugin/scripts/learn-capture.sh +1 -0
- package/plugin/scripts/project-progression-reader.mjs +10 -0
- package/plugin/scripts/project-progression-sources.mjs +16 -4
- package/plugin/scripts/project-progression-store.mjs +201 -6
- package/plugin/scripts/protect-brain-state.sh +1 -0
- package/plugin/scripts/route-dispatch.sh +1 -0
- package/plugin/scripts/session-snapshot-hook.mjs +383 -37
- package/plugin/scripts/session-start-health.mjs +24 -3
- package/plugin/scripts/session-start-update-plane.mjs +1 -1
- package/plugin/scripts/update-apply.mjs +22 -2
- package/scripts/console-instances.mjs +203 -0
- package/scripts/console-runtime-identity.mjs +2 -0
- package/scripts/corpus-canary.mjs +130 -18
- package/scripts/customer-seams.mjs +84 -0
- package/scripts/customer-state-matrix.mjs +363 -0
- package/scripts/full-suite-gate.mjs +162 -0
- package/scripts/grounding-turn-replay.mjs +11 -3
- package/scripts/hook-qualify-core.mjs +346 -0
- package/scripts/hook-qualify-hosts.mjs +115 -0
- package/scripts/hook-qualify.mjs +101 -0
- package/scripts/host-cli.mjs +115 -0
- package/scripts/qe/agentic-qe-4.3.mjs +0 -1
- package/scripts/route-gold-rank.mjs +156 -0
- package/scripts/route-index-memory.mjs +51 -0
- package/scripts/route-latency-warm.mjs +123 -0
- package/scripts/wired-check.mjs +17 -3
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import fs from 'node:fs';
|
|
2
2
|
import path from 'node:path';
|
|
3
|
-
import { spawnSync } from 'node:child_process';
|
|
3
|
+
import { spawn, spawnSync } from 'node:child_process';
|
|
4
|
+
import { fileURLToPath } from 'node:url';
|
|
5
|
+
import { ProgressionOutbox } from './project-progression-outbox.mjs';
|
|
4
6
|
import {
|
|
5
7
|
captureProjectTransition,
|
|
6
8
|
hasProjectProgression,
|
|
@@ -19,6 +21,23 @@ import { captureTurnOutcome } from './turn-outcome-capture.mjs';
|
|
|
19
21
|
*/
|
|
20
22
|
export const CAPTURE_BUDGET_MS = 8_000;
|
|
21
23
|
|
|
24
|
+
/**
|
|
25
|
+
* Replaying an interrupted session's outbox costs one `ruflo` write per pending snapshot, each ~3s
|
|
26
|
+
* cold (project-progression-store.mjs). Under this budget there is room for the NEW snapshot or the
|
|
27
|
+
* old ones, not both — and the new one is the one nothing else will ever write.
|
|
28
|
+
*/
|
|
29
|
+
export const REPLAY_MIN_BUDGET_MS = 4_000;
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* The budget this invocation really has. The Codex wrapper hands its own kill deadline down as
|
|
33
|
+
* RUVNET_CODEX_BUDGET_MS (2200ms at SessionEnd, which Codex caps at 3s); planning for 8s there meant
|
|
34
|
+
* being SIGKILLed mid-write with nothing reported. 300ms is left for the adapter → shim → body spawns.
|
|
35
|
+
*/
|
|
36
|
+
export function effectiveBudgetMs(env = process.env) {
|
|
37
|
+
const handed = Number(env.RUVNET_CODEX_BUDGET_MS);
|
|
38
|
+
return Number.isFinite(handed) && handed > 0 ? Math.max(0, Math.min(CAPTURE_BUDGET_MS, handed - 300)) : CAPTURE_BUDGET_MS;
|
|
39
|
+
}
|
|
40
|
+
|
|
22
41
|
function regularOrAbsent(file) {
|
|
23
42
|
try {
|
|
24
43
|
const stat = fs.lstatSync(file);
|
|
@@ -96,11 +115,16 @@ export function runSessionSnapshotHook(projectDir, event, {
|
|
|
96
115
|
host = process.env.RUVNET_HOOK_HOST || 'claude',
|
|
97
116
|
captureProgression = captureProjectTransition,
|
|
98
117
|
produce = buildProjectProgression,
|
|
99
|
-
budgetMs =
|
|
118
|
+
budgetMs = effectiveBudgetMs(),
|
|
100
119
|
now = Date.now,
|
|
101
120
|
captureTurn = captureTurnOutcome,
|
|
121
|
+
writeMetadata = true,
|
|
122
|
+
makeStoreFactory = boundedStoreFactory,
|
|
123
|
+
spawnReplay = replayOutboxDetached,
|
|
124
|
+
ordered = null,
|
|
102
125
|
} = {}) {
|
|
103
|
-
|
|
126
|
+
// The detached worker re-runs a QUEUED boundary; its session receipt was already written then.
|
|
127
|
+
const metadataWritten = writeMetadata ? writeSessionSnapshot(projectDir, event) : false;
|
|
104
128
|
let payload;
|
|
105
129
|
try { payload = rawInput ? JSON.parse(rawInput) : {}; } catch { payload = {}; }
|
|
106
130
|
// TURN OUTCOMES FIRST, and independent of `.swarm`: every turn in every repository is recorded
|
|
@@ -134,51 +158,373 @@ export function runSessionSnapshotHook(projectDir, event, {
|
|
|
134
158
|
return { ...idle, skipped: 'project has not adopted the canonical store' };
|
|
135
159
|
}
|
|
136
160
|
|
|
137
|
-
const
|
|
138
|
-
const
|
|
161
|
+
const root = resolution.projectRoot;
|
|
162
|
+
const pendingCount = () => {
|
|
163
|
+
try { return new ProgressionOutbox({ projectRoot: root }).pendingSnapshots().length; } catch { return 0; }
|
|
164
|
+
};
|
|
139
165
|
|
|
140
|
-
//
|
|
141
|
-
//
|
|
142
|
-
//
|
|
143
|
-
|
|
166
|
+
// CAUSAL ORDER. The producer links a new snapshot to the COMMITTED heads
|
|
167
|
+
// (project-progression-producer.mjs), so a snapshot produced while older work is uncommitted would
|
|
168
|
+
// not descend from it and the project would end with two unrelated heads
|
|
169
|
+
// (tests/acceptance/cross-host-project-resume.test.mjs). "Older work" is BOTH the outbox (captures
|
|
170
|
+
// interrupted after their fsync) AND the capture queue (whole boundaries waiting for a worker).
|
|
171
|
+
//
|
|
172
|
+
// THE REPLAY LOCK IS THE RIGHT TO COMMIT IN ORDER (4.4.0 re-review S-A). Every boundary takes it
|
|
173
|
+
// before doing anything that commits. If it cannot — a live worker holds it — or older work is
|
|
174
|
+
// queued, or (on a short budget) outbox debt cannot be replayed here, this boundary QUEUES itself
|
|
175
|
+
// behind that work and the lock passes to a DETACHED, bounded worker that drains everything in
|
|
176
|
+
// order. On Codex no boundary has the replay budget (Stop 3700ms effective, SessionEnd 1900ms, no
|
|
177
|
+
// PreCompact). Any boundary, of any budget, therefore also drains a queue a dead worker stranded.
|
|
178
|
+
// `ordered` is the worker's own re-entry: it already holds the lock and is draining in order.
|
|
179
|
+
let token = ordered;
|
|
180
|
+
const handOff = (why) => {
|
|
181
|
+
const queued = queueCapture({ projectDir: root, event, host, payload });
|
|
182
|
+
const handed = queued ? spawnReplay({ projectDir: root, token }) : false;
|
|
183
|
+
if (!handed && token && token !== ordered) releaseReplayLock(root, token);
|
|
184
|
+
return { ...idle, replayed: 0, progressionCaptured: false, deferredToReplayer: Boolean(queued),
|
|
185
|
+
replaySkipped: `${why}; this capture ${queued ? 'queued behind it' : 'NOT queued (queue unwritable)'}`
|
|
186
|
+
+ `${queued ? (handed ? ', handed to a detached worker' : ' (the current lock holder hands the queue to a worker when it releases)') : ''}` };
|
|
187
|
+
};
|
|
188
|
+
if (!ordered) {
|
|
189
|
+
token = takeReplayLock(root);
|
|
190
|
+
if (!token) return handOff('a worker is committing older work');
|
|
191
|
+
const queuedAhead = queuedWork(root);
|
|
192
|
+
if (queuedAhead) return handOff(`${queuedAhead} older capture(s) queued`);
|
|
193
|
+
if (budgetMs < REPLAY_MIN_BUDGET_MS) {
|
|
194
|
+
const pending = pendingCount();
|
|
195
|
+
if (pending) return handOff(`outbox replay deferred: budget ${budgetMs}ms < ${REPLAY_MIN_BUDGET_MS}ms; ${pending} pending`);
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
let handedLock = false;
|
|
144
200
|
try {
|
|
145
|
-
|
|
146
|
-
|
|
201
|
+
const deadlineAt = now() + budgetMs;
|
|
202
|
+
const storeFactory = makeStoreFactory(deadlineAt);
|
|
203
|
+
let replayed = 0;
|
|
204
|
+
if (budgetMs >= REPLAY_MIN_BUDGET_MS) {
|
|
205
|
+
try {
|
|
206
|
+
replayed = storeFactory({ projectDir, requestedStorePath: resolution.canonicalAgentDbPath }).replay().length;
|
|
207
|
+
} catch { /* the debt stays durable in the outbox; this capture is still worth attempting */ }
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
// Before COMMITTING anything, re-check the lock is still ours: with three racers, a stale-lock
|
|
211
|
+
// put-back can leave a holder that no longer owns it. One that lost it queues itself instead.
|
|
212
|
+
if (!ordered && !refreshReplayLock(root, token)) {
|
|
213
|
+
token = null;
|
|
214
|
+
handedLock = true; // nothing of ours to release
|
|
215
|
+
return handOff('the lock was taken over before this capture committed');
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
let produced;
|
|
219
|
+
try {
|
|
220
|
+
produced = produce({ resolution, payload, host, trigger: event });
|
|
221
|
+
} catch (error) {
|
|
222
|
+
return { ...idle, replayed, skipped: `producer failed: ${error.message}` };
|
|
223
|
+
}
|
|
224
|
+
if (produced.skipped) return { ...idle, replayed, skipped: produced.skipped.reason };
|
|
225
|
+
|
|
226
|
+
let result;
|
|
227
|
+
try {
|
|
228
|
+
result = captureProgression({
|
|
229
|
+
host,
|
|
230
|
+
payload: { ...payload, hook_event_name: event, projectProgression: produced.projectProgression },
|
|
231
|
+
projectDir,
|
|
232
|
+
storeFactory,
|
|
233
|
+
});
|
|
234
|
+
} catch (error) {
|
|
235
|
+
// NOT LOST — DEFERRED. capture() fsyncs the snapshot to the durable outbox BEFORE it writes to
|
|
236
|
+
// the store, so a budget overrun leaves the evidence on disk. On a short budget nothing later in
|
|
237
|
+
// this process can settle it, so the lock goes straight to a detached worker.
|
|
238
|
+
handedLock = !ordered && budgetMs < REPLAY_MIN_BUDGET_MS && pendingCount() > 0 && spawnReplay({ projectDir: root, token });
|
|
239
|
+
return { ...idle, replayed, skipped: `capture deferred: ${error.message}`,
|
|
240
|
+
...(handedLock ? { replaySkipped: 'deferred capture handed to a detached worker' } : {}) };
|
|
241
|
+
}
|
|
242
|
+
return {
|
|
243
|
+
metadataWritten,
|
|
244
|
+
progressionCaptured: true,
|
|
245
|
+
turn,
|
|
246
|
+
replayed,
|
|
247
|
+
receipt: result.receipt,
|
|
248
|
+
provenance: produced.provenance,
|
|
249
|
+
};
|
|
250
|
+
} finally {
|
|
251
|
+
// A boundary that fired while this one held the lock queued itself and could not start a worker
|
|
252
|
+
// (this lock was in the way). Releasing without looking stranded it until the next boundary — two
|
|
253
|
+
// simultaneous SessionEnds lost the second one's final state (4.4.1). So: queued work → hand THIS
|
|
254
|
+
// lock to a worker; and re-check after releasing, for a boundary that queued in between.
|
|
255
|
+
if (!ordered && !handedLock) {
|
|
256
|
+
if (!(queuedWork(root) && spawnReplay({ projectDir: root, token }))) {
|
|
257
|
+
releaseReplayLock(root, token);
|
|
258
|
+
if (queuedWork(root)) spawnReplay({ projectDir: root });
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* How long the detached worker may spend per step. The lock is refreshed between steps and goes stale
|
|
266
|
+
* after REPLAY_LOCK_STALE_MS, which is more than twice a step, so a live worker never looks dead.
|
|
267
|
+
*/
|
|
268
|
+
export const DETACHED_REPLAY_BUDGET_MS = 45_000;
|
|
269
|
+
export const REPLAY_LOCK_STALE_MS = 120_000;
|
|
270
|
+
const REPLAY_LOCK = '.progression-replay.lock';
|
|
271
|
+
const QUEUE_PREFIX = '.progression-capture-queue-';
|
|
272
|
+
const lockPath = (projectDir) => path.join(projectDir, '.swarm', REPLAY_LOCK);
|
|
273
|
+
// The lock's FIRST line is the owner token; a second `pid <n>` line names the process holding it.
|
|
274
|
+
const readLock = (projectDir) => { try { return fs.readFileSync(lockPath(projectDir), 'utf8').split('\n')[0].trim(); } catch { return null; } };
|
|
275
|
+
const CLAIM_PREFIX = '.progression-capture-claimed-';
|
|
276
|
+
/** After this long a stale lock is taken over even if its holder pid looks alive (pid reuse, a wedged process). */
|
|
277
|
+
export const REPLAY_LOCK_ABANDON_MS = 30 * 60_000;
|
|
278
|
+
|
|
279
|
+
/** Is a process with this pid alive? EPERM means alive but not ours. Never throws. */
|
|
280
|
+
export function pidAlive(pid) {
|
|
281
|
+
if (!Number.isSafeInteger(pid) || pid <= 0) return false;
|
|
282
|
+
try { process.kill(pid, 0); return true; } catch (error) { return error?.code === 'EPERM'; }
|
|
283
|
+
}
|
|
284
|
+
const seqOf = (name) => Number((/(\d{12})\.json$/.exec(name) || [])[1] ?? 0);
|
|
285
|
+
const swarmEntries = (projectDir) => { try { return fs.readdirSync(path.join(projectDir, '.swarm')); } catch { return []; } };
|
|
286
|
+
// 4.4.0 named queue files by wall clock: `<prefix><15-digit ms>-<hrtime>-<pid>-<n>.json`. Open 4.4.0
|
|
287
|
+
// sessions keep queuing in that format after the update, so the two formats coexist for a while.
|
|
288
|
+
const LEGACY_QUEUE = /^\d{15}-/;
|
|
289
|
+
const queueTail = (name) => (name.startsWith(QUEUE_PREFIX) ? name.slice(QUEUE_PREFIX.length) : name.slice(CLAIM_PREFIX.length).replace(/^\d+-[A-Za-z0-9]*-\d+-/, '')); // <pid>-<start>-<queuedAt>-
|
|
290
|
+
const mtimeOf = (projectDir, name) => { try { return fs.statSync(path.join(projectDir, '.swarm', name)).mtimeMs; } catch { return Infinity; } };
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* Queue one boundary's capture for the worker (0600, inside the project's own .swarm). ORDER IS THE
|
|
294
|
+
* ORDER OF EXCLUSIVE CREATION: the name is the next sequence number after every queued or claimed one,
|
|
295
|
+
* created with O_EXCL and retried on collision — never a clock, which can step backwards or wrap.
|
|
296
|
+
*/
|
|
297
|
+
export function queueCapture({ projectDir, event, host, payload }) {
|
|
298
|
+
const body = JSON.stringify({ event, host, payload });
|
|
299
|
+
for (let attempt = 0; attempt < 64; attempt += 1) {
|
|
300
|
+
const seq = Math.max(0, ...swarmEntries(projectDir).filter((n) => n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)).map(seqOf)) + 1;
|
|
301
|
+
const file = path.join(projectDir, '.swarm', `${QUEUE_PREFIX}${String(seq).padStart(12, '0')}.json`);
|
|
302
|
+
try { fs.writeFileSync(file, body, { flag: 'wx', mode: 0o600 }); return file; } catch (error) {
|
|
303
|
+
if (error?.code !== 'EEXIST') return null;
|
|
304
|
+
}
|
|
305
|
+
}
|
|
306
|
+
return null;
|
|
307
|
+
}
|
|
147
308
|
|
|
148
|
-
|
|
309
|
+
/**
|
|
310
|
+
* Unclaimed queued captures, in queue order. While any 4.4.0 (timestamp-named) entry is present the
|
|
311
|
+
* order is creation time (mtime, then name) — by NAME every 4.4.1 sequence file would sort before every
|
|
312
|
+
* 4.4.0 one, replaying newer captures before older ones across the upgrade window. Once the legacy
|
|
313
|
+
* entries are gone the order is the sequence alone, independent of any clock.
|
|
314
|
+
*/
|
|
315
|
+
export function queuedCaptures(projectDir) {
|
|
316
|
+
const all = swarmEntries(projectDir).filter((n) => (n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)) && n.endsWith('.json'));
|
|
317
|
+
const queued = all.filter((n) => n.startsWith(QUEUE_PREFIX));
|
|
318
|
+
const mixed = all.some((n) => LEGACY_QUEUE.test(queueTail(n)));
|
|
319
|
+
const ordered = mixed
|
|
320
|
+
? queued.map((n) => [mtimeOf(projectDir, n), n]).sort((a, b) => a[0] - b[0] || (a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0)).map(([, n]) => n)
|
|
321
|
+
: queued.sort();
|
|
322
|
+
return ordered.map((n) => path.join(projectDir, '.swarm', n));
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/** All older work a boundary must wait behind: unclaimed captures plus captures a worker has claimed. */
|
|
326
|
+
export function queuedWork(projectDir) {
|
|
327
|
+
return swarmEntries(projectDir).filter((n) => (n.startsWith(QUEUE_PREFIX) || n.startsWith(CLAIM_PREFIX)) && n.endsWith('.json')).length;
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
/**
|
|
331
|
+
* A process's START TIME, as a filename-safe token, or null where it cannot be read (no `ps`, e.g.
|
|
332
|
+
* Windows). With the pid it identifies the process: a reused pid has a different start time.
|
|
333
|
+
*/
|
|
334
|
+
export function processStart(pid) {
|
|
149
335
|
try {
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
336
|
+
// TZ and locale PINNED: `lstart` prints local time in the locale's format, so two workers with
|
|
337
|
+
// different settings would record the same live process differently and read it as pid reuse.
|
|
338
|
+
const r = spawnSync('ps', ['-o', 'lstart=', '-p', String(pid)], { encoding: 'utf8', timeout: 2000, windowsHide: true,
|
|
339
|
+
env: { ...process.env, TZ: 'UTC', LC_ALL: 'C', LANG: 'C' } });
|
|
340
|
+
const s = String(r.stdout || '').replace(/[^A-Za-z0-9]/g, '');
|
|
341
|
+
return r.status === 0 && s ? s : null;
|
|
342
|
+
} catch { return null; }
|
|
343
|
+
}
|
|
344
|
+
let selfStart;
|
|
345
|
+
const ownStart = () => (selfStart === undefined ? (selfStart = processStart(process.pid)) : selfStart);
|
|
346
|
+
|
|
347
|
+
/**
|
|
348
|
+
* Claim a queued capture by atomic rename; null if taken. The claim's name records pid, start time and
|
|
349
|
+
* the queue file's ORIGINAL mtime (its creation order, which the mixed upgrade window sorts by); the
|
|
350
|
+
* claim file's own mtime is then set to the claim time, from which the orphan ceiling counts.
|
|
351
|
+
*/
|
|
352
|
+
function claimQueued(file) {
|
|
353
|
+
let queuedAt = 0;
|
|
354
|
+
try { queuedAt = Math.floor(fs.statSync(file).mtimeMs); } catch { return null; }
|
|
355
|
+
const claimed = path.join(path.dirname(file), `${CLAIM_PREFIX}${process.pid}-${ownStart() || 'na'}-${queuedAt}-${path.basename(file).slice(QUEUE_PREFIX.length)}`);
|
|
356
|
+
try { fs.renameSync(file, claimed); } catch { return null; }
|
|
357
|
+
try { const t = new Date(); fs.utimesSync(claimed, t, t); } catch { /* the ceiling then counts from queue time: earlier, never later */ }
|
|
358
|
+
return claimed;
|
|
359
|
+
}
|
|
360
|
+
const unclaimedName = (claimed) => path.join(path.dirname(claimed), `${QUEUE_PREFIX}${queueTail(path.basename(claimed))}`);
|
|
361
|
+
/** Put a claim back in the queue: rename (atomic, needs no hard links), restoring its creation order. */
|
|
362
|
+
const returnClaim = (claimed) => {
|
|
363
|
+
const queuedAt = Number(path.basename(claimed).slice(CLAIM_PREFIX.length).split('-')[2]);
|
|
364
|
+
const back = unclaimedName(claimed);
|
|
365
|
+
try { fs.renameSync(claimed, back); } catch { return false; }
|
|
366
|
+
if (Number.isFinite(queuedAt) && queuedAt > 0) { try { const t = new Date(queuedAt); fs.utimesSync(back, t, t); } catch { /* best effort */ } }
|
|
367
|
+
return true;
|
|
368
|
+
};
|
|
369
|
+
|
|
370
|
+
/**
|
|
371
|
+
* Return to the queue every claim whose worker is gone: its pid is dead, OR the pid now belongs to a
|
|
372
|
+
* different process (start time differs — pid reuse), OR the claim is older than REPLAY_LOCK_ABANDON_MS
|
|
373
|
+
* whatever the pid says (a reused pid where no start time can be read, a wedged worker). Without the
|
|
374
|
+
* last two a reused pid stranded a claim forever and every Stop spawned a worker that could not run it.
|
|
375
|
+
*/
|
|
376
|
+
export function reclaimOrphans(projectDir, { isAlive = pidAlive, startOf = processStart, now = Date.now() } = {}) {
|
|
377
|
+
let n = 0;
|
|
378
|
+
for (const name of swarmEntries(projectDir).filter((x) => x.startsWith(CLAIM_PREFIX))) {
|
|
379
|
+
const [pidText, start] = name.slice(CLAIM_PREFIX.length).split('-');
|
|
380
|
+
const pid = Number(pidText);
|
|
381
|
+
const claimed = path.join(projectDir, '.swarm', name);
|
|
382
|
+
const abandoned = now - mtimeOf(projectDir, name) > REPLAY_LOCK_ABANDON_MS;
|
|
383
|
+
let gone = abandoned || !isAlive(pid);
|
|
384
|
+
if (!gone && start && start !== 'na') {
|
|
385
|
+
const current = startOf(pid);
|
|
386
|
+
gone = Boolean(current) && current !== start;
|
|
387
|
+
}
|
|
388
|
+
if (gone && returnClaim(claimed)) n += 1;
|
|
153
389
|
}
|
|
154
|
-
|
|
390
|
+
return n;
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
const lockFacts = (file) => { const st = fs.statSync(file); return { content: fs.readFileSync(file, 'utf8'), mtimeMs: st.mtimeMs, ino: st.ino }; };
|
|
394
|
+
const sameFacts = (a, b) => a.content === b.content && a.mtimeMs === b.mtimeMs && a.ino === b.ino;
|
|
395
|
+
/** The pid ACTUALLY holding a lock: the `pid <n>` line a worker writes for itself, else the token's pid. */
|
|
396
|
+
const holderPid = (content) => {
|
|
397
|
+
const line = /^pid (\d+)$/m.exec(String(content));
|
|
398
|
+
return Number(line ? line[1] : String(content).trim().split('-')[0]);
|
|
399
|
+
};
|
|
155
400
|
|
|
156
|
-
|
|
401
|
+
/**
|
|
402
|
+
* Take the lock. Returns this holder's TOKEN (`<pid>-<time>-<random>`), or null.
|
|
403
|
+
* • Free → exclusive create.
|
|
404
|
+
* • Fresh (refreshed within REPLAY_LOCK_STALE_MS) → null.
|
|
405
|
+
* • Stale but its holder pid is ALIVE → null until REPLAY_LOCK_ABANDON_MS: a laptop asleep mid-step,
|
|
406
|
+
* or a long step, is not a dead worker, and taking over would put two workers on one job. The holder
|
|
407
|
+
* pid is the WORKER's own (it rewrites the lock on start), not the hook that spawned it and exited.
|
|
408
|
+
* • Otherwise taken over: the stale file is renamed aside and VERIFIED to be the very file judged
|
|
409
|
+
* stale (content, mtime, inode). If a successor's fresh lock was moved instead (it took over between
|
|
410
|
+
* our check and our rename), it is put back — never over a third lock — and we back off. One winner.
|
|
411
|
+
*/
|
|
412
|
+
export function takeReplayLock(projectDir, now = Date.now(), { isAlive = pidAlive, beforeRename = null } = {}) {
|
|
413
|
+
const lock = lockPath(projectDir);
|
|
414
|
+
const token = `${process.pid}-${now}-${Math.random().toString(36).slice(2, 10)}`;
|
|
415
|
+
const create = () => { fs.writeFileSync(lock, `${token}\npid ${process.pid}\n`, { flag: 'wx', mode: 0o600 }); return token; };
|
|
416
|
+
try { return create(); } catch { /* held, or stale */ }
|
|
417
|
+
let seen;
|
|
418
|
+
try { seen = lockFacts(lock); } catch { try { return create(); } catch { return null; } }
|
|
419
|
+
const age = now - seen.mtimeMs;
|
|
420
|
+
if (age <= REPLAY_LOCK_STALE_MS) return null;
|
|
421
|
+
if (age <= REPLAY_LOCK_ABANDON_MS && isAlive(holderPid(seen.content))) return null;
|
|
422
|
+
beforeRename?.();
|
|
423
|
+
const aside = `${lock}.stale-${token}`;
|
|
424
|
+
try { fs.renameSync(lock, aside); } catch { return null; }
|
|
425
|
+
let moved = null;
|
|
426
|
+
try { moved = lockFacts(aside); } catch { /* vanished */ }
|
|
427
|
+
if (!moved || !sameFacts(moved, seen)) {
|
|
428
|
+
// Put the successor's lock back without ever overwriting a third holder's: a hard link where the
|
|
429
|
+
// filesystem has them, else an exclusive copy.
|
|
430
|
+
try { fs.linkSync(aside, lock); } catch {
|
|
431
|
+
try { fs.copyFileSync(aside, lock, fs.constants.COPYFILE_EXCL); } catch { /* a third holder exists; the successor sees it lost the lock and stops */ }
|
|
432
|
+
}
|
|
433
|
+
try { fs.rmSync(aside, { force: true }); } catch { /* best effort */ }
|
|
434
|
+
return null;
|
|
435
|
+
}
|
|
436
|
+
try { fs.rmSync(aside, { force: true }); } catch { /* best effort */ }
|
|
437
|
+
try { return create(); } catch { return null; }
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
/** Heartbeat: refresh the lock's mtime if (and only if) this holder still owns it. */
|
|
441
|
+
export function refreshReplayLock(projectDir, token) {
|
|
442
|
+
if (!token || readLock(projectDir) !== token) return false;
|
|
443
|
+
try { const t = new Date(); fs.utimesSync(lockPath(projectDir), t, t); return true; } catch { return false; }
|
|
444
|
+
}
|
|
445
|
+
|
|
446
|
+
/** A worker that inherited the lock records ITS OWN pid on it, keeping the owner token. */
|
|
447
|
+
export function adoptReplayLock(projectDir, token) {
|
|
448
|
+
if (!token || readLock(projectDir) !== token) return false;
|
|
449
|
+
try { fs.writeFileSync(lockPath(projectDir), `${token}\npid ${process.pid}\n`, { mode: 0o600 }); return true; } catch { return false; }
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
/** Release ONLY a lock this holder owns; a successor's lock is never deleted. */
|
|
453
|
+
export function releaseReplayLock(projectDir, token) {
|
|
454
|
+
if (!token || readLock(projectDir) !== token) return false;
|
|
455
|
+
try { fs.rmSync(lockPath(projectDir), { force: true }); return true; } catch { return false; }
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
/** Hand the lock (or take it, if free) to a detached worker. Returns whether one was started. Never throws. */
|
|
459
|
+
export function replayOutboxDetached({ projectDir, token = null, spawnFn = spawn } = {}) {
|
|
460
|
+
const held = token || takeReplayLock(projectDir);
|
|
461
|
+
if (!held) return false;
|
|
157
462
|
try {
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
projectDir,
|
|
162
|
-
storeFactory,
|
|
463
|
+
const child = spawnFn(process.execPath, [fileURLToPath(import.meta.url), '--replay-outbox'], {
|
|
464
|
+
cwd: projectDir, detached: true, stdio: 'ignore', windowsHide: true,
|
|
465
|
+
env: { ...process.env, RUVNET_REPLAY_LOCK_TOKEN: held },
|
|
163
466
|
});
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
return { ...idle, replayed, skipped: `capture deferred: ${error.message}` };
|
|
467
|
+
child.unref?.();
|
|
468
|
+
return true;
|
|
469
|
+
} catch {
|
|
470
|
+
releaseReplayLock(projectDir, held);
|
|
471
|
+
return false;
|
|
170
472
|
}
|
|
171
|
-
return {
|
|
172
|
-
metadataWritten,
|
|
173
|
-
progressionCaptured: true,
|
|
174
|
-
turn,
|
|
175
|
-
replayed,
|
|
176
|
-
receipt: result.receipt,
|
|
177
|
-
provenance: produced.provenance,
|
|
178
|
-
};
|
|
179
473
|
}
|
|
180
474
|
|
|
181
|
-
|
|
475
|
+
/**
|
|
476
|
+
* The detached worker's body, holding the lock `token`: record its own pid on the lock, return orphaned
|
|
477
|
+
* claims to the queue, replay the outbox, then run every queued capture IN ORDER — each CLAIMED by
|
|
478
|
+
* atomic rename first, so no other worker can run it too, and each re-entering the boundary as
|
|
479
|
+
* `ordered`, so it replays before it produces. Ownership is re-checked before every step and right
|
|
480
|
+
* after each claim; a worker that lost the lock puts an unstarted claim back and stops. A finished
|
|
481
|
+
* claim is the claimer's own and is deleted. Releases only its own lock, then re-checks for captures
|
|
482
|
+
* queued while it held it.
|
|
483
|
+
*/
|
|
484
|
+
export function runOutboxReplay({ projectDir, token = process.env.RUVNET_REPLAY_LOCK_TOKEN || null, budgetMs = DETACHED_REPLAY_BUDGET_MS,
|
|
485
|
+
makeStoreFactory = boundedStoreFactory, now = Date.now, runCapture = runSessionSnapshotHook, onClaim = null } = {}) {
|
|
486
|
+
let held = token || takeReplayLock(projectDir);
|
|
487
|
+
let replayed = 0;
|
|
488
|
+
for (let round = 0; held && round < 8; round += 1) {
|
|
489
|
+
try {
|
|
490
|
+
if (!adoptReplayLock(projectDir, held)) return replayed;
|
|
491
|
+
reclaimOrphans(projectDir);
|
|
492
|
+
const resolution = resolveProjectStore({ projectDir });
|
|
493
|
+
const store = makeStoreFactory(now() + budgetMs)({ projectDir, requestedStorePath: resolution.canonicalAgentDbPath });
|
|
494
|
+
for (const snapshot of store.outbox.pendingSnapshots()) {
|
|
495
|
+
if (!refreshReplayLock(projectDir, held)) return replayed;
|
|
496
|
+
store.outbox.markCommitted(store.appendExact(snapshot));
|
|
497
|
+
replayed += 1;
|
|
498
|
+
}
|
|
499
|
+
for (const file of queuedCaptures(projectDir)) {
|
|
500
|
+
if (!refreshReplayLock(projectDir, held)) return replayed;
|
|
501
|
+
const claimed = claimQueued(file);
|
|
502
|
+
if (!claimed) continue;
|
|
503
|
+
onClaim?.(claimed);
|
|
504
|
+
if (!refreshReplayLock(projectDir, held)) {
|
|
505
|
+
returnClaim(claimed);
|
|
506
|
+
return replayed;
|
|
507
|
+
}
|
|
508
|
+
let job = null;
|
|
509
|
+
try { job = JSON.parse(fs.readFileSync(claimed, 'utf8')); } catch { /* torn: dropped below */ }
|
|
510
|
+
try {
|
|
511
|
+
if (job) runCapture(projectDir, job.event, { rawInput: JSON.stringify(job.payload), host: job.host,
|
|
512
|
+
budgetMs, makeStoreFactory, now, ordered: held, writeMetadata: false,
|
|
513
|
+
captureTurn: () => ({ recorded: false, skipped: 'detached replay' }) });
|
|
514
|
+
} catch { /* a failed capture leaves its own snapshot durable in the outbox */ }
|
|
515
|
+
try { fs.rmSync(claimed, { force: true }); } catch { /* best effort */ }
|
|
516
|
+
}
|
|
517
|
+
} catch { /* the debt stays durable; the next boundary hands it on again */ } finally {
|
|
518
|
+
releaseReplayLock(projectDir, held);
|
|
519
|
+
}
|
|
520
|
+
held = queuedWork(projectDir) ? takeReplayLock(projectDir) : null;
|
|
521
|
+
}
|
|
522
|
+
return replayed;
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
if (process.argv[1] && path.resolve(process.argv[1]).endsWith('session-snapshot-hook.mjs') && process.argv[2] === '--replay-outbox') {
|
|
526
|
+
try { runOutboxReplay({ projectDir: process.cwd() }); } catch { /* the debt stays durable in the outbox */ }
|
|
527
|
+
} else if (process.argv[1] && path.resolve(process.argv[1]).endsWith('session-snapshot-hook.mjs')) {
|
|
182
528
|
// projectDirectory() is the SAME derivation the Console's detector uses. Deriving it here
|
|
183
529
|
// independently is what let this hook write a receipt the Console then reported as missing (#85).
|
|
184
530
|
const rawInput = fs.readFileSync(0, 'utf8');
|
|
@@ -93,13 +93,29 @@ export const knowledgeFacts = ({ env = process.env, home, now = Date.now() } = {
|
|
|
93
93
|
lockMs: Date.parse(json(auto.lockFile)?.at || '') || mtimeMs(auto.lockFile) };
|
|
94
94
|
};
|
|
95
95
|
|
|
96
|
+
/**
|
|
97
|
+
* agentic-kit ownership is a CLAIM in kit.json, not a delivery: on the owner's Mac (2026-09-30)
|
|
98
|
+
* kit.json said ruvnetBrain:true while agentic-kit scheduled nothing, so the self-heal stood down
|
|
99
|
+
* forever and the knowledge base aged by hand only. Ownership is honoured only while an update is
|
|
100
|
+
* PROVEN inside this window (a successful refresh receipt or a CURRENT --check verdict); 36h leaves
|
|
101
|
+
* the self-heal 12h to land one before the 48h invariant breaks.
|
|
102
|
+
*/
|
|
103
|
+
export const AGENTIC_KIT_PROOF_HOURS = 36;
|
|
104
|
+
/** 'none' | 'delivering' (kit.json claims it AND an update is proven) | 'not-delivering'. */
|
|
105
|
+
export const agenticKitUpdates = ({ home, facts }) => {
|
|
106
|
+
if (!updateOwnedByAgenticKit(home)) return 'none';
|
|
107
|
+
return facts.provenWithin(AGENTIC_KIT_PROOF_HOURS) ? 'delivering' : 'not-delivering';
|
|
108
|
+
};
|
|
109
|
+
|
|
96
110
|
/** Why the SessionStart knowledge auto-update may NEVER run on this machine ('' = it may). */
|
|
97
111
|
export const autoUpdateOptOut = ({ env = process.env, home, facts }) => {
|
|
98
112
|
const flag = String(env.RUVNET_AUTO_UPDATE || '').toLowerCase();
|
|
99
113
|
if (flag === 'off') return 'RUVNET_AUTO_UPDATE=off';
|
|
100
114
|
if (env.RUVNET_BRAIN_TEST === '1' && flag !== 'on') return 'test mode';
|
|
101
115
|
if (read(path.join(facts.brainHome, '.auto-update-pref')).trim() === 'no') return 'you answered no to background auto-update';
|
|
102
|
-
if (
|
|
116
|
+
if (agenticKitUpdates({ home, facts }) === 'delivering') {
|
|
117
|
+
return `agentic-kit owns updates and one is proven within ${AGENTIC_KIT_PROOF_HOURS}h: ak sync`;
|
|
118
|
+
}
|
|
103
119
|
if (!exists(path.join(facts.kbDir, 'forge-update.mjs'))) return 'this install predates the self-updater';
|
|
104
120
|
return '';
|
|
105
121
|
};
|
|
@@ -122,8 +138,10 @@ export const knowledgeCurrency = ({ env = process.env, home, now = Date.now(), w
|
|
|
122
138
|
const ageKnown = Number.isFinite(builtMs);
|
|
123
139
|
if (!failing && proven) return '';
|
|
124
140
|
if (!failing && ageKnown && hours(builtMs) <= windowHours) return '';
|
|
125
|
-
const
|
|
126
|
-
const
|
|
141
|
+
const kit = agenticKitUpdates({ home, facts });
|
|
142
|
+
const agentKit = kit === 'delivering';
|
|
143
|
+
// An agentic-kit machine must never be told to also --enable-nightly (one owner per machine).
|
|
144
|
+
const scheduled = kit !== 'none' || readNightlyRegistration({ brainHome }).ok;
|
|
127
145
|
const parts = [ageKnown ? `knowledge base built ${day(builtMs)} (${age(builtMs)})`
|
|
128
146
|
: 'knowledge base age UNKNOWN (SOURCE.json missing or unreadable)'];
|
|
129
147
|
if (autoFailed) {
|
|
@@ -138,6 +156,9 @@ export const knowledgeCurrency = ({ env = process.env, home, now = Date.now(), w
|
|
|
138
156
|
parts.push(history.receipts
|
|
139
157
|
? `${history.failuresSinceSuccess} failed run(s) since the last success (${history.lastSuccess ? day(history.lastSuccess.at) : 'none recorded'})`
|
|
140
158
|
: 'no refresh has ever run on this machine');
|
|
159
|
+
if (kit === 'not-delivering') {
|
|
160
|
+
parts.push(`agentic-kit claims updates (kit.json ruvnetBrain:true) but no update is proven in ${AGENTIC_KIT_PROOF_HOURS}h, so the Brain's own self-heal runs instead`);
|
|
161
|
+
}
|
|
141
162
|
if (!scheduled) parts.push('no nightly refresh is scheduled');
|
|
142
163
|
if (history.unreadable) parts.push(`${history.unreadable} unreadable receipt(s)`);
|
|
143
164
|
const fix = agentKit ? 'ak sync' : scheduled ? 'npx ruvnet-brain@latest --update'
|
|
@@ -145,7 +145,7 @@ export const heartbeat = ({ env, hookDir, stateDir, home, running, seedDispatche
|
|
|
145
145
|
if (pref === 'yes' && exists(path.join(kbDir, 'forge-update.mjs'))) {
|
|
146
146
|
const kbLog = path.join(stateDir, '.last-kb-check.log');
|
|
147
147
|
if (/\bBEHIND\b/.test(read(kbLog))) {
|
|
148
|
-
emit('[RuvNet Brain — a newer knowledge bundle is available. It is signed (Ed25519) and the updater verifies that signature before extracting anything. We do NOT auto-apply it: applying replaces executable tool files, which is your call. To update:
|
|
148
|
+
emit('[RuvNet Brain — a newer knowledge bundle is available. It is signed (Ed25519) and the updater verifies that signature before extracting anything. We do NOT auto-apply it: applying replaces executable tool files, which is your call. To update: npx ruvnet-brain@latest --update]');
|
|
149
149
|
}
|
|
150
150
|
// S2 (ONE CURRENCY VERDICT): --result-file records the SAME structured verdict --check/--apply
|
|
151
151
|
// and bin/install.mjs already read (forge-update.mjs's currencyVerdict()), at the well-known path
|
|
@@ -53,10 +53,19 @@ const RECEIPTS = path.join(BRAIN_HOME, 'update-receipts.jsonl');
|
|
|
53
53
|
const LEASES = path.join(BRAIN_HOME, 'leases');
|
|
54
54
|
const DEV = path.join(BRAIN_HOME, 'dev.json');
|
|
55
55
|
const SEEDED = path.join(BRAIN_HOME, '.spine-seeded');
|
|
56
|
+
// Claude Code honours CLAUDE_CONFIG_DIR for its whole config tree, plugins included; Codex honours CODEX_HOME.
|
|
57
|
+
const CLAUDE_CONFIG = process.env.CLAUDE_CONFIG_DIR || path.join(os.homedir(), '.claude');
|
|
58
|
+
const CODEX_CONFIG = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
|
|
56
59
|
const PLUGIN_CACHES = [
|
|
57
|
-
path.join(
|
|
58
|
-
path.join(
|
|
60
|
+
path.join(CLAUDE_CONFIG, 'plugins', 'cache', 'ruvnet-brain', 'ruvnet-brain'),
|
|
61
|
+
path.join(CODEX_CONFIG, 'plugins', 'cache', 'ruvnet-brain', 'ruvnet-brain'),
|
|
59
62
|
];
|
|
63
|
+
// Evidence that a host has been pointed at the Brain's plugin at all (marketplace registered or plugin
|
|
64
|
+
// cache created) — even when `plugin install` then failed and staged nothing.
|
|
65
|
+
const HOST_PLUGIN_EVIDENCE = [CLAUDE_CONFIG, CODEX_CONFIG].flatMap((root) => [
|
|
66
|
+
path.join(root, 'plugins', 'marketplaces', 'ruvnet-brain'),
|
|
67
|
+
path.join(root, 'plugins', 'cache', 'ruvnet-brain'),
|
|
68
|
+
]);
|
|
60
69
|
const LEASE_FRESH_MS = 6 * 3600_000; // a lease older than 6h is stale (its process is long gone)
|
|
61
70
|
|
|
62
71
|
const argv = process.argv.slice(2);
|
|
@@ -425,6 +434,17 @@ function main() {
|
|
|
425
434
|
console.log(`already on ${activeNow.version}, at or above requested ${expectedVersion} — nothing to apply.`);
|
|
426
435
|
return 0;
|
|
427
436
|
}
|
|
437
|
+
// NO HOST AT ALL is not a stale spine. With no active spine and no payload of ANY version in
|
|
438
|
+
// any host cache, nothing was ever seeded, so nothing can be behind (a desktop-app/IDE-extension
|
|
439
|
+
// customer whose shell has no host CLI). A host cache holding some OTHER version still fails
|
|
440
|
+
// closed below — issue #64's exact-selection guard is untouched.
|
|
441
|
+
// Existence, not a parse: a corrupt active.json is a damaged spine and still fails closed below.
|
|
442
|
+
// A host that registered the Brain's plugin but staged nothing (its `plugin install` failed) is
|
|
443
|
+
// NOT "no host": that is a failed install and must keep failing.
|
|
444
|
+
if (!fs.existsSync(ACTIVE) && !newestStagedCC() && !HOST_PLUGIN_EVIDENCE.some((dir) => fs.existsSync(dir))) {
|
|
445
|
+
console.log(`no host has staged a payload and no spine is active — nothing to converge for ${expectedVersion}.`);
|
|
446
|
+
return 0;
|
|
447
|
+
}
|
|
428
448
|
console.error(`✗ no staged host payload exactly matches expected version ${expectedVersion} — spine unchanged`);
|
|
429
449
|
return 1;
|
|
430
450
|
}
|