ruvnet-brain 4.3.26 → 4.3.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +2 -2
  2. package/bin/install.mjs +19 -4
  3. package/data/model-catalog.json +1 -1
  4. package/docs/RELEASE-NOTES-4.0.md +4 -3
  5. package/kb/capability-only.mjs +27 -0
  6. package/kb/capability-summaries/cognitum-ruos/CAPABILITIES.md +26 -0
  7. package/kb/verify-citation.mjs +16 -4
  8. package/package.json +5 -1
  9. package/plugin/.claude-plugin/plugin.json +1 -1
  10. package/plugin/.codex-plugin/plugin.json +1 -1
  11. package/plugin/docs/RELEASE-NOTES-4.0.md +4 -3
  12. package/plugin/hooks/codex-hooks.json +6 -1
  13. package/plugin/hooks/hook-contracts.json +24 -2
  14. package/plugin/hooks/hooks.json +6 -1
  15. package/plugin/scripts/capacity-aware-parallel-work.mjs +200 -0
  16. package/plugin/scripts/codex-hook-adapter.mjs +37 -0
  17. package/plugin/scripts/continuity-hook-policy.mjs +4 -0
  18. package/plugin/scripts/coverage-integrity.mjs +17 -0
  19. package/plugin/scripts/hook-shim.mjs +1 -0
  20. package/plugin/scripts/lesson-gate.mjs +4 -1
  21. package/plugin/scripts/project-progression-reader.mjs +12 -1
  22. package/plugin/scripts/project-progression-session-start.mjs +24 -6
  23. package/plugin/scripts/project-progression-store.mjs +76 -4
  24. package/plugin/skills/release-proof/SKILL.md +28 -4
  25. package/plugin/skills/release-proof/scripts/release-proof.mjs +48 -24
  26. package/scripts/brain-novice-50.mjs +14 -16
  27. package/scripts/build-bundle.mjs +2 -0
  28. package/scripts/candidate-host-evidence.mjs +44 -16
  29. package/scripts/corpus-dispatch-receipt.mjs +22 -0
  30. package/scripts/corpus-reconcile.mjs +26 -4
  31. package/scripts/doc-currency.mjs +12 -4
  32. package/scripts/eval-brain.mjs +7 -5
  33. package/scripts/gist-git-transport.mjs +218 -0
  34. package/scripts/gist-receipts.mjs +168 -38
  35. package/scripts/host-install-matrix.mjs +54 -11
  36. package/scripts/ingest-gists.mjs +5 -2
  37. package/scripts/installed-brain-health.mjs +99 -0
  38. package/scripts/prepublication-evidence.mjs +8 -0
  39. package/scripts/public-inputs.mjs +2 -1
  40. package/scripts/public-verification-inputs.mjs +14 -6
  41. package/scripts/publication-receipt.mjs +71 -30
  42. package/scripts/qe/ux-suite.mjs +17 -5
  43. package/scripts/refresh-capability-only-store.mjs +143 -0
  44. package/scripts/release-qualification-contract.mjs +11 -1
  45. package/scripts/release-vector.mjs +44 -20
  46. package/scripts/run-operational-benchmark.mjs +151 -0
  47. package/scripts/run-operational-benchmark.v3.mjs +194 -0
  48. package/scripts/self-update.mjs +2 -0
  49. package/scripts/source-coverage.mjs +6 -1
  50. package/scripts/staged-host-verifier.mjs +2 -1
  51. package/scripts/sync-version.mjs +10 -2
  52. package/scripts/wired-check.mjs +73 -45
@@ -113,6 +113,18 @@ function exactNamedFile(root, name, label) {
113
113
  return trustedFile(path.join(root, name), label);
114
114
  }
115
115
 
116
+ function validHistoricalGenerationLedger(ledger, names) {
117
+ const legacy = ledger?.schemaVersion === 1;
118
+ const runtime = ledger?.schemaVersion === 2
119
+ && ledger?.kind === 'ruvnet-brain-runtime-generation-ledger'
120
+ && HEX40.test(String(ledger?.sourceSnapshot || ''));
121
+ return (legacy || runtime)
122
+ && typeof ledger?.brainVersion === 'string' && Boolean(ledger.brainVersion)
123
+ && typeof ledger?.releaseTag === 'string' && Boolean(ledger.releaseTag)
124
+ && names.length > 0
125
+ && new Set(names.map((name) => name.toLowerCase())).size === names.length;
126
+ }
127
+
116
128
  function validateCandidate({ root, bundleFile, packageFile }) {
117
129
  const coverageFile = trustedFile(path.join(root, 'COVERAGE.json'), 'candidate release coverage');
118
130
  const coverageBytes = fs.readFileSync(coverageFile);
@@ -176,9 +188,7 @@ function retrospectiveBaselineFromTree({ extractedRoot, bundleFile, expectedTag,
176
188
  const root = path.dirname(ledgerFile);
177
189
  const { value: ledger } = readJson(ledgerFile, 'historical baseline generation ledger');
178
190
  const names = Object.keys(ledger?.stores || {}).sort();
179
- if (ledger.schemaVersion !== 1 || typeof ledger.brainVersion !== 'string' || !ledger.brainVersion
180
- || typeof ledger.releaseTag !== 'string' || !ledger.releaseTag || !names.length
181
- || new Set(names.map((name) => name.toLowerCase())).size !== names.length) {
191
+ if (!validHistoricalGenerationLedger(ledger, names)) {
182
192
  fail('historical baseline generation ledger is malformed');
183
193
  }
184
194
  const archive = namedIdentity(bundleFile);
@@ -245,9 +255,7 @@ function observedBaselineFromTree({ extractedRoot, bundleFile, expectedTag, expe
245
255
  const root = path.dirname(ledgerFile);
246
256
  const { value: ledger } = readJson(ledgerFile, 'historical baseline generation ledger');
247
257
  const ledgerNames = Object.keys(ledger?.stores || {}).sort();
248
- if (ledger.schemaVersion !== 1 || typeof ledger.brainVersion !== 'string' || !ledger.brainVersion
249
- || typeof ledger.releaseTag !== 'string' || !ledger.releaseTag || !ledgerNames.length
250
- || new Set(ledgerNames.map((name) => name.toLowerCase())).size !== ledgerNames.length) {
258
+ if (!validHistoricalGenerationLedger(ledger, ledgerNames)) {
251
259
  fail('historical baseline generation ledger is malformed');
252
260
  }
253
261
  const archive = namedIdentity(bundleFile);
@@ -15,7 +15,9 @@ import { spawn, spawnSync } from 'node:child_process';
15
15
  // `fixtures.claude` described the same fixture and nothing could tell. The richer
16
16
  // post-publication proofs below (payload assertions, MCP wiring, SOURCE.json, rpcSearch)
17
17
  // stay here — they are this side's job, not duplication.
18
- import { HOST_MODES, RECEIPT_MODE_NAMES, MODE_FROM_RECEIPT_NAME, classifyDoctor, VARIANTS, createInstalledMcpSession } from './host-install-matrix.mjs';
18
+ import { HOST_MODES, RECEIPT_MODE_NAMES, MODE_FROM_RECEIPT_NAME, classifyDoctor, VARIANTS,
19
+ createInstalledMcpSession, HOST_WARMUP_TIMEOUT_MS, RELEASE_SEARCH_DEADLINE_MS, SELF_STORE_PROOF_QUERY,
20
+ SELF_STORE_PROOF_K } from './host-install-matrix.mjs';
19
21
  import { fileURLToPath, pathToFileURL } from 'node:url';
20
22
  import { evaluateCandidateReceipt, evaluatePublicationReceipt } from './release-proof.mjs';
21
23
  import { verifyPayload } from './release-payload.mjs';
@@ -30,8 +32,21 @@ import { parseRetrievalResult } from '../kb/retrieval-result.mjs';
30
32
 
31
33
  const REPO = 'stuinfla/ruvnet-brain';
32
34
  const PACKAGE = 'ruvnet-brain';
33
- const DEADLINE_MS = 30_000;
34
- const WARMUP_TIMEOUT_MS = 300_000;
35
+ const DEADLINE_MS = RELEASE_SEARCH_DEADLINE_MS;
36
+ const WARMUP_TIMEOUT_MS = HOST_WARMUP_TIMEOUT_MS;
37
+
38
+ export async function runMeasuredHostSearches(hosts, search, { warmup, after } = {}) {
39
+ const results = new Map();
40
+ for (const host of hosts) {
41
+ try {
42
+ await warmup?.(host);
43
+ results.set(host.mode, await search(host, DEADLINE_MS));
44
+ } finally {
45
+ await after?.(host);
46
+ }
47
+ }
48
+ return results;
49
+ }
35
50
 
36
51
  const sha256 = (file) => crypto.createHash('sha256').update(fs.readFileSync(file)).digest('hex');
37
52
  const readJson = (file) => JSON.parse(fs.readFileSync(file, 'utf8'));
@@ -524,40 +539,55 @@ export function livePublicationAdapter({ root = process.cwd(), candidateRoot = r
524
539
  };
525
540
  }));
526
541
 
527
- // Keep one MCP worker per host. The first search on each worker may load the local model,
528
- // but all later proofs reuse the initialized process instead of paying that startup cost
529
- // again. The fixed deadline remains strict for each actual search operation.
530
- for (const { mode, context } of hostResults) {
531
- const publicMode = MODE_FROM_RECEIPT_NAME[mode];
532
- mcpSessions.set(publicMode, createInstalledMcpSession({
533
- serverPath: findMcpServer(context.home), env: context.env, timeout: WARMUP_TIMEOUT_MS,
534
- }));
535
- }
536
542
  const searched = new Map();
537
543
  const searchInstalledHost = async ({ mode }, timeoutMs) => {
538
544
  const publicMode = MODE_FROM_RECEIPT_NAME[mode];
539
545
  const session = mcpSessions.get(publicMode);
546
+ const phase = timeoutMs > DEADLINE_MS ? 'warmup' : 'measured search';
547
+ console.log(`Public verification: ${mode} ${phase} (limit ${timeoutMs}ms)`);
540
548
  const result = await session.search({
541
- query: 'How does RuvNet Brain prove a public release artifact?', k: 5,
549
+ query: SELF_STORE_PROOF_QUERY, k: SELF_STORE_PROOF_K,
542
550
  timeoutMs,
543
551
  });
544
552
  if (result.error || !result.mcpResult || (Object.hasOwn(result, 'status') && result.status !== 0)) {
545
- throw new Error(`installed Brain search failed for ${mode}: ${result.error?.message || 'no MCP result'}`);
553
+ throw new Error(`installed Brain ${phase} failed for ${mode} after ${timeoutMs}ms: ${result.error?.message || 'no MCP result'}`);
546
554
  }
547
- if (!/repo=/i.test(result.stdout) || !/path\s*:/i.test(result.stdout)) {
555
+ if (!/repo\s*=\s*ruvnet-brain/i.test(result.stdout) || !/path\s*:/i.test(result.stdout)) {
548
556
  throw new Error(`installed Brain search returned no source citation for ${mode}`);
549
557
  }
550
558
  return result;
551
559
  };
552
- // Warm each worker serially. The model cache is shared, but simultaneous first-loads can
553
- // contend on slower runners (the macOS Codex-only timeout that motivated this path). The
554
- // warm-up is bounded generously and is never used as the release latency measurement.
555
- for (const host of hostResults) await searchInstalledHost(host, WARMUP_TIMEOUT_MS);
556
- // Once initialized, the three steady-state checks can run in parallel and are each held to
557
- // the strict public deadline that the receipt and aggregate validators enforce.
558
- await Promise.all(hostResults.map(async (host) => {
559
- searched.set(host.mode, await searchInstalledHost(host, DEADLINE_MS));
560
- }));
560
+ // Warm and measure one isolated host at a time, then retire its model-backed worker before
561
+ // moving to the next host. Serial RPCs alone are insufficient: keeping three independent
562
+ // workers resident still competes for memory and CPU during the measured query on macOS.
563
+ const measuredSearches = await runMeasuredHostSearches(hostResults, searchInstalledHost, {
564
+ warmup: async ({ mode, context }) => {
565
+ for (const [openMode, session] of mcpSessions) {
566
+ if (openMode !== MODE_FROM_RECEIPT_NAME[mode]) {
567
+ await session.close();
568
+ mcpSessions.delete(openMode);
569
+ }
570
+ }
571
+ const publicMode = MODE_FROM_RECEIPT_NAME[mode];
572
+ if (!mcpSessions.has(publicMode)) {
573
+ mcpSessions.set(publicMode, createInstalledMcpSession({
574
+ serverPath: findMcpServer(context.home), env: context.env, timeout: WARMUP_TIMEOUT_MS,
575
+ }));
576
+ }
577
+ // The generous bound is for one-time readiness/model warmup only; the following
578
+ // measured query still has the unchanged strict 30-second deadline.
579
+ await searchInstalledHost({ mode }, WARMUP_TIMEOUT_MS);
580
+ },
581
+ after: async ({ mode }) => {
582
+ const publicMode = MODE_FROM_RECEIPT_NAME[mode];
583
+ if (mode !== 'dual') {
584
+ const session = mcpSessions.get(publicMode);
585
+ if (session) await session.close();
586
+ mcpSessions.delete(publicMode);
587
+ }
588
+ },
589
+ });
590
+ for (const [mode, result] of measuredSearches) searched.set(mode, result);
561
591
  await Promise.all(hostResults.map(async ({ context, installer }) => {
562
592
  await commandAsync(process.execPath, [installer, '--doctor', '--hooks'], {
563
593
  env: context.env, cwd: packageRoot, timeout: 300_000, stdio: 'inherit',
@@ -590,9 +620,9 @@ export function livePublicationAdapter({ root = process.cwd(), candidateRoot = r
590
620
  // installed process that passed the host canary instead of starting a cold verifier child.
591
621
  const session = mcpSessions.get(mode);
592
622
  const result = session
593
- ? await session.search({ query: 'How does RuvNet Brain prove a public release artifact?', k: 5, timeoutMs: DEADLINE_MS })
623
+ ? await session.search({ query: SELF_STORE_PROOF_QUERY, k: SELF_STORE_PROOF_K, timeoutMs: DEADLINE_MS })
594
624
  : await rpcSearch(server, installContext.env,
595
- 'How does RuvNet Brain prove a public release artifact?', 5, DEADLINE_MS);
625
+ SELF_STORE_PROOF_QUERY, SELF_STORE_PROOF_K, DEADLINE_MS);
596
626
  if (result.error || !result.mcpResult || (Object.hasOwn(result, 'status') && result.status !== 0)) {
597
627
  throw new Error(`installed Brain search failed for ${mode}: ${result.error?.message || 'no MCP result'}`);
598
628
  }
@@ -604,13 +634,24 @@ export function livePublicationAdapter({ root = process.cwd(), candidateRoot = r
604
634
  async searchInstalled({ mode, query, k }) {
605
635
  const context = installContexts.get(mode);
606
636
  if (!context) throw new Error(`${mode} public host is not installed`);
607
- // Keep one MCP worker per installed host. Starting a fresh worker for every canary reloads
608
- // the local embedding model repeatedly and can exceed the fixed 30-second public deadline
609
- // on macOS, even when the installed Brain itself is healthy.
637
+ // Retrieval canaries run mode-by-mode. Retire the prior mode before opening this host so
638
+ // only one model-backed MCP worker can consume resources at a time.
639
+ for (const [openMode, openSession] of mcpSessions) {
640
+ if (openMode !== mode) {
641
+ await openSession.close();
642
+ mcpSessions.delete(openMode);
643
+ }
644
+ }
610
645
  let session = mcpSessions.get(mode);
611
646
  if (!session) {
612
- session = createInstalledMcpSession({ serverPath: findMcpServer(context.home), env: context.env, timeout: DEADLINE_MS });
647
+ session = createInstalledMcpSession({ serverPath: findMcpServer(context.home), env: context.env, timeout: WARMUP_TIMEOUT_MS });
613
648
  mcpSessions.set(mode, session);
649
+ const warmed = await session.search({
650
+ query: SELF_STORE_PROOF_QUERY, k: SELF_STORE_PROOF_K, timeoutMs: WARMUP_TIMEOUT_MS,
651
+ });
652
+ if (warmed.error || !warmed.mcpResult || (Object.hasOwn(warmed, 'status') && warmed.status !== 0)) {
653
+ throw new Error(`installed Brain warmup failed for ${mode}: ${warmed.error?.message || 'no MCP result'}`);
654
+ }
614
655
  }
615
656
  const result = await session.search({ query, k, timeoutMs: DEADLINE_MS });
616
657
  return parseRetrievalResult(result.mcpResult, { query, k });
@@ -175,13 +175,22 @@ export function overBudgetRows(results, budgets) {
175
175
  return (results || []).filter((r) => timingFailure(r.label, r.ms, budgets[r.label]) !== null);
176
176
  }
177
177
 
178
+ /** Failed or missing acceptance rows are hard failures and must participate in attempt ranking. */
179
+ export function failedAcceptanceRows(acceptance) {
180
+ if (!Array.isArray(acceptance) || acceptance.length === 0) return [{ label: 'acceptance: NOT RUN' }];
181
+ return acceptance.filter((row) => row?.pass !== true);
182
+ }
183
+
178
184
  /**
179
- * Rank two attempts: fewer over-budget rows wins; ties break on lower total measured ms, so a
180
- * genuinely faster run is preferred over a marginally-less-bad one.
185
+ * Rank two attempts: fewer failed hard acceptance checks wins first, then fewer over-budget
186
+ * timings, with the lower measured total breaking remaining ties.
181
187
  */
182
188
  export function betterAttempt(a, b, budgets) {
183
189
  if (!a) return b;
184
190
  if (!b) return a;
191
+ const aa = failedAcceptanceRows(a.acceptance).length;
192
+ const ba = failedAcceptanceRows(b.acceptance).length;
193
+ if (aa !== ba) return aa < ba ? a : b;
185
194
  const oa = overBudgetRows(a.results, budgets).length;
186
195
  const ob = overBudgetRows(b.results, budgets).length;
187
196
  if (oa !== ob) return oa < ob ? a : b;
@@ -190,8 +199,8 @@ export function betterAttempt(a, b, budgets) {
190
199
  }
191
200
 
192
201
  /**
193
- * Run the render probe until an attempt clears every budget, or ATTEMPTS is exhausted; return the
194
- * best attempt seen, annotated with how many attempts it took.
202
+ * Run the render probe until timings, hard acceptance checks and harness notes are clean, or
203
+ * ATTEMPTS is exhausted; return the best attempt seen, annotated with how many attempts it took.
195
204
  */
196
205
  export async function runRenderProbeBestOf(budgets, {
197
206
  attempts = RENDER_ATTEMPTS,
@@ -200,9 +209,12 @@ export async function runRenderProbeBestOf(budgets, {
200
209
  let best = null;
201
210
  for (let i = 1; i <= attempts; i++) {
202
211
  const attempt = await run();
212
+ const acceptanceFailures = failedAcceptanceRows(attempt.acceptance);
203
213
  // `notes` means the probe could not produce a reading at all — a harness failure, not slowness.
204
214
  // Retrying it is legitimate for the same reason, but it must never be silently swallowed.
205
- if (!overBudgetRows(attempt.results, budgets).length && !(attempt.notes || []).length) {
215
+ if (!overBudgetRows(attempt.results, budgets).length
216
+ && acceptanceFailures.length === 0
217
+ && !(attempt.notes || []).length) {
206
218
  return { ...attempt, attemptsUsed: i, attemptsAllowed: attempts };
207
219
  }
208
220
  best = betterAttempt(best, attempt, budgets);
@@ -0,0 +1,143 @@
1
+ #!/usr/bin/env node
2
+ // Rebuild ruOS's release store from the curated capability summary, never from repository source.
3
+ import crypto from 'node:crypto';
4
+ import fs from 'node:fs';
5
+ import path from 'node:path';
6
+ import { spawnSync } from 'node:child_process';
7
+ import { fileURLToPath, pathToFileURL } from 'node:url';
8
+ import { buildCorpus } from '../kb/forge-corpus.mjs';
9
+ import { loadRvf } from '../kb/resolve-deps.mjs';
10
+ import { CAPABILITY_RETIRED_SUFFIXES, isCapabilityOnly } from '../kb/capability-only.mjs';
11
+
12
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
13
+ const NAME = 'cognitum-ruos';
14
+ const MODEL = 'Xenova/bge-base-en-v1.5';
15
+ const DIMENSIONS = 768;
16
+
17
+ function writeJson(file, value) {
18
+ fs.writeFileSync(file, `${JSON.stringify(value, null, 2)}\n`);
19
+ }
20
+
21
+ export function prepareCapabilityOnlyInputs(assetsDir) {
22
+ const dir = path.resolve(assetsDir);
23
+ if (!fs.statSync(dir).isDirectory()) throw new Error(`assets directory is invalid: ${dir}`);
24
+ const corpus = buildCorpus({ repo: dir, name: NAME });
25
+ if (corpus.chunks.length !== 1 || corpus.chunks[0].path !== 'CAPABILITIES.md'
26
+ || corpus.chunks[0].kind !== 'doc') {
27
+ throw new Error('curated ruOS source must produce exactly one CAPABILITIES.md document');
28
+ }
29
+
30
+ for (const suffix of CAPABILITY_RETIRED_SUFFIXES) fs.rmSync(path.join(dir, `${NAME}${suffix}`), { force: true });
31
+ for (const file of fs.readdirSync(dir)) {
32
+ if (file.startsWith(`${NAME}.big.vecs.`) || file.startsWith(`${NAME}.big.progress.`)) {
33
+ fs.rmSync(path.join(dir, file), { force: true });
34
+ }
35
+ }
36
+ const chunk = corpus.chunks[0];
37
+ writeJsonLines(path.join(dir, `${NAME}.passages.jsonl`), [{
38
+ id: chunk.id, text: chunk.text, path: chunk.path, title: chunk.title,
39
+ }]);
40
+ writeJson(path.join(dir, `${NAME}.meta.json`), {
41
+ model: MODEL, dimensions: DIMENSIONS, metric: 'cosine', generated: new Date().toISOString(),
42
+ entries: { [chunk.id]: { path: chunk.path, kind: chunk.kind, title: chunk.title, chunk: '1/1', preview: chunk.preview } },
43
+ });
44
+ return { dir, sourceText: chunk.text, id: chunk.id };
45
+ }
46
+
47
+ // Historical v4.3.26 Brain self-store embedded the implementation primer. Remove those
48
+ // vectors from the copied seed and bind the rewritten bytes before packaging.
49
+ export async function pruneCapabilityOnlySelfStore(assetsDir, { RvfDatabase = null } = {}) {
50
+ const dir = path.resolve(assetsDir);
51
+ const passagesFile = path.join(dir, 'ruvnet-brain.passages.jsonl');
52
+ const metaFile = path.join(dir, 'ruvnet-brain.meta.json');
53
+ const rvfFile = path.join(dir, 'ruvnet-brain.big.rvf');
54
+ const ledgerFile = path.join(dir, 'RVF-GENERATIONS.json');
55
+ const rows = fs.readFileSync(passagesFile, 'utf8').trim().split('\n').filter(Boolean).map(JSON.parse);
56
+ const removed = rows.filter(row => /(?:^|\/)cognitum-ruos-primer\.md$/i.test(row.path || ''));
57
+ if (!removed.length) return { removed: 0 };
58
+ if (!RvfDatabase) ({ mod: { RvfDatabase } } = loadRvf());
59
+ const db = await RvfDatabase.open(rvfFile);
60
+ try {
61
+ await db.delete(removed.map(row => row.id));
62
+ await db.compact();
63
+ } finally { await db.close(); }
64
+ const kept = rows.filter(row => !removed.some(item => item.id === row.id));
65
+ writeJsonLines(passagesFile, kept);
66
+ const meta = JSON.parse(fs.readFileSync(metaFile, 'utf8'));
67
+ const removedIds = new Set(removed.map(row => row.id));
68
+ for (const row of removed) {
69
+ const entry = meta.entries?.[row.path];
70
+ if (!entry) continue;
71
+ entry.chunkIds = (entry.chunkIds || []).filter(id => !removedIds.has(id));
72
+ if (!entry.chunkIds.length) delete meta.entries[row.path];
73
+ }
74
+ writeJson(metaFile, meta);
75
+ const ledger = JSON.parse(fs.readFileSync(ledgerFile, 'utf8'));
76
+ const bytes = fs.readFileSync(rvfFile);
77
+ ledger.stores['ruvnet-brain'] = {
78
+ ...ledger.stores['ruvnet-brain'],
79
+ sha256: crypto.createHash('sha256').update(bytes).digest('hex'), bytes: bytes.length,
80
+ builtUtc: new Date().toISOString(),
81
+ };
82
+ writeJson(ledgerFile, ledger);
83
+ return { removed: removed.length };
84
+ }
85
+
86
+ function writeJsonLines(file, rows) {
87
+ fs.writeFileSync(file, `${rows.map((row) => JSON.stringify(row)).join('\n')}\n`);
88
+ }
89
+
90
+ export function bindCapabilityOnlyGeneration(assetsDir, { builtUtc = new Date().toISOString() } = {}) {
91
+ const dir = path.resolve(assetsDir);
92
+ const rvfFile = path.join(dir, `${NAME}.big.rvf`);
93
+ const metaFile = path.join(dir, `${NAME}.meta.json`);
94
+ const ledgerFile = path.join(dir, 'RVF-GENERATIONS.json');
95
+ const embedFile = path.join(dir, `${NAME}.big.rvf.embed.json`);
96
+ const rvf = fs.readFileSync(rvfFile);
97
+ const embed = JSON.parse(fs.readFileSync(embedFile, 'utf8'));
98
+ const meta = JSON.parse(fs.readFileSync(metaFile, 'utf8'));
99
+ if (embed.model !== MODEL || embed.dimensions !== DIMENSIONS || Object.keys(meta.entries || {}).length !== 1) {
100
+ throw new Error('rebuilt ruOS store does not match the 768-dim, one-summary contract');
101
+ }
102
+ const ledger = JSON.parse(fs.readFileSync(ledgerFile, 'utf8'));
103
+ if (!ledger || ![1, 2].includes(ledger.schemaVersion) || typeof ledger.stores !== 'object' || !ledger.stores) {
104
+ throw new Error('seed runtime generation ledger is malformed');
105
+ }
106
+ const previous = ledger.stores[NAME] || {};
107
+ ledger.stores[NAME] = {
108
+ file: `${NAME}.big.rvf`,
109
+ sha256: crypto.createHash('sha256').update(rvf).digest('hex'),
110
+ bytes: rvf.length,
111
+ model: MODEL,
112
+ dimensions: DIMENSIONS,
113
+ sourceCommit: previous.sourceCommit ?? null,
114
+ builtUtc,
115
+ };
116
+ writeJson(ledgerFile, ledger);
117
+ return ledger.stores[NAME];
118
+ }
119
+
120
+ export async function refreshCapabilityOnlyStore(assetsDir) {
121
+ const { dir } = prepareCapabilityOnlyInputs(assetsDir);
122
+ const selfStore = await pruneCapabilityOnlySelfStore(dir);
123
+ const script = path.join(ROOT, 'kb/forge-big.mjs');
124
+ const result = spawnSync(process.execPath, [script, 'both', '--dir', dir, '--name', NAME], {
125
+ cwd: ROOT, encoding: 'utf8', stdio: 'inherit', env: process.env,
126
+ });
127
+ if (result.error) throw result.error;
128
+ if (result.status !== 0) throw new Error(`ruOS capability-only RVF build exited ${result.status}`);
129
+ return { ...bindCapabilityOnlyGeneration(dir), selfStore };
130
+ }
131
+
132
+ async function main(argv) {
133
+ const index = argv.indexOf('--assets');
134
+ const assetsDir = index >= 0 ? argv[index + 1] : null;
135
+ if (!assetsDir) throw new Error('Usage: refresh-capability-only-store.mjs --assets <seed-kb-directory>');
136
+ if (!isCapabilityOnly(NAME)) throw new Error('configured store is not capability-only');
137
+ const result = await refreshCapabilityOnlyStore(assetsDir);
138
+ console.log(JSON.stringify({ store: NAME, ...result }));
139
+ }
140
+
141
+ if (process.argv[1] && pathToFileURL(path.resolve(process.argv[1])).href === import.meta.url) {
142
+ try { await main(process.argv.slice(2)); } catch (error) { console.error(`[capability-store] ${error.message}`); process.exitCode = 1; }
143
+ }
@@ -17,10 +17,13 @@ export const RELEASE_REQUIREMENTS = Object.freeze({
17
17
  "tests/unit/release-identity-invariants.test.mjs",
18
18
  "tests/unit/release-transaction.test.mjs",
19
19
  "tests/unit/prepublication-evidence.test.mjs",
20
+ "tests/unit/candidate-host-evidence.test.mjs",
21
+ "tests/unit/host-install-matrix-concurrency.test.mjs",
20
22
  "tests/unit/integration-evidence.test.mjs",
21
23
  "tests/unit/qualified-candidate-check.test.mjs",
22
24
  "tests/unit/release-qualification.test.mjs",
23
- "tests/unit/development-push-boundary.test.mjs"
25
+ "tests/unit/development-push-boundary.test.mjs",
26
+ "tests/unit/protected-release-workflow.test.mjs"
24
27
  ]
25
28
  },
26
29
  {
@@ -52,6 +55,13 @@ export const RELEASE_REQUIREMENTS = Object.freeze({
52
55
  "tests/unit/nightly-refresh-launcher.test.mjs",
53
56
  "tests/unit/nightly-two-run-proof.test.mjs"
54
57
  ]
58
+ },
59
+ {
60
+ "id": "ux-hard-acceptance",
61
+ "reason": "Retry accounting preserves hard UI acceptance failures and chooses only a clean measured attempt",
62
+ "files": [
63
+ "tests/unit/ux-render-best-of-n.test.mjs"
64
+ ]
55
65
  }
56
66
  ],
57
67
  "integration": [
@@ -21,6 +21,7 @@ import { spawnSync } from 'node:child_process';
21
21
  import fs from 'node:fs';
22
22
  import path from 'node:path';
23
23
  import { fileURLToPath } from 'node:url';
24
+ import { performance } from 'node:perf_hooks';
24
25
 
25
26
  export const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
26
27
 
@@ -276,35 +277,58 @@ export function headSha() {
276
277
  /** Strings a non-PASS verdict mechanically bans from release surfaces (ADR-058). */
277
278
  export const BANNED_WHEN_DEGRADED = ['healthy', 'proven', 'all pass'];
278
279
 
279
- export async function evaluate(invariants = INVARIANTS, options = {}) {
280
+ export async function evaluate(invariants = INVARIANTS, options = {}, { onInvariantStart, onInvariantComplete } = {}) {
280
281
  const lineage = candidateLineage(options.root || ROOT);
281
282
  const sha = lineage.sha;
282
283
  const results = [];
283
284
  for (const inv of invariants) {
285
+ onInvariantStart?.({ name: inv.name, dimension: inv.dimension });
286
+ const started = performance.now();
284
287
  const r = await inv.detect(options);
285
- results.push({ name: inv.name, dimension: inv.dimension, state: r.state, why: r.why, sha });
288
+ const elapsedMs = Math.max(0, Math.round(performance.now() - started));
289
+ const result = { name: inv.name, dimension: inv.dimension, state: r.state, why: r.why, sha, elapsedMs };
290
+ results.push(result);
291
+ onInvariantComplete?.(result);
286
292
  }
287
293
  return { sha, lineage, results, verdict: verdictWithLineage(results, lineage) };
288
294
  }
289
295
 
296
+ /** Map the vector verdict to the process contract: only PASS exits successfully. */
297
+ export function exitCodeForVerdict(verdict) {
298
+ return verdict === 'PASS' ? 0 : 1;
299
+ }
300
+
301
+ /** Render either CLI format from one already-evaluated result; rendering never reruns detectors. */
302
+ export function formatVectorOutput(result, { json = false } = {}) {
303
+ if (json) return JSON.stringify(result, null, 2);
304
+ const mark = { PASS: '✓', FAIL: '✗', UNKNOWN: '?' };
305
+ const rows = result.results.map((r) =>
306
+ ` ${mark[r.state]} ${r.state.padEnd(7)} ${r.dimension.padEnd(3)} ${r.name.padEnd(22)} (${r.elapsedMs ?? '—'}ms) ${r.why}`);
307
+ const lines = [
308
+ '',
309
+ ` release vector @ ${result.sha.slice(0, 7)}`,
310
+ '',
311
+ ...rows,
312
+ ` lineage: tree ${result.lineage.tree.slice(0, 12)} · ${result.lineage.dirty ? 'DIRTY (release-blocking)' : 'clean'}`,
313
+ '',
314
+ ` verdict: ${result.verdict} (vector MINIMUM over ${result.results.length} invariants — never an average)`,
315
+ ];
316
+ if (result.verdict !== 'PASS') {
317
+ lines.push(` release metadata must read DEGRADED; these strings are banned: ${BANNED_WHEN_DEGRADED.map((s) => `"${s}"`).join(', ')}`);
318
+ }
319
+ return `${lines.join('\n')}\n`;
320
+ }
321
+
290
322
  if (process.argv[1] && fileURLToPath(import.meta.url) === path.resolve(process.argv[1])) {
291
- const { sha, lineage, results, verdict } = await evaluate();
323
+ const timings = process.argv.includes('--timings');
324
+ const emitTiming = (phase, data) => process.stderr.write(`${JSON.stringify({
325
+ schema: 'ruvnet-brain.release-vector.timing', phase, ...data,
326
+ })}\n`);
327
+ const result = await evaluate(INVARIANTS, {}, timings ? {
328
+ onInvariantStart: ({ name, dimension }) => emitTiming('start', { name, dimension }),
329
+ onInvariantComplete: ({ name, dimension, elapsedMs }) => emitTiming('complete', { name, dimension, elapsedMs }),
330
+ } : {});
292
331
  const json = process.argv.includes('--json');
293
- if (json) {
294
- console.log(JSON.stringify({ sha, lineage, verdict, results }, null, 2));
295
- } else {
296
- const mark = { PASS: '✓', FAIL: '✗', UNKNOWN: '?' };
297
- console.log(`\n release vector @ ${sha.slice(0, 7)}\n`);
298
- for (const r of results) {
299
- console.log(` ${mark[r.state]} ${r.state.padEnd(7)} ${r.dimension.padEnd(3)} ${r.name.padEnd(22)} ${r.why}`);
300
- }
301
- console.log(` lineage: tree ${lineage.tree.slice(0, 12)} · ${lineage.dirty ? 'DIRTY (release-blocking)' : 'clean'}`);
302
- console.log(`\n verdict: ${verdict} (vector MINIMUM over ${results.length} invariants — never an average)`);
303
- if (verdict !== 'PASS') {
304
- console.log(` release metadata must read DEGRADED; these strings are banned: ${BANNED_WHEN_DEGRADED.map((s) => `"${s}"`).join(', ')}\n`);
305
- } else {
306
- console.log('');
307
- }
308
- }
309
- process.exitCode = verdict === 'PASS' ? 0 : 1;
332
+ console.log(formatVectorOutput(result, { json }));
333
+ process.exitCode = exitCodeForVerdict(result.verdict);
310
334
  }
@@ -0,0 +1,151 @@
1
+ #!/usr/bin/env node
2
+ // Source-span operational benchmark. Each receipt preserves the exact query, raw customer-path
3
+ // output, citation resolution, oracle match, and latency. It never edits historical eval baselines.
4
+ import fs from 'node:fs';
5
+ import os from 'node:os';
6
+ import path from 'node:path';
7
+ import { execFile } from 'node:child_process';
8
+ import { promisify } from 'node:util';
9
+ import { performance } from 'node:perf_hooks';
10
+ import { fileURLToPath, pathToFileURL } from 'node:url';
11
+ import {
12
+ OPERATIONAL_FIXTURES,
13
+ gradeOperationalFixture,
14
+ latencyDistribution,
15
+ verifyFixtureSourceSupport,
16
+ } from '../evals/operational-benchmark.v2.mjs';
17
+
18
+ const execFileAsync = promisify(execFile);
19
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
20
+ const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
21
+
22
+ export async function runOperationalBenchmark({ fixtures = OPERATIONAL_FIXTURES, kb = KB, timeoutMs = 240_000 } = {}) {
23
+ const reader = path.join(kb, 'forge-ask-all.mjs');
24
+ const verifierPath = path.join(kb, 'verify-citation.mjs');
25
+ if (!fs.existsSync(reader) || !fs.existsSync(verifierPath)) throw new Error(`RuvNet Brain runtime or verifier missing under ${kb}`);
26
+ const { verifyGrounding } = await import(pathToFileURL(verifierPath).href);
27
+ const oracleFiles = new Map();
28
+ for (const fixture of fixtures) {
29
+ if (!fixture.expectedFact) continue;
30
+ const oracleFile = path.join(ROOT, fixture.oraclePath || 'kb/capability-cards.md');
31
+ if (!fs.existsSync(oracleFile)) throw new Error(`Operational source oracle missing for ${fixture.id}: ${oracleFile}`);
32
+ oracleFiles.set(oracleFile, true);
33
+ const oracleText = fs.readFileSync(oracleFile, 'utf8');
34
+ const normalize = (value) => String(value).replace(/\s+/g, ' ').trim();
35
+ if (fixture.expectedFact && !normalize(oracleText).includes(normalize(fixture.expectedFact))) {
36
+ throw new Error(`Source oracle drift for ${fixture.id}: expected fact is absent from ${oracleFile}`);
37
+ }
38
+ }
39
+ const receipts = [];
40
+ for (const fixture of fixtures) {
41
+ if (fixture.availability) {
42
+ receipts.push({ fixtureId: fixture.id, class: fixture.class, query: fixture.query, availability: fixture.availability, grade: { pass: false, reason: 'fixture-not-measurable-with-available-source-oracle' } });
43
+ process.stderr.write(`UNAVAILABLE ${fixture.id} ${fixture.availability}\n`);
44
+ continue;
45
+ }
46
+ const started = performance.now();
47
+ let output = '';
48
+ let stderr = '';
49
+ let error = null;
50
+ let processOk = false;
51
+ let processExitCode = null;
52
+ try {
53
+ const args = [reader, '--dir', kb, '--q', fixture.query, '--k', '5'];
54
+ if (process.env.EVAL_FULL_CORPUS !== '1') args.push('--bounded');
55
+ const result = await execFileAsync(process.execPath, args, { cwd: kb, timeout: timeoutMs, maxBuffer: 64 * 1024 * 1024, env: process.env });
56
+ output = String(result.stdout ?? '');
57
+ stderr = String(result.stderr ?? '');
58
+ processOk = true;
59
+ processExitCode = 0;
60
+ } catch (caught) {
61
+ error = caught?.message ?? String(caught);
62
+ output = String(caught?.stdout ?? '');
63
+ stderr = String(caught?.stderr ?? '');
64
+ processExitCode = Number.isInteger(caught?.code) ? caught.code : null;
65
+ }
66
+ const elapsedMs = Math.round(performance.now() - started);
67
+ const verification = await verifyGrounding(output, kb);
68
+ const sourceSupport = processOk ? await verifyFixtureSourceSupport(fixture, verification, kb) : null;
69
+ const grade = gradeOperationalFixture(fixture, { output, verification, sourceSupport, processOk });
70
+ receipts.push({
71
+ fixtureId: fixture.id,
72
+ class: fixture.class,
73
+ query: fixture.query,
74
+ expectedRepos: fixture.expectedRepos,
75
+ expectedFact: fixture.expectedFact,
76
+ elapsedMs,
77
+ processOk,
78
+ processExitCode,
79
+ error,
80
+ stderr,
81
+ verification,
82
+ sourceSupport,
83
+ grade,
84
+ rawOutput: output,
85
+ });
86
+ process.stderr.write(`${grade.pass ? 'PASS' : 'FAIL'} ${fixture.id} ${elapsedMs}ms\n`);
87
+ }
88
+ const classes = Object.fromEntries([...new Set(fixtures.map((fixture) => fixture.class))].map((name) => [
89
+ name,
90
+ {
91
+ pass: receipts.filter((row) => row.class === name && !row.availability && row.grade.pass).length,
92
+ n: receipts.filter((row) => row.class === name && !row.availability).length,
93
+ unavailable: receipts.filter((row) => row.class === name && row.availability).length,
94
+ latency: latencyDistribution(receipts.filter((row) => row.class === name && !row.availability).map((row) => row.elapsedMs)),
95
+ },
96
+ ]));
97
+ return {
98
+ schema: 'ruvnet-brain-operational-benchmark/v2',
99
+ generatedAt: new Date().toISOString(),
100
+ runtime: {
101
+ kb: path.resolve(kb),
102
+ nodeExecutable: process.execPath,
103
+ nodeVersion: process.version,
104
+ readerSha256: await hashFile(reader),
105
+ verifierSha256: await hashFile(verifierPath),
106
+ fixtureSha256: await hashFile(path.join(ROOT, 'evals', 'operational-benchmark.v2.mjs')),
107
+ oracleSha256: Object.fromEntries(await Promise.all([...oracleFiles.keys()].map(async (file) => [path.relative(ROOT, file), await hashFile(file)]))),
108
+ sourceManifestSha256: await hashExistingFiles(kb, ['SOURCE.json', 'RVF-GENERATIONS.json', 'PUBLIC-RVF-GENERATIONS.json']),
109
+ },
110
+ evaluationConfig: { lane: process.env.EVAL_FULL_CORPUS === '1' ? 'full-corpus' : 'bounded', k: 5, timeoutMs, sequential: true },
111
+ identityScope: 'Entry-point, verifier, checked manifests, capability-card oracle, and each evidence-bearing passage store are SHA-256 bound. This is not a signed release archive or a hash of every transitive package/model input.',
112
+ claimBoundary: { sourceSupportedRetrievalUtility: 'measured', generatedAnswerUsefulness: 'UNKNOWN; this retrieval tool does not provide a generated answer to grade' },
113
+ classes,
114
+ passed: receipts.filter((row) => row.grade.pass).length,
115
+ total: receipts.filter((row) => !row.availability).length,
116
+ unavailable: receipts.filter((row) => row.availability).map(({ fixtureId, availability }) => ({ fixtureId, availability })),
117
+ receipts,
118
+ };
119
+ }
120
+
121
+ async function hashFile(file) {
122
+ const { createHash } = await import('node:crypto');
123
+ return createHash('sha256').update(fs.readFileSync(file)).digest('hex');
124
+ }
125
+
126
+ async function hashExistingFiles(dir, names) {
127
+ const result = {};
128
+ for (const name of names) {
129
+ const file = path.join(dir, name);
130
+ if (fs.existsSync(file)) result[name] = await hashFile(file);
131
+ }
132
+ return result;
133
+ }
134
+
135
+ async function main() {
136
+ const classIndex = process.argv.indexOf('--class');
137
+ const selectedClass = classIndex >= 0 ? process.argv[classIndex + 1] : null;
138
+ const fixtures = selectedClass ? OPERATIONAL_FIXTURES.filter((fixture) => fixture.class === selectedClass) : OPERATIONAL_FIXTURES;
139
+ if (!fixtures.length) throw new Error(`No operational fixtures for class ${selectedClass}`);
140
+ const report = await runOperationalBenchmark({ fixtures });
141
+ const outDir = path.join(ROOT, 'evals', 'operational-runs');
142
+ fs.mkdirSync(outDir, { recursive: true });
143
+ const out = path.join(outDir, `${report.generatedAt.replace(/[:.]/g, '-')}.json`);
144
+ fs.writeFileSync(out, `${JSON.stringify(report, null, 2)}\n`);
145
+ console.log(JSON.stringify({ pass: report.passed === report.total && report.unavailable.length === 0, unavailable: report.unavailable.length, passed: report.passed, total: report.total, classes: report.classes, receipt: out }, null, 2));
146
+ if (report.passed !== report.total || report.unavailable.length) process.exitCode = 1;
147
+ }
148
+
149
+ if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
150
+ main().catch((error) => { console.error(error.stack || error.message); process.exitCode = 2; });
151
+ }