ruvnet-brain 4.3.26 → 4.3.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/bin/install.mjs +19 -4
- package/data/model-catalog.json +1 -1
- package/docs/RELEASE-NOTES-4.0.md +4 -3
- package/kb/capability-only.mjs +27 -0
- package/kb/capability-summaries/cognitum-ruos/CAPABILITIES.md +26 -0
- package/kb/verify-citation.mjs +16 -4
- package/package.json +5 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/docs/RELEASE-NOTES-4.0.md +4 -3
- package/plugin/hooks/codex-hooks.json +6 -1
- package/plugin/hooks/hook-contracts.json +24 -2
- package/plugin/hooks/hooks.json +6 -1
- package/plugin/scripts/capacity-aware-parallel-work.mjs +200 -0
- package/plugin/scripts/codex-hook-adapter.mjs +37 -0
- package/plugin/scripts/continuity-hook-policy.mjs +4 -0
- package/plugin/scripts/coverage-integrity.mjs +17 -0
- package/plugin/scripts/hook-shim.mjs +1 -0
- package/plugin/scripts/lesson-gate.mjs +4 -1
- package/plugin/scripts/project-progression-reader.mjs +12 -1
- package/plugin/scripts/project-progression-session-start.mjs +24 -6
- package/plugin/scripts/project-progression-store.mjs +76 -4
- package/plugin/skills/release-proof/SKILL.md +28 -4
- package/plugin/skills/release-proof/scripts/release-proof.mjs +48 -24
- package/scripts/brain-novice-50.mjs +14 -16
- package/scripts/build-bundle.mjs +2 -0
- package/scripts/corpus-dispatch-receipt.mjs +22 -0
- package/scripts/corpus-reconcile.mjs +26 -4
- package/scripts/doc-currency.mjs +12 -4
- package/scripts/eval-brain.mjs +7 -5
- package/scripts/gist-git-transport.mjs +218 -0
- package/scripts/gist-receipts.mjs +168 -38
- package/scripts/ingest-gists.mjs +5 -2
- package/scripts/installed-brain-health.mjs +99 -0
- package/scripts/public-inputs.mjs +2 -1
- package/scripts/public-verification-inputs.mjs +14 -6
- package/scripts/refresh-capability-only-store.mjs +143 -0
- package/scripts/release-vector.mjs +44 -20
- package/scripts/run-operational-benchmark.mjs +151 -0
- package/scripts/run-operational-benchmark.v3.mjs +194 -0
- package/scripts/self-update.mjs +2 -0
- package/scripts/source-coverage.mjs +6 -1
- package/scripts/sync-version.mjs +10 -2
- package/scripts/wired-check.mjs +73 -45
|
@@ -21,6 +21,7 @@ import { spawnSync } from 'node:child_process';
|
|
|
21
21
|
import fs from 'node:fs';
|
|
22
22
|
import path from 'node:path';
|
|
23
23
|
import { fileURLToPath } from 'node:url';
|
|
24
|
+
import { performance } from 'node:perf_hooks';
|
|
24
25
|
|
|
25
26
|
export const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
26
27
|
|
|
@@ -276,35 +277,58 @@ export function headSha() {
|
|
|
276
277
|
/** Strings a non-PASS verdict mechanically bans from release surfaces (ADR-058). */
|
|
277
278
|
export const BANNED_WHEN_DEGRADED = ['healthy', 'proven', 'all pass'];
|
|
278
279
|
|
|
279
|
-
export async function evaluate(invariants = INVARIANTS, options = {}) {
|
|
280
|
+
export async function evaluate(invariants = INVARIANTS, options = {}, { onInvariantStart, onInvariantComplete } = {}) {
|
|
280
281
|
const lineage = candidateLineage(options.root || ROOT);
|
|
281
282
|
const sha = lineage.sha;
|
|
282
283
|
const results = [];
|
|
283
284
|
for (const inv of invariants) {
|
|
285
|
+
onInvariantStart?.({ name: inv.name, dimension: inv.dimension });
|
|
286
|
+
const started = performance.now();
|
|
284
287
|
const r = await inv.detect(options);
|
|
285
|
-
|
|
288
|
+
const elapsedMs = Math.max(0, Math.round(performance.now() - started));
|
|
289
|
+
const result = { name: inv.name, dimension: inv.dimension, state: r.state, why: r.why, sha, elapsedMs };
|
|
290
|
+
results.push(result);
|
|
291
|
+
onInvariantComplete?.(result);
|
|
286
292
|
}
|
|
287
293
|
return { sha, lineage, results, verdict: verdictWithLineage(results, lineage) };
|
|
288
294
|
}
|
|
289
295
|
|
|
296
|
+
/** Map the vector verdict to the process contract: only PASS exits successfully. */
|
|
297
|
+
export function exitCodeForVerdict(verdict) {
|
|
298
|
+
return verdict === 'PASS' ? 0 : 1;
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
/** Render either CLI format from one already-evaluated result; rendering never reruns detectors. */
|
|
302
|
+
export function formatVectorOutput(result, { json = false } = {}) {
|
|
303
|
+
if (json) return JSON.stringify(result, null, 2);
|
|
304
|
+
const mark = { PASS: '✓', FAIL: '✗', UNKNOWN: '?' };
|
|
305
|
+
const rows = result.results.map((r) =>
|
|
306
|
+
` ${mark[r.state]} ${r.state.padEnd(7)} ${r.dimension.padEnd(3)} ${r.name.padEnd(22)} (${r.elapsedMs ?? '—'}ms) ${r.why}`);
|
|
307
|
+
const lines = [
|
|
308
|
+
'',
|
|
309
|
+
` release vector @ ${result.sha.slice(0, 7)}`,
|
|
310
|
+
'',
|
|
311
|
+
...rows,
|
|
312
|
+
` lineage: tree ${result.lineage.tree.slice(0, 12)} · ${result.lineage.dirty ? 'DIRTY (release-blocking)' : 'clean'}`,
|
|
313
|
+
'',
|
|
314
|
+
` verdict: ${result.verdict} (vector MINIMUM over ${result.results.length} invariants — never an average)`,
|
|
315
|
+
];
|
|
316
|
+
if (result.verdict !== 'PASS') {
|
|
317
|
+
lines.push(` release metadata must read DEGRADED; these strings are banned: ${BANNED_WHEN_DEGRADED.map((s) => `"${s}"`).join(', ')}`);
|
|
318
|
+
}
|
|
319
|
+
return `${lines.join('\n')}\n`;
|
|
320
|
+
}
|
|
321
|
+
|
|
290
322
|
if (process.argv[1] && fileURLToPath(import.meta.url) === path.resolve(process.argv[1])) {
|
|
291
|
-
const
|
|
323
|
+
const timings = process.argv.includes('--timings');
|
|
324
|
+
const emitTiming = (phase, data) => process.stderr.write(`${JSON.stringify({
|
|
325
|
+
schema: 'ruvnet-brain.release-vector.timing', phase, ...data,
|
|
326
|
+
})}\n`);
|
|
327
|
+
const result = await evaluate(INVARIANTS, {}, timings ? {
|
|
328
|
+
onInvariantStart: ({ name, dimension }) => emitTiming('start', { name, dimension }),
|
|
329
|
+
onInvariantComplete: ({ name, dimension, elapsedMs }) => emitTiming('complete', { name, dimension, elapsedMs }),
|
|
330
|
+
} : {});
|
|
292
331
|
const json = process.argv.includes('--json');
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
} else {
|
|
296
|
-
const mark = { PASS: '✓', FAIL: '✗', UNKNOWN: '?' };
|
|
297
|
-
console.log(`\n release vector @ ${sha.slice(0, 7)}\n`);
|
|
298
|
-
for (const r of results) {
|
|
299
|
-
console.log(` ${mark[r.state]} ${r.state.padEnd(7)} ${r.dimension.padEnd(3)} ${r.name.padEnd(22)} ${r.why}`);
|
|
300
|
-
}
|
|
301
|
-
console.log(` lineage: tree ${lineage.tree.slice(0, 12)} · ${lineage.dirty ? 'DIRTY (release-blocking)' : 'clean'}`);
|
|
302
|
-
console.log(`\n verdict: ${verdict} (vector MINIMUM over ${results.length} invariants — never an average)`);
|
|
303
|
-
if (verdict !== 'PASS') {
|
|
304
|
-
console.log(` release metadata must read DEGRADED; these strings are banned: ${BANNED_WHEN_DEGRADED.map((s) => `"${s}"`).join(', ')}\n`);
|
|
305
|
-
} else {
|
|
306
|
-
console.log('');
|
|
307
|
-
}
|
|
308
|
-
}
|
|
309
|
-
process.exitCode = verdict === 'PASS' ? 0 : 1;
|
|
332
|
+
console.log(formatVectorOutput(result, { json }));
|
|
333
|
+
process.exitCode = exitCodeForVerdict(result.verdict);
|
|
310
334
|
}
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Source-span operational benchmark. Each receipt preserves the exact query, raw customer-path
|
|
3
|
+
// output, citation resolution, oracle match, and latency. It never edits historical eval baselines.
|
|
4
|
+
import fs from 'node:fs';
|
|
5
|
+
import os from 'node:os';
|
|
6
|
+
import path from 'node:path';
|
|
7
|
+
import { execFile } from 'node:child_process';
|
|
8
|
+
import { promisify } from 'node:util';
|
|
9
|
+
import { performance } from 'node:perf_hooks';
|
|
10
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
11
|
+
import {
|
|
12
|
+
OPERATIONAL_FIXTURES,
|
|
13
|
+
gradeOperationalFixture,
|
|
14
|
+
latencyDistribution,
|
|
15
|
+
verifyFixtureSourceSupport,
|
|
16
|
+
} from '../evals/operational-benchmark.v2.mjs';
|
|
17
|
+
|
|
18
|
+
const execFileAsync = promisify(execFile);
|
|
19
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
20
|
+
const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
|
|
21
|
+
|
|
22
|
+
export async function runOperationalBenchmark({ fixtures = OPERATIONAL_FIXTURES, kb = KB, timeoutMs = 240_000 } = {}) {
|
|
23
|
+
const reader = path.join(kb, 'forge-ask-all.mjs');
|
|
24
|
+
const verifierPath = path.join(kb, 'verify-citation.mjs');
|
|
25
|
+
if (!fs.existsSync(reader) || !fs.existsSync(verifierPath)) throw new Error(`RuvNet Brain runtime or verifier missing under ${kb}`);
|
|
26
|
+
const { verifyGrounding } = await import(pathToFileURL(verifierPath).href);
|
|
27
|
+
const oracleFiles = new Map();
|
|
28
|
+
for (const fixture of fixtures) {
|
|
29
|
+
if (!fixture.expectedFact) continue;
|
|
30
|
+
const oracleFile = path.join(ROOT, fixture.oraclePath || 'kb/capability-cards.md');
|
|
31
|
+
if (!fs.existsSync(oracleFile)) throw new Error(`Operational source oracle missing for ${fixture.id}: ${oracleFile}`);
|
|
32
|
+
oracleFiles.set(oracleFile, true);
|
|
33
|
+
const oracleText = fs.readFileSync(oracleFile, 'utf8');
|
|
34
|
+
const normalize = (value) => String(value).replace(/\s+/g, ' ').trim();
|
|
35
|
+
if (fixture.expectedFact && !normalize(oracleText).includes(normalize(fixture.expectedFact))) {
|
|
36
|
+
throw new Error(`Source oracle drift for ${fixture.id}: expected fact is absent from ${oracleFile}`);
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
const receipts = [];
|
|
40
|
+
for (const fixture of fixtures) {
|
|
41
|
+
if (fixture.availability) {
|
|
42
|
+
receipts.push({ fixtureId: fixture.id, class: fixture.class, query: fixture.query, availability: fixture.availability, grade: { pass: false, reason: 'fixture-not-measurable-with-available-source-oracle' } });
|
|
43
|
+
process.stderr.write(`UNAVAILABLE ${fixture.id} ${fixture.availability}\n`);
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
46
|
+
const started = performance.now();
|
|
47
|
+
let output = '';
|
|
48
|
+
let stderr = '';
|
|
49
|
+
let error = null;
|
|
50
|
+
let processOk = false;
|
|
51
|
+
let processExitCode = null;
|
|
52
|
+
try {
|
|
53
|
+
const args = [reader, '--dir', kb, '--q', fixture.query, '--k', '5'];
|
|
54
|
+
if (process.env.EVAL_FULL_CORPUS !== '1') args.push('--bounded');
|
|
55
|
+
const result = await execFileAsync(process.execPath, args, { cwd: kb, timeout: timeoutMs, maxBuffer: 64 * 1024 * 1024, env: process.env });
|
|
56
|
+
output = String(result.stdout ?? '');
|
|
57
|
+
stderr = String(result.stderr ?? '');
|
|
58
|
+
processOk = true;
|
|
59
|
+
processExitCode = 0;
|
|
60
|
+
} catch (caught) {
|
|
61
|
+
error = caught?.message ?? String(caught);
|
|
62
|
+
output = String(caught?.stdout ?? '');
|
|
63
|
+
stderr = String(caught?.stderr ?? '');
|
|
64
|
+
processExitCode = Number.isInteger(caught?.code) ? caught.code : null;
|
|
65
|
+
}
|
|
66
|
+
const elapsedMs = Math.round(performance.now() - started);
|
|
67
|
+
const verification = await verifyGrounding(output, kb);
|
|
68
|
+
const sourceSupport = processOk ? await verifyFixtureSourceSupport(fixture, verification, kb) : null;
|
|
69
|
+
const grade = gradeOperationalFixture(fixture, { output, verification, sourceSupport, processOk });
|
|
70
|
+
receipts.push({
|
|
71
|
+
fixtureId: fixture.id,
|
|
72
|
+
class: fixture.class,
|
|
73
|
+
query: fixture.query,
|
|
74
|
+
expectedRepos: fixture.expectedRepos,
|
|
75
|
+
expectedFact: fixture.expectedFact,
|
|
76
|
+
elapsedMs,
|
|
77
|
+
processOk,
|
|
78
|
+
processExitCode,
|
|
79
|
+
error,
|
|
80
|
+
stderr,
|
|
81
|
+
verification,
|
|
82
|
+
sourceSupport,
|
|
83
|
+
grade,
|
|
84
|
+
rawOutput: output,
|
|
85
|
+
});
|
|
86
|
+
process.stderr.write(`${grade.pass ? 'PASS' : 'FAIL'} ${fixture.id} ${elapsedMs}ms\n`);
|
|
87
|
+
}
|
|
88
|
+
const classes = Object.fromEntries([...new Set(fixtures.map((fixture) => fixture.class))].map((name) => [
|
|
89
|
+
name,
|
|
90
|
+
{
|
|
91
|
+
pass: receipts.filter((row) => row.class === name && !row.availability && row.grade.pass).length,
|
|
92
|
+
n: receipts.filter((row) => row.class === name && !row.availability).length,
|
|
93
|
+
unavailable: receipts.filter((row) => row.class === name && row.availability).length,
|
|
94
|
+
latency: latencyDistribution(receipts.filter((row) => row.class === name && !row.availability).map((row) => row.elapsedMs)),
|
|
95
|
+
},
|
|
96
|
+
]));
|
|
97
|
+
return {
|
|
98
|
+
schema: 'ruvnet-brain-operational-benchmark/v2',
|
|
99
|
+
generatedAt: new Date().toISOString(),
|
|
100
|
+
runtime: {
|
|
101
|
+
kb: path.resolve(kb),
|
|
102
|
+
nodeExecutable: process.execPath,
|
|
103
|
+
nodeVersion: process.version,
|
|
104
|
+
readerSha256: await hashFile(reader),
|
|
105
|
+
verifierSha256: await hashFile(verifierPath),
|
|
106
|
+
fixtureSha256: await hashFile(path.join(ROOT, 'evals', 'operational-benchmark.v2.mjs')),
|
|
107
|
+
oracleSha256: Object.fromEntries(await Promise.all([...oracleFiles.keys()].map(async (file) => [path.relative(ROOT, file), await hashFile(file)]))),
|
|
108
|
+
sourceManifestSha256: await hashExistingFiles(kb, ['SOURCE.json', 'RVF-GENERATIONS.json', 'PUBLIC-RVF-GENERATIONS.json']),
|
|
109
|
+
},
|
|
110
|
+
evaluationConfig: { lane: process.env.EVAL_FULL_CORPUS === '1' ? 'full-corpus' : 'bounded', k: 5, timeoutMs, sequential: true },
|
|
111
|
+
identityScope: 'Entry-point, verifier, checked manifests, capability-card oracle, and each evidence-bearing passage store are SHA-256 bound. This is not a signed release archive or a hash of every transitive package/model input.',
|
|
112
|
+
claimBoundary: { sourceSupportedRetrievalUtility: 'measured', generatedAnswerUsefulness: 'UNKNOWN; this retrieval tool does not provide a generated answer to grade' },
|
|
113
|
+
classes,
|
|
114
|
+
passed: receipts.filter((row) => row.grade.pass).length,
|
|
115
|
+
total: receipts.filter((row) => !row.availability).length,
|
|
116
|
+
unavailable: receipts.filter((row) => row.availability).map(({ fixtureId, availability }) => ({ fixtureId, availability })),
|
|
117
|
+
receipts,
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
async function hashFile(file) {
|
|
122
|
+
const { createHash } = await import('node:crypto');
|
|
123
|
+
return createHash('sha256').update(fs.readFileSync(file)).digest('hex');
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
async function hashExistingFiles(dir, names) {
|
|
127
|
+
const result = {};
|
|
128
|
+
for (const name of names) {
|
|
129
|
+
const file = path.join(dir, name);
|
|
130
|
+
if (fs.existsSync(file)) result[name] = await hashFile(file);
|
|
131
|
+
}
|
|
132
|
+
return result;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
async function main() {
|
|
136
|
+
const classIndex = process.argv.indexOf('--class');
|
|
137
|
+
const selectedClass = classIndex >= 0 ? process.argv[classIndex + 1] : null;
|
|
138
|
+
const fixtures = selectedClass ? OPERATIONAL_FIXTURES.filter((fixture) => fixture.class === selectedClass) : OPERATIONAL_FIXTURES;
|
|
139
|
+
if (!fixtures.length) throw new Error(`No operational fixtures for class ${selectedClass}`);
|
|
140
|
+
const report = await runOperationalBenchmark({ fixtures });
|
|
141
|
+
const outDir = path.join(ROOT, 'evals', 'operational-runs');
|
|
142
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
143
|
+
const out = path.join(outDir, `${report.generatedAt.replace(/[:.]/g, '-')}.json`);
|
|
144
|
+
fs.writeFileSync(out, `${JSON.stringify(report, null, 2)}\n`);
|
|
145
|
+
console.log(JSON.stringify({ pass: report.passed === report.total && report.unavailable.length === 0, unavailable: report.unavailable.length, passed: report.passed, total: report.total, classes: report.classes, receipt: out }, null, 2));
|
|
146
|
+
if (report.passed !== report.total || report.unavailable.length) process.exitCode = 1;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
150
|
+
main().catch((error) => { console.error(error.stack || error.message); process.exitCode = 2; });
|
|
151
|
+
}
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// v3 validates independently frozen source facts against corpus bytes before running retrieval.
|
|
3
|
+
// Missing corpus facts remain explicit CORPUS_GAP outcomes and block qualification.
|
|
4
|
+
import fs from 'node:fs';
|
|
5
|
+
import os from 'node:os';
|
|
6
|
+
import path from 'node:path';
|
|
7
|
+
import { execFile } from 'node:child_process';
|
|
8
|
+
import { promisify } from 'node:util';
|
|
9
|
+
import { performance } from 'node:perf_hooks';
|
|
10
|
+
import { fileURLToPath } from 'node:url';
|
|
11
|
+
import {
|
|
12
|
+
OPERATIONAL_FIXTURES_V3,
|
|
13
|
+
ORACLE_CATALOG,
|
|
14
|
+
gradeOperationalFixtureV3,
|
|
15
|
+
matchClaimSlots,
|
|
16
|
+
preflightOperationalOracle,
|
|
17
|
+
} from '../evals/operational-benchmark.v3.mjs';
|
|
18
|
+
import { verifyGrounding as evaluatorVerifyGrounding } from '../kb/verify-citation.mjs';
|
|
19
|
+
|
|
20
|
+
const execFileAsync = promisify(execFile);
|
|
21
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
22
|
+
const DEFAULT_KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
|
|
23
|
+
const { createHash } = await import('node:crypto');
|
|
24
|
+
const digestBytes = (bytes) => createHash('sha256').update(bytes).digest('hex');
|
|
25
|
+
const hashBytes = async (file) => digestBytes(fs.readFileSync(file));
|
|
26
|
+
const hashIfPresent = async (file) => fs.existsSync(file) ? hashBytes(file) : null;
|
|
27
|
+
|
|
28
|
+
async function readCatalog(catalog, catalogFile) {
|
|
29
|
+
if (catalog) return { value: catalog, file: null, sha256: digestBytes(Buffer.from(JSON.stringify(catalog))) };
|
|
30
|
+
const file = path.resolve(ROOT, catalogFile || ORACLE_CATALOG);
|
|
31
|
+
try {
|
|
32
|
+
const bytes = fs.readFileSync(file);
|
|
33
|
+
return { value: JSON.parse(bytes.toString('utf8')), file, sha256: digestBytes(bytes) };
|
|
34
|
+
}
|
|
35
|
+
catch (error) { return { value: null, file, error: error.message }; }
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export async function runOperationalBenchmarkV3({ fixtures = OPERATIONAL_FIXTURES_V3, kb = DEFAULT_KB,
|
|
39
|
+
catalog = null, catalogFile = ORACLE_CATALOG, timeoutMs = 240_000, runQuery = null,
|
|
40
|
+
verify = null, now = () => new Date().toISOString() } = {}) {
|
|
41
|
+
const reader = path.join(kb, 'forge-ask-all.mjs');
|
|
42
|
+
const verifierPath = path.join(kb, 'verify-citation.mjs');
|
|
43
|
+
if (!fs.existsSync(reader) || !fs.existsSync(verifierPath)) throw new Error(`RuvNet Brain runtime or verifier missing under ${kb}`);
|
|
44
|
+
const loaded = await readCatalog(catalog, catalogFile);
|
|
45
|
+
const catalogSha256AtStart = loaded.sha256 ?? null;
|
|
46
|
+
// Both baseline and candidate use the evaluator checkout's fixed parser/verifier. The runtime
|
|
47
|
+
// archive verifier is hashed as runtime metadata, never imported as the grader implementation.
|
|
48
|
+
const verifier = verify || evaluatorVerifyGrounding;
|
|
49
|
+
const preflight = loaded.error
|
|
50
|
+
? new Map(fixtures.map((fixture) => [fixture.id, { status: 'INVALID_ORACLE', reason: `oracle catalog cannot be read: ${loaded.error}` }]))
|
|
51
|
+
: await preflightOperationalOracle({ fixtures, catalog: loaded.value, catalogPath: loaded.file, kbDir: kb });
|
|
52
|
+
const receipts = [];
|
|
53
|
+
for (const fixture of fixtures) {
|
|
54
|
+
const oracle = preflight.get(fixture.id) || { status: 'INVALID_ORACLE', reason: 'fixture preflight result is missing' };
|
|
55
|
+
if (oracle.status !== 'PASS') {
|
|
56
|
+
const grade = gradeOperationalFixtureV3(fixture, { processOk: true, preflightStatus: oracle.status });
|
|
57
|
+
receipts.push({ fixtureId: fixture.id, class: fixture.class, query: fixture.query, preflight: oracle,
|
|
58
|
+
processOk: false, elapsedMs: null, grade });
|
|
59
|
+
process.stderr.write(`${oracle.status} ${fixture.id} ${oracle.reason}\n`);
|
|
60
|
+
continue;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
const started = performance.now();
|
|
64
|
+
let output = '';
|
|
65
|
+
let stderr = '';
|
|
66
|
+
let error = null;
|
|
67
|
+
let processExitCode = null;
|
|
68
|
+
let processOk = false;
|
|
69
|
+
try {
|
|
70
|
+
const result = runQuery
|
|
71
|
+
? await runQuery(fixture, { kb, timeoutMs })
|
|
72
|
+
: await execFileAsync(process.execPath, [reader, '--dir', kb, '--q', fixture.query, '--k', '5',
|
|
73
|
+
...(process.env.EVAL_FULL_CORPUS === '1' ? [] : ['--bounded'])],
|
|
74
|
+
{ cwd: kb, timeout: timeoutMs, maxBuffer: 64 * 1024 * 1024, env: process.env });
|
|
75
|
+
output = String(result?.stdout ?? '');
|
|
76
|
+
stderr = String(result?.stderr ?? '');
|
|
77
|
+
processOk = result?.code === undefined ? result?.status === undefined || result.status === 0 : result.code === 0;
|
|
78
|
+
processExitCode = result?.code ?? result?.status ?? 0;
|
|
79
|
+
if (!processOk) error = 'retrieval process returned a nonzero status';
|
|
80
|
+
} catch (caught) {
|
|
81
|
+
error = caught?.message ?? String(caught);
|
|
82
|
+
output = String(caught?.stdout ?? '');
|
|
83
|
+
stderr = String(caught?.stderr ?? '');
|
|
84
|
+
processExitCode = Number.isInteger(caught?.code) ? caught.code : null;
|
|
85
|
+
}
|
|
86
|
+
const elapsedMs = Math.round(performance.now() - started);
|
|
87
|
+
const verification = processOk ? await verifier(output, kb) : null;
|
|
88
|
+
const sourceSupport = fixture.class === 'negative' || fixture.class === 'ambiguity'
|
|
89
|
+
? null : matchClaimSlots(oracle, verification);
|
|
90
|
+
const grade = gradeOperationalFixtureV3(fixture, { output, verification, sourceSupport,
|
|
91
|
+
processOk, preflightStatus: oracle.status });
|
|
92
|
+
receipts.push({ fixtureId: fixture.id, class: fixture.class, query: fixture.query,
|
|
93
|
+
preflight: { status: oracle.status, resolvedSlots: oracle.resolvedSlots?.map((slot) => ({
|
|
94
|
+
id: slot.id, alternatives: slot.alternatives.map(({ repo, path: sourcePath, passageSha256,
|
|
95
|
+
store, storeSha256, sourceRowsByStore }) => ({ repo, path: sourcePath, passageSha256, store,
|
|
96
|
+
storeSha256, sourceStores: sourceRowsByStore.map(({ store: sourceStore, storeSha256: sourceStoreSha256 }) =>
|
|
97
|
+
({ store: sourceStore, storeSha256: sourceStoreSha256 })) })),
|
|
98
|
+
})) ?? [] }, elapsedMs, processOk, processExitCode, error, stderr, verification, sourceSupport,
|
|
99
|
+
grade, rawOutput: output });
|
|
100
|
+
process.stderr.write(`${grade.status} ${fixture.id} ${elapsedMs}ms\n`);
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
// Close the receipt over the inputs after replay as well as before it. If any pinned corpus
|
|
104
|
+
// bytes or oracle catalog moved mid-run, demote every affected row so a race cannot pass.
|
|
105
|
+
const corpusChanged = loaded.value?.corpus && (
|
|
106
|
+
await hashIfPresent(path.join(kb, 'ARCHIVE-MANIFEST.json')) !== loaded.value.corpus.archiveManifestSha256
|
|
107
|
+
|| await hashIfPresent(path.join(kb, 'SOURCE.json')) !== loaded.value.corpus.sourceManifestSha256);
|
|
108
|
+
const oracleChanged = loaded.file && await hashIfPresent(loaded.file) !== catalogSha256AtStart;
|
|
109
|
+
if (corpusChanged || oracleChanged) {
|
|
110
|
+
for (const row of receipts) {
|
|
111
|
+
if (row.preflight?.status !== 'PASS') continue;
|
|
112
|
+
row.preflight = { status: corpusChanged ? 'CORPUS_GAP' : 'INVALID_ORACLE',
|
|
113
|
+
reason: corpusChanged ? 'pinned corpus identity changed during replay' : 'oracle catalog bytes changed during replay' };
|
|
114
|
+
row.grade = { pass: false, status: row.preflight.status, reason: row.preflight.reason };
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
const storeIdentities = new Map();
|
|
118
|
+
for (const row of receipts) {
|
|
119
|
+
for (const slot of row.preflight?.resolvedSlots ?? []) {
|
|
120
|
+
for (const alt of slot.alternatives ?? []) {
|
|
121
|
+
storeIdentities.set(alt.store, alt.storeSha256);
|
|
122
|
+
for (const sourceStore of alt.sourceStores ?? []) storeIdentities.set(sourceStore.store, sourceStore.storeSha256);
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
for (const [store, expectedSha256] of storeIdentities) {
|
|
127
|
+
const actual = await hashIfPresent(path.join(kb, store));
|
|
128
|
+
if (actual === expectedSha256) continue;
|
|
129
|
+
for (const row of receipts) {
|
|
130
|
+
const bound = (row.preflight?.resolvedSlots ?? []).some((slot) => slot.alternatives?.some((alt) =>
|
|
131
|
+
alt.store === store || alt.sourceStores?.some((sourceStore) => sourceStore.store === store)));
|
|
132
|
+
if (!bound || row.preflight.status !== 'PASS') continue;
|
|
133
|
+
row.preflight = { status: 'CORPUS_GAP', reason: `source passage store ${store} changed or disappeared during replay` };
|
|
134
|
+
row.grade = { pass: false, status: 'CORPUS_GAP', reason: row.preflight.reason };
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
const classes = Object.fromEntries([...new Set(fixtures.map(({ class: fixtureClass }) => fixtureClass))].map((name) => {
|
|
139
|
+
const rows = receipts.filter((row) => row.class === name);
|
|
140
|
+
const measured = rows.filter((row) => row.preflight?.status === 'PASS');
|
|
141
|
+
const latencies = measured.filter((row) => Number.isFinite(row.elapsedMs)).map((row) => row.elapsedMs).sort((a, b) => a - b);
|
|
142
|
+
const percentile = (f) => latencies.length ? latencies[Math.max(0, Math.ceil(latencies.length * f) - 1)] : null;
|
|
143
|
+
return [name, { pass: measured.filter((row) => row.grade.pass).length, n: measured.length,
|
|
144
|
+
corpusGap: rows.filter((row) => row.preflight?.status === 'CORPUS_GAP').length,
|
|
145
|
+
invalidOracle: rows.filter((row) => row.preflight?.status === 'INVALID_ORACLE').length,
|
|
146
|
+
latency: { n: latencies.length, p50Ms: percentile(.5), p95Ms: percentile(.95), p99Ms: percentile(.99), maxMs: latencies.at(-1) ?? null } }];
|
|
147
|
+
}));
|
|
148
|
+
const [readerSha256, runtimeVerifierSha256, evaluatorParserSha256, archiveManifestSha256, sourceManifestSha256,
|
|
149
|
+
runnerSha256, graderSha256, querySetSha256, oracleCatalogSha256] = await Promise.all([
|
|
150
|
+
hashBytes(reader), hashBytes(verifierPath), hashBytes(path.join(ROOT, 'kb/verify-citation.mjs')),
|
|
151
|
+
hashIfPresent(path.join(kb, 'ARCHIVE-MANIFEST.json')),
|
|
152
|
+
hashIfPresent(path.join(kb, 'SOURCE.json')),
|
|
153
|
+
hashBytes(fileURLToPath(import.meta.url)), hashBytes(path.join(ROOT, 'evals/operational-benchmark.v3.mjs')),
|
|
154
|
+
Promise.resolve(digestBytes(Buffer.from(JSON.stringify(fixtures)))),
|
|
155
|
+
Promise.resolve(catalogSha256AtStart),
|
|
156
|
+
]);
|
|
157
|
+
const measured = receipts.filter((row) => row.preflight?.status === 'PASS');
|
|
158
|
+
const unavailable = receipts.filter((row) => row.preflight?.status === 'CORPUS_GAP')
|
|
159
|
+
.map(({ fixtureId, preflight: result }) => ({ fixtureId, reason: result.reason }));
|
|
160
|
+
const invalidOracle = receipts.filter((row) => row.preflight?.status === 'INVALID_ORACLE')
|
|
161
|
+
.map(({ fixtureId, preflight: result }) => ({ fixtureId, reason: result.reason }));
|
|
162
|
+
return {
|
|
163
|
+
schema: 'ruvnet-brain-operational-benchmark/v3', generatedAt: now(),
|
|
164
|
+
runtime: { kb: path.resolve(kb), nodeExecutable: process.execPath, nodeVersion: process.version,
|
|
165
|
+
readerSha256, runtimeVerifierSha256, evaluatorParserSha256, archiveManifestSha256, sourceManifestSha256,
|
|
166
|
+
runnerSha256, graderSha256, querySetSha256, oracleCatalogSha256 },
|
|
167
|
+
evaluationConfig: { lane: process.env.EVAL_FULL_CORPUS === '1' ? 'full-corpus' : 'bounded',
|
|
168
|
+
k: 5, timeoutMs, sequential: true, preflightBeforeSearch: true },
|
|
169
|
+
claimBoundary: { sourceSupportedRetrieval: 'measured', generatedAnswerUsefulness: 'UNKNOWN; this is a retrieval tool evaluation' },
|
|
170
|
+
classes, passed: measured.filter((row) => row.grade.pass).length,
|
|
171
|
+
measured: measured.length, total: fixtures.length, corpusGaps: unavailable,
|
|
172
|
+
invalidOracles: invalidOracle,
|
|
173
|
+
qualificationPass: measured.length === fixtures.length && invalidOracle.length === 0
|
|
174
|
+
&& unavailable.length === 0 && measured.every((row) => row.grade.pass),
|
|
175
|
+
receipts,
|
|
176
|
+
};
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
async function main() {
|
|
180
|
+
const kb = process.env.RUVNET_BRAIN_KB || DEFAULT_KB;
|
|
181
|
+
const report = await runOperationalBenchmarkV3({ fixtures: OPERATIONAL_FIXTURES_V3, kb });
|
|
182
|
+
const outDir = path.join(ROOT, 'evals', 'operational-runs');
|
|
183
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
184
|
+
const out = path.join(outDir, `v3-${report.generatedAt.replace(/[:.]/g, '-')}.json`);
|
|
185
|
+
fs.writeFileSync(out, `${JSON.stringify(report, null, 2)}\n`);
|
|
186
|
+
console.log(JSON.stringify({ qualificationPass: report.qualificationPass, measured: report.measured,
|
|
187
|
+
total: report.total, corpusGaps: report.corpusGaps.length, invalidOracles: report.invalidOracles.length,
|
|
188
|
+
classes: report.classes, receipt: out }, null, 2));
|
|
189
|
+
if (!report.qualificationPass) process.exitCode = 1;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
193
|
+
main().catch((error) => { console.error(error.stack || error.message); process.exitCode = 2; });
|
|
194
|
+
}
|
package/scripts/self-update.mjs
CHANGED
|
@@ -167,6 +167,8 @@ for (const r of inScope) {
|
|
|
167
167
|
else if (built === 'unknown') action = live ? 'rebuild (changed)' : 'unverified (probe failed)';
|
|
168
168
|
else if (live === null) action = 'unverified (probe failed)';
|
|
169
169
|
else if (live !== built) action = 'rebuild (changed)';
|
|
170
|
+
// The curated capability summary can change without a change to upstream code.
|
|
171
|
+
if (r.name === 'cognitum-ruos' && live) action = 'rebuild (changed)';
|
|
170
172
|
plan.push({ name: r.name, owner: r.owner, repo: r.repo, tier: r.tier, built: built?.slice(0, 12) || '—', live: live?.slice(0, 12) || '?', action });
|
|
171
173
|
}
|
|
172
174
|
if (inScope.length > 0 && probeFailures * 2 >= inScope.length) {
|
|
@@ -215,6 +215,9 @@ export function canonicalGistRows(rows = []) {
|
|
|
215
215
|
id: gist?.id,
|
|
216
216
|
updated_at: gist?.updated_at ?? null,
|
|
217
217
|
html_url: gist?.html_url,
|
|
218
|
+
// Preserve GitHub's top-level completeness indicator. `files.truncated` is a different
|
|
219
|
+
// per-file content flag; neither omission nor a missing value means the inventory is complete.
|
|
220
|
+
...(Object.hasOwn(gist || {}, 'truncated') ? { truncated: gist.truncated } : {}),
|
|
218
221
|
files: Object.fromEntries(Object.entries(gist?.files || {}).sort(([a], [b]) => a.localeCompare(b))
|
|
219
222
|
.map(([key, file]) => [key, {
|
|
220
223
|
filename: file?.filename,
|
|
@@ -409,6 +412,7 @@ export function classifyGist(gist, evidence) {
|
|
|
409
412
|
let status = 'CURRENT';
|
|
410
413
|
const reasons = [];
|
|
411
414
|
if (!evidence.rvfPresent) { status = 'MISSING'; reasons.push('ruv-gists RVF is absent'); }
|
|
415
|
+
else if (gist.truncated !== false) { status = 'UNVERIFIED'; reasons.push('GitHub gist file inventory is truncated or its completeness flag is unknown'); }
|
|
412
416
|
else if (!evidence.receipt || !source || !version) { status = 'UNVERIFIED'; reasons.push('per-gist source receipt is absent'); }
|
|
413
417
|
else if (!evidence.bytesVerified) { status = 'FAILED'; reasons.push('ruv-gists RVF bytes do not match the generation receipt'); }
|
|
414
418
|
else if (!evidence.passagesBound) { status = 'FAILED'; reasons.push('gist passages do not match the source receipt'); }
|
|
@@ -421,7 +425,8 @@ export function classifyGist(gist, evidence) {
|
|
|
421
425
|
name: Object.values(gist.files || {})[0]?.filename || gist.id,
|
|
422
426
|
url: gist.html_url,
|
|
423
427
|
disposition: 'eligible',
|
|
424
|
-
upstream: { sha: version, updatedAt: gist.updated_at,
|
|
428
|
+
upstream: { sha: version, updatedAt: gist.updated_at, truncated: gist.truncated ?? null,
|
|
429
|
+
fileCount: filenames.length, files: filenames },
|
|
425
430
|
artifact: { store: 'ruv-gists', sourceCommit: source?.versionSha || null, ingestedAt: source?.ingestedAt || ingestedAt,
|
|
426
431
|
contentDigest: source?.contentDigest || null, fileCount: source?.files?.length || null,
|
|
427
432
|
rvfSha256: evidence.receipt?.sha256 || null, bytesVerified: evidence.bytesVerified },
|
package/scripts/sync-version.mjs
CHANGED
|
@@ -33,6 +33,14 @@ export function writeLockVersion(s, version) {
|
|
|
33
33
|
return `${JSON.stringify(doc, null, 2)}\n`;
|
|
34
34
|
}
|
|
35
35
|
|
|
36
|
+
export function readExplainerBadgeVersion(s) {
|
|
37
|
+
return s.match(/·\s*v(\d+\.\d+\.\d+(?:-[\w.-]+)?)(?=\s*(?:·|<))/)?.[1] ?? null;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function writeExplainerBadgeVersion(s, version) {
|
|
41
|
+
return s.replace(/(·\s*v)\d+\.\d+\.\d+(?:-[\w.-]+)?/, `$1${version}`);
|
|
42
|
+
}
|
|
43
|
+
|
|
36
44
|
// Each target: a file, a regex to find the version-bearing line, and the corrected line.
|
|
37
45
|
const targets = [
|
|
38
46
|
{ // Codex plugin manifest — same product, same release train as the Claude manifest
|
|
@@ -71,12 +79,12 @@ const targets = [
|
|
|
71
79
|
file: 'explainer/index.html',
|
|
72
80
|
get: (s) => {
|
|
73
81
|
const ld = s.match(/"softwareVersion"\s*:\s*"([^"]+)"/)?.[1] ?? null;
|
|
74
|
-
const badge =
|
|
82
|
+
const badge = readExplainerBadgeVersion(s);
|
|
75
83
|
return ld === badge ? ld : `mixed(${ld}, ${badge})`;
|
|
76
84
|
},
|
|
77
85
|
set: (s) => s
|
|
78
86
|
.replace(/("softwareVersion"\s*:\s*)"[^"]+"/, `$1"${V}"`)
|
|
79
|
-
.replace(
|
|
87
|
+
.replace(/·\s*v\d+\.\d+\.\d+(?:-[\w.-]+)?/, (label) => writeExplainerBadgeVersion(label, V)),
|
|
80
88
|
},
|
|
81
89
|
];
|
|
82
90
|
|