ruvnet-brain 4.4.1 → 4.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/bin/install.mjs +612 -113
- package/console/app.js +178 -81
- package/console/index.html +1 -1
- package/console/install-architecture.html +1 -0
- package/console/scope.css +4 -1
- package/console/style.css +13 -0
- package/console/tips.html +4 -4
- package/kb/brain-profile.mjs +1 -0
- package/kb/corpus-release-identity.mjs +1 -1
- package/kb/forge-update.mjs +41 -17
- package/kb/model-requirements.mjs +4 -1
- package/kb/update-storage-transaction.mjs +79 -0
- package/kb/zip-extract.mjs +22 -0
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +5 -4
- package/plugin/commands/configure.md +5 -4
- package/plugin/commands/rnb-brief.md +41 -0
- package/plugin/commands/rnb.md +80 -0
- package/plugin/commands/rnbc.md +80 -0
- package/plugin/commands/rvbc.md +5 -4
- package/plugin/commands/rvcb.md +5 -4
- package/plugin/commands/whats-new.md +4 -4
- package/plugin/mcp/server.mjs +10 -2
- package/plugin/scripts/advocacy-route.mjs +59 -23
- package/plugin/scripts/anticipate.sh +4 -0
- package/plugin/scripts/brain-confirmation.mjs +258 -0
- package/plugin/scripts/brain-footprint.mjs +494 -0
- package/plugin/scripts/brain-location.mjs +47 -0
- package/plugin/scripts/capability-registry.mjs +11 -1
- package/plugin/scripts/continuity-brief.mjs +324 -0
- package/plugin/scripts/continuity-events.mjs +327 -0
- package/plugin/scripts/continuity-journal.mjs +500 -0
- package/plugin/scripts/decision-gate.mjs +56 -3
- package/plugin/scripts/footprint-io.mjs +186 -0
- package/plugin/scripts/ground-before-write.sh +8 -1
- package/plugin/scripts/ground-ruvnet.sh +103 -12
- package/plugin/scripts/grounding-answer.mjs +2 -1
- package/plugin/scripts/grounding-stamp.sh +3 -0
- package/plugin/scripts/grounding-substance.mjs +1 -1
- package/plugin/scripts/grounding-turn-evidence.mjs +17 -2
- package/plugin/scripts/hook-input.mjs +78 -4
- package/plugin/scripts/kb-copy-proof.mjs +148 -0
- package/plugin/scripts/lesson-bridge.mjs +6 -2
- package/plugin/scripts/nightly-controller.mjs +8 -1
- package/plugin/scripts/node-sqlite.mjs +41 -0
- package/plugin/scripts/package-cards.json +797 -0
- package/plugin/scripts/package-cards.rvf +0 -0
- package/plugin/scripts/package-cards.rvf.idmap.json +1 -0
- package/plugin/scripts/package-cards.rvf.meta.json +1 -0
- package/plugin/scripts/package-recommender-client.mjs +138 -0
- package/plugin/scripts/package-recommender-flag.mjs +30 -0
- package/plugin/scripts/package-recommender.mjs +391 -0
- package/plugin/scripts/project-progression-outbox.mjs +26 -8
- package/plugin/scripts/project-progression-reader.mjs +4 -2
- package/plugin/scripts/protect-brain-state.sh +4 -1
- package/plugin/scripts/session-snapshot-hook.mjs +27 -3
- package/plugin/scripts/session-start-budget.mjs +1 -0
- package/plugin/scripts/session-start-core.mjs +34 -4
- package/plugin/scripts/session-start-health.mjs +7 -1
- package/plugin/scripts/session-start-update-plane.mjs +35 -0
- package/plugin/scripts/turn-outcome-capture.mjs +12 -1
- package/plugin/scripts/unprompted-runtime.mjs +2 -2
- package/plugin/skills/brain-console/SKILL.md +3 -3
- package/plugin/skills/rnbc/SKILL.md +24 -0
- package/plugin/skills/rvbc/SKILL.md +2 -2
- package/scripts/approved-runtime.mjs +2 -2
- package/scripts/ci/warm-brain-models.mjs +28 -0
- package/scripts/codex-hook-trust.mjs +94 -0
- package/scripts/console-runtime-identity.mjs +5 -0
- package/scripts/corpus-canary.mjs +46 -6
- package/scripts/corpus-dispatch-decision.mjs +2 -2
- package/scripts/corpus-promotion.mjs +1 -1
- package/scripts/hook-qualify-hosts.mjs +15 -3
- package/scripts/host-install-matrix.mjs +63 -2
- package/scripts/human-approval-phrases.mjs +46 -0
- package/scripts/installed-brain-health.mjs +53 -0
- package/scripts/move-brain.mjs +310 -0
- package/scripts/onboarding-console.mjs +93 -10
- package/scripts/oracle/abstain-threshold-sweep.mjs +62 -0
- package/scripts/oracle/abstain-trace.mjs +139 -0
- package/scripts/oracle/doc2query-generate.mjs +162 -0
- package/scripts/oracle/doc2query-reach.mjs +110 -0
- package/scripts/oracle/judge-train.mjs +158 -0
- package/scripts/oracle/need-set-split.mjs +48 -0
- package/scripts/oracle/sona-query-adapter-eval.mjs +139 -0
- package/scripts/package-cards.mjs +374 -0
- package/scripts/publication-receipt.mjs +37 -9
- package/scripts/recommendation-e2e.mjs +110 -0
- package/scripts/recommendation-eval.mjs +105 -0
- package/scripts/recommendation-floor.mjs +56 -0
- package/scripts/recommendation-judge-score.mjs +74 -0
- package/scripts/recommendation-latency.mjs +95 -0
- package/scripts/recommendation-real-host-score.mjs +76 -0
- package/scripts/recommendation-real-host.mjs +137 -0
- package/scripts/release-channel-kind.mjs +1 -1
- package/scripts/release-environment-policy.mjs +33 -0
- package/scripts/single-source-check.mjs +15 -10
- package/scripts/sync-commands.mjs +5 -2
- package/scripts/wired-check.mjs +17 -2
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/doc2query-reach.mjs — ADR-099 arm A, measurement: do generated entry points bring
|
|
4
|
+
* the gold file into the pool?
|
|
5
|
+
*
|
|
6
|
+
* Builds one RVF "entry" index per store from doc2query-generate output (every question embedded with
|
|
7
|
+
* the production query embedder, so a newcomer's question is matched question-to-question), then, for
|
|
8
|
+
* every need, asks the gold store's entry index for its top --k questions and collapses them to files.
|
|
9
|
+
* Pool reach is reported against the existing pool (dense 64 + keyword lane, from an abstain-trace
|
|
10
|
+
* run's poolRanks), on the TRAIN and HELD-OUT splits separately, with Wilson intervals:
|
|
11
|
+
* baselineReach gold already pooled by dense + keyword
|
|
12
|
+
* entryReach gold among the entry lane's files
|
|
13
|
+
* unionReach either
|
|
14
|
+
* Restricted to needs whose gold repository the router searched (the trace's eligible set); over all
|
|
15
|
+
* needs only entryReach is reported (store-level, independent of routing).
|
|
16
|
+
*
|
|
17
|
+
* node scripts/oracle/doc2query-reach.mjs --runtime <kbDir with forge-ask.mjs + node_modules>
|
|
18
|
+
* --d2q <generated.jsonl> --set <need-set.json> --split <split.json> --trace <abstain-trace.json>
|
|
19
|
+
* [--k 24] [--files 8] [--index-dir <dir>] [--out <file>]
|
|
20
|
+
*/
|
|
21
|
+
import fs from 'node:fs';
|
|
22
|
+
import path from 'node:path';
|
|
23
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
24
|
+
import { wilson } from '../eval-brain.mjs';
|
|
25
|
+
|
|
26
|
+
/** Collapse ranked entry hits (each pointing at a file) to the first `files` distinct files. */
|
|
27
|
+
export function filesFromEntries(hits, pathOfId, files) {
|
|
28
|
+
const out = [];
|
|
29
|
+
for (const h of hits) {
|
|
30
|
+
const p = pathOfId(h.id);
|
|
31
|
+
if (p && !out.includes(p)) out.push(p);
|
|
32
|
+
if (out.length >= files) break;
|
|
33
|
+
}
|
|
34
|
+
return out;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function reachSummary(rows) {
|
|
38
|
+
const r = (pred) => { const k = rows.filter(pred).length; const w = wilson(k, rows.length); return { k, n: rows.length, lo: +w.lo.toFixed(4), hi: +w.hi.toFixed(4) }; };
|
|
39
|
+
return { baselineReach: r((x) => x.baseline), entryReach: r((x) => x.entry), unionReach: r((x) => x.baseline || x.entry),
|
|
40
|
+
gained: rows.filter((x) => x.entry && !x.baseline).length };
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
async function main() {
|
|
44
|
+
const args = process.argv.slice(2);
|
|
45
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
46
|
+
const runtime = path.resolve(arg('--runtime'));
|
|
47
|
+
const { __embedInternals } = await import(pathToFileURL(path.join(runtime, 'forge-ask.mjs')).href);
|
|
48
|
+
const { loadRvf } = await import(pathToFileURL(path.join(runtime, 'resolve-deps.mjs')).href);
|
|
49
|
+
const { RvfDatabase } = loadRvf().mod;
|
|
50
|
+
const embed = __embedInternals.embed;
|
|
51
|
+
const k = Number(arg('--k', 24));
|
|
52
|
+
const nFiles = Number(arg('--files', 8));
|
|
53
|
+
const indexDir = arg('--index-dir', fs.mkdtempSync(path.join(runtime, '..', 'd2q-index-')));
|
|
54
|
+
fs.mkdirSync(indexDir, { recursive: true });
|
|
55
|
+
const gen = fs.readFileSync(arg('--d2q'), 'utf8').split('\n').filter(Boolean).map((l) => JSON.parse(l));
|
|
56
|
+
const stores = [...new Set(gen.map((g) => g.store))];
|
|
57
|
+
const indexes = new Map();
|
|
58
|
+
// The entry index uses the store's own query embedder (big variant when present: bge + query prefix),
|
|
59
|
+
// exactly what searchKb computes for the question, so production needs one query embedding.
|
|
60
|
+
const cfgFor = (store) => {
|
|
61
|
+
for (const v of [`${store}.big.rvf.embed.json`, `${store}.rvf.embed.json`]) {
|
|
62
|
+
try { return JSON.parse(fs.readFileSync(path.join(runtime, v), 'utf8')); } catch { /* next */ }
|
|
63
|
+
}
|
|
64
|
+
return undefined;
|
|
65
|
+
};
|
|
66
|
+
for (const store of stores) {
|
|
67
|
+
const cfg = cfgFor(store);
|
|
68
|
+
const dims = cfg?.dimensions || 384;
|
|
69
|
+
const rows = gen.filter((g) => g.store === store).flatMap((g) => g.questions.map((q) => ({ q, path: g.path })));
|
|
70
|
+
const file = path.join(indexDir, `${store}.entry.rvf`);
|
|
71
|
+
fs.rmSync(file, { force: true });
|
|
72
|
+
const db = await RvfDatabase.create(file, { dimensions: dims, metric: 'cosine' });
|
|
73
|
+
const batch = [];
|
|
74
|
+
for (let i = 0; i < rows.length; i++) {
|
|
75
|
+
batch.push({ id: i + 1, vector: Array.from(await embed(rows[i].q, cfg)) });
|
|
76
|
+
if (batch.length === 64 || i === rows.length - 1) { await db.ingestBatch(batch.splice(0)); }
|
|
77
|
+
if (i % 500 === 0) process.stderr.write(`\r[d2q-reach] ${store} ${i}/${rows.length}`);
|
|
78
|
+
}
|
|
79
|
+
indexes.set(store, { db, cfg, paths: rows.map((r) => r.path), questions: rows.length, model: cfg?.model || 'default' });
|
|
80
|
+
}
|
|
81
|
+
const set = JSON.parse(fs.readFileSync(arg('--set'), 'utf8')).questions;
|
|
82
|
+
const split = JSON.parse(fs.readFileSync(arg('--split'), 'utf8'));
|
|
83
|
+
const trace = JSON.parse(fs.readFileSync(arg('--trace'), 'utf8'));
|
|
84
|
+
const pooled = new Map(trace.poolRanks.map((p) => [p.id, p.poolRank != null]));
|
|
85
|
+
const rows = [];
|
|
86
|
+
for (const q of set) {
|
|
87
|
+
const store = q.repo.toLowerCase();
|
|
88
|
+
const idx = indexes.get(store);
|
|
89
|
+
if (!idx) continue;
|
|
90
|
+
const hits = await idx.db.query(Array.from(await embed(q.need, idx.cfg)), k);
|
|
91
|
+
const files = filesFromEntries(hits, (id) => idx.paths[Number(id) - 1], nFiles);
|
|
92
|
+
rows.push({ id: q.id, split: split.train.includes(q.id) ? 'train' : 'heldout', routed: pooled.has(q.id),
|
|
93
|
+
baseline: pooled.get(q.id) === true, entry: files.includes(q.path), entryRank: files.indexOf(q.path) + 1 || null });
|
|
94
|
+
}
|
|
95
|
+
const report = { kind: 'ruvnet-brain-doc2query-reach', k, files: nFiles,
|
|
96
|
+
indexes: Object.fromEntries([...indexes].map(([s, v]) => [s, { questions: v.questions, model: v.model }])),
|
|
97
|
+
generatedFiles: gen.length, keptQuestions: gen.reduce((s, g) => s + g.questions.length, 0),
|
|
98
|
+
rejectedQuestions: gen.reduce((s, g) => s + (g.rejected?.length || 0), 0), results: {} };
|
|
99
|
+
for (const sp of ['train', 'heldout']) {
|
|
100
|
+
report.results[sp] = { routedNeeds: reachSummary(rows.filter((r) => r.split === sp && r.routed)),
|
|
101
|
+
// the trace covers only routed needs, so over all needs only the entry lane itself is known
|
|
102
|
+
allNeedsEntryReach: reachSummary(rows.filter((r) => r.split === sp)).entryReach };
|
|
103
|
+
}
|
|
104
|
+
report.rows = rows;
|
|
105
|
+
if (arg('--out')) fs.writeFileSync(arg('--out'), `${JSON.stringify(report, null, 1)}\n`);
|
|
106
|
+
console.log(JSON.stringify({ ...report, rows: undefined }, null, 1));
|
|
107
|
+
process.exit(0);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/judge-train.mjs — fit the learned judge (kb/judge-rank.mjs, ADR-099 arm C) on the
|
|
4
|
+
* need-set TRAIN split and report it on the HELD-OUT split, offline.
|
|
5
|
+
*
|
|
6
|
+
* Input is the scored pool the search already recorded (KB_CE_TRACE: one line per question, every
|
|
7
|
+
* pooled candidate with its cross-encoder logit, dense distance, lane, title, path and length), so
|
|
8
|
+
* training and evaluation replay exactly what the cross-encoder saw; no model is run here.
|
|
9
|
+
*
|
|
10
|
+
* Model: logistic regression over JUDGE_FEATURES, standardised on train, L2 1.0, class-balanced,
|
|
11
|
+
* full-batch gradient descent (deterministic). Positive = the candidate is the gold file or a
|
|
12
|
+
* pre-registered alternative. Operating point: the smallest threshold on the judge's top score at
|
|
13
|
+
* which TRAIN precision of answered questions reaches --precision (default 0.8); it is folded into
|
|
14
|
+
* the bias so the product's existing "score < 0 abstains" rule applies unchanged.
|
|
15
|
+
*
|
|
16
|
+
* node scripts/oracle/judge-train.mjs --set <need-set.json> --split <split.json> --trace <cetrace.jsonl>
|
|
17
|
+
* [--precision 0.8] [--weights <out.json>] [--report <out.json>]
|
|
18
|
+
* [--recall-trace <recall.cetrace.jsonl> --recall-fixture data/retrieval-query-evidence.json]
|
|
19
|
+
*/
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
23
|
+
import { wilson } from '../eval-brain.mjs';
|
|
24
|
+
|
|
25
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
|
|
26
|
+
const { JUDGE_FEATURES, judgeFeatures, judgeLogit } = await import(pathToFileURL(path.join(ROOT, 'kb/judge-rank.mjs')).href);
|
|
27
|
+
|
|
28
|
+
const norm = (p) => String(p || '').replace(/^\.\//, '');
|
|
29
|
+
export function isTarget(q, cand) {
|
|
30
|
+
const targets = [{ repo: q.repo, path: q.path }, ...(q.alternatives || [])];
|
|
31
|
+
return targets.some((t) => String(t.repo).toLowerCase() === String(cand.repo).toLowerCase() && norm(t.path) === norm(cand.path));
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** Map need id -> its last recorded pool, from KB_CE_TRACE lines (unparsable lines are skipped). */
|
|
35
|
+
export function poolsByNeed(questions, traceText) {
|
|
36
|
+
const byNeed = new Map(questions.map((q) => [q.need, q.id]));
|
|
37
|
+
const pools = new Map();
|
|
38
|
+
for (const line of traceText.split('\n')) {
|
|
39
|
+
if (!line.trim()) continue;
|
|
40
|
+
let row;
|
|
41
|
+
try { row = JSON.parse(line); } catch { continue; }
|
|
42
|
+
const id = byNeed.get(row.query);
|
|
43
|
+
if (id && Array.isArray(row.cands) && row.cands.length) pools.set(id, row.cands);
|
|
44
|
+
}
|
|
45
|
+
return pools;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function fitLogistic(X, y, { l2 = 1, iters = 3000, lr = 0.1 } = {}) {
|
|
49
|
+
const d = X[0].length;
|
|
50
|
+
const mean = Array.from({ length: d }, (_, j) => X.reduce((s, x) => s + x[j], 0) / X.length);
|
|
51
|
+
const std = Array.from({ length: d }, (_, j) => Math.sqrt(X.reduce((s, x) => s + (x[j] - mean[j]) ** 2, 0) / X.length) || 1);
|
|
52
|
+
const Z = X.map((x) => x.map((v, j) => (v - mean[j]) / std[j]));
|
|
53
|
+
const pos = y.filter(Boolean).length;
|
|
54
|
+
const wPos = pos ? (y.length - pos) / pos : 1;
|
|
55
|
+
const w = new Array(d).fill(0);
|
|
56
|
+
let b = 0;
|
|
57
|
+
for (let it = 0; it < iters; it++) {
|
|
58
|
+
const g = new Array(d).fill(0);
|
|
59
|
+
let gb = 0;
|
|
60
|
+
let W = 0;
|
|
61
|
+
for (let i = 0; i < Z.length; i++) {
|
|
62
|
+
const s = b + Z[i].reduce((a, v, j) => a + v * w[j], 0);
|
|
63
|
+
const p = 1 / (1 + Math.exp(-s));
|
|
64
|
+
const cw = y[i] ? wPos : 1;
|
|
65
|
+
const e = cw * (p - (y[i] ? 1 : 0));
|
|
66
|
+
for (let j = 0; j < d; j++) g[j] += e * Z[i][j];
|
|
67
|
+
gb += e;
|
|
68
|
+
W += cw;
|
|
69
|
+
}
|
|
70
|
+
for (let j = 0; j < d; j++) w[j] -= lr * (g[j] / W + (l2 * w[j]) / Z.length);
|
|
71
|
+
b -= lr * (gb / W);
|
|
72
|
+
}
|
|
73
|
+
return { weights: w, bias: b, mean, std };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Per question: the top candidate under a scorer, whether a target is in the top 1 / top 5, top score. */
|
|
77
|
+
export function decide(questions, pools, score) {
|
|
78
|
+
return questions.filter((q) => pools.has(q.id)).map((q) => {
|
|
79
|
+
const cands = pools.get(q.id);
|
|
80
|
+
const s = score(q, cands);
|
|
81
|
+
const order = s.map((v, i) => [v, i]).sort((a, b) => b[0] - a[0]).map(([, i]) => i);
|
|
82
|
+
return { id: q.id, top: s[order[0]], at1: isTarget(q, cands[order[0]]),
|
|
83
|
+
within5: order.slice(0, 5).some((i) => isTarget(q, cands[i])) };
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export function metrics(rows, n, threshold = 0) {
|
|
88
|
+
const r = (k, m) => { const w = wilson(k, m); return { k, n: m, lo: +w.lo.toFixed(4), hi: +w.hi.toFixed(4) }; };
|
|
89
|
+
const answered = rows.filter((x) => x.top >= threshold);
|
|
90
|
+
return {
|
|
91
|
+
traced: rows.length,
|
|
92
|
+
goldWithin5: r(rows.filter((x) => x.within5).length, n),
|
|
93
|
+
goldAt1: r(rows.filter((x) => x.at1).length, n),
|
|
94
|
+
// measure-need-set's definitions (target within 5 AND not abstained), so baselines compare
|
|
95
|
+
confidentHit: r(answered.filter((x) => x.within5).length, n),
|
|
96
|
+
confidentWrong: r(answered.filter((x) => !x.within5).length, n),
|
|
97
|
+
// the stricter reading: the single top answer is the target
|
|
98
|
+
confidentHitTop1: r(answered.filter((x) => x.at1).length, n),
|
|
99
|
+
abstained: r(n - answered.length, n),
|
|
100
|
+
precision: r(answered.filter((x) => x.at1).length, answered.length),
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function chooseThreshold(rows, precision) {
|
|
105
|
+
const tops = [...new Set(rows.map((x) => x.top))].sort((a, b) => a - b);
|
|
106
|
+
for (const t of tops) {
|
|
107
|
+
const ans = rows.filter((x) => x.top >= t);
|
|
108
|
+
if (ans.length && ans.filter((x) => x.at1).length / ans.length >= precision) return t;
|
|
109
|
+
}
|
|
110
|
+
return Infinity;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
async function main() {
|
|
114
|
+
const args = process.argv.slice(2);
|
|
115
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
116
|
+
const set = JSON.parse(fs.readFileSync(arg('--set'), 'utf8')).questions;
|
|
117
|
+
const split = JSON.parse(fs.readFileSync(arg('--split'), 'utf8'));
|
|
118
|
+
const pools = poolsByNeed(set, fs.readFileSync(arg('--trace'), 'utf8'));
|
|
119
|
+
const train = set.filter((q) => split.train.includes(q.id));
|
|
120
|
+
const held = set.filter((q) => split.heldout.includes(q.id));
|
|
121
|
+
const X = [];
|
|
122
|
+
const y = [];
|
|
123
|
+
for (const q of train) {
|
|
124
|
+
if (!pools.has(q.id)) continue;
|
|
125
|
+
const cands = pools.get(q.id);
|
|
126
|
+
judgeFeatures(q.need, cands).forEach((x, i) => { X.push(x); y.push(isTarget(q, cands[i])); });
|
|
127
|
+
}
|
|
128
|
+
const fit = fitLogistic(X, y);
|
|
129
|
+
const model0 = { ...fit, features: JUDGE_FEATURES };
|
|
130
|
+
const judgeScore = (m) => (q, cands) => judgeFeatures(q.need, cands).map((x) => judgeLogit(m, x));
|
|
131
|
+
const ceScore = (q, cands) => cands.map((c) => (typeof c.ce === 'number' ? c.ce : -Infinity));
|
|
132
|
+
const precision = Number(arg('--precision', 0.8));
|
|
133
|
+
const tau = chooseThreshold(decide(train, pools, judgeScore(model0)), precision);
|
|
134
|
+
const model = { ...model0, bias: Number.isFinite(tau) ? fit.bias - tau : -1e9, threshold: tau,
|
|
135
|
+
trainedOn: { split: split.salt, trainNeeds: train.length, tracedTrain: train.filter((q) => pools.has(q.id)).length,
|
|
136
|
+
candidates: X.length, positives: y.filter(Boolean).length, precisionTarget: precision } };
|
|
137
|
+
const report = {
|
|
138
|
+
kind: 'ruvnet-brain-judge-report', features: JUDGE_FEATURES, trainedOn: model.trainedOn, threshold: tau,
|
|
139
|
+
weights: Object.fromEntries(JUDGE_FEATURES.map((f, j) => [f, +model.weights[j].toFixed(4)])),
|
|
140
|
+
train: { crossEncoder: metrics(decide(train, pools, ceScore), train.length), judge: metrics(decide(train, pools, judgeScore(model)), train.length) },
|
|
141
|
+
heldout: { crossEncoder: metrics(decide(held, pools, ceScore), held.length), judge: metrics(decide(held, pools, judgeScore(model)), held.length) },
|
|
142
|
+
};
|
|
143
|
+
// Offline replay on the recall-gate fixture's recorded pools: does re-ranking hurt questions the
|
|
144
|
+
// cross-encoder already answers? (Approximate: selectResults' name boosts are not replayed here, so
|
|
145
|
+
// the full path is the authority; this only flags gross damage early.)
|
|
146
|
+
if (arg('--recall-trace') && arg('--recall-fixture')) {
|
|
147
|
+
const fx = JSON.parse(fs.readFileSync(arg('--recall-fixture'), 'utf8')).queries;
|
|
148
|
+
const items = Object.entries(fx).map(([store, v]) => ({ id: store, need: v.query, repo: store, path: v.expected.path }));
|
|
149
|
+
const rpools = poolsByNeed(items, fs.readFileSync(arg('--recall-trace'), 'utf8'));
|
|
150
|
+
report.recallReplay = { crossEncoder: metrics(decide(items, rpools, ceScore), items.length),
|
|
151
|
+
judge: metrics(decide(items, rpools, judgeScore(model)), items.length) };
|
|
152
|
+
}
|
|
153
|
+
if (arg('--weights')) fs.writeFileSync(arg('--weights'), `${JSON.stringify(model, null, 1)}\n`);
|
|
154
|
+
if (arg('--report')) fs.writeFileSync(arg('--report'), `${JSON.stringify(report, null, 1)}\n`);
|
|
155
|
+
console.log(JSON.stringify({ threshold: tau, trainedOn: model.trainedOn, heldout: report.heldout }, null, 1));
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/need-set-split.mjs — a frozen, stratified train / held-out split of a need set.
|
|
4
|
+
*
|
|
5
|
+
* Anything that LEARNS from novice needs (a query adapter, a learned abstain model) may train only
|
|
6
|
+
* on `train`; every claim about it is measured on `heldout`. The split is a pure function of the
|
|
7
|
+
* question ids and a salt: within each repository the ids are ordered by sha256(salt + id) and the
|
|
8
|
+
* first round(n * trainFraction) go to train. Re-running it on the same set yields the same split.
|
|
9
|
+
*
|
|
10
|
+
* node scripts/oracle/need-set-split.mjs --set <need-set.json> [--salt v1] [--train 0.5] [--out <file>]
|
|
11
|
+
*/
|
|
12
|
+
import { createHash } from 'node:crypto';
|
|
13
|
+
import fs from 'node:fs';
|
|
14
|
+
import path from 'node:path';
|
|
15
|
+
import { fileURLToPath } from 'node:url';
|
|
16
|
+
|
|
17
|
+
export function splitNeedSet(questions, { salt = 'v1', trainFraction = 0.5 } = {}) {
|
|
18
|
+
const byRepo = new Map();
|
|
19
|
+
for (const q of questions) {
|
|
20
|
+
if (!byRepo.has(q.repo)) byRepo.set(q.repo, []);
|
|
21
|
+
byRepo.get(q.repo).push(q.id);
|
|
22
|
+
}
|
|
23
|
+
const train = [];
|
|
24
|
+
const heldout = [];
|
|
25
|
+
for (const repo of [...byRepo.keys()].sort()) {
|
|
26
|
+
const ids = byRepo.get(repo)
|
|
27
|
+
.map((id) => ({ id, h: createHash('sha256').update(`${salt}\n${id}`).digest('hex') }))
|
|
28
|
+
.sort((a, b) => (a.h < b.h ? -1 : a.h > b.h ? 1 : 0))
|
|
29
|
+
.map((x) => x.id);
|
|
30
|
+
const cut = Math.round(ids.length * trainFraction);
|
|
31
|
+
train.push(...ids.slice(0, cut));
|
|
32
|
+
heldout.push(...ids.slice(cut));
|
|
33
|
+
}
|
|
34
|
+
return { salt, trainFraction, train, heldout };
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
async function main() {
|
|
38
|
+
const args = process.argv.slice(2);
|
|
39
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
40
|
+
const set = JSON.parse(fs.readFileSync(arg('--set'), 'utf8'));
|
|
41
|
+
const split = splitNeedSet(set.questions, { salt: arg('--salt', 'v1'), trainFraction: Number(arg('--train', 0.5)) });
|
|
42
|
+
const out = { kind: 'ruvnet-brain-need-set-split', set: { kind: set.kind, version: set.version, n: set.questions.length,
|
|
43
|
+
contentHash: set.contentHash }, ...split };
|
|
44
|
+
if (arg('--out')) fs.writeFileSync(arg('--out'), `${JSON.stringify(out, null, 1)}\n`);
|
|
45
|
+
console.log(JSON.stringify({ train: split.train.length, heldout: split.heldout.length, contentHash: set.contentHash }));
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/sona-query-adapter-eval.mjs — ADR-099 arm B, an EXPERIMENT: can a SONA MicroLoRA
|
|
4
|
+
* query adapter (@ruvector/sona, rank 1-2) move a novice question's embedding so the gold file ranks
|
|
5
|
+
* higher in the store's own dense index?
|
|
6
|
+
*
|
|
7
|
+
* Train (need-set TRAIN split only): one SONA trajectory per need, beginning at the question's
|
|
8
|
+
* embedding, with one step whose activation is the gold passage's embedding (reward 1) and one step
|
|
9
|
+
* per top dense non-gold passage (reward 0), ended with quality 1. SONA's REINFORCE gradient
|
|
10
|
+
* ((reward - baseline) x activation, ruvector/crates/sona/src/types.rs) then pushes the adapter toward
|
|
11
|
+
* the gold direction. forceLearn() applies it.
|
|
12
|
+
* Evaluate (HELD-OUT, and TRAIN for fit): the gold file's rank among the store's top --depth dense
|
|
13
|
+
* hits for the raw question vs applyMicroLora(question); and the same on the recall-gate fixture, which
|
|
14
|
+
* the adapter must not hurt. An untrained adapter is measured too, because it is not the identity.
|
|
15
|
+
*
|
|
16
|
+
* node scripts/oracle/sona-query-adapter-eval.mjs --kb <kbDir with forge-ask.mjs + node_modules>
|
|
17
|
+
* --sona <dir containing node_modules/@ruvector/sona> --set <need-set.json> --split <split.json>
|
|
18
|
+
* [--recall data/retrieval-query-evidence.json] [--depth 64] [--negatives 4] [--out <file>]
|
|
19
|
+
*/
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
import { createRequire } from 'node:module';
|
|
23
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
24
|
+
import { wilson } from '../eval-brain.mjs';
|
|
25
|
+
|
|
26
|
+
/** Rank (1-based) of the first hit whose path is `gold`, or null within the list. */
|
|
27
|
+
export function rankOf(paths, gold) {
|
|
28
|
+
const i = paths.indexOf(gold);
|
|
29
|
+
return i < 0 ? null : i + 1;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** Paired comparison of two rank lists (null = not within depth). */
|
|
33
|
+
export function compareRanks(base, adapted, cut = 5) {
|
|
34
|
+
const within = (r) => r != null && r <= cut;
|
|
35
|
+
const n = base.length;
|
|
36
|
+
const r = (k) => { const w = wilson(k, n); return { k, n, lo: +w.lo.toFixed(4), hi: +w.hi.toFixed(4) }; };
|
|
37
|
+
let gained = 0;
|
|
38
|
+
let lost = 0;
|
|
39
|
+
for (let i = 0; i < n; i++) {
|
|
40
|
+
if (!within(base[i]) && within(adapted[i])) gained++;
|
|
41
|
+
if (within(base[i]) && !within(adapted[i])) lost++;
|
|
42
|
+
}
|
|
43
|
+
return { base: r(base.filter(within).length), adapted: r(adapted.filter(within).length), gained, lost,
|
|
44
|
+
inDepthBase: base.filter((x) => x != null).length, inDepthAdapted: adapted.filter((x) => x != null).length };
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
async function main() {
|
|
48
|
+
const args = process.argv.slice(2);
|
|
49
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
50
|
+
const kb = path.resolve(arg('--kb'));
|
|
51
|
+
const depth = Number(arg('--depth', 64));
|
|
52
|
+
const negatives = Number(arg('--negatives', 4));
|
|
53
|
+
const require = createRequire(path.join(path.resolve(arg('--sona')), 'index.js'));
|
|
54
|
+
const { SonaEngine } = require('@ruvector/sona');
|
|
55
|
+
const { __embedInternals } = await import(pathToFileURL(path.join(kb, 'forge-ask.mjs')).href);
|
|
56
|
+
const { loadRvf } = await import(pathToFileURL(path.join(kb, 'resolve-deps.mjs')).href);
|
|
57
|
+
const { RvfDatabase } = loadRvf().mod;
|
|
58
|
+
const embed = __embedInternals.embed;
|
|
59
|
+
const set = JSON.parse(fs.readFileSync(arg('--set'), 'utf8')).questions;
|
|
60
|
+
const split = JSON.parse(fs.readFileSync(arg('--split'), 'utf8'));
|
|
61
|
+
const stores = new Map();
|
|
62
|
+
const storeOf = async (name) => {
|
|
63
|
+
if (stores.has(name)) return stores.get(name);
|
|
64
|
+
const big = fs.existsSync(path.join(kb, `${name}.big.rvf`));
|
|
65
|
+
const base = path.join(kb, big ? `${name}.big.rvf` : `${name}.rvf`);
|
|
66
|
+
const cfg = JSON.parse(fs.readFileSync(`${base}.embed.json`, 'utf8'));
|
|
67
|
+
const idmap = JSON.parse(fs.readFileSync(`${base}.idmap.json`, 'utf8')).idToLabel;
|
|
68
|
+
const labelToId = new Map(Object.entries(idmap).map(([id, label]) => [Number(label), id]));
|
|
69
|
+
const passages = path.join(kb, big && fs.existsSync(path.join(kb, `${name}.big.passages.jsonl`)) ? `${name}.big.passages.jsonl` : `${name}.passages.jsonl`);
|
|
70
|
+
const pathOf = new Map();
|
|
71
|
+
const textOf = new Map();
|
|
72
|
+
for (const l of fs.readFileSync(passages, 'utf8').split('\n')) {
|
|
73
|
+
if (!l) continue;
|
|
74
|
+
const r = JSON.parse(l);
|
|
75
|
+
pathOf.set(String(r.id), r.path);
|
|
76
|
+
if (!textOf.has(r.path)) textOf.set(r.path, String(r.text || ''));
|
|
77
|
+
}
|
|
78
|
+
const db = await RvfDatabase.openReadonly(base);
|
|
79
|
+
const s = { name, cfg, db, labelToId, pathOf, textOf, engine: new SonaEngine(cfg.dimensions) };
|
|
80
|
+
stores.set(name, s);
|
|
81
|
+
return s;
|
|
82
|
+
};
|
|
83
|
+
const search = async (s, vec) => {
|
|
84
|
+
const hits = await s.db.query(Array.from(vec), depth);
|
|
85
|
+
const paths = [];
|
|
86
|
+
for (const h of hits) {
|
|
87
|
+
const p = s.pathOf.get(String(s.labelToId.get(Number(h.id)) ?? h.id));
|
|
88
|
+
if (p && !paths.includes(p)) paths.push(p);
|
|
89
|
+
}
|
|
90
|
+
return paths;
|
|
91
|
+
};
|
|
92
|
+
const passageEmbed = (s, text) => embed(text, { ...s.cfg, queryPrefix: '' });
|
|
93
|
+
const qs = set.map((q) => ({ ...q, store: q.repo.toLowerCase(), isTrain: split.train.includes(q.id) }));
|
|
94
|
+
// TRAIN
|
|
95
|
+
let trained = 0;
|
|
96
|
+
for (const q of qs.filter((x) => x.isTrain)) {
|
|
97
|
+
const s = await storeOf(q.store);
|
|
98
|
+
const qv = Array.from(await embed(q.need, s.cfg));
|
|
99
|
+
const gold = s.textOf.get(q.path);
|
|
100
|
+
if (!gold) continue;
|
|
101
|
+
const t = s.engine.beginTrajectory(qv);
|
|
102
|
+
const zeros = new Array(64).fill(0);
|
|
103
|
+
s.engine.addTrajectoryStep(t, Array.from(await passageEmbed(s, q.span || gold)), zeros, 1);
|
|
104
|
+
for (const p of (await search(s, qv)).filter((p) => p !== q.path).slice(0, negatives)) {
|
|
105
|
+
s.engine.addTrajectoryStep(t, Array.from(await passageEmbed(s, s.textOf.get(p) || '')), zeros, 0);
|
|
106
|
+
}
|
|
107
|
+
s.engine.endTrajectory(t, 1);
|
|
108
|
+
trained++;
|
|
109
|
+
process.stderr.write(`\r[sona-eval] trained ${trained}`);
|
|
110
|
+
}
|
|
111
|
+
const learn = {};
|
|
112
|
+
for (const s of stores.values()) learn[s.name] = String(s.engine.forceLearn());
|
|
113
|
+
// EVALUATE: raw vs adapted, on held-out and train needs, and on the recall fixture
|
|
114
|
+
const evalSet = async (items) => {
|
|
115
|
+
const base = [];
|
|
116
|
+
const adapted = [];
|
|
117
|
+
for (const it of items) {
|
|
118
|
+
const s = await storeOf(it.store);
|
|
119
|
+
const qv = Array.from(await embed(it.query, s.cfg));
|
|
120
|
+
base.push(rankOf(await search(s, qv), it.gold));
|
|
121
|
+
adapted.push(rankOf(await search(s, s.engine.applyMicroLora(qv)), it.gold));
|
|
122
|
+
}
|
|
123
|
+
return compareRanks(base, adapted);
|
|
124
|
+
};
|
|
125
|
+
const needItems = (isTrain) => qs.filter((q) => q.isTrain === isTrain).map((q) => ({ store: q.store, query: q.need, gold: q.path }));
|
|
126
|
+
const report = { kind: 'ruvnet-brain-sona-query-adapter-eval', depth, negatives, trained, learn,
|
|
127
|
+
heldout: await evalSet(needItems(false)), train: await evalSet(needItems(true)) };
|
|
128
|
+
if (arg('--recall')) {
|
|
129
|
+
const fx = JSON.parse(fs.readFileSync(arg('--recall'), 'utf8')).queries;
|
|
130
|
+
const items = Object.entries(fx).filter(([store]) => stores.has(store))
|
|
131
|
+
.map(([store, v]) => ({ store, query: v.query, gold: v.expected.path }));
|
|
132
|
+
report.recallFixtureSameStores = await evalSet(items);
|
|
133
|
+
}
|
|
134
|
+
if (arg('--out')) fs.writeFileSync(arg('--out'), `${JSON.stringify(report, null, 1)}\n`);
|
|
135
|
+
console.log(JSON.stringify(report, null, 1));
|
|
136
|
+
process.exit(0);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|