ruvnet-brain 4.4.0 → 4.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/bin/install.mjs +679 -121
- package/console/app.js +178 -81
- package/console/index.html +1 -1
- package/console/install-architecture.html +1 -0
- package/console/scope.css +4 -1
- package/console/style.css +13 -0
- package/console/tips.html +4 -4
- package/kb/brain-profile.mjs +1 -0
- package/kb/corpus-release-identity.mjs +1 -1
- package/kb/forge-update.mjs +41 -17
- package/kb/model-requirements.mjs +4 -1
- package/kb/update-storage-transaction.mjs +79 -0
- package/kb/zip-extract.mjs +22 -0
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +5 -4
- package/plugin/commands/configure.md +5 -4
- package/plugin/commands/rnb-brief.md +41 -0
- package/plugin/commands/rnb.md +80 -0
- package/plugin/commands/rnbc.md +80 -0
- package/plugin/commands/rvbc.md +5 -4
- package/plugin/commands/rvcb.md +5 -4
- package/plugin/commands/whats-new.md +4 -4
- package/plugin/mcp/server.mjs +10 -2
- package/plugin/scripts/advocacy-route.mjs +59 -23
- package/plugin/scripts/anticipate.sh +4 -0
- package/plugin/scripts/brain-confirmation.mjs +258 -0
- package/plugin/scripts/brain-footprint.mjs +494 -0
- package/plugin/scripts/brain-location.mjs +47 -0
- package/plugin/scripts/capability-registry.mjs +11 -1
- package/plugin/scripts/continuity-brief.mjs +324 -0
- package/plugin/scripts/continuity-events.mjs +327 -0
- package/plugin/scripts/continuity-journal.mjs +500 -0
- package/plugin/scripts/decision-gate.mjs +56 -3
- package/plugin/scripts/footprint-io.mjs +186 -0
- package/plugin/scripts/ground-before-write.sh +8 -1
- package/plugin/scripts/ground-ruvnet.sh +103 -12
- package/plugin/scripts/grounding-answer.mjs +2 -1
- package/plugin/scripts/grounding-stamp.sh +3 -0
- package/plugin/scripts/grounding-substance.mjs +1 -1
- package/plugin/scripts/grounding-turn-evidence.mjs +68 -6
- package/plugin/scripts/hook-input.mjs +78 -4
- package/plugin/scripts/kb-copy-proof.mjs +148 -0
- package/plugin/scripts/lesson-bridge.mjs +6 -2
- package/plugin/scripts/nightly-controller.mjs +8 -1
- package/plugin/scripts/node-sqlite.mjs +41 -0
- package/plugin/scripts/package-cards.json +797 -0
- package/plugin/scripts/package-cards.rvf +0 -0
- package/plugin/scripts/package-cards.rvf.idmap.json +1 -0
- package/plugin/scripts/package-cards.rvf.meta.json +1 -0
- package/plugin/scripts/package-recommender-client.mjs +138 -0
- package/plugin/scripts/package-recommender-flag.mjs +30 -0
- package/plugin/scripts/package-recommender.mjs +391 -0
- package/plugin/scripts/project-progression-outbox.mjs +26 -8
- package/plugin/scripts/project-progression-reader.mjs +14 -2
- package/plugin/scripts/project-progression-store.mjs +153 -6
- package/plugin/scripts/protect-brain-state.sh +4 -1
- package/plugin/scripts/session-snapshot-hook.mjs +225 -41
- package/plugin/scripts/session-start-budget.mjs +1 -0
- package/plugin/scripts/session-start-core.mjs +34 -4
- package/plugin/scripts/session-start-health.mjs +7 -1
- package/plugin/scripts/session-start-update-plane.mjs +35 -0
- package/plugin/scripts/turn-outcome-capture.mjs +12 -1
- package/plugin/scripts/unprompted-runtime.mjs +2 -2
- package/plugin/skills/brain-console/SKILL.md +3 -3
- package/plugin/skills/rnbc/SKILL.md +24 -0
- package/plugin/skills/rvbc/SKILL.md +2 -2
- package/scripts/approved-runtime.mjs +2 -2
- package/scripts/ci/warm-brain-models.mjs +28 -0
- package/scripts/codex-hook-trust.mjs +94 -0
- package/scripts/console-instances.mjs +70 -12
- package/scripts/console-runtime-identity.mjs +5 -0
- package/scripts/corpus-canary.mjs +46 -6
- package/scripts/corpus-dispatch-decision.mjs +2 -2
- package/scripts/corpus-promotion.mjs +1 -1
- package/scripts/full-suite-gate.mjs +9 -2
- package/scripts/hook-qualify-hosts.mjs +15 -3
- package/scripts/host-install-matrix.mjs +63 -2
- package/scripts/human-approval-phrases.mjs +46 -0
- package/scripts/installed-brain-health.mjs +53 -0
- package/scripts/move-brain.mjs +310 -0
- package/scripts/onboarding-console.mjs +93 -10
- package/scripts/oracle/abstain-threshold-sweep.mjs +62 -0
- package/scripts/oracle/abstain-trace.mjs +139 -0
- package/scripts/oracle/doc2query-generate.mjs +162 -0
- package/scripts/oracle/doc2query-reach.mjs +110 -0
- package/scripts/oracle/judge-train.mjs +158 -0
- package/scripts/oracle/need-set-split.mjs +48 -0
- package/scripts/oracle/sona-query-adapter-eval.mjs +139 -0
- package/scripts/package-cards.mjs +374 -0
- package/scripts/publication-receipt.mjs +37 -9
- package/scripts/recommendation-e2e.mjs +110 -0
- package/scripts/recommendation-eval.mjs +105 -0
- package/scripts/recommendation-floor.mjs +56 -0
- package/scripts/recommendation-judge-score.mjs +74 -0
- package/scripts/recommendation-latency.mjs +95 -0
- package/scripts/recommendation-real-host-score.mjs +76 -0
- package/scripts/recommendation-real-host.mjs +137 -0
- package/scripts/release-channel-kind.mjs +1 -1
- package/scripts/release-environment-policy.mjs +33 -0
- package/scripts/single-source-check.mjs +15 -10
- package/scripts/sync-commands.mjs +5 -2
- package/scripts/wired-check.mjs +17 -2
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/abstain-trace.mjs — WHY does the reranker abstain on a novice need whose gold
|
|
4
|
+
* repository WAS searched?
|
|
5
|
+
*
|
|
6
|
+
* For each need it measures, in the gold store, with the production models:
|
|
7
|
+
* poolRank rank of the gold file among the store's dense pool (searchKb, depth --pool)
|
|
8
|
+
* ceProduction cross-encoder logit on the text production reranks (the assembled document, which
|
|
9
|
+
* the reranker reads only through its first 3000 chars / 512 tokens)
|
|
10
|
+
* ceBestChunk best logit over the gold file's own sidecar chunks, each read on its own
|
|
11
|
+
* ceSpan logit on the verbatim gold span the question was written from (an upper bound:
|
|
12
|
+
* the answer text itself, nothing else)
|
|
13
|
+
* ceTopProd production's top logit for this need (from the measured needs run)
|
|
14
|
+
* and classifies the abstention:
|
|
15
|
+
* not-in-pool dense retrieval never surfaced the gold file
|
|
16
|
+
* window gold scores >= 0 on some chunk but < 0 on the text production reads
|
|
17
|
+
* calibration even the verbatim span scores < 0: the model, not the window, says "irrelevant"
|
|
18
|
+
* chunking span >= 0 but every stored chunk < 0 (the answer is split across chunks)
|
|
19
|
+
* outranked production text >= 0, yet the top citation is another file
|
|
20
|
+
*
|
|
21
|
+
* node scripts/oracle/abstain-trace.mjs --kb <kbDir> --set <need-set.json> --rows <needs.json>
|
|
22
|
+
* [--runtime <dir>] [--keyword 8] [--sample 20] [--pool 64] [--out <file>]
|
|
23
|
+
*
|
|
24
|
+
* --rows is a measure-need-set output (it supplies reposSearched and the production top logit);
|
|
25
|
+
* only needs whose gold repository was searched are traced. Nothing written contains a local path.
|
|
26
|
+
*/
|
|
27
|
+
import fs from 'node:fs';
|
|
28
|
+
import path from 'node:path';
|
|
29
|
+
import readline from 'node:readline';
|
|
30
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
31
|
+
|
|
32
|
+
const args = process.argv.slice(2);
|
|
33
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
34
|
+
|
|
35
|
+
export function classify({ poolRank, ceProduction, ceBestChunk, ceSpan }) {
|
|
36
|
+
if (poolRank == null) return 'not-in-pool';
|
|
37
|
+
if (ceProduction >= 0) return 'outranked';
|
|
38
|
+
if (ceBestChunk >= 0) return 'window';
|
|
39
|
+
if (ceSpan < 0) return 'calibration';
|
|
40
|
+
return 'chunking';
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Deterministic sample: every step-th need of the eligible list, spread over the whole list. */
|
|
44
|
+
export function sampleEvenly(list, n) {
|
|
45
|
+
if (list.length <= n) return list;
|
|
46
|
+
const step = list.length / n;
|
|
47
|
+
return Array.from({ length: n }, (_, i) => list[Math.floor(i * step)]);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
async function chunksOf(kb, store, file) {
|
|
51
|
+
const out = [];
|
|
52
|
+
for (const name of [`${store}.passages.jsonl`, `${store}.big.passages.jsonl`]) {
|
|
53
|
+
const p = path.join(kb, name);
|
|
54
|
+
if (!fs.existsSync(p)) continue;
|
|
55
|
+
const rl = readline.createInterface({ input: fs.createReadStream(p), crlfDelay: Infinity });
|
|
56
|
+
for await (const line of rl) {
|
|
57
|
+
if (!line.includes(file)) continue;
|
|
58
|
+
try { const r = JSON.parse(line); if (r.path === file) out.push(String(r.text || '')); } catch { /* skip */ }
|
|
59
|
+
}
|
|
60
|
+
if (out.length) break;
|
|
61
|
+
}
|
|
62
|
+
return out;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
async function main() {
|
|
66
|
+
const kb = arg('--kb');
|
|
67
|
+
const set = JSON.parse(fs.readFileSync(arg('--set'), 'utf8'));
|
|
68
|
+
const rows = JSON.parse(fs.readFileSync(arg('--rows'), 'utf8')).rows;
|
|
69
|
+
const poolDepth = Number(arg('--pool', 64));
|
|
70
|
+
const want = Number(arg('--sample', 20));
|
|
71
|
+
const byId = new Map(set.questions.map((q) => [q.id, q]));
|
|
72
|
+
// The runtime is the KB directory's own copy of the reader (as in production), unless --runtime
|
|
73
|
+
// names another directory holding forge-ask.mjs / forge-rerank.mjs and their node_modules.
|
|
74
|
+
const runtime = path.resolve(arg('--runtime', kb));
|
|
75
|
+
const { searchKb } = await import(pathToFileURL(path.join(runtime, 'forge-ask.mjs')).href);
|
|
76
|
+
const { rerankPairs } = await import(pathToFileURL(path.join(runtime, 'forge-rerank.mjs')).href);
|
|
77
|
+
// --keyword N also counts the keyword lane (keyword-lane.mjs in the runtime) as part of the pool.
|
|
78
|
+
const keywordTopN = Number(arg('--keyword', 0));
|
|
79
|
+
const keywordCandidates = keywordTopN > 0
|
|
80
|
+
? (await import(pathToFileURL(path.join(runtime, 'keyword-lane.mjs')).href)).keywordCandidates : null;
|
|
81
|
+
const ce = async (query, texts) => {
|
|
82
|
+
if (!texts.length) return [];
|
|
83
|
+
const scored = await rerankPairs(query, texts.map((t, i) => ({ fullText: t, i })));
|
|
84
|
+
const out = new Array(texts.length);
|
|
85
|
+
for (const s of scored) out[s.i] = s.ceScore;
|
|
86
|
+
return out;
|
|
87
|
+
};
|
|
88
|
+
const eligible = rows.filter((r) => (r.reposSearched || []).some((s) => s.toLowerCase() === r.repo.toLowerCase()));
|
|
89
|
+
const traced = [];
|
|
90
|
+
for (const r of eligible) {
|
|
91
|
+
const q = byId.get(r.id);
|
|
92
|
+
const store = r.repo.toLowerCase();
|
|
93
|
+
const hits = await searchKb({ dir: kb, name: store, query: q.need, k: poolDepth, n: poolDepth });
|
|
94
|
+
const idx = hits.findIndex((h) => h.path === q.path);
|
|
95
|
+
let lane = idx < 0 ? null : 'dense';
|
|
96
|
+
let gold = idx < 0 ? null : hits[idx];
|
|
97
|
+
if (!gold && keywordCandidates) {
|
|
98
|
+
const kw = keywordCandidates(kb, store, q.need, { topN: keywordTopN, exclude: new Set(hits.map((h) => h.path)) });
|
|
99
|
+
const k = kw.findIndex((c) => c.path === q.path);
|
|
100
|
+
if (k >= 0) { gold = kw[k]; lane = 'keyword'; }
|
|
101
|
+
}
|
|
102
|
+
traced.push({ r, q, store, gold, lane, poolRank: idx < 0 ? (gold ? poolDepth + 1 : null) : idx + 1 });
|
|
103
|
+
process.stderr.write(`\r[abstain-trace] pool ${traced.length}/${eligible.length}`);
|
|
104
|
+
}
|
|
105
|
+
const inPool = traced.filter((t) => t.poolRank != null);
|
|
106
|
+
const sample = sampleEvenly(inPool, want);
|
|
107
|
+
const out = [];
|
|
108
|
+
for (const t of sample) {
|
|
109
|
+
const { gold } = t;
|
|
110
|
+
const chunks = await chunksOf(kb, t.store, t.q.path);
|
|
111
|
+
const [ceProduction] = await ce(t.q.need, [gold.fullText || gold.text || '']);
|
|
112
|
+
const chunkScores = await ce(t.q.need, chunks);
|
|
113
|
+
const [ceSpan] = await ce(t.q.need, [t.q.span || '']);
|
|
114
|
+
const row = {
|
|
115
|
+
id: t.q.id, repo: t.r.repo, need: t.q.need, poolRank: t.poolRank, lane: t.lane,
|
|
116
|
+
ceProduction: +ceProduction.toFixed(3),
|
|
117
|
+
ceBestChunk: chunkScores.length ? +Math.max(...chunkScores).toFixed(3) : null,
|
|
118
|
+
chunks: chunks.length, goldDocChars: (gold.fullText || '').length,
|
|
119
|
+
ceSpan: +ceSpan.toFixed(3), ceTopProd: t.r.topCe, prodTop: t.r.topPath, prodAbstained: t.r.abstained,
|
|
120
|
+
};
|
|
121
|
+
row.cause = classify(row);
|
|
122
|
+
out.push(row);
|
|
123
|
+
process.stderr.write(`\r[abstain-trace] ce ${out.length}/${sample.length} `);
|
|
124
|
+
}
|
|
125
|
+
const count = (k) => out.filter((x) => x.cause === k).length;
|
|
126
|
+
const report = {
|
|
127
|
+
kind: 'ruvnet-brain-abstain-trace', poolDepth,
|
|
128
|
+
eligible: eligible.length, goldInPool: inPool.length, goldViaKeyword: traced.filter((t) => t.lane === 'keyword').length, notInPool: traced.length - inPool.length, sampled: out.length,
|
|
129
|
+
causes: Object.fromEntries(['not-in-pool', 'window', 'calibration', 'chunking', 'outranked'].map((k) => [k, count(k)])),
|
|
130
|
+
rows: out,
|
|
131
|
+
poolRanks: traced.map((t) => ({ id: t.q.id, poolRank: t.poolRank })),
|
|
132
|
+
};
|
|
133
|
+
const file = arg('--out');
|
|
134
|
+
if (file) fs.writeFileSync(file, `${JSON.stringify(report, null, 1)}\n`);
|
|
135
|
+
console.log(JSON.stringify({ ...report, rows: undefined, poolRanks: undefined }, null, 1));
|
|
136
|
+
process.exit(0);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/doc2query-generate.mjs — ADR-099 arm A, generation: newcomer-style questions per
|
|
4
|
+
* documentation file ("doc2query"), produced ONCE where the corpus is built, never on a customer
|
|
5
|
+
* machine.
|
|
6
|
+
*
|
|
7
|
+
* The generator sees only the file's title, path and opening text, never the need set. Each question
|
|
8
|
+
* must describe a NEED in plain words. Deterministic checks drop any question that:
|
|
9
|
+
* - names the product, the repository or a code identifier, or
|
|
10
|
+
* - shares more than 3 consecutive words with the excerpt
|
|
11
|
+
* (the need-set producer's leak rules). Output is append-only JSONL
|
|
12
|
+
* {store, path, questions[], rejected}, so a run can be resumed and interrupted runs keep their work.
|
|
13
|
+
*
|
|
14
|
+
* node scripts/oracle/doc2query-generate.mjs --kb <kbDir> --stores ruflo,ruvector,ruview --out <file.jsonl>
|
|
15
|
+
* [--kind md] [--per-file 3] [--batch 25] [--model haiku] [--limit N] [--conc 2]
|
|
16
|
+
*
|
|
17
|
+
* Children run with scripts/subscription-hosts.mjs#subscriptionOnlyEnv (no API billing keys) and the
|
|
18
|
+
* same minimised-context claude flags as scripts/oracle/producer-hosts.mjs.
|
|
19
|
+
*/
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
import os from 'node:os';
|
|
23
|
+
import { fileURLToPath } from 'node:url';
|
|
24
|
+
import { claudeArgs, spawnHost, isQuotaRefusal } from './producer-hosts.mjs';
|
|
25
|
+
import { subscriptionOnlyEnv } from '../subscription-hosts.mjs';
|
|
26
|
+
|
|
27
|
+
export const EXCERPT_CHARS = 1500;
|
|
28
|
+
export const D2Q_SCHEMA = Object.freeze({
|
|
29
|
+
type: 'object', additionalProperties: false,
|
|
30
|
+
properties: { items: { type: 'array', items: { type: 'object', additionalProperties: false,
|
|
31
|
+
properties: { docId: { type: 'string' }, questions: { type: 'array', items: { type: 'string' } } },
|
|
32
|
+
required: ['docId', 'questions'] } } },
|
|
33
|
+
required: ['items'],
|
|
34
|
+
});
|
|
35
|
+
export const D2Q_SYSTEM = 'You write search questions for documentation. You see only the documents in the message, '
|
|
36
|
+
+ 'you have no tools, and you use no outside knowledge. Return only the structured output.';
|
|
37
|
+
|
|
38
|
+
export function d2qPrompt(docs, perFile) {
|
|
39
|
+
return [
|
|
40
|
+
`For each DOC below write ${perFile} different questions that a newcomer might type into a search box, `
|
|
41
|
+
+ 'where this document is what would help them.',
|
|
42
|
+
'Rules for every question:',
|
|
43
|
+
'- 12-35 words, plain everyday language, describing the person\'s NEED or problem, not the document.',
|
|
44
|
+
'- The person has never heard of this project: no product, project, repository, package, library or tool names, no code, no identifiers, no file names.',
|
|
45
|
+
'- Do not copy more than 3 consecutive words from the document.',
|
|
46
|
+
'- The three questions should cover different things the document helps with.',
|
|
47
|
+
'Return one item per DOC with its docId copied exactly.',
|
|
48
|
+
'',
|
|
49
|
+
...docs.map((d) => `===DOC docId=${d.docId}\ntitle: ${d.title}\n${d.excerpt}\n===END`),
|
|
50
|
+
].join('\n');
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
const PRODUCT_NAME = /^(?:ruv\w*|ruflo|rufl\w*|ruview|claudeflow|agentdb|agentic|sona|rvf|ruvllm|cognitum|densepose|metaharness|reasoningbank)$/;
|
|
54
|
+
const words = (s) => String(s).toLowerCase().match(/[a-z0-9]+/g) || [];
|
|
55
|
+
/** Longest run of consecutive words a question shares with the excerpt. */
|
|
56
|
+
export function sharedRun(question, excerpt) {
|
|
57
|
+
const q = words(question);
|
|
58
|
+
const grams = new Set();
|
|
59
|
+
const e = words(excerpt);
|
|
60
|
+
let best = 0;
|
|
61
|
+
for (let n = 1; n <= q.length; n++) {
|
|
62
|
+
grams.clear();
|
|
63
|
+
for (let i = 0; i + n <= e.length; i++) grams.add(e.slice(i, i + n).join(' '));
|
|
64
|
+
let found = false;
|
|
65
|
+
for (let i = 0; i + n <= q.length; i++) if (grams.has(q.slice(i, i + n).join(' '))) { found = true; break; }
|
|
66
|
+
if (!found) break;
|
|
67
|
+
best = n;
|
|
68
|
+
}
|
|
69
|
+
return best;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** Keep a question only if it obeys the leak rules; return the reason when it does not. */
|
|
73
|
+
export function leakReason(question, { excerpt, store, path: p }) {
|
|
74
|
+
const q = String(question || '').trim();
|
|
75
|
+
const n = words(q).length;
|
|
76
|
+
if (n < 8 || n > 45) return 'length';
|
|
77
|
+
if (/[`{}<>=]|::|\w\(|\b[a-z]+[A-Z][A-Za-z]+\b|\b\w+\.(?:md|js|ts|rs|py|json|toml)\b|@[a-z0-9-]+\//.test(q)) return 'identifier';
|
|
78
|
+
// Product names: the store itself and the rUv family's own names (a newcomer has heard none of them).
|
|
79
|
+
if (words(q).some((w) => w === String(store).toLowerCase() || PRODUCT_NAME.test(w))) return 'names-source';
|
|
80
|
+
void p;
|
|
81
|
+
if (sharedRun(q, excerpt) > 3) return 'copies-source';
|
|
82
|
+
return null;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export function excerptOf(chunks) {
|
|
86
|
+
return chunks.join('\n\n').slice(0, EXCERPT_CHARS);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
function readStoreDocs(kb, store, kind) {
|
|
90
|
+
const big = path.join(kb, `${store}.big.passages.jsonl`);
|
|
91
|
+
const file = fs.existsSync(big) ? big : path.join(kb, `${store}.passages.jsonl`);
|
|
92
|
+
const byPath = new Map();
|
|
93
|
+
for (const line of fs.readFileSync(file, 'utf8').split('\n')) {
|
|
94
|
+
if (!line) continue;
|
|
95
|
+
let r;
|
|
96
|
+
try { r = JSON.parse(line); } catch { continue; }
|
|
97
|
+
if (kind === 'md' && !/\.md$/i.test(r.path)) continue;
|
|
98
|
+
if (!byPath.has(r.path)) byPath.set(r.path, { title: r.title || path.basename(r.path), chunks: [] });
|
|
99
|
+
const d = byPath.get(r.path);
|
|
100
|
+
if (d.chunks.join('').length < EXCERPT_CHARS) d.chunks.push(String(r.text || ''));
|
|
101
|
+
}
|
|
102
|
+
return [...byPath.entries()].sort(([a], [b]) => a.localeCompare(b))
|
|
103
|
+
.map(([p, d]) => ({ store, path: p, title: d.title, excerpt: excerptOf(d.chunks) }));
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
async function main() {
|
|
107
|
+
const args = process.argv.slice(2);
|
|
108
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
109
|
+
const kb = arg('--kb');
|
|
110
|
+
const out = arg('--out');
|
|
111
|
+
const perFile = Number(arg('--per-file', 3));
|
|
112
|
+
const batchSize = Number(arg('--batch', 25));
|
|
113
|
+
const conc = Number(arg('--conc', 2));
|
|
114
|
+
const model = arg('--model', 'haiku');
|
|
115
|
+
const done = new Set();
|
|
116
|
+
if (fs.existsSync(out)) {
|
|
117
|
+
for (const l of fs.readFileSync(out, 'utf8').split('\n')) { try { const r = JSON.parse(l); done.add(`${r.store}|${r.path}`); } catch { /* partial */ } }
|
|
118
|
+
}
|
|
119
|
+
let docs = String(arg('--stores')).split(',').flatMap((s) => readStoreDocs(kb, s.trim(), arg('--kind', 'md')))
|
|
120
|
+
.filter((d) => !done.has(`${d.store}|${d.path}`));
|
|
121
|
+
if (arg('--limit')) docs = docs.slice(0, Number(arg('--limit')));
|
|
122
|
+
const batches = [];
|
|
123
|
+
for (let i = 0; i < docs.length; i += batchSize) batches.push(docs.slice(i, i + batchSize));
|
|
124
|
+
const env = { ...subscriptionOnlyEnv(), CLAUDE_HOOK: '/usr/bin/true' };
|
|
125
|
+
const cwd = fs.mkdtempSync(path.join(os.tmpdir(), 'd2q-'));
|
|
126
|
+
let next = 0;
|
|
127
|
+
let stop = false;
|
|
128
|
+
let written = 0;
|
|
129
|
+
const worker = async () => {
|
|
130
|
+
while (!stop && next < batches.length) {
|
|
131
|
+
const b = batches[next++];
|
|
132
|
+
const withIds = b.map((d, i) => ({ ...d, docId: `d${i}` }));
|
|
133
|
+
const res = await spawnHost('claude', claudeArgs({ model, effort: 'low', schema: D2Q_SCHEMA, systemPrompt: D2Q_SYSTEM }),
|
|
134
|
+
{ cwd, env, timeoutMs: 600_000 }, d2qPrompt(withIds, perFile));
|
|
135
|
+
let items = null;
|
|
136
|
+
try {
|
|
137
|
+
const env2 = JSON.parse(res.stdout);
|
|
138
|
+
items = (env2.structured_output || JSON.parse(env2.result)).items;
|
|
139
|
+
} catch { items = null; }
|
|
140
|
+
if (!items) {
|
|
141
|
+
if (isQuotaRefusal(res.stdout + res.stderr)) { stop = true; process.stderr.write('\n[d2q] quota refusal: stopping\n'); }
|
|
142
|
+
else process.stderr.write(`\n[d2q] batch failed (status ${res.status}${res.timedOut ? ', timeout' : ''})\n`);
|
|
143
|
+
continue;
|
|
144
|
+
}
|
|
145
|
+
const byId = new Map(items.map((it) => [it.docId, it.questions || []]));
|
|
146
|
+
const lines = withIds.map((d) => {
|
|
147
|
+
const qs = byId.get(d.docId) || [];
|
|
148
|
+
const kept = [];
|
|
149
|
+
const rejected = [];
|
|
150
|
+
for (const q of qs) { const why = leakReason(q, d); if (why) rejected.push({ q, why }); else kept.push(q); }
|
|
151
|
+
return JSON.stringify({ store: d.store, path: d.path, questions: kept, rejected });
|
|
152
|
+
});
|
|
153
|
+
fs.appendFileSync(out, `${lines.join('\n')}\n`);
|
|
154
|
+
written += lines.length;
|
|
155
|
+
process.stderr.write(`\r[d2q] ${written}/${docs.length} files`);
|
|
156
|
+
}
|
|
157
|
+
};
|
|
158
|
+
await Promise.all(Array.from({ length: conc }, worker));
|
|
159
|
+
console.log(JSON.stringify({ files: docs.length, written, stopped: stop }));
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/doc2query-reach.mjs — ADR-099 arm A, measurement: do generated entry points bring
|
|
4
|
+
* the gold file into the pool?
|
|
5
|
+
*
|
|
6
|
+
* Builds one RVF "entry" index per store from doc2query-generate output (every question embedded with
|
|
7
|
+
* the production query embedder, so a newcomer's question is matched question-to-question), then, for
|
|
8
|
+
* every need, asks the gold store's entry index for its top --k questions and collapses them to files.
|
|
9
|
+
* Pool reach is reported against the existing pool (dense 64 + keyword lane, from an abstain-trace
|
|
10
|
+
* run's poolRanks), on the TRAIN and HELD-OUT splits separately, with Wilson intervals:
|
|
11
|
+
* baselineReach gold already pooled by dense + keyword
|
|
12
|
+
* entryReach gold among the entry lane's files
|
|
13
|
+
* unionReach either
|
|
14
|
+
* Restricted to needs whose gold repository the router searched (the trace's eligible set); over all
|
|
15
|
+
* needs only entryReach is reported (store-level, independent of routing).
|
|
16
|
+
*
|
|
17
|
+
* node scripts/oracle/doc2query-reach.mjs --runtime <kbDir with forge-ask.mjs + node_modules>
|
|
18
|
+
* --d2q <generated.jsonl> --set <need-set.json> --split <split.json> --trace <abstain-trace.json>
|
|
19
|
+
* [--k 24] [--files 8] [--index-dir <dir>] [--out <file>]
|
|
20
|
+
*/
|
|
21
|
+
import fs from 'node:fs';
|
|
22
|
+
import path from 'node:path';
|
|
23
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
24
|
+
import { wilson } from '../eval-brain.mjs';
|
|
25
|
+
|
|
26
|
+
/** Collapse ranked entry hits (each pointing at a file) to the first `files` distinct files. */
|
|
27
|
+
export function filesFromEntries(hits, pathOfId, files) {
|
|
28
|
+
const out = [];
|
|
29
|
+
for (const h of hits) {
|
|
30
|
+
const p = pathOfId(h.id);
|
|
31
|
+
if (p && !out.includes(p)) out.push(p);
|
|
32
|
+
if (out.length >= files) break;
|
|
33
|
+
}
|
|
34
|
+
return out;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function reachSummary(rows) {
|
|
38
|
+
const r = (pred) => { const k = rows.filter(pred).length; const w = wilson(k, rows.length); return { k, n: rows.length, lo: +w.lo.toFixed(4), hi: +w.hi.toFixed(4) }; };
|
|
39
|
+
return { baselineReach: r((x) => x.baseline), entryReach: r((x) => x.entry), unionReach: r((x) => x.baseline || x.entry),
|
|
40
|
+
gained: rows.filter((x) => x.entry && !x.baseline).length };
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
async function main() {
|
|
44
|
+
const args = process.argv.slice(2);
|
|
45
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
46
|
+
const runtime = path.resolve(arg('--runtime'));
|
|
47
|
+
const { __embedInternals } = await import(pathToFileURL(path.join(runtime, 'forge-ask.mjs')).href);
|
|
48
|
+
const { loadRvf } = await import(pathToFileURL(path.join(runtime, 'resolve-deps.mjs')).href);
|
|
49
|
+
const { RvfDatabase } = loadRvf().mod;
|
|
50
|
+
const embed = __embedInternals.embed;
|
|
51
|
+
const k = Number(arg('--k', 24));
|
|
52
|
+
const nFiles = Number(arg('--files', 8));
|
|
53
|
+
const indexDir = arg('--index-dir', fs.mkdtempSync(path.join(runtime, '..', 'd2q-index-')));
|
|
54
|
+
fs.mkdirSync(indexDir, { recursive: true });
|
|
55
|
+
const gen = fs.readFileSync(arg('--d2q'), 'utf8').split('\n').filter(Boolean).map((l) => JSON.parse(l));
|
|
56
|
+
const stores = [...new Set(gen.map((g) => g.store))];
|
|
57
|
+
const indexes = new Map();
|
|
58
|
+
// The entry index uses the store's own query embedder (big variant when present: bge + query prefix),
|
|
59
|
+
// exactly what searchKb computes for the question, so production needs one query embedding.
|
|
60
|
+
const cfgFor = (store) => {
|
|
61
|
+
for (const v of [`${store}.big.rvf.embed.json`, `${store}.rvf.embed.json`]) {
|
|
62
|
+
try { return JSON.parse(fs.readFileSync(path.join(runtime, v), 'utf8')); } catch { /* next */ }
|
|
63
|
+
}
|
|
64
|
+
return undefined;
|
|
65
|
+
};
|
|
66
|
+
for (const store of stores) {
|
|
67
|
+
const cfg = cfgFor(store);
|
|
68
|
+
const dims = cfg?.dimensions || 384;
|
|
69
|
+
const rows = gen.filter((g) => g.store === store).flatMap((g) => g.questions.map((q) => ({ q, path: g.path })));
|
|
70
|
+
const file = path.join(indexDir, `${store}.entry.rvf`);
|
|
71
|
+
fs.rmSync(file, { force: true });
|
|
72
|
+
const db = await RvfDatabase.create(file, { dimensions: dims, metric: 'cosine' });
|
|
73
|
+
const batch = [];
|
|
74
|
+
for (let i = 0; i < rows.length; i++) {
|
|
75
|
+
batch.push({ id: i + 1, vector: Array.from(await embed(rows[i].q, cfg)) });
|
|
76
|
+
if (batch.length === 64 || i === rows.length - 1) { await db.ingestBatch(batch.splice(0)); }
|
|
77
|
+
if (i % 500 === 0) process.stderr.write(`\r[d2q-reach] ${store} ${i}/${rows.length}`);
|
|
78
|
+
}
|
|
79
|
+
indexes.set(store, { db, cfg, paths: rows.map((r) => r.path), questions: rows.length, model: cfg?.model || 'default' });
|
|
80
|
+
}
|
|
81
|
+
const set = JSON.parse(fs.readFileSync(arg('--set'), 'utf8')).questions;
|
|
82
|
+
const split = JSON.parse(fs.readFileSync(arg('--split'), 'utf8'));
|
|
83
|
+
const trace = JSON.parse(fs.readFileSync(arg('--trace'), 'utf8'));
|
|
84
|
+
const pooled = new Map(trace.poolRanks.map((p) => [p.id, p.poolRank != null]));
|
|
85
|
+
const rows = [];
|
|
86
|
+
for (const q of set) {
|
|
87
|
+
const store = q.repo.toLowerCase();
|
|
88
|
+
const idx = indexes.get(store);
|
|
89
|
+
if (!idx) continue;
|
|
90
|
+
const hits = await idx.db.query(Array.from(await embed(q.need, idx.cfg)), k);
|
|
91
|
+
const files = filesFromEntries(hits, (id) => idx.paths[Number(id) - 1], nFiles);
|
|
92
|
+
rows.push({ id: q.id, split: split.train.includes(q.id) ? 'train' : 'heldout', routed: pooled.has(q.id),
|
|
93
|
+
baseline: pooled.get(q.id) === true, entry: files.includes(q.path), entryRank: files.indexOf(q.path) + 1 || null });
|
|
94
|
+
}
|
|
95
|
+
const report = { kind: 'ruvnet-brain-doc2query-reach', k, files: nFiles,
|
|
96
|
+
indexes: Object.fromEntries([...indexes].map(([s, v]) => [s, { questions: v.questions, model: v.model }])),
|
|
97
|
+
generatedFiles: gen.length, keptQuestions: gen.reduce((s, g) => s + g.questions.length, 0),
|
|
98
|
+
rejectedQuestions: gen.reduce((s, g) => s + (g.rejected?.length || 0), 0), results: {} };
|
|
99
|
+
for (const sp of ['train', 'heldout']) {
|
|
100
|
+
report.results[sp] = { routedNeeds: reachSummary(rows.filter((r) => r.split === sp && r.routed)),
|
|
101
|
+
// the trace covers only routed needs, so over all needs only the entry lane itself is known
|
|
102
|
+
allNeedsEntryReach: reachSummary(rows.filter((r) => r.split === sp)).entryReach };
|
|
103
|
+
}
|
|
104
|
+
report.rows = rows;
|
|
105
|
+
if (arg('--out')) fs.writeFileSync(arg('--out'), `${JSON.stringify(report, null, 1)}\n`);
|
|
106
|
+
console.log(JSON.stringify({ ...report, rows: undefined }, null, 1));
|
|
107
|
+
process.exit(0);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/judge-train.mjs — fit the learned judge (kb/judge-rank.mjs, ADR-099 arm C) on the
|
|
4
|
+
* need-set TRAIN split and report it on the HELD-OUT split, offline.
|
|
5
|
+
*
|
|
6
|
+
* Input is the scored pool the search already recorded (KB_CE_TRACE: one line per question, every
|
|
7
|
+
* pooled candidate with its cross-encoder logit, dense distance, lane, title, path and length), so
|
|
8
|
+
* training and evaluation replay exactly what the cross-encoder saw; no model is run here.
|
|
9
|
+
*
|
|
10
|
+
* Model: logistic regression over JUDGE_FEATURES, standardised on train, L2 1.0, class-balanced,
|
|
11
|
+
* full-batch gradient descent (deterministic). Positive = the candidate is the gold file or a
|
|
12
|
+
* pre-registered alternative. Operating point: the smallest threshold on the judge's top score at
|
|
13
|
+
* which TRAIN precision of answered questions reaches --precision (default 0.8); it is folded into
|
|
14
|
+
* the bias so the product's existing "score < 0 abstains" rule applies unchanged.
|
|
15
|
+
*
|
|
16
|
+
* node scripts/oracle/judge-train.mjs --set <need-set.json> --split <split.json> --trace <cetrace.jsonl>
|
|
17
|
+
* [--precision 0.8] [--weights <out.json>] [--report <out.json>]
|
|
18
|
+
* [--recall-trace <recall.cetrace.jsonl> --recall-fixture data/retrieval-query-evidence.json]
|
|
19
|
+
*/
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
23
|
+
import { wilson } from '../eval-brain.mjs';
|
|
24
|
+
|
|
25
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
|
|
26
|
+
const { JUDGE_FEATURES, judgeFeatures, judgeLogit } = await import(pathToFileURL(path.join(ROOT, 'kb/judge-rank.mjs')).href);
|
|
27
|
+
|
|
28
|
+
const norm = (p) => String(p || '').replace(/^\.\//, '');
|
|
29
|
+
export function isTarget(q, cand) {
|
|
30
|
+
const targets = [{ repo: q.repo, path: q.path }, ...(q.alternatives || [])];
|
|
31
|
+
return targets.some((t) => String(t.repo).toLowerCase() === String(cand.repo).toLowerCase() && norm(t.path) === norm(cand.path));
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** Map need id -> its last recorded pool, from KB_CE_TRACE lines (unparsable lines are skipped). */
|
|
35
|
+
export function poolsByNeed(questions, traceText) {
|
|
36
|
+
const byNeed = new Map(questions.map((q) => [q.need, q.id]));
|
|
37
|
+
const pools = new Map();
|
|
38
|
+
for (const line of traceText.split('\n')) {
|
|
39
|
+
if (!line.trim()) continue;
|
|
40
|
+
let row;
|
|
41
|
+
try { row = JSON.parse(line); } catch { continue; }
|
|
42
|
+
const id = byNeed.get(row.query);
|
|
43
|
+
if (id && Array.isArray(row.cands) && row.cands.length) pools.set(id, row.cands);
|
|
44
|
+
}
|
|
45
|
+
return pools;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function fitLogistic(X, y, { l2 = 1, iters = 3000, lr = 0.1 } = {}) {
|
|
49
|
+
const d = X[0].length;
|
|
50
|
+
const mean = Array.from({ length: d }, (_, j) => X.reduce((s, x) => s + x[j], 0) / X.length);
|
|
51
|
+
const std = Array.from({ length: d }, (_, j) => Math.sqrt(X.reduce((s, x) => s + (x[j] - mean[j]) ** 2, 0) / X.length) || 1);
|
|
52
|
+
const Z = X.map((x) => x.map((v, j) => (v - mean[j]) / std[j]));
|
|
53
|
+
const pos = y.filter(Boolean).length;
|
|
54
|
+
const wPos = pos ? (y.length - pos) / pos : 1;
|
|
55
|
+
const w = new Array(d).fill(0);
|
|
56
|
+
let b = 0;
|
|
57
|
+
for (let it = 0; it < iters; it++) {
|
|
58
|
+
const g = new Array(d).fill(0);
|
|
59
|
+
let gb = 0;
|
|
60
|
+
let W = 0;
|
|
61
|
+
for (let i = 0; i < Z.length; i++) {
|
|
62
|
+
const s = b + Z[i].reduce((a, v, j) => a + v * w[j], 0);
|
|
63
|
+
const p = 1 / (1 + Math.exp(-s));
|
|
64
|
+
const cw = y[i] ? wPos : 1;
|
|
65
|
+
const e = cw * (p - (y[i] ? 1 : 0));
|
|
66
|
+
for (let j = 0; j < d; j++) g[j] += e * Z[i][j];
|
|
67
|
+
gb += e;
|
|
68
|
+
W += cw;
|
|
69
|
+
}
|
|
70
|
+
for (let j = 0; j < d; j++) w[j] -= lr * (g[j] / W + (l2 * w[j]) / Z.length);
|
|
71
|
+
b -= lr * (gb / W);
|
|
72
|
+
}
|
|
73
|
+
return { weights: w, bias: b, mean, std };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Per question: the top candidate under a scorer, whether a target is in the top 1 / top 5, top score. */
|
|
77
|
+
export function decide(questions, pools, score) {
|
|
78
|
+
return questions.filter((q) => pools.has(q.id)).map((q) => {
|
|
79
|
+
const cands = pools.get(q.id);
|
|
80
|
+
const s = score(q, cands);
|
|
81
|
+
const order = s.map((v, i) => [v, i]).sort((a, b) => b[0] - a[0]).map(([, i]) => i);
|
|
82
|
+
return { id: q.id, top: s[order[0]], at1: isTarget(q, cands[order[0]]),
|
|
83
|
+
within5: order.slice(0, 5).some((i) => isTarget(q, cands[i])) };
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export function metrics(rows, n, threshold = 0) {
|
|
88
|
+
const r = (k, m) => { const w = wilson(k, m); return { k, n: m, lo: +w.lo.toFixed(4), hi: +w.hi.toFixed(4) }; };
|
|
89
|
+
const answered = rows.filter((x) => x.top >= threshold);
|
|
90
|
+
return {
|
|
91
|
+
traced: rows.length,
|
|
92
|
+
goldWithin5: r(rows.filter((x) => x.within5).length, n),
|
|
93
|
+
goldAt1: r(rows.filter((x) => x.at1).length, n),
|
|
94
|
+
// measure-need-set's definitions (target within 5 AND not abstained), so baselines compare
|
|
95
|
+
confidentHit: r(answered.filter((x) => x.within5).length, n),
|
|
96
|
+
confidentWrong: r(answered.filter((x) => !x.within5).length, n),
|
|
97
|
+
// the stricter reading: the single top answer is the target
|
|
98
|
+
confidentHitTop1: r(answered.filter((x) => x.at1).length, n),
|
|
99
|
+
abstained: r(n - answered.length, n),
|
|
100
|
+
precision: r(answered.filter((x) => x.at1).length, answered.length),
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function chooseThreshold(rows, precision) {
|
|
105
|
+
const tops = [...new Set(rows.map((x) => x.top))].sort((a, b) => a - b);
|
|
106
|
+
for (const t of tops) {
|
|
107
|
+
const ans = rows.filter((x) => x.top >= t);
|
|
108
|
+
if (ans.length && ans.filter((x) => x.at1).length / ans.length >= precision) return t;
|
|
109
|
+
}
|
|
110
|
+
return Infinity;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
async function main() {
|
|
114
|
+
const args = process.argv.slice(2);
|
|
115
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
116
|
+
const set = JSON.parse(fs.readFileSync(arg('--set'), 'utf8')).questions;
|
|
117
|
+
const split = JSON.parse(fs.readFileSync(arg('--split'), 'utf8'));
|
|
118
|
+
const pools = poolsByNeed(set, fs.readFileSync(arg('--trace'), 'utf8'));
|
|
119
|
+
const train = set.filter((q) => split.train.includes(q.id));
|
|
120
|
+
const held = set.filter((q) => split.heldout.includes(q.id));
|
|
121
|
+
const X = [];
|
|
122
|
+
const y = [];
|
|
123
|
+
for (const q of train) {
|
|
124
|
+
if (!pools.has(q.id)) continue;
|
|
125
|
+
const cands = pools.get(q.id);
|
|
126
|
+
judgeFeatures(q.need, cands).forEach((x, i) => { X.push(x); y.push(isTarget(q, cands[i])); });
|
|
127
|
+
}
|
|
128
|
+
const fit = fitLogistic(X, y);
|
|
129
|
+
const model0 = { ...fit, features: JUDGE_FEATURES };
|
|
130
|
+
const judgeScore = (m) => (q, cands) => judgeFeatures(q.need, cands).map((x) => judgeLogit(m, x));
|
|
131
|
+
const ceScore = (q, cands) => cands.map((c) => (typeof c.ce === 'number' ? c.ce : -Infinity));
|
|
132
|
+
const precision = Number(arg('--precision', 0.8));
|
|
133
|
+
const tau = chooseThreshold(decide(train, pools, judgeScore(model0)), precision);
|
|
134
|
+
const model = { ...model0, bias: Number.isFinite(tau) ? fit.bias - tau : -1e9, threshold: tau,
|
|
135
|
+
trainedOn: { split: split.salt, trainNeeds: train.length, tracedTrain: train.filter((q) => pools.has(q.id)).length,
|
|
136
|
+
candidates: X.length, positives: y.filter(Boolean).length, precisionTarget: precision } };
|
|
137
|
+
const report = {
|
|
138
|
+
kind: 'ruvnet-brain-judge-report', features: JUDGE_FEATURES, trainedOn: model.trainedOn, threshold: tau,
|
|
139
|
+
weights: Object.fromEntries(JUDGE_FEATURES.map((f, j) => [f, +model.weights[j].toFixed(4)])),
|
|
140
|
+
train: { crossEncoder: metrics(decide(train, pools, ceScore), train.length), judge: metrics(decide(train, pools, judgeScore(model)), train.length) },
|
|
141
|
+
heldout: { crossEncoder: metrics(decide(held, pools, ceScore), held.length), judge: metrics(decide(held, pools, judgeScore(model)), held.length) },
|
|
142
|
+
};
|
|
143
|
+
// Offline replay on the recall-gate fixture's recorded pools: does re-ranking hurt questions the
|
|
144
|
+
// cross-encoder already answers? (Approximate: selectResults' name boosts are not replayed here, so
|
|
145
|
+
// the full path is the authority; this only flags gross damage early.)
|
|
146
|
+
if (arg('--recall-trace') && arg('--recall-fixture')) {
|
|
147
|
+
const fx = JSON.parse(fs.readFileSync(arg('--recall-fixture'), 'utf8')).queries;
|
|
148
|
+
const items = Object.entries(fx).map(([store, v]) => ({ id: store, need: v.query, repo: store, path: v.expected.path }));
|
|
149
|
+
const rpools = poolsByNeed(items, fs.readFileSync(arg('--recall-trace'), 'utf8'));
|
|
150
|
+
report.recallReplay = { crossEncoder: metrics(decide(items, rpools, ceScore), items.length),
|
|
151
|
+
judge: metrics(decide(items, rpools, judgeScore(model)), items.length) };
|
|
152
|
+
}
|
|
153
|
+
if (arg('--weights')) fs.writeFileSync(arg('--weights'), `${JSON.stringify(model, null, 1)}\n`);
|
|
154
|
+
if (arg('--report')) fs.writeFileSync(arg('--report'), `${JSON.stringify(report, null, 1)}\n`);
|
|
155
|
+
console.log(JSON.stringify({ threshold: tau, trainedOn: model.trainedOn, heldout: report.heldout }, null, 1));
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/need-set-split.mjs — a frozen, stratified train / held-out split of a need set.
|
|
4
|
+
*
|
|
5
|
+
* Anything that LEARNS from novice needs (a query adapter, a learned abstain model) may train only
|
|
6
|
+
* on `train`; every claim about it is measured on `heldout`. The split is a pure function of the
|
|
7
|
+
* question ids and a salt: within each repository the ids are ordered by sha256(salt + id) and the
|
|
8
|
+
* first round(n * trainFraction) go to train. Re-running it on the same set yields the same split.
|
|
9
|
+
*
|
|
10
|
+
* node scripts/oracle/need-set-split.mjs --set <need-set.json> [--salt v1] [--train 0.5] [--out <file>]
|
|
11
|
+
*/
|
|
12
|
+
import { createHash } from 'node:crypto';
|
|
13
|
+
import fs from 'node:fs';
|
|
14
|
+
import path from 'node:path';
|
|
15
|
+
import { fileURLToPath } from 'node:url';
|
|
16
|
+
|
|
17
|
+
export function splitNeedSet(questions, { salt = 'v1', trainFraction = 0.5 } = {}) {
|
|
18
|
+
const byRepo = new Map();
|
|
19
|
+
for (const q of questions) {
|
|
20
|
+
if (!byRepo.has(q.repo)) byRepo.set(q.repo, []);
|
|
21
|
+
byRepo.get(q.repo).push(q.id);
|
|
22
|
+
}
|
|
23
|
+
const train = [];
|
|
24
|
+
const heldout = [];
|
|
25
|
+
for (const repo of [...byRepo.keys()].sort()) {
|
|
26
|
+
const ids = byRepo.get(repo)
|
|
27
|
+
.map((id) => ({ id, h: createHash('sha256').update(`${salt}\n${id}`).digest('hex') }))
|
|
28
|
+
.sort((a, b) => (a.h < b.h ? -1 : a.h > b.h ? 1 : 0))
|
|
29
|
+
.map((x) => x.id);
|
|
30
|
+
const cut = Math.round(ids.length * trainFraction);
|
|
31
|
+
train.push(...ids.slice(0, cut));
|
|
32
|
+
heldout.push(...ids.slice(cut));
|
|
33
|
+
}
|
|
34
|
+
return { salt, trainFraction, train, heldout };
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
async function main() {
|
|
38
|
+
const args = process.argv.slice(2);
|
|
39
|
+
const arg = (n, d) => { const i = args.indexOf(n); return i >= 0 ? args[i + 1] : d; };
|
|
40
|
+
const set = JSON.parse(fs.readFileSync(arg('--set'), 'utf8'));
|
|
41
|
+
const split = splitNeedSet(set.questions, { salt: arg('--salt', 'v1'), trainFraction: Number(arg('--train', 0.5)) });
|
|
42
|
+
const out = { kind: 'ruvnet-brain-need-set-split', set: { kind: set.kind, version: set.version, n: set.questions.length,
|
|
43
|
+
contentHash: set.contentHash }, ...split };
|
|
44
|
+
if (arg('--out')) fs.writeFileSync(arg('--out'), `${JSON.stringify(out, null, 1)}\n`);
|
|
45
|
+
console.log(JSON.stringify({ train: split.train.length, heldout: split.heldout.length, contentHash: set.contentHash }));
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) await main();
|