ruvnet-brain 4.4.0 → 4.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/README.md +3 -3
  2. package/bin/install.mjs +679 -121
  3. package/console/app.js +178 -81
  4. package/console/index.html +1 -1
  5. package/console/install-architecture.html +1 -0
  6. package/console/scope.css +4 -1
  7. package/console/style.css +13 -0
  8. package/console/tips.html +4 -4
  9. package/kb/brain-profile.mjs +1 -0
  10. package/kb/corpus-release-identity.mjs +1 -1
  11. package/kb/forge-update.mjs +41 -17
  12. package/kb/model-requirements.mjs +4 -1
  13. package/kb/update-storage-transaction.mjs +79 -0
  14. package/kb/zip-extract.mjs +22 -0
  15. package/package.json +1 -1
  16. package/plugin/.claude-plugin/plugin.json +1 -1
  17. package/plugin/.codex-plugin/plugin.json +1 -1
  18. package/plugin/commands/brain-console.md +5 -4
  19. package/plugin/commands/configure.md +5 -4
  20. package/plugin/commands/rnb-brief.md +41 -0
  21. package/plugin/commands/rnb.md +80 -0
  22. package/plugin/commands/rnbc.md +80 -0
  23. package/plugin/commands/rvbc.md +5 -4
  24. package/plugin/commands/rvcb.md +5 -4
  25. package/plugin/commands/whats-new.md +4 -4
  26. package/plugin/mcp/server.mjs +10 -2
  27. package/plugin/scripts/advocacy-route.mjs +59 -23
  28. package/plugin/scripts/anticipate.sh +4 -0
  29. package/plugin/scripts/brain-confirmation.mjs +258 -0
  30. package/plugin/scripts/brain-footprint.mjs +494 -0
  31. package/plugin/scripts/brain-location.mjs +47 -0
  32. package/plugin/scripts/capability-registry.mjs +11 -1
  33. package/plugin/scripts/continuity-brief.mjs +324 -0
  34. package/plugin/scripts/continuity-events.mjs +327 -0
  35. package/plugin/scripts/continuity-journal.mjs +500 -0
  36. package/plugin/scripts/decision-gate.mjs +56 -3
  37. package/plugin/scripts/footprint-io.mjs +186 -0
  38. package/plugin/scripts/ground-before-write.sh +8 -1
  39. package/plugin/scripts/ground-ruvnet.sh +103 -12
  40. package/plugin/scripts/grounding-answer.mjs +2 -1
  41. package/plugin/scripts/grounding-stamp.sh +3 -0
  42. package/plugin/scripts/grounding-substance.mjs +1 -1
  43. package/plugin/scripts/grounding-turn-evidence.mjs +68 -6
  44. package/plugin/scripts/hook-input.mjs +78 -4
  45. package/plugin/scripts/kb-copy-proof.mjs +148 -0
  46. package/plugin/scripts/lesson-bridge.mjs +6 -2
  47. package/plugin/scripts/nightly-controller.mjs +8 -1
  48. package/plugin/scripts/node-sqlite.mjs +41 -0
  49. package/plugin/scripts/package-cards.json +797 -0
  50. package/plugin/scripts/package-cards.rvf +0 -0
  51. package/plugin/scripts/package-cards.rvf.idmap.json +1 -0
  52. package/plugin/scripts/package-cards.rvf.meta.json +1 -0
  53. package/plugin/scripts/package-recommender-client.mjs +138 -0
  54. package/plugin/scripts/package-recommender-flag.mjs +30 -0
  55. package/plugin/scripts/package-recommender.mjs +391 -0
  56. package/plugin/scripts/project-progression-outbox.mjs +26 -8
  57. package/plugin/scripts/project-progression-reader.mjs +14 -2
  58. package/plugin/scripts/project-progression-store.mjs +153 -6
  59. package/plugin/scripts/protect-brain-state.sh +4 -1
  60. package/plugin/scripts/session-snapshot-hook.mjs +225 -41
  61. package/plugin/scripts/session-start-budget.mjs +1 -0
  62. package/plugin/scripts/session-start-core.mjs +34 -4
  63. package/plugin/scripts/session-start-health.mjs +7 -1
  64. package/plugin/scripts/session-start-update-plane.mjs +35 -0
  65. package/plugin/scripts/turn-outcome-capture.mjs +12 -1
  66. package/plugin/scripts/unprompted-runtime.mjs +2 -2
  67. package/plugin/skills/brain-console/SKILL.md +3 -3
  68. package/plugin/skills/rnbc/SKILL.md +24 -0
  69. package/plugin/skills/rvbc/SKILL.md +2 -2
  70. package/scripts/approved-runtime.mjs +2 -2
  71. package/scripts/ci/warm-brain-models.mjs +28 -0
  72. package/scripts/codex-hook-trust.mjs +94 -0
  73. package/scripts/console-instances.mjs +70 -12
  74. package/scripts/console-runtime-identity.mjs +5 -0
  75. package/scripts/corpus-canary.mjs +46 -6
  76. package/scripts/corpus-dispatch-decision.mjs +2 -2
  77. package/scripts/corpus-promotion.mjs +1 -1
  78. package/scripts/full-suite-gate.mjs +9 -2
  79. package/scripts/hook-qualify-hosts.mjs +15 -3
  80. package/scripts/host-install-matrix.mjs +63 -2
  81. package/scripts/human-approval-phrases.mjs +46 -0
  82. package/scripts/installed-brain-health.mjs +53 -0
  83. package/scripts/move-brain.mjs +310 -0
  84. package/scripts/onboarding-console.mjs +93 -10
  85. package/scripts/oracle/abstain-threshold-sweep.mjs +62 -0
  86. package/scripts/oracle/abstain-trace.mjs +139 -0
  87. package/scripts/oracle/doc2query-generate.mjs +162 -0
  88. package/scripts/oracle/doc2query-reach.mjs +110 -0
  89. package/scripts/oracle/judge-train.mjs +158 -0
  90. package/scripts/oracle/need-set-split.mjs +48 -0
  91. package/scripts/oracle/sona-query-adapter-eval.mjs +139 -0
  92. package/scripts/package-cards.mjs +374 -0
  93. package/scripts/publication-receipt.mjs +37 -9
  94. package/scripts/recommendation-e2e.mjs +110 -0
  95. package/scripts/recommendation-eval.mjs +105 -0
  96. package/scripts/recommendation-floor.mjs +56 -0
  97. package/scripts/recommendation-judge-score.mjs +74 -0
  98. package/scripts/recommendation-latency.mjs +95 -0
  99. package/scripts/recommendation-real-host-score.mjs +76 -0
  100. package/scripts/recommendation-real-host.mjs +137 -0
  101. package/scripts/release-channel-kind.mjs +1 -1
  102. package/scripts/release-environment-policy.mjs +33 -0
  103. package/scripts/single-source-check.mjs +15 -10
  104. package/scripts/sync-commands.mjs +5 -2
  105. package/scripts/wired-check.mjs +17 -2
@@ -0,0 +1,110 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * recommendation-e2e.mjs — the package recommender AS SHIPPED, end to end (ADR-093 rev 2).
4
+ *
5
+ * 1. Spawns the real search worker (kb/forge-mcp-all.mjs) in a throwaway brain home with the flag on,
6
+ * and runs its real `brain/warmup`, which starts the recommend endpoint.
7
+ * 2. For every eval prompt, runs the real hook producer (plugin/scripts/advocacy-route.mjs) as a
8
+ * cold process — flag OFF then flag ON, interleaved — and records wall time and the exact hint
9
+ * the host model would receive.
10
+ * 3. Writes a judge packet (qid, prompt, hint — no labels) and a separate key, so a model standing in
11
+ * for the host can decide what it would say without seeing the answers.
12
+ *
13
+ * Nothing touches the real ~/.claude, ~/.codex or ~/.cache/ruvnet-brain: HOME and RUVNET_BRAIN_HOME
14
+ * are a temp dir. The embedder needs a model cache and @xenova/transformers:
15
+ * --models <dir> (KB_MODEL_CACHE) --xenova <dir> (XENOVA_PATH)
16
+ * Refuses to measure when the 1-minute load average exceeds --max-load (default 60).
17
+ *
18
+ * node scripts/recommendation-e2e.mjs --models <d> --xenova <d> --out <dir> [--set f.json]... [--max-load 60] [--budget ms]
19
+ *
20
+ * --budget raises the hook's semantic budget for a QUALITY run (what the lane says when it answers);
21
+ * the default-budget run is the LATENCY run (how often, and how fast, it answers under its real limit).
22
+ */
23
+ import { spawn, spawnSync } from 'node:child_process';
24
+ import fs from 'node:fs';
25
+ import os from 'node:os';
26
+ import path from 'node:path';
27
+ import readline from 'node:readline';
28
+ import { fileURLToPath } from 'node:url';
29
+
30
+ const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
31
+ const arg = (n, d) => { const i = process.argv.indexOf(n); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
32
+ const sets = process.argv.flatMap((a, i) => (a === '--set' ? [process.argv[i + 1]] : []));
33
+ const outDir = arg('--out', null);
34
+ const maxLoad = Number(arg('--max-load', '60'));
35
+ if (!outDir) { console.error('--out <dir> is required'); process.exit(2); }
36
+ if (os.loadavg()[0] > maxLoad) { console.error(`load average ${os.loadavg()[0].toFixed(1)} > ${maxLoad}; refusing to measure`); process.exit(3); }
37
+
38
+ const home = fs.mkdtempSync(path.join(os.tmpdir(), 'reco-e2e-'));
39
+ const kb = path.join(home, 'kb');
40
+ fs.mkdirSync(kb);
41
+ const base = {
42
+ PATH: process.env.PATH, HOME: home, USERPROFILE: home, RUVNET_BRAIN_HOME: home, RUVNET_HOME_OVERRIDE: home,
43
+ KB_DIR: kb, KB_MODEL_CACHE: arg('--models', ''), XENOVA_PATH: arg('--xenova', ''),
44
+ };
45
+ const items = (sets.length ? sets : ['evals/recommendation-eval.v1.json'])
46
+ .flatMap((f) => JSON.parse(fs.readFileSync(path.resolve(ROOT, f), 'utf8')).items.map((i) => ({ ...i, set: path.basename(f) })));
47
+
48
+ const worker = spawn(process.execPath, [path.join(ROOT, 'kb', 'forge-mcp-all.mjs')],
49
+ { env: { ...base, RUVNET_PACKAGE_RECOMMENDER: '1' }, stdio: ['pipe', 'pipe', 'ignore'] });
50
+ const rl = readline.createInterface({ input: worker.stdout });
51
+ const waiters = new Map();
52
+ rl.on('line', (l) => { try { const m = JSON.parse(l); waiters.get(m.id)?.(m); } catch { /* not ours */ } });
53
+ const call = (id, method) => new Promise((r) => { waiters.set(id, r); worker.stdin.write(`${JSON.stringify({ jsonrpc: '2.0', id, method, params: {} })}\n`); });
54
+
55
+ function produce(prompt, flag, n) {
56
+ const env = {
57
+ ...base, RUVNET_EMIT_CANDIDATES: '1',
58
+ RUVNET_ADVOCACY_ROUTE_STATE: path.join(home, `state-${flag}-${n}.json`),
59
+ RUVNET_ADVOCACY_OUTCOMES: path.join(home, `outcomes-${flag}-${n}.jsonl`),
60
+ RUVNET_ADVOCACY_ROUTE_ROOTS: path.join(home, 'none'),
61
+ ...(flag === 'on' ? { RUVNET_PACKAGE_RECOMMENDER: '1' } : {}),
62
+ ...(arg('--budget', null) ? { RUVNET_PACKAGE_RECOMMENDER_BUDGET_MS: arg('--budget', null) } : {}),
63
+ // --floor 0 records every hint so a floor can be derived afterwards on the tuning set only.
64
+ ...(arg('--floor', null) !== null ? { RUVNET_PACKAGE_RECOMMENDER_MIN_SIMILARITY: arg('--floor', null) } : {}),
65
+ };
66
+ const t = process.hrtime.bigint();
67
+ const r = spawnSync(process.execPath, [path.join(ROOT, 'plugin', 'scripts', 'advocacy-route.mjs')],
68
+ { input: JSON.stringify({ session_id: `e2e-${flag}-${n}`, prompt }), env, encoding: 'utf8', timeout: 10000 });
69
+ const ms = Number(process.hrtime.bigint() - t) / 1e6;
70
+ let cand = null;
71
+ try { cand = r.stdout.trim() ? JSON.parse(r.stdout.trim()) : null; } catch { cand = null; }
72
+ return { ms, status: r.status, cand };
73
+ }
74
+
75
+ const q = (a, p) => { const s = [...a].sort((x, y) => x - y); return +s[Math.min(s.length - 1, Math.ceil(p * s.length) - 1)].toFixed(1); };
76
+ try {
77
+ await call(1, 'initialize');
78
+ const warm = await call(2, 'brain/warmup');
79
+ if (!warm.result?.ready) throw new Error(`warmup failed: ${JSON.stringify(warm.error || warm.result)}`);
80
+ for (let i = 0; i < 50 && !fs.existsSync(path.join(home, 'run')); i++) await new Promise((r) => setTimeout(r, 100));
81
+ produce('warm the page cache once before measuring', 'on', 'warm');
82
+ const packet = []; const key = []; const off = []; const on = []; const added = [];
83
+ items.forEach((it, i) => {
84
+ const a = produce(it.prompt, 'off', i);
85
+ const b = produce(it.prompt, 'on', i);
86
+ off.push(a.ms); on.push(b.ms); added.push(b.ms - a.ms);
87
+ const qid = `Q${String(i + 1).padStart(3, '0')}`;
88
+ key.push({ qid, set: it.set, id: it.id, category: it.category, accept: it.accept || [], acceptStores: it.acceptStores || [],
89
+ lane: b.cand ? (b.cand.candidates ? 'semantic' : String(b.cand.findingId).startsWith('recommend:pkg:') ? 'lexical' : 'catalogue') : null,
90
+ offered: b.cand ? (b.cand.candidates || [b.cand.package || b.cand.capability]) : [], status: b.status,
91
+ topSimilarity: Array.isArray(b.cand?.similarities) ? b.cand.similarities[0] : null });
92
+ if (b.cand) packet.push({ qid, prompt: it.prompt, hint: b.cand.copy });
93
+ });
94
+ const timing = {
95
+ samples: items.length, loadAvg1m: +os.loadavg()[0].toFixed(1), host: `${os.platform()} ${os.arch()} ${os.cpus().length} vCPU node ${process.version}`,
96
+ flagOffMs: { p50: q(off, 0.5), p90: q(off, 0.9) }, flagOnMs: { p50: q(on, 0.5), p90: q(on, 0.9), max: q(on, 1) },
97
+ addedMs: { p50: q(added, 0.5), p90: q(added, 0.9), max: q(added, 1) },
98
+ };
99
+ fs.mkdirSync(outDir, { recursive: true });
100
+ fs.writeFileSync(path.join(outDir, 'judge-packet.json'), JSON.stringify(packet.sort(() => 0), null, 1));
101
+ fs.writeFileSync(path.join(outDir, 'judge-key.json'), JSON.stringify(key, null, 1));
102
+ fs.writeFileSync(path.join(outDir, 'timing.json'), JSON.stringify(timing, null, 1));
103
+ console.log(JSON.stringify(timing));
104
+ console.log(`hints emitted for ${packet.length}/${items.length} prompts; lanes: ${JSON.stringify(key.reduce((m, k) => ({ ...m, [k.lane]: (m[k.lane] || 0) + 1 }), {}))}`);
105
+ } finally {
106
+ worker.kill('SIGTERM');
107
+ await new Promise((r) => setTimeout(r, 300));
108
+ fs.rmSync(home, { recursive: true, force: true });
109
+ }
110
+ process.exit(0);
@@ -0,0 +1,105 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * recommendation-eval.mjs — score the package recommender against a need-prompt eval set (ADR-0093).
4
+ *
5
+ * node scripts/recommendation-eval.mjs [--set evals/recommendation-eval.v1.json]... [--cards <file>] [--json] [--verbose]
6
+ *
7
+ * Reports, per split and overall: recall on positives (fired AND an accepted package), precision over
8
+ * all firings, false-firing rate on negatives (off-topic + no-fit) — each with a Wilson 95% interval,
9
+ * because 30 prompts is a small sample and a bare percentage would overstate what it shows.
10
+ * An `accept` id that is not in the card set is reported, never silently counted as a miss.
11
+ */
12
+ import fs from 'node:fs';
13
+ import path from 'node:path';
14
+ import { fileURLToPath } from 'node:url';
15
+ import { indexCards, rank, SNAPSHOT_FILE } from '../plugin/scripts/package-recommender.mjs';
16
+
17
+ const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
18
+
19
+ /** Wilson score interval, 95%. Returns [lo, hi] as fractions; [0, 0] for n = 0. */
20
+ export function wilson(k, n, z = 1.96) {
21
+ if (!n) return [0, 0];
22
+ const p = k / n;
23
+ const den = 1 + (z * z) / n;
24
+ const centre = (p + (z * z) / (2 * n)) / den;
25
+ const half = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / den;
26
+ return [Math.max(0, centre - half), Math.min(1, centre + half)];
27
+ }
28
+
29
+ const POSITIVE = new Set(['design', 'diagnosis']);
30
+
31
+ export function scoreItem(item, decision) {
32
+ const fired = Boolean(decision);
33
+ const pkg = decision?.card?.id || null;
34
+ const store = decision?.card?.store || null;
35
+ if (!POSITIVE.has(item.category)) return { fired, pkg, correct: !fired, firedCorrect: false };
36
+ const accepted = fired && ((item.accept || []).includes(pkg) || (item.acceptStores || []).includes(store));
37
+ return { fired, pkg, correct: accepted, firedCorrect: accepted };
38
+ }
39
+
40
+ export function summarize(rows) {
41
+ const pos = rows.filter((r) => POSITIVE.has(r.category));
42
+ const neg = rows.filter((r) => !POSITIVE.has(r.category));
43
+ const fired = rows.filter((r) => r.fired);
44
+ const firedCorrect = rows.filter((r) => r.firedCorrect).length;
45
+ const hits = pos.filter((r) => r.correct).length;
46
+ const falseFires = neg.filter((r) => r.fired).length;
47
+ const pct = ([lo, hi]) => [+(lo * 100).toFixed(1), +(hi * 100).toFixed(1)];
48
+ return {
49
+ n: rows.length,
50
+ positives: pos.length,
51
+ negatives: neg.length,
52
+ recall: { k: hits, n: pos.length, pct: pos.length ? +((hits / pos.length) * 100).toFixed(1) : null, ci95: pct(wilson(hits, pos.length)) },
53
+ precision: { k: firedCorrect, n: fired.length, pct: fired.length ? +((firedCorrect / fired.length) * 100).toFixed(1) : null, ci95: pct(wilson(firedCorrect, fired.length)) },
54
+ falseFiring: { k: falseFires, n: neg.length, pct: neg.length ? +((falseFires / neg.length) * 100).toFixed(1) : null, ci95: pct(wilson(falseFires, neg.length)) },
55
+ wrongPackage: pos.filter((r) => r.fired && !r.correct).length,
56
+ };
57
+ }
58
+
59
+ export function evaluate(items, index) {
60
+ const ids = new Set(index.entries.map((e) => e.card.id));
61
+ const missing = [...new Set(items.flatMap((i) => i.accept || []).filter((id) => !ids.has(id)))];
62
+ const rows = items.map((item) => {
63
+ const { decision, reason, ranked } = rank(item.prompt, index);
64
+ return {
65
+ id: item.id, category: item.category, split: item.split, prompt: item.prompt, accept: item.accept,
66
+ ...scoreItem(item, decision),
67
+ reason, top: ranked.slice(0, 3).map((r) => `${r.card.id}:${r.score.toFixed(2)}[${r.matched.join(',')}]`),
68
+ };
69
+ });
70
+ const splits = [...new Set(rows.map((r) => r.split))];
71
+ return {
72
+ missingAcceptIds: missing,
73
+ overall: summarize(rows),
74
+ bySplit: Object.fromEntries(splits.map((s) => [s, summarize(rows.filter((r) => r.split === s))])),
75
+ rows,
76
+ };
77
+ }
78
+
79
+ const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
80
+ if (isMain) {
81
+ const sets = [];
82
+ let cards = SNAPSHOT_FILE;
83
+ for (let i = 2; i < process.argv.length; i++) {
84
+ if (process.argv[i] === '--set') sets.push(process.argv[++i]);
85
+ if (process.argv[i] === '--cards') cards = process.argv[++i];
86
+ }
87
+ if (!sets.length) sets.push(path.join(ROOT, 'evals', 'recommendation-eval.v1.json'));
88
+ // --split dev restricts BOTH scoring and output to one split, so tuning never shows held-out rows.
89
+ const splitArg = process.argv.includes('--split') ? process.argv[process.argv.indexOf('--split') + 1] : null;
90
+ const items = sets.flatMap((f) => JSON.parse(fs.readFileSync(f, 'utf8')).items)
91
+ .filter((i) => !splitArg || i.split === splitArg);
92
+ const index = indexCards(JSON.parse(fs.readFileSync(cards, 'utf8')));
93
+ const result = evaluate(items, index);
94
+ if (process.argv.includes('--json')) { process.stdout.write(`${JSON.stringify(result, null, 1)}\n`); process.exit(0); }
95
+ const line = (name, s) => console.log(`${name.padEnd(9)} n=${s.n} recall ${s.recall.k}/${s.recall.n} ${s.recall.pct}% [${s.recall.ci95.join('–')}] precision ${s.precision.k}/${s.precision.n} ${s.precision.pct}% [${s.precision.ci95.join('–')}] false-firing ${s.falseFiring.k}/${s.falseFiring.n} ${s.falseFiring.pct}% [${s.falseFiring.ci95.join('–')}] wrong-pkg ${s.wrongPackage}`);
96
+ if (result.missingAcceptIds.length) console.log(`accept ids not in card set: ${result.missingAcceptIds.join(', ')}`);
97
+ line('overall', result.overall);
98
+ for (const [s, v] of Object.entries(result.bySplit)) line(s, v);
99
+ if (process.argv.includes('--verbose')) {
100
+ for (const r of result.rows) {
101
+ const mark = r.correct ? 'ok ' : 'XX ';
102
+ console.log(`${mark}${r.id} ${r.category.padEnd(9)} -> ${r.fired ? r.pkg : '(silent:' + r.reason + ')'} | ${r.top.join(' ')}`);
103
+ }
104
+ }
105
+ }
@@ -0,0 +1,56 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * recommendation-floor.mjs — derive the semantic similarity floor on the TUNING set only, freeze it, and
4
+ * apply it to every set (ADR-093 rev 3). The rule is fixed in advance: the floor is the 5th percentile
5
+ * of the nearest card's similarity among the tuning set's correctly-judged recommendations (family-aware).
6
+ *
7
+ * node scripts/recommendation-floor.mjs --dir <e2e out dir with judge-key.json + judge-picks.json>
8
+ * [--tune recommendation-eval.v1.json] [--floor <value to apply instead of deriving>] [--json]
9
+ *
10
+ * The e2e run must be made with --floor 0 so every hint (and its top similarity) is recorded.
11
+ */
12
+ import fs from 'node:fs';
13
+ import path from 'node:path';
14
+ import { fileURLToPath } from 'node:url';
15
+ import { judgeScore } from './recommendation-judge-score.mjs';
16
+ import { wilson } from './recommendation-eval.mjs';
17
+
18
+ const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
19
+ const POS = new Set(['design', 'diagnosis']);
20
+
21
+ export function deriveFloor(key, picks, cards, tuneSet, pct = 0.05) {
22
+ const productOf = new Map(cards.map((c) => [c.id, c.product || c.id]));
23
+ const right = (k, p) => k.accept.includes(p) || k.accept.some((a) => productOf.has(p) && productOf.get(a) === productOf.get(p));
24
+ const sims = key.filter((k) => k.set === tuneSet && k.lane === 'semantic' && POS.has(k.category) && picks[k.qid] && right(k, picks[k.qid]))
25
+ .map((k) => k.topSimilarity).filter(Number.isFinite).sort((a, b) => a - b);
26
+ return sims.length ? +sims[Math.floor(sims.length * pct)].toFixed(3) : null;
27
+ }
28
+
29
+ /** Picks as the shipped hook would produce them under `floor`: below it, no hint, so no pick. */
30
+ export function applyFloor(key, picks, floor) {
31
+ return Object.fromEntries(key.map((k) => [k.qid, k.lane === 'semantic' && !(k.topSimilarity >= floor) ? null : picks[k.qid] ?? null]));
32
+ }
33
+
34
+ const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
35
+ if (isMain) {
36
+ const arg = (n, d) => { const i = process.argv.indexOf(n); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
37
+ const dir = arg('--dir', null);
38
+ const key = JSON.parse(fs.readFileSync(path.join(dir, 'judge-key.json'), 'utf8'));
39
+ const picks = JSON.parse(fs.readFileSync(path.join(dir, 'judge-picks.json'), 'utf8')).picks;
40
+ const cards = JSON.parse(fs.readFileSync(path.join(ROOT, 'plugin', 'scripts', 'package-cards.json'), 'utf8')).cards;
41
+ const floor = arg('--floor', null) !== null ? Number(arg('--floor')) : deriveFloor(key, picks, cards, arg('--tune', 'recommendation-eval.v1.json'));
42
+ const gated = applyFloor(key, picks, floor);
43
+ const strict = judgeScore(key, gated, cards);
44
+ const family = judgeScore(key, gated, cards, { familyAware: true });
45
+ const negHinted = (set) => key.filter((k) => set(k) && !POS.has(k.category) && k.lane === 'semantic' && k.topSimilarity >= floor).length;
46
+ const negs = key.filter((k) => /blind/.test(k.set) && !POS.has(k.category)).length;
47
+ const out = { floor, strict, family, hintsOnBlindNegatives: { k: negHinted((k) => /blind/.test(k.set)), n: negs, ci95: wilson(negHinted((k) => /blind/.test(k.set)), negs).map((x) => +(x * 100).toFixed(1)) } };
48
+ if (process.argv.includes('--json')) { process.stdout.write(`${JSON.stringify(out, null, 1)}\n`); process.exit(0); }
49
+ console.log(`floor ${floor}`);
50
+ const line = (name, s) => console.log(`${name.padEnd(40)} recall ${s.recall.k}/${s.recall.n} ${s.recall.pct}% [${s.recall.ci95.join('–')}] precision ${s.precision.k}/${s.precision.n} ${s.precision.pct}% [${s.precision.ci95.join('–')}] false-firing ${s.falseFiring.k}/${s.falseFiring.n} [${s.falseFiring.ci95.join('–')}]`);
51
+ for (const [mode, r] of [['strict', strict], ['family', family]]) {
52
+ line(`${mode} blinds`, r.blinds);
53
+ for (const [s, v] of Object.entries(r.bySet)) line(`${mode} ${s}`, v);
54
+ }
55
+ console.log(`hints injected on blind negatives: ${out.hintsOnBlindNegatives.k}/${out.hintsOnBlindNegatives.n} [${out.hintsOnBlindNegatives.ci95.join('–')}]`);
56
+ }
@@ -0,0 +1,74 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * recommendation-judge-score.mjs — score what the host model SAID, given the hints the shipped
4
+ * pipeline emitted (ADR-093 rev 2). Inputs come from scripts/recommendation-e2e.mjs (judge-key.json)
5
+ * and a judge that saw only prompts + hints (judge-picks.json: { picks: { qid: id|null } }).
6
+ *
7
+ * node scripts/recommendation-judge-score.mjs --dir <e2e out dir> [--cards plugin/scripts/package-cards.json] [--family] [--json]
8
+ *
9
+ * A prompt the pipeline stayed silent on counts as silent. Precision is over what the model said.
10
+ * Wilson 95% intervals throughout.
11
+ */
12
+ import fs from 'node:fs';
13
+ import path from 'node:path';
14
+ import { fileURLToPath } from 'node:url';
15
+ import { wilson } from './recommendation-eval.mjs';
16
+
17
+ const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
18
+ const POS = new Set(['design', 'diagnosis']);
19
+ // The closed catalogue speaks in building-block names; these are the packages each one names.
20
+ export const CATALOGUE_PACKAGES = Object.freeze({
21
+ ruvector: ['ruvector', '@ruvector/rvf', '@ruvector/core', '@ruvector/node'],
22
+ agentdb: ['agentdb'],
23
+ ruflo: ['ruflo', 'claude-flow', '@claude-flow/swarm'],
24
+ aidefence: ['@claude-flow/aidefence', 'aidefence-core'],
25
+ 'agentic-qe': ['agentic-qe'],
26
+ 'agentic-flow': ['agentic-flow'],
27
+ rulake: ['rulake'],
28
+ });
29
+
30
+ export function judgeScore(key, picks, cards, { familyAware = false } = {}) {
31
+ const storeOf = new Map(cards.map((c) => [c.id, c.store]));
32
+ // FAMILY-AWARE (ADR-093 rev 3): a pick is right when it is the same PRODUCT as an accepted id, per the
33
+ // corpus-derived product map on the cards (scripts/package-cards.mjs deriveProducts). Off by default so
34
+ // the strict number is always the one reported first.
35
+ const productOf = new Map(cards.map((c) => [c.id, c.product || c.id]));
36
+ const sameProduct = (a, b) => familyAware && productOf.has(a) && productOf.get(a) === productOf.get(b);
37
+ const sets = [...new Set(key.map((k) => k.set))];
38
+ const pct = (k, n) => ({ k, n, pct: n ? +((100 * k) / n).toFixed(1) : null, ci95: wilson(k, n).map((x) => +(x * 100).toFixed(1)) });
39
+ const summarize = (rows) => {
40
+ let hit = 0; let said = 0; let correct = 0; let ff = 0; let wrong = 0;
41
+ const pos = rows.filter((r) => POS.has(r.category)).length;
42
+ for (const r of rows) {
43
+ const p = picks[r.qid] ?? null;
44
+ if (!p) continue;
45
+ said++;
46
+ const ids = [p, ...(CATALOGUE_PACKAGES[p] || [])];
47
+ const ok = ids.some((id) => r.accept.includes(id) || r.acceptStores.includes(storeOf.get(id))
48
+ || r.accept.some((a) => sameProduct(id, a)));
49
+ if (POS.has(r.category)) { if (ok) { hit++; correct++; } else wrong++; } else ff++;
50
+ }
51
+ return { n: rows.length, recall: pct(hit, pos), precision: pct(correct, said), falseFiring: pct(ff, rows.length - pos), wrongPackage: wrong };
52
+ };
53
+ return {
54
+ overall: summarize(key),
55
+ blinds: summarize(key.filter((k) => /blind/.test(k.set))),
56
+ bySet: Object.fromEntries(sets.map((s) => [s, summarize(key.filter((k) => k.set === s))])),
57
+ };
58
+ }
59
+
60
+ const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
61
+ if (isMain) {
62
+ const arg = (n, d) => { const i = process.argv.indexOf(n); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
63
+ const dir = arg('--dir', null);
64
+ if (!dir) { console.error('--dir required'); process.exit(2); }
65
+ const key = JSON.parse(fs.readFileSync(path.join(dir, 'judge-key.json'), 'utf8'));
66
+ const picks = JSON.parse(fs.readFileSync(path.join(dir, 'judge-picks.json'), 'utf8')).picks || {};
67
+ const cards = JSON.parse(fs.readFileSync(arg('--cards', path.join(ROOT, 'plugin', 'scripts', 'package-cards.json')), 'utf8')).cards;
68
+ const r = judgeScore(key, picks, cards, { familyAware: process.argv.includes('--family') });
69
+ if (process.argv.includes('--json')) { process.stdout.write(`${JSON.stringify(r, null, 1)}\n`); process.exit(0); }
70
+ const line = (name, s) => console.log(`${name.padEnd(36)} recall ${s.recall.k}/${s.recall.n} ${s.recall.pct}% [${s.recall.ci95.join('–')}] precision ${s.precision.k}/${s.precision.n} ${s.precision.pct}% [${s.precision.ci95.join('–')}] false-firing ${s.falseFiring.k}/${s.falseFiring.n} ${s.falseFiring.pct}% [${s.falseFiring.ci95.join('–')}] wrong ${s.wrongPackage}`);
71
+ line('overall', r.overall);
72
+ line('blinds (1+2)', r.blinds);
73
+ for (const [s, v] of Object.entries(r.bySet)) line(s, v);
74
+ }
@@ -0,0 +1,95 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * recommendation-latency.mjs — what the package recommender ADDS to every prompt (ADR-0093).
4
+ *
5
+ * Measured where the cost is paid: a COLD `node advocacy-route.mjs` process per prompt, exactly as
6
+ * unprompted-runtime.mjs spawns it on UserPromptSubmit. For each eval prompt the producer runs twice,
7
+ * flag OFF then flag ON, interleaved so machine drift lands on both arms; the per-prompt difference is
8
+ * the added latency. Every path the producer could write (state, ledger, HOME) is a fresh temp dir.
9
+ *
10
+ * node scripts/recommendation-latency.mjs [--set <eval.json>]... [--rounds 1] [--json]
11
+ */
12
+ import { spawnSync } from 'node:child_process';
13
+ import fs from 'node:fs';
14
+ import os from 'node:os';
15
+ import path from 'node:path';
16
+ import { fileURLToPath } from 'node:url';
17
+
18
+ const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
19
+ const ROUTE = path.join(ROOT, 'plugin', 'scripts', 'advocacy-route.mjs');
20
+ const CARDS = path.join(ROOT, 'plugin', 'scripts', 'package-cards.json');
21
+
22
+ const sets = [];
23
+ let rounds = 1;
24
+ for (let i = 2; i < process.argv.length; i++) {
25
+ if (process.argv[i] === '--set') sets.push(process.argv[++i]);
26
+ if (process.argv[i] === '--rounds') rounds = Math.max(1, Number(process.argv[++i]) || 1);
27
+ }
28
+ if (!sets.length) sets.push(path.join(ROOT, 'evals', 'recommendation-eval.v1.json'));
29
+ const prompts = sets.flatMap((f) => JSON.parse(fs.readFileSync(f, 'utf8')).items.map((i) => i.prompt));
30
+
31
+ const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'reco-latency-'));
32
+ // Only what the producer needs; nothing inherited that could point it at a real store.
33
+ const baseEnv = {
34
+ PATH: process.env.PATH,
35
+ HOME: tmp,
36
+ USERPROFILE: tmp,
37
+ RUVNET_HOME_OVERRIDE: tmp,
38
+ RUVNET_EMIT_CANDIDATES: '1',
39
+ RUVNET_PACKAGE_CARDS: CARDS,
40
+ RUVNET_ADVOCACY_ROUTE_ROOTS: path.join(tmp, 'no-modules'),
41
+ };
42
+
43
+ function run(prompt, flag, n) {
44
+ const env = {
45
+ ...baseEnv,
46
+ RUVNET_ADVOCACY_ROUTE_STATE: path.join(tmp, `state-${flag}-${n}.json`),
47
+ RUVNET_ADVOCACY_OUTCOMES: path.join(tmp, `outcomes-${flag}-${n}.jsonl`),
48
+ ...(flag === 'on' ? { RUVNET_PACKAGE_RECOMMENDER: '1' } : {}),
49
+ };
50
+ const input = JSON.stringify({ session_id: `lat-${flag}-${n}`, cwd: ROOT, hook_event_name: 'UserPromptSubmit', prompt });
51
+ const t0 = process.hrtime.bigint();
52
+ const r = spawnSync(process.execPath, [ROUTE], { input, env, encoding: 'utf8', timeout: 10000 });
53
+ const ms = Number(process.hrtime.bigint() - t0) / 1e6;
54
+ return { ms, status: r.status, emitted: Boolean((r.stdout || '').trim()) };
55
+ }
56
+
57
+ const q = (arr, p) => { const s = [...arr].sort((a, b) => a - b); return s[Math.min(s.length - 1, Math.ceil(p * s.length) - 1)]; };
58
+ const off = []; const on = []; const added = []; let failures = 0;
59
+ let n = 0;
60
+ try {
61
+ run('warm up the filesystem cache once', 'off', 'warm');
62
+ for (let r = 0; r < rounds; r++) {
63
+ for (const prompt of prompts) {
64
+ n++;
65
+ const a = run(prompt, 'off', n);
66
+ const b = run(prompt, 'on', n);
67
+ if (a.status !== 0 || b.status !== 0) failures++;
68
+ off.push(a.ms); on.push(b.ms); added.push(b.ms - a.ms);
69
+ }
70
+ }
71
+ } finally { fs.rmSync(tmp, { recursive: true, force: true }); }
72
+
73
+ const fmt = (arr) => ({ p50: +q(arr, 0.5).toFixed(1), p90: +q(arr, 0.9).toFixed(1), max: +Math.max(...arr).toFixed(1) });
74
+ // Percentile bootstrap (2000 resamples, fixed seed) for the ADDED p50/p90, so the report carries an
75
+ // interval rather than one number from one noisy run.
76
+ function bootstrap(arr, p, B = 2000) {
77
+ let seed = 0x9e3779b9;
78
+ const rnd = () => { seed ^= seed << 13; seed ^= seed >>> 17; seed ^= seed << 5; return ((seed >>> 0) % 1e9) / 1e9; };
79
+ const stats = [];
80
+ for (let b = 0; b < B; b++) stats.push(q(Array.from(arr, () => arr[Math.floor(rnd() * arr.length)]), p));
81
+ return [+q(stats, 0.025).toFixed(1), +q(stats, 0.975).toFixed(1)];
82
+ }
83
+ const out = {
84
+ samples: n, failures,
85
+ host: `${os.platform()} ${os.arch()} ${os.cpus().length} vCPU node ${process.version}`,
86
+ loadAvg1m: +os.loadavg()[0].toFixed(1),
87
+ flagOffMs: fmt(off), flagOnMs: fmt(on), addedMs: { ...fmt(added), p50ci95: bootstrap(added, 0.5), p90ci95: bootstrap(added, 0.9) },
88
+ };
89
+ if (process.argv.includes('--json')) process.stdout.write(`${JSON.stringify(out, null, 1)}\n`);
90
+ else {
91
+ console.log(`[latency] ${out.samples} paired cold runs on ${out.host}, ${failures} non-zero exits`);
92
+ console.log(`[latency] flag off : p50 ${out.flagOffMs.p50} ms p90 ${out.flagOffMs.p90} ms max ${out.flagOffMs.max} ms`);
93
+ console.log(`[latency] flag on : p50 ${out.flagOnMs.p50} ms p90 ${out.flagOnMs.p90} ms max ${out.flagOnMs.max} ms`);
94
+ console.log(`[latency] added : p50 ${out.addedMs.p50} ms p90 ${out.addedMs.p90} ms max ${out.addedMs.max} ms`);
95
+ }
@@ -0,0 +1,76 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * recommendation-real-host-score.mjs — score a real-host run (scripts/recommendation-real-host.mjs) and
4
+ * its agreement with the simulated host (ADR-093 rev 3).
5
+ *
6
+ * node scripts/recommendation-real-host-score.mjs --real <real-host.json> --dir <e2e dir> [--floor 0.532] [--json]
7
+ *
8
+ * Correctness is family-aware (same product as an accepted id). Agreement compares, per prompt, what the
9
+ * real host named with what the simulated host would have named under the same floor: both silent, or
10
+ * the same product. A prompt the real hook did NOT inject a hint for is reported separately, because
11
+ * then the two hosts did not see the same input.
12
+ */
13
+ import fs from 'node:fs';
14
+ import path from 'node:path';
15
+ import { fileURLToPath } from 'node:url';
16
+ import { wilson } from './recommendation-eval.mjs';
17
+ import { applyFloor } from './recommendation-floor.mjs';
18
+
19
+ const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
20
+ const POS = new Set(['design', 'diagnosis']);
21
+
22
+ export function scoreRealHost(real, key, simPicks, cards) {
23
+ const productOf = new Map(cards.map((c) => [c.id, c.product || c.id]));
24
+ const same = (a, b) => a === b || (productOf.has(a) && productOf.get(a) === productOf.get(b));
25
+ const byQid = new Map(key.map((k) => [k.qid, k]));
26
+ const rows = real.rows.map((r) => {
27
+ const k = byQid.get(r.qid);
28
+ const right = r.said ? (k.accept || []).some((a) => same(r.said, a)) : null;
29
+ const sim = simPicks[r.qid] ?? null;
30
+ const agree = (!r.said && !sim) || (r.said && sim && same(r.said, sim));
31
+ return { ...r, accept: k.accept, sim, right, agree };
32
+ });
33
+ const pos = rows.filter((r) => POS.has(r.category));
34
+ const neg = rows.filter((r) => !POS.has(r.category));
35
+ const said = rows.filter((r) => r.said);
36
+ const pct = (a, n) => ({ k: a, n, pct: n ? +((100 * a) / n).toFixed(1) : null, ci95: wilson(a, n).map((x) => +(x * 100).toFixed(1)) });
37
+ const sameInput = rows.filter((r) => r.injected === Boolean(byQid.get(r.qid).lane));
38
+ // Hint delivery against the 1-minute load each prompt ran at: how the lane behaves on a busy laptop.
39
+ const buckets = [[0, 60], [60, 120], [120, 240], [240, Infinity]];
40
+ const deliveryByLoad = buckets.map(([lo, hi]) => {
41
+ const inB = rows.filter((r) => r.expectHint && Number.isFinite(r.load1m) && r.load1m >= lo && r.load1m < hi);
42
+ return { load: hi === Infinity ? `>=${lo}` : `${lo}-${hi}`, ...pct(inB.filter((r) => r.injected).length, inB.length) };
43
+ });
44
+ return {
45
+ n: rows.length,
46
+ deliveryByLoad,
47
+ recall: pct(pos.filter((r) => r.right).length, pos.length),
48
+ precision: pct(said.filter((r) => POS.has(r.category) && r.right).length, said.length),
49
+ falseFiring: pct(neg.filter((r) => r.said).length, neg.length),
50
+ hintDeliveredWhereExpected: pct(rows.filter((r) => r.expectHint && r.injected).length, rows.filter((r) => r.expectHint).length),
51
+ hintsNotExpected: rows.filter((r) => !r.expectHint && r.injected).length, // catalogue/lexical lanes, which the floor does not govern
52
+ agreement: pct(rows.filter((r) => r.agree).length, rows.length),
53
+ agreementSameInput: pct(sameInput.filter((r) => r.agree).length, sameInput.length),
54
+ rows,
55
+ };
56
+ }
57
+
58
+ const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
59
+ if (isMain) {
60
+ const arg = (n, d) => { const i = process.argv.indexOf(n); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
61
+ const real = JSON.parse(fs.readFileSync(arg('--real'), 'utf8'));
62
+ const dir = arg('--dir');
63
+ const key = JSON.parse(fs.readFileSync(path.join(dir, 'judge-key.json'), 'utf8'));
64
+ const floor = Number(arg('--floor', '0.532'));
65
+ const sim = applyFloor(key, JSON.parse(fs.readFileSync(path.join(dir, 'judge-picks.json'), 'utf8')).picks, floor);
66
+ const cards = JSON.parse(fs.readFileSync(path.join(ROOT, 'plugin', 'scripts', 'package-cards.json'), 'utf8')).cards;
67
+ const keyed = new Map(key.map((k) => [k.qid, k]));
68
+ real.rows.forEach((r) => { const k = keyed.get(r.qid); r.expectHint = Boolean(k.lane) && (k.lane !== 'semantic' || k.topSimilarity >= floor); });
69
+ const out = scoreRealHost(real, key, sim, cards);
70
+ if (process.argv.includes('--json')) { process.stdout.write(`${JSON.stringify(out, null, 1)}\n`); process.exit(0); }
71
+ const f = (n, s) => console.log(`${n.padEnd(30)} ${s.k}/${s.n} = ${s.pct}% [${s.ci95.join('–')}]`);
72
+ f('recall (family-aware)', out.recall); f('precision (family-aware)', out.precision); f('false firing', out.falseFiring);
73
+ f('hints delivered / expected', out.hintDeliveredWhereExpected);
74
+ for (const b of out.deliveryByLoad) f(` delivered at load ${b.load}`, b); f('agreement with simulated host', out.agreement); f(' …where both saw a hint', out.agreementSameInput);
75
+ for (const r of out.rows) console.log(`${r.qid} ${r.category.padEnd(9)} hint=${r.injected ? 'y' : 'n'}${r.expectHint ? '' : '(not expected)'} real=${r.said || '-'} sim=${r.sim || '-'} ${r.said ? (r.right ? 'RIGHT' : 'WRONG') : ''} ${r.agree ? '' : 'DISAGREE'}`);
76
+ }