ruvnet-brain 4.4.1 → 4.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/bin/install.mjs +612 -113
- package/console/app.js +178 -81
- package/console/index.html +1 -1
- package/console/install-architecture.html +1 -0
- package/console/scope.css +4 -1
- package/console/style.css +13 -0
- package/console/tips.html +4 -4
- package/kb/brain-profile.mjs +1 -0
- package/kb/corpus-release-identity.mjs +1 -1
- package/kb/forge-update.mjs +41 -17
- package/kb/model-requirements.mjs +4 -1
- package/kb/update-storage-transaction.mjs +79 -0
- package/kb/zip-extract.mjs +22 -0
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +5 -4
- package/plugin/commands/configure.md +5 -4
- package/plugin/commands/rnb-brief.md +41 -0
- package/plugin/commands/rnb.md +80 -0
- package/plugin/commands/rnbc.md +80 -0
- package/plugin/commands/rvbc.md +5 -4
- package/plugin/commands/rvcb.md +5 -4
- package/plugin/commands/whats-new.md +4 -4
- package/plugin/mcp/server.mjs +10 -2
- package/plugin/scripts/advocacy-route.mjs +59 -23
- package/plugin/scripts/anticipate.sh +4 -0
- package/plugin/scripts/brain-confirmation.mjs +258 -0
- package/plugin/scripts/brain-footprint.mjs +494 -0
- package/plugin/scripts/brain-location.mjs +47 -0
- package/plugin/scripts/capability-registry.mjs +11 -1
- package/plugin/scripts/continuity-brief.mjs +324 -0
- package/plugin/scripts/continuity-events.mjs +327 -0
- package/plugin/scripts/continuity-journal.mjs +500 -0
- package/plugin/scripts/decision-gate.mjs +56 -3
- package/plugin/scripts/footprint-io.mjs +186 -0
- package/plugin/scripts/ground-before-write.sh +8 -1
- package/plugin/scripts/ground-ruvnet.sh +103 -12
- package/plugin/scripts/grounding-answer.mjs +2 -1
- package/plugin/scripts/grounding-stamp.sh +3 -0
- package/plugin/scripts/grounding-substance.mjs +1 -1
- package/plugin/scripts/grounding-turn-evidence.mjs +17 -2
- package/plugin/scripts/hook-input.mjs +78 -4
- package/plugin/scripts/kb-copy-proof.mjs +148 -0
- package/plugin/scripts/lesson-bridge.mjs +6 -2
- package/plugin/scripts/nightly-controller.mjs +8 -1
- package/plugin/scripts/node-sqlite.mjs +41 -0
- package/plugin/scripts/package-cards.json +797 -0
- package/plugin/scripts/package-cards.rvf +0 -0
- package/plugin/scripts/package-cards.rvf.idmap.json +1 -0
- package/plugin/scripts/package-cards.rvf.meta.json +1 -0
- package/plugin/scripts/package-recommender-client.mjs +138 -0
- package/plugin/scripts/package-recommender-flag.mjs +30 -0
- package/plugin/scripts/package-recommender.mjs +391 -0
- package/plugin/scripts/project-progression-outbox.mjs +26 -8
- package/plugin/scripts/project-progression-reader.mjs +4 -2
- package/plugin/scripts/protect-brain-state.sh +4 -1
- package/plugin/scripts/session-snapshot-hook.mjs +27 -3
- package/plugin/scripts/session-start-budget.mjs +1 -0
- package/plugin/scripts/session-start-core.mjs +34 -4
- package/plugin/scripts/session-start-health.mjs +7 -1
- package/plugin/scripts/session-start-update-plane.mjs +35 -0
- package/plugin/scripts/turn-outcome-capture.mjs +12 -1
- package/plugin/scripts/unprompted-runtime.mjs +2 -2
- package/plugin/skills/brain-console/SKILL.md +3 -3
- package/plugin/skills/rnbc/SKILL.md +24 -0
- package/plugin/skills/rvbc/SKILL.md +2 -2
- package/scripts/approved-runtime.mjs +2 -2
- package/scripts/ci/warm-brain-models.mjs +28 -0
- package/scripts/codex-hook-trust.mjs +94 -0
- package/scripts/console-runtime-identity.mjs +5 -0
- package/scripts/corpus-canary.mjs +46 -6
- package/scripts/corpus-dispatch-decision.mjs +2 -2
- package/scripts/corpus-promotion.mjs +1 -1
- package/scripts/hook-qualify-hosts.mjs +15 -3
- package/scripts/host-install-matrix.mjs +63 -2
- package/scripts/human-approval-phrases.mjs +46 -0
- package/scripts/installed-brain-health.mjs +53 -0
- package/scripts/move-brain.mjs +310 -0
- package/scripts/onboarding-console.mjs +93 -10
- package/scripts/oracle/abstain-threshold-sweep.mjs +62 -0
- package/scripts/oracle/abstain-trace.mjs +139 -0
- package/scripts/oracle/doc2query-generate.mjs +162 -0
- package/scripts/oracle/doc2query-reach.mjs +110 -0
- package/scripts/oracle/judge-train.mjs +158 -0
- package/scripts/oracle/need-set-split.mjs +48 -0
- package/scripts/oracle/sona-query-adapter-eval.mjs +139 -0
- package/scripts/package-cards.mjs +374 -0
- package/scripts/publication-receipt.mjs +37 -9
- package/scripts/recommendation-e2e.mjs +110 -0
- package/scripts/recommendation-eval.mjs +105 -0
- package/scripts/recommendation-floor.mjs +56 -0
- package/scripts/recommendation-judge-score.mjs +74 -0
- package/scripts/recommendation-latency.mjs +95 -0
- package/scripts/recommendation-real-host-score.mjs +76 -0
- package/scripts/recommendation-real-host.mjs +137 -0
- package/scripts/release-channel-kind.mjs +1 -1
- package/scripts/release-environment-policy.mjs +33 -0
- package/scripts/single-source-check.mjs +15 -10
- package/scripts/sync-commands.mjs +5 -2
- package/scripts/wired-check.mjs +17 -2
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* recommendation-eval.mjs — score the package recommender against a need-prompt eval set (ADR-0093).
|
|
4
|
+
*
|
|
5
|
+
* node scripts/recommendation-eval.mjs [--set evals/recommendation-eval.v1.json]... [--cards <file>] [--json] [--verbose]
|
|
6
|
+
*
|
|
7
|
+
* Reports, per split and overall: recall on positives (fired AND an accepted package), precision over
|
|
8
|
+
* all firings, false-firing rate on negatives (off-topic + no-fit) — each with a Wilson 95% interval,
|
|
9
|
+
* because 30 prompts is a small sample and a bare percentage would overstate what it shows.
|
|
10
|
+
* An `accept` id that is not in the card set is reported, never silently counted as a miss.
|
|
11
|
+
*/
|
|
12
|
+
import fs from 'node:fs';
|
|
13
|
+
import path from 'node:path';
|
|
14
|
+
import { fileURLToPath } from 'node:url';
|
|
15
|
+
import { indexCards, rank, SNAPSHOT_FILE } from '../plugin/scripts/package-recommender.mjs';
|
|
16
|
+
|
|
17
|
+
const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
|
|
18
|
+
|
|
19
|
+
/** Wilson score interval, 95%. Returns [lo, hi] as fractions; [0, 0] for n = 0. */
|
|
20
|
+
export function wilson(k, n, z = 1.96) {
|
|
21
|
+
if (!n) return [0, 0];
|
|
22
|
+
const p = k / n;
|
|
23
|
+
const den = 1 + (z * z) / n;
|
|
24
|
+
const centre = (p + (z * z) / (2 * n)) / den;
|
|
25
|
+
const half = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / den;
|
|
26
|
+
return [Math.max(0, centre - half), Math.min(1, centre + half)];
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
const POSITIVE = new Set(['design', 'diagnosis']);
|
|
30
|
+
|
|
31
|
+
export function scoreItem(item, decision) {
|
|
32
|
+
const fired = Boolean(decision);
|
|
33
|
+
const pkg = decision?.card?.id || null;
|
|
34
|
+
const store = decision?.card?.store || null;
|
|
35
|
+
if (!POSITIVE.has(item.category)) return { fired, pkg, correct: !fired, firedCorrect: false };
|
|
36
|
+
const accepted = fired && ((item.accept || []).includes(pkg) || (item.acceptStores || []).includes(store));
|
|
37
|
+
return { fired, pkg, correct: accepted, firedCorrect: accepted };
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function summarize(rows) {
|
|
41
|
+
const pos = rows.filter((r) => POSITIVE.has(r.category));
|
|
42
|
+
const neg = rows.filter((r) => !POSITIVE.has(r.category));
|
|
43
|
+
const fired = rows.filter((r) => r.fired);
|
|
44
|
+
const firedCorrect = rows.filter((r) => r.firedCorrect).length;
|
|
45
|
+
const hits = pos.filter((r) => r.correct).length;
|
|
46
|
+
const falseFires = neg.filter((r) => r.fired).length;
|
|
47
|
+
const pct = ([lo, hi]) => [+(lo * 100).toFixed(1), +(hi * 100).toFixed(1)];
|
|
48
|
+
return {
|
|
49
|
+
n: rows.length,
|
|
50
|
+
positives: pos.length,
|
|
51
|
+
negatives: neg.length,
|
|
52
|
+
recall: { k: hits, n: pos.length, pct: pos.length ? +((hits / pos.length) * 100).toFixed(1) : null, ci95: pct(wilson(hits, pos.length)) },
|
|
53
|
+
precision: { k: firedCorrect, n: fired.length, pct: fired.length ? +((firedCorrect / fired.length) * 100).toFixed(1) : null, ci95: pct(wilson(firedCorrect, fired.length)) },
|
|
54
|
+
falseFiring: { k: falseFires, n: neg.length, pct: neg.length ? +((falseFires / neg.length) * 100).toFixed(1) : null, ci95: pct(wilson(falseFires, neg.length)) },
|
|
55
|
+
wrongPackage: pos.filter((r) => r.fired && !r.correct).length,
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export function evaluate(items, index) {
|
|
60
|
+
const ids = new Set(index.entries.map((e) => e.card.id));
|
|
61
|
+
const missing = [...new Set(items.flatMap((i) => i.accept || []).filter((id) => !ids.has(id)))];
|
|
62
|
+
const rows = items.map((item) => {
|
|
63
|
+
const { decision, reason, ranked } = rank(item.prompt, index);
|
|
64
|
+
return {
|
|
65
|
+
id: item.id, category: item.category, split: item.split, prompt: item.prompt, accept: item.accept,
|
|
66
|
+
...scoreItem(item, decision),
|
|
67
|
+
reason, top: ranked.slice(0, 3).map((r) => `${r.card.id}:${r.score.toFixed(2)}[${r.matched.join(',')}]`),
|
|
68
|
+
};
|
|
69
|
+
});
|
|
70
|
+
const splits = [...new Set(rows.map((r) => r.split))];
|
|
71
|
+
return {
|
|
72
|
+
missingAcceptIds: missing,
|
|
73
|
+
overall: summarize(rows),
|
|
74
|
+
bySplit: Object.fromEntries(splits.map((s) => [s, summarize(rows.filter((r) => r.split === s))])),
|
|
75
|
+
rows,
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
80
|
+
if (isMain) {
|
|
81
|
+
const sets = [];
|
|
82
|
+
let cards = SNAPSHOT_FILE;
|
|
83
|
+
for (let i = 2; i < process.argv.length; i++) {
|
|
84
|
+
if (process.argv[i] === '--set') sets.push(process.argv[++i]);
|
|
85
|
+
if (process.argv[i] === '--cards') cards = process.argv[++i];
|
|
86
|
+
}
|
|
87
|
+
if (!sets.length) sets.push(path.join(ROOT, 'evals', 'recommendation-eval.v1.json'));
|
|
88
|
+
// --split dev restricts BOTH scoring and output to one split, so tuning never shows held-out rows.
|
|
89
|
+
const splitArg = process.argv.includes('--split') ? process.argv[process.argv.indexOf('--split') + 1] : null;
|
|
90
|
+
const items = sets.flatMap((f) => JSON.parse(fs.readFileSync(f, 'utf8')).items)
|
|
91
|
+
.filter((i) => !splitArg || i.split === splitArg);
|
|
92
|
+
const index = indexCards(JSON.parse(fs.readFileSync(cards, 'utf8')));
|
|
93
|
+
const result = evaluate(items, index);
|
|
94
|
+
if (process.argv.includes('--json')) { process.stdout.write(`${JSON.stringify(result, null, 1)}\n`); process.exit(0); }
|
|
95
|
+
const line = (name, s) => console.log(`${name.padEnd(9)} n=${s.n} recall ${s.recall.k}/${s.recall.n} ${s.recall.pct}% [${s.recall.ci95.join('–')}] precision ${s.precision.k}/${s.precision.n} ${s.precision.pct}% [${s.precision.ci95.join('–')}] false-firing ${s.falseFiring.k}/${s.falseFiring.n} ${s.falseFiring.pct}% [${s.falseFiring.ci95.join('–')}] wrong-pkg ${s.wrongPackage}`);
|
|
96
|
+
if (result.missingAcceptIds.length) console.log(`accept ids not in card set: ${result.missingAcceptIds.join(', ')}`);
|
|
97
|
+
line('overall', result.overall);
|
|
98
|
+
for (const [s, v] of Object.entries(result.bySplit)) line(s, v);
|
|
99
|
+
if (process.argv.includes('--verbose')) {
|
|
100
|
+
for (const r of result.rows) {
|
|
101
|
+
const mark = r.correct ? 'ok ' : 'XX ';
|
|
102
|
+
console.log(`${mark}${r.id} ${r.category.padEnd(9)} -> ${r.fired ? r.pkg : '(silent:' + r.reason + ')'} | ${r.top.join(' ')}`);
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* recommendation-floor.mjs — derive the semantic similarity floor on the TUNING set only, freeze it, and
|
|
4
|
+
* apply it to every set (ADR-093 rev 3). The rule is fixed in advance: the floor is the 5th percentile
|
|
5
|
+
* of the nearest card's similarity among the tuning set's correctly-judged recommendations (family-aware).
|
|
6
|
+
*
|
|
7
|
+
* node scripts/recommendation-floor.mjs --dir <e2e out dir with judge-key.json + judge-picks.json>
|
|
8
|
+
* [--tune recommendation-eval.v1.json] [--floor <value to apply instead of deriving>] [--json]
|
|
9
|
+
*
|
|
10
|
+
* The e2e run must be made with --floor 0 so every hint (and its top similarity) is recorded.
|
|
11
|
+
*/
|
|
12
|
+
import fs from 'node:fs';
|
|
13
|
+
import path from 'node:path';
|
|
14
|
+
import { fileURLToPath } from 'node:url';
|
|
15
|
+
import { judgeScore } from './recommendation-judge-score.mjs';
|
|
16
|
+
import { wilson } from './recommendation-eval.mjs';
|
|
17
|
+
|
|
18
|
+
const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
|
|
19
|
+
const POS = new Set(['design', 'diagnosis']);
|
|
20
|
+
|
|
21
|
+
export function deriveFloor(key, picks, cards, tuneSet, pct = 0.05) {
|
|
22
|
+
const productOf = new Map(cards.map((c) => [c.id, c.product || c.id]));
|
|
23
|
+
const right = (k, p) => k.accept.includes(p) || k.accept.some((a) => productOf.has(p) && productOf.get(a) === productOf.get(p));
|
|
24
|
+
const sims = key.filter((k) => k.set === tuneSet && k.lane === 'semantic' && POS.has(k.category) && picks[k.qid] && right(k, picks[k.qid]))
|
|
25
|
+
.map((k) => k.topSimilarity).filter(Number.isFinite).sort((a, b) => a - b);
|
|
26
|
+
return sims.length ? +sims[Math.floor(sims.length * pct)].toFixed(3) : null;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Picks as the shipped hook would produce them under `floor`: below it, no hint, so no pick. */
|
|
30
|
+
export function applyFloor(key, picks, floor) {
|
|
31
|
+
return Object.fromEntries(key.map((k) => [k.qid, k.lane === 'semantic' && !(k.topSimilarity >= floor) ? null : picks[k.qid] ?? null]));
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
35
|
+
if (isMain) {
|
|
36
|
+
const arg = (n, d) => { const i = process.argv.indexOf(n); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
|
|
37
|
+
const dir = arg('--dir', null);
|
|
38
|
+
const key = JSON.parse(fs.readFileSync(path.join(dir, 'judge-key.json'), 'utf8'));
|
|
39
|
+
const picks = JSON.parse(fs.readFileSync(path.join(dir, 'judge-picks.json'), 'utf8')).picks;
|
|
40
|
+
const cards = JSON.parse(fs.readFileSync(path.join(ROOT, 'plugin', 'scripts', 'package-cards.json'), 'utf8')).cards;
|
|
41
|
+
const floor = arg('--floor', null) !== null ? Number(arg('--floor')) : deriveFloor(key, picks, cards, arg('--tune', 'recommendation-eval.v1.json'));
|
|
42
|
+
const gated = applyFloor(key, picks, floor);
|
|
43
|
+
const strict = judgeScore(key, gated, cards);
|
|
44
|
+
const family = judgeScore(key, gated, cards, { familyAware: true });
|
|
45
|
+
const negHinted = (set) => key.filter((k) => set(k) && !POS.has(k.category) && k.lane === 'semantic' && k.topSimilarity >= floor).length;
|
|
46
|
+
const negs = key.filter((k) => /blind/.test(k.set) && !POS.has(k.category)).length;
|
|
47
|
+
const out = { floor, strict, family, hintsOnBlindNegatives: { k: negHinted((k) => /blind/.test(k.set)), n: negs, ci95: wilson(negHinted((k) => /blind/.test(k.set)), negs).map((x) => +(x * 100).toFixed(1)) } };
|
|
48
|
+
if (process.argv.includes('--json')) { process.stdout.write(`${JSON.stringify(out, null, 1)}\n`); process.exit(0); }
|
|
49
|
+
console.log(`floor ${floor}`);
|
|
50
|
+
const line = (name, s) => console.log(`${name.padEnd(40)} recall ${s.recall.k}/${s.recall.n} ${s.recall.pct}% [${s.recall.ci95.join('–')}] precision ${s.precision.k}/${s.precision.n} ${s.precision.pct}% [${s.precision.ci95.join('–')}] false-firing ${s.falseFiring.k}/${s.falseFiring.n} [${s.falseFiring.ci95.join('–')}]`);
|
|
51
|
+
for (const [mode, r] of [['strict', strict], ['family', family]]) {
|
|
52
|
+
line(`${mode} blinds`, r.blinds);
|
|
53
|
+
for (const [s, v] of Object.entries(r.bySet)) line(`${mode} ${s}`, v);
|
|
54
|
+
}
|
|
55
|
+
console.log(`hints injected on blind negatives: ${out.hintsOnBlindNegatives.k}/${out.hintsOnBlindNegatives.n} [${out.hintsOnBlindNegatives.ci95.join('–')}]`);
|
|
56
|
+
}
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* recommendation-judge-score.mjs — score what the host model SAID, given the hints the shipped
|
|
4
|
+
* pipeline emitted (ADR-093 rev 2). Inputs come from scripts/recommendation-e2e.mjs (judge-key.json)
|
|
5
|
+
* and a judge that saw only prompts + hints (judge-picks.json: { picks: { qid: id|null } }).
|
|
6
|
+
*
|
|
7
|
+
* node scripts/recommendation-judge-score.mjs --dir <e2e out dir> [--cards plugin/scripts/package-cards.json] [--family] [--json]
|
|
8
|
+
*
|
|
9
|
+
* A prompt the pipeline stayed silent on counts as silent. Precision is over what the model said.
|
|
10
|
+
* Wilson 95% intervals throughout.
|
|
11
|
+
*/
|
|
12
|
+
import fs from 'node:fs';
|
|
13
|
+
import path from 'node:path';
|
|
14
|
+
import { fileURLToPath } from 'node:url';
|
|
15
|
+
import { wilson } from './recommendation-eval.mjs';
|
|
16
|
+
|
|
17
|
+
const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
|
|
18
|
+
const POS = new Set(['design', 'diagnosis']);
|
|
19
|
+
// The closed catalogue speaks in building-block names; these are the packages each one names.
|
|
20
|
+
export const CATALOGUE_PACKAGES = Object.freeze({
|
|
21
|
+
ruvector: ['ruvector', '@ruvector/rvf', '@ruvector/core', '@ruvector/node'],
|
|
22
|
+
agentdb: ['agentdb'],
|
|
23
|
+
ruflo: ['ruflo', 'claude-flow', '@claude-flow/swarm'],
|
|
24
|
+
aidefence: ['@claude-flow/aidefence', 'aidefence-core'],
|
|
25
|
+
'agentic-qe': ['agentic-qe'],
|
|
26
|
+
'agentic-flow': ['agentic-flow'],
|
|
27
|
+
rulake: ['rulake'],
|
|
28
|
+
});
|
|
29
|
+
|
|
30
|
+
export function judgeScore(key, picks, cards, { familyAware = false } = {}) {
|
|
31
|
+
const storeOf = new Map(cards.map((c) => [c.id, c.store]));
|
|
32
|
+
// FAMILY-AWARE (ADR-093 rev 3): a pick is right when it is the same PRODUCT as an accepted id, per the
|
|
33
|
+
// corpus-derived product map on the cards (scripts/package-cards.mjs deriveProducts). Off by default so
|
|
34
|
+
// the strict number is always the one reported first.
|
|
35
|
+
const productOf = new Map(cards.map((c) => [c.id, c.product || c.id]));
|
|
36
|
+
const sameProduct = (a, b) => familyAware && productOf.has(a) && productOf.get(a) === productOf.get(b);
|
|
37
|
+
const sets = [...new Set(key.map((k) => k.set))];
|
|
38
|
+
const pct = (k, n) => ({ k, n, pct: n ? +((100 * k) / n).toFixed(1) : null, ci95: wilson(k, n).map((x) => +(x * 100).toFixed(1)) });
|
|
39
|
+
const summarize = (rows) => {
|
|
40
|
+
let hit = 0; let said = 0; let correct = 0; let ff = 0; let wrong = 0;
|
|
41
|
+
const pos = rows.filter((r) => POS.has(r.category)).length;
|
|
42
|
+
for (const r of rows) {
|
|
43
|
+
const p = picks[r.qid] ?? null;
|
|
44
|
+
if (!p) continue;
|
|
45
|
+
said++;
|
|
46
|
+
const ids = [p, ...(CATALOGUE_PACKAGES[p] || [])];
|
|
47
|
+
const ok = ids.some((id) => r.accept.includes(id) || r.acceptStores.includes(storeOf.get(id))
|
|
48
|
+
|| r.accept.some((a) => sameProduct(id, a)));
|
|
49
|
+
if (POS.has(r.category)) { if (ok) { hit++; correct++; } else wrong++; } else ff++;
|
|
50
|
+
}
|
|
51
|
+
return { n: rows.length, recall: pct(hit, pos), precision: pct(correct, said), falseFiring: pct(ff, rows.length - pos), wrongPackage: wrong };
|
|
52
|
+
};
|
|
53
|
+
return {
|
|
54
|
+
overall: summarize(key),
|
|
55
|
+
blinds: summarize(key.filter((k) => /blind/.test(k.set))),
|
|
56
|
+
bySet: Object.fromEntries(sets.map((s) => [s, summarize(key.filter((k) => k.set === s))])),
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
61
|
+
if (isMain) {
|
|
62
|
+
const arg = (n, d) => { const i = process.argv.indexOf(n); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
|
|
63
|
+
const dir = arg('--dir', null);
|
|
64
|
+
if (!dir) { console.error('--dir required'); process.exit(2); }
|
|
65
|
+
const key = JSON.parse(fs.readFileSync(path.join(dir, 'judge-key.json'), 'utf8'));
|
|
66
|
+
const picks = JSON.parse(fs.readFileSync(path.join(dir, 'judge-picks.json'), 'utf8')).picks || {};
|
|
67
|
+
const cards = JSON.parse(fs.readFileSync(arg('--cards', path.join(ROOT, 'plugin', 'scripts', 'package-cards.json')), 'utf8')).cards;
|
|
68
|
+
const r = judgeScore(key, picks, cards, { familyAware: process.argv.includes('--family') });
|
|
69
|
+
if (process.argv.includes('--json')) { process.stdout.write(`${JSON.stringify(r, null, 1)}\n`); process.exit(0); }
|
|
70
|
+
const line = (name, s) => console.log(`${name.padEnd(36)} recall ${s.recall.k}/${s.recall.n} ${s.recall.pct}% [${s.recall.ci95.join('–')}] precision ${s.precision.k}/${s.precision.n} ${s.precision.pct}% [${s.precision.ci95.join('–')}] false-firing ${s.falseFiring.k}/${s.falseFiring.n} ${s.falseFiring.pct}% [${s.falseFiring.ci95.join('–')}] wrong ${s.wrongPackage}`);
|
|
71
|
+
line('overall', r.overall);
|
|
72
|
+
line('blinds (1+2)', r.blinds);
|
|
73
|
+
for (const [s, v] of Object.entries(r.bySet)) line(s, v);
|
|
74
|
+
}
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* recommendation-latency.mjs — what the package recommender ADDS to every prompt (ADR-0093).
|
|
4
|
+
*
|
|
5
|
+
* Measured where the cost is paid: a COLD `node advocacy-route.mjs` process per prompt, exactly as
|
|
6
|
+
* unprompted-runtime.mjs spawns it on UserPromptSubmit. For each eval prompt the producer runs twice,
|
|
7
|
+
* flag OFF then flag ON, interleaved so machine drift lands on both arms; the per-prompt difference is
|
|
8
|
+
* the added latency. Every path the producer could write (state, ledger, HOME) is a fresh temp dir.
|
|
9
|
+
*
|
|
10
|
+
* node scripts/recommendation-latency.mjs [--set <eval.json>]... [--rounds 1] [--json]
|
|
11
|
+
*/
|
|
12
|
+
import { spawnSync } from 'node:child_process';
|
|
13
|
+
import fs from 'node:fs';
|
|
14
|
+
import os from 'node:os';
|
|
15
|
+
import path from 'node:path';
|
|
16
|
+
import { fileURLToPath } from 'node:url';
|
|
17
|
+
|
|
18
|
+
const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
|
|
19
|
+
const ROUTE = path.join(ROOT, 'plugin', 'scripts', 'advocacy-route.mjs');
|
|
20
|
+
const CARDS = path.join(ROOT, 'plugin', 'scripts', 'package-cards.json');
|
|
21
|
+
|
|
22
|
+
const sets = [];
|
|
23
|
+
let rounds = 1;
|
|
24
|
+
for (let i = 2; i < process.argv.length; i++) {
|
|
25
|
+
if (process.argv[i] === '--set') sets.push(process.argv[++i]);
|
|
26
|
+
if (process.argv[i] === '--rounds') rounds = Math.max(1, Number(process.argv[++i]) || 1);
|
|
27
|
+
}
|
|
28
|
+
if (!sets.length) sets.push(path.join(ROOT, 'evals', 'recommendation-eval.v1.json'));
|
|
29
|
+
const prompts = sets.flatMap((f) => JSON.parse(fs.readFileSync(f, 'utf8')).items.map((i) => i.prompt));
|
|
30
|
+
|
|
31
|
+
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'reco-latency-'));
|
|
32
|
+
// Only what the producer needs; nothing inherited that could point it at a real store.
|
|
33
|
+
const baseEnv = {
|
|
34
|
+
PATH: process.env.PATH,
|
|
35
|
+
HOME: tmp,
|
|
36
|
+
USERPROFILE: tmp,
|
|
37
|
+
RUVNET_HOME_OVERRIDE: tmp,
|
|
38
|
+
RUVNET_EMIT_CANDIDATES: '1',
|
|
39
|
+
RUVNET_PACKAGE_CARDS: CARDS,
|
|
40
|
+
RUVNET_ADVOCACY_ROUTE_ROOTS: path.join(tmp, 'no-modules'),
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
function run(prompt, flag, n) {
|
|
44
|
+
const env = {
|
|
45
|
+
...baseEnv,
|
|
46
|
+
RUVNET_ADVOCACY_ROUTE_STATE: path.join(tmp, `state-${flag}-${n}.json`),
|
|
47
|
+
RUVNET_ADVOCACY_OUTCOMES: path.join(tmp, `outcomes-${flag}-${n}.jsonl`),
|
|
48
|
+
...(flag === 'on' ? { RUVNET_PACKAGE_RECOMMENDER: '1' } : {}),
|
|
49
|
+
};
|
|
50
|
+
const input = JSON.stringify({ session_id: `lat-${flag}-${n}`, cwd: ROOT, hook_event_name: 'UserPromptSubmit', prompt });
|
|
51
|
+
const t0 = process.hrtime.bigint();
|
|
52
|
+
const r = spawnSync(process.execPath, [ROUTE], { input, env, encoding: 'utf8', timeout: 10000 });
|
|
53
|
+
const ms = Number(process.hrtime.bigint() - t0) / 1e6;
|
|
54
|
+
return { ms, status: r.status, emitted: Boolean((r.stdout || '').trim()) };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const q = (arr, p) => { const s = [...arr].sort((a, b) => a - b); return s[Math.min(s.length - 1, Math.ceil(p * s.length) - 1)]; };
|
|
58
|
+
const off = []; const on = []; const added = []; let failures = 0;
|
|
59
|
+
let n = 0;
|
|
60
|
+
try {
|
|
61
|
+
run('warm up the filesystem cache once', 'off', 'warm');
|
|
62
|
+
for (let r = 0; r < rounds; r++) {
|
|
63
|
+
for (const prompt of prompts) {
|
|
64
|
+
n++;
|
|
65
|
+
const a = run(prompt, 'off', n);
|
|
66
|
+
const b = run(prompt, 'on', n);
|
|
67
|
+
if (a.status !== 0 || b.status !== 0) failures++;
|
|
68
|
+
off.push(a.ms); on.push(b.ms); added.push(b.ms - a.ms);
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
} finally { fs.rmSync(tmp, { recursive: true, force: true }); }
|
|
72
|
+
|
|
73
|
+
const fmt = (arr) => ({ p50: +q(arr, 0.5).toFixed(1), p90: +q(arr, 0.9).toFixed(1), max: +Math.max(...arr).toFixed(1) });
|
|
74
|
+
// Percentile bootstrap (2000 resamples, fixed seed) for the ADDED p50/p90, so the report carries an
|
|
75
|
+
// interval rather than one number from one noisy run.
|
|
76
|
+
function bootstrap(arr, p, B = 2000) {
|
|
77
|
+
let seed = 0x9e3779b9;
|
|
78
|
+
const rnd = () => { seed ^= seed << 13; seed ^= seed >>> 17; seed ^= seed << 5; return ((seed >>> 0) % 1e9) / 1e9; };
|
|
79
|
+
const stats = [];
|
|
80
|
+
for (let b = 0; b < B; b++) stats.push(q(Array.from(arr, () => arr[Math.floor(rnd() * arr.length)]), p));
|
|
81
|
+
return [+q(stats, 0.025).toFixed(1), +q(stats, 0.975).toFixed(1)];
|
|
82
|
+
}
|
|
83
|
+
const out = {
|
|
84
|
+
samples: n, failures,
|
|
85
|
+
host: `${os.platform()} ${os.arch()} ${os.cpus().length} vCPU node ${process.version}`,
|
|
86
|
+
loadAvg1m: +os.loadavg()[0].toFixed(1),
|
|
87
|
+
flagOffMs: fmt(off), flagOnMs: fmt(on), addedMs: { ...fmt(added), p50ci95: bootstrap(added, 0.5), p90ci95: bootstrap(added, 0.9) },
|
|
88
|
+
};
|
|
89
|
+
if (process.argv.includes('--json')) process.stdout.write(`${JSON.stringify(out, null, 1)}\n`);
|
|
90
|
+
else {
|
|
91
|
+
console.log(`[latency] ${out.samples} paired cold runs on ${out.host}, ${failures} non-zero exits`);
|
|
92
|
+
console.log(`[latency] flag off : p50 ${out.flagOffMs.p50} ms p90 ${out.flagOffMs.p90} ms max ${out.flagOffMs.max} ms`);
|
|
93
|
+
console.log(`[latency] flag on : p50 ${out.flagOnMs.p50} ms p90 ${out.flagOnMs.p90} ms max ${out.flagOnMs.max} ms`);
|
|
94
|
+
console.log(`[latency] added : p50 ${out.addedMs.p50} ms p90 ${out.addedMs.p90} ms max ${out.addedMs.max} ms`);
|
|
95
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* recommendation-real-host-score.mjs — score a real-host run (scripts/recommendation-real-host.mjs) and
|
|
4
|
+
* its agreement with the simulated host (ADR-093 rev 3).
|
|
5
|
+
*
|
|
6
|
+
* node scripts/recommendation-real-host-score.mjs --real <real-host.json> --dir <e2e dir> [--floor 0.532] [--json]
|
|
7
|
+
*
|
|
8
|
+
* Correctness is family-aware (same product as an accepted id). Agreement compares, per prompt, what the
|
|
9
|
+
* real host named with what the simulated host would have named under the same floor: both silent, or
|
|
10
|
+
* the same product. A prompt the real hook did NOT inject a hint for is reported separately, because
|
|
11
|
+
* then the two hosts did not see the same input.
|
|
12
|
+
*/
|
|
13
|
+
import fs from 'node:fs';
|
|
14
|
+
import path from 'node:path';
|
|
15
|
+
import { fileURLToPath } from 'node:url';
|
|
16
|
+
import { wilson } from './recommendation-eval.mjs';
|
|
17
|
+
import { applyFloor } from './recommendation-floor.mjs';
|
|
18
|
+
|
|
19
|
+
const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
|
|
20
|
+
const POS = new Set(['design', 'diagnosis']);
|
|
21
|
+
|
|
22
|
+
export function scoreRealHost(real, key, simPicks, cards) {
|
|
23
|
+
const productOf = new Map(cards.map((c) => [c.id, c.product || c.id]));
|
|
24
|
+
const same = (a, b) => a === b || (productOf.has(a) && productOf.get(a) === productOf.get(b));
|
|
25
|
+
const byQid = new Map(key.map((k) => [k.qid, k]));
|
|
26
|
+
const rows = real.rows.map((r) => {
|
|
27
|
+
const k = byQid.get(r.qid);
|
|
28
|
+
const right = r.said ? (k.accept || []).some((a) => same(r.said, a)) : null;
|
|
29
|
+
const sim = simPicks[r.qid] ?? null;
|
|
30
|
+
const agree = (!r.said && !sim) || (r.said && sim && same(r.said, sim));
|
|
31
|
+
return { ...r, accept: k.accept, sim, right, agree };
|
|
32
|
+
});
|
|
33
|
+
const pos = rows.filter((r) => POS.has(r.category));
|
|
34
|
+
const neg = rows.filter((r) => !POS.has(r.category));
|
|
35
|
+
const said = rows.filter((r) => r.said);
|
|
36
|
+
const pct = (a, n) => ({ k: a, n, pct: n ? +((100 * a) / n).toFixed(1) : null, ci95: wilson(a, n).map((x) => +(x * 100).toFixed(1)) });
|
|
37
|
+
const sameInput = rows.filter((r) => r.injected === Boolean(byQid.get(r.qid).lane));
|
|
38
|
+
// Hint delivery against the 1-minute load each prompt ran at: how the lane behaves on a busy laptop.
|
|
39
|
+
const buckets = [[0, 60], [60, 120], [120, 240], [240, Infinity]];
|
|
40
|
+
const deliveryByLoad = buckets.map(([lo, hi]) => {
|
|
41
|
+
const inB = rows.filter((r) => r.expectHint && Number.isFinite(r.load1m) && r.load1m >= lo && r.load1m < hi);
|
|
42
|
+
return { load: hi === Infinity ? `>=${lo}` : `${lo}-${hi}`, ...pct(inB.filter((r) => r.injected).length, inB.length) };
|
|
43
|
+
});
|
|
44
|
+
return {
|
|
45
|
+
n: rows.length,
|
|
46
|
+
deliveryByLoad,
|
|
47
|
+
recall: pct(pos.filter((r) => r.right).length, pos.length),
|
|
48
|
+
precision: pct(said.filter((r) => POS.has(r.category) && r.right).length, said.length),
|
|
49
|
+
falseFiring: pct(neg.filter((r) => r.said).length, neg.length),
|
|
50
|
+
hintDeliveredWhereExpected: pct(rows.filter((r) => r.expectHint && r.injected).length, rows.filter((r) => r.expectHint).length),
|
|
51
|
+
hintsNotExpected: rows.filter((r) => !r.expectHint && r.injected).length, // catalogue/lexical lanes, which the floor does not govern
|
|
52
|
+
agreement: pct(rows.filter((r) => r.agree).length, rows.length),
|
|
53
|
+
agreementSameInput: pct(sameInput.filter((r) => r.agree).length, sameInput.length),
|
|
54
|
+
rows,
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
59
|
+
if (isMain) {
|
|
60
|
+
const arg = (n, d) => { const i = process.argv.indexOf(n); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
|
|
61
|
+
const real = JSON.parse(fs.readFileSync(arg('--real'), 'utf8'));
|
|
62
|
+
const dir = arg('--dir');
|
|
63
|
+
const key = JSON.parse(fs.readFileSync(path.join(dir, 'judge-key.json'), 'utf8'));
|
|
64
|
+
const floor = Number(arg('--floor', '0.532'));
|
|
65
|
+
const sim = applyFloor(key, JSON.parse(fs.readFileSync(path.join(dir, 'judge-picks.json'), 'utf8')).picks, floor);
|
|
66
|
+
const cards = JSON.parse(fs.readFileSync(path.join(ROOT, 'plugin', 'scripts', 'package-cards.json'), 'utf8')).cards;
|
|
67
|
+
const keyed = new Map(key.map((k) => [k.qid, k]));
|
|
68
|
+
real.rows.forEach((r) => { const k = keyed.get(r.qid); r.expectHint = Boolean(k.lane) && (k.lane !== 'semantic' || k.topSimilarity >= floor); });
|
|
69
|
+
const out = scoreRealHost(real, key, sim, cards);
|
|
70
|
+
if (process.argv.includes('--json')) { process.stdout.write(`${JSON.stringify(out, null, 1)}\n`); process.exit(0); }
|
|
71
|
+
const f = (n, s) => console.log(`${n.padEnd(30)} ${s.k}/${s.n} = ${s.pct}% [${s.ci95.join('–')}]`);
|
|
72
|
+
f('recall (family-aware)', out.recall); f('precision (family-aware)', out.precision); f('false firing', out.falseFiring);
|
|
73
|
+
f('hints delivered / expected', out.hintDeliveredWhereExpected);
|
|
74
|
+
for (const b of out.deliveryByLoad) f(` delivered at load ${b.load}`, b); f('agreement with simulated host', out.agreement); f(' …where both saw a hint', out.agreementSameInput);
|
|
75
|
+
for (const r of out.rows) console.log(`${r.qid} ${r.category.padEnd(9)} hint=${r.injected ? 'y' : 'n'}${r.expectHint ? '' : '(not expected)'} real=${r.said || '-'} sim=${r.sim || '-'} ${r.said ? (r.right ? 'RIGHT' : 'WRONG') : ''} ${r.agree ? '' : 'DISAGREE'}`);
|
|
76
|
+
}
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* recommendation-real-host.mjs — the package recommender in a REAL Claude Code session (ADR-093 rev 3).
|
|
4
|
+
*
|
|
5
|
+
* For each selected eval prompt: a fresh `claude -p` with this checkout's UserPromptSubmit runtime as
|
|
6
|
+
* its only hook (route producer only, flag on), a warm search worker in a throwaway brain home, then the
|
|
7
|
+
* model's actual answer. Records whether the hook injected a hint, which packages it carried, and which
|
|
8
|
+
* one (if any) the model named — so the simulated-host judge can be checked against a real host.
|
|
9
|
+
*
|
|
10
|
+
* Isolation, following scripts/hook-qualify-hosts.mjs: --no-session-persistence, --setting-sources ''
|
|
11
|
+
* (no user/project settings, plugins or hooks), --strict-mcp-config (no MCP servers), --tools '' (no
|
|
12
|
+
* tool can touch anything), a temp cwd, every RUVNET_* state path in a temp dir. Auth is the user's own
|
|
13
|
+
* login; the run snapshots ~/.claude mtimes before/after and reports anything that changed.
|
|
14
|
+
*
|
|
15
|
+
* node scripts/recommendation-real-host.mjs --key <e2e judge-key.json> --models <d> --xenova <d> --out <dir>
|
|
16
|
+
* [--n-pos 12 --n-neg-hinted 6 --n-other 4 | --all-blinds] [--max-load 60 --batch 11] [--claude <bin>]
|
|
17
|
+
*/
|
|
18
|
+
import { spawn, spawnSync } from 'node:child_process';
|
|
19
|
+
import fs from 'node:fs';
|
|
20
|
+
import os from 'node:os';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
import readline from 'node:readline';
|
|
23
|
+
import { fileURLToPath } from 'node:url';
|
|
24
|
+
|
|
25
|
+
const ROOT = path.dirname(path.dirname(fileURLToPath(import.meta.url)));
|
|
26
|
+
const arg = (n, d) => { const i = process.argv.indexOf(n); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
|
|
27
|
+
const CLAUDE = arg('--claude', path.join(os.homedir(), '.npm-global', 'bin', 'claude'));
|
|
28
|
+
const POS = new Set(['design', 'diagnosis']);
|
|
29
|
+
|
|
30
|
+
/** Deterministic stratified sample from the blind sets of an e2e key. */
|
|
31
|
+
export function stratify(key, { nPos = 12, nNegHinted = 6, nOther = 4, floor = 0 } = {}) {
|
|
32
|
+
const hinted = (k) => k.lane && (k.lane !== 'semantic' || k.topSimilarity >= floor);
|
|
33
|
+
const blind = key.filter((k) => /blind/.test(k.set)).sort((a, b) => a.qid.localeCompare(b.qid));
|
|
34
|
+
const every = (list, n) => (list.length <= n ? list : Array.from({ length: n }, (_, i) => list[Math.floor((i * list.length) / n)]));
|
|
35
|
+
const pos = every(blind.filter((k) => POS.has(k.category) && hinted(k)), nPos);
|
|
36
|
+
const negHinted = every(blind.filter((k) => !POS.has(k.category) && hinted(k)), nNegHinted);
|
|
37
|
+
const other = every(blind.filter((k) => !hinted(k)), nOther);
|
|
38
|
+
return [...pos, ...negHinted, ...other];
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Which offered package (by full id or short name) the answer names, if any. Pure. */
|
|
42
|
+
export function mentioned(answer, offered) {
|
|
43
|
+
const text = String(answer || '');
|
|
44
|
+
const sentences = text.split(/(?<=[.!?])\s+/);
|
|
45
|
+
for (const id of offered || []) {
|
|
46
|
+
const short = id.replace(/^@[^/]+\//, '');
|
|
47
|
+
const re = (s) => new RegExp(`(?<![\\w@/-])${s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}(?![\\w/-])`, 'i');
|
|
48
|
+
// A scoped or hyphenated id is unambiguous anywhere. A bare short name ("migration", "typesafe")
|
|
49
|
+
// counts only in a sentence that also names rUv — "write the migration" is not a recommendation.
|
|
50
|
+
if (re(id).test(text) && (id.startsWith('@') || id.includes('-'))) return id;
|
|
51
|
+
if (sentences.some((s) => /\brUv\b|ruvector|ruvnet/i.test(s) && (re(id).test(s) || (short.length >= 5 && re(short).test(s))))) return id;
|
|
52
|
+
}
|
|
53
|
+
return null;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
57
|
+
if (isMain) {
|
|
58
|
+
const outDir = arg('--out', null);
|
|
59
|
+
if (!outDir || !arg('--key', null)) { console.error('--key and --out are required'); process.exit(2); }
|
|
60
|
+
const key = JSON.parse(fs.readFileSync(arg('--key'), 'utf8'));
|
|
61
|
+
const items = new Map();
|
|
62
|
+
for (const f of ['recommendation-eval.blind.v1.json', 'recommendation-eval.blind.v2.json']) {
|
|
63
|
+
for (const it of JSON.parse(fs.readFileSync(path.join(ROOT, 'evals', f), 'utf8')).items) items.set(`${f}:${it.id}`, it);
|
|
64
|
+
}
|
|
65
|
+
// --all-blinds: every blind prompt, in qid order (the decision run); otherwise the stratified sample.
|
|
66
|
+
const sample = process.argv.includes('--all-blinds')
|
|
67
|
+
? key.filter((k) => /blind/.test(k.set)).sort((a, b) => a.qid.localeCompare(b.qid))
|
|
68
|
+
: stratify(key, { nPos: Number(arg('--n-pos', 12)), nNegHinted: Number(arg('--n-neg-hinted', 6)), nOther: Number(arg('--n-other', 4)), floor: Number(arg('--floor', 0.532)) });
|
|
69
|
+
const home = fs.mkdtempSync(path.join(os.tmpdir(), 'reco-host-'));
|
|
70
|
+
const kb = path.join(home, 'kb'); fs.mkdirSync(kb);
|
|
71
|
+
const cwd = path.join(home, 'cwd'); fs.mkdirSync(cwd);
|
|
72
|
+
const marker = path.join(home, 'marker'); fs.writeFileSync(marker, '');
|
|
73
|
+
const brainEnv = { RUVNET_BRAIN_HOME: home, KB_DIR: kb, KB_MODEL_CACHE: arg('--models', ''), XENOVA_PATH: arg('--xenova', '') };
|
|
74
|
+
const worker = spawn(process.execPath, [path.join(ROOT, 'kb', 'forge-mcp-all.mjs')],
|
|
75
|
+
{ env: { PATH: process.env.PATH, HOME: home, ...brainEnv, RUVNET_PACKAGE_RECOMMENDER: '1', RUVNET_BRAIN_IDLE_EXIT_MS: '0' }, stdio: ['pipe', 'pipe', 'ignore'] }); // idle exit off: load-gate waits must not retire the worker mid-run
|
|
76
|
+
for (const sig of ['SIGTERM', 'SIGINT']) process.once(sig, () => { worker.kill('SIGTERM'); fs.rmSync(home, { recursive: true, force: true }); process.exit(1); });
|
|
77
|
+
const rl = readline.createInterface({ input: worker.stdout });
|
|
78
|
+
const waiters = new Map();
|
|
79
|
+
rl.on('line', (l) => { try { const m = JSON.parse(l); waiters.get(m.id)?.(m); } catch { /* not ours */ } });
|
|
80
|
+
const call = (id, method) => new Promise((r) => { waiters.set(id, r); worker.stdin.write(`${JSON.stringify({ jsonrpc: '2.0', id, method, params: {} })}\n`); });
|
|
81
|
+
const settings = path.join(home, 'settings.json');
|
|
82
|
+
fs.writeFileSync(settings, JSON.stringify({ autoMemoryEnabled: false, hooks: { UserPromptSubmit: [{ hooks: [{ type: 'command', command: `"${process.execPath}" "${path.join(ROOT, 'plugin', 'scripts', 'unprompted-runtime.mjs')}" UserPromptSubmit`, timeout: 3 }] }] } }));
|
|
83
|
+
const rows = [];
|
|
84
|
+
try {
|
|
85
|
+
await call(1, 'initialize');
|
|
86
|
+
const warm = await call(2, 'brain/warmup');
|
|
87
|
+
if (!warm.result?.ready) throw new Error('worker warmup failed');
|
|
88
|
+
// LOAD GATE: never start a batch while the 1-minute load is above --max-load; wait (polling) instead.
|
|
89
|
+
// Each row records the load it ran at, so delivery can be read against load afterwards.
|
|
90
|
+
const maxLoad = Number(arg('--max-load', 1e9));
|
|
91
|
+
const batch = Number(arg('--batch', 11));
|
|
92
|
+
const waitForLoad = async () => {
|
|
93
|
+
let waited = 0;
|
|
94
|
+
while (os.loadavg()[0] > maxLoad) { await new Promise((res) => setTimeout(res, 20_000)); waited += 20; }
|
|
95
|
+
if (waited) console.log(`[load-gate] waited ${waited}s for load < ${maxLoad}`);
|
|
96
|
+
};
|
|
97
|
+
for (const [n, k] of sample.entries()) {
|
|
98
|
+
if (n % batch === 0) await waitForLoad();
|
|
99
|
+
const it = items.get(`${k.set}:${k.id}`);
|
|
100
|
+
const env = {
|
|
101
|
+
...process.env, ...brainEnv, CLAUDE_HOOK: '/usr/bin/true', CLAUDE_CODE_DISABLE_AUTO_MEMORY: '1', RUVNET_PACKAGE_RECOMMENDER: '1',
|
|
102
|
+
RUVNET_UNPROMPTED_PRODUCERS: JSON.stringify([{ argv: [process.execPath, path.join(ROOT, 'plugin', 'scripts', 'advocacy-route.mjs')], feedStdin: true, channels: ['advocacy'] }]),
|
|
103
|
+
RUVNET_ADVOCACY_ROUTE_STATE: path.join(home, `state-${n}.json`), RUVNET_ADVOCACY_OUTCOMES: path.join(home, `outcomes-${n}.jsonl`),
|
|
104
|
+
RUVNET_ADVOCACY_ROUTE_ROOTS: path.join(home, 'none'), RUVNET_SETTINGS_FILE: path.join(home, 'user-settings.json'),
|
|
105
|
+
};
|
|
106
|
+
const r = spawnSync(CLAUDE, ['-p', it.prompt, '--output-format', 'stream-json', '--verbose', '--include-hook-events',
|
|
107
|
+
'--no-session-persistence', '--setting-sources', '', '--settings', settings, '--strict-mcp-config', '--tools', '',
|
|
108
|
+
'--max-turns', '1', '--max-budget-usd', '0.30',
|
|
109
|
+
'--append-system-prompt', 'This is a quick planning exchange with no tools: answer in at most five sentences with how you would approach the request.'],
|
|
110
|
+
{ cwd, env, encoding: 'utf8', timeout: 240_000, maxBuffer: 64e6 });
|
|
111
|
+
let injected = ''; let answer = '';
|
|
112
|
+
for (const line of String(r.stdout || '').split('\n')) {
|
|
113
|
+
let o; try { o = JSON.parse(line); } catch { continue; }
|
|
114
|
+
const blob = JSON.stringify(o);
|
|
115
|
+
// All three advocacy copies: the package lanes AND the closed catalogue ("capability advocacy").
|
|
116
|
+
const m = blob.match(/\[RuvNet Brain — (?:rUv (?:may already ship|already ships) this|capability advocacy)\][^"]*/);
|
|
117
|
+
if (m && !injected) injected = m[0];
|
|
118
|
+
if (o.type === 'result' && typeof o.result === 'string') answer = o.result;
|
|
119
|
+
}
|
|
120
|
+
const offered = injected ? [...new Set([...injected.matchAll(/(?:Consider )?(@[a-z0-9-]+\/[a-z0-9._-]+|[a-z0-9][a-z0-9._-]+) — /gi)].map((x) => x[1]))] : [];
|
|
121
|
+
rows.push({ qid: k.qid, set: k.set, id: k.id, category: k.category, load1m: +os.loadavg()[0].toFixed(1), exit: r.status, injected: Boolean(injected), offered, said: mentioned(answer, offered), answer: answer.slice(0, 1200) });
|
|
122
|
+
console.log(`${k.qid} ${k.category.padEnd(9)} hint=${injected ? 'yes' : 'no '} said=${rows.at(-1).said || '-'}`);
|
|
123
|
+
}
|
|
124
|
+
} finally {
|
|
125
|
+
worker.kill('SIGTERM');
|
|
126
|
+
// Other sessions write under ~/.claude all the time; what THIS run could own is a project entry for
|
|
127
|
+
// its own temp cwd, so that is checked by name as well as the raw mtime list.
|
|
128
|
+
const ours = [path.join(os.homedir(), '.claude', 'projects')].flatMap((d) => { try { return fs.readdirSync(d).filter((n) => n.includes('reco-host')); } catch { return []; } });
|
|
129
|
+
let inDotClaudeJson = false; try { inDotClaudeJson = fs.readFileSync(path.join(os.homedir(), '.claude.json'), 'utf8').includes(path.basename(home)); } catch { /* absent */ }
|
|
130
|
+
const touched = spawnSync('find', [path.join(os.homedir(), '.claude'), '-newer', marker, '-maxdepth', '3'], { encoding: 'utf8' }).stdout.trim().split('\n').filter(Boolean);
|
|
131
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
132
|
+
fs.writeFileSync(path.join(outDir, 'real-host.json'), JSON.stringify({ claude: spawnSync(CLAUDE, ['--version'], { encoding: 'utf8' }).stdout.trim(), sample: rows.length, projectEntriesForThisRun: ours, cwdRecordedInDotClaudeJson: inDotClaudeJson, modifiedUnderDotClaudeByAnyProcess: touched.length, rows }, null, 1));
|
|
133
|
+
console.log(`project entries for this run under ~/.claude/projects: ${ours.length}; temp cwd in ~/.claude.json: ${inDotClaudeJson}; ~/.claude entries modified by ANY process meanwhile: ${touched.length}`);
|
|
134
|
+
fs.rmSync(home, { recursive: true, force: true });
|
|
135
|
+
}
|
|
136
|
+
process.exit(0);
|
|
137
|
+
}
|
|
@@ -27,7 +27,7 @@
|
|
|
27
27
|
|
|
28
28
|
/** Content-addressed corpus generation: the tag IS the archive's sha256. */
|
|
29
29
|
export const CORPUS_TAG_PATTERN = /^corpus-sha256-[0-9a-f]{64}$/;
|
|
30
|
-
/**
|
|
30
|
+
/** Product (code) release: a plain semver tag. */
|
|
31
31
|
export const CODE_TAG_PATTERN = /^v\d+\.\d+\.\d+$/;
|
|
32
32
|
|
|
33
33
|
export function releaseKind(tag) {
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
// release-environment-policy.mjs — the ONE verdict on the npm-scoped Production environment
|
|
2
|
+
// ("Production – ruvnet-brain"), as GitHub's environments API reports it.
|
|
3
|
+
//
|
|
4
|
+
// The design (CONTRIBUTING.md, "What replaces a human approval"): the environment is a scoping boundary —
|
|
5
|
+
// a branch policy (protected branches only) and admins cannot bypass — and NO person is part of it. A
|
|
6
|
+
// required reviewer coming back (someone re-adds one in the GitHub UI) silently re-inserts a human click
|
|
7
|
+
// into every release; it must turn single-source C3 red, not pass because the other two rules still hold.
|
|
8
|
+
export const PRODUCTION_ENVIRONMENT = 'Production – ruvnet-brain';
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* @param {object|null} environment one element of `GET /repos/{o}/{r}/environments` `.environments[]`
|
|
12
|
+
* @returns {{ ok: boolean, problems: string[], detail: string }}
|
|
13
|
+
*/
|
|
14
|
+
export function productionEnvironmentVerdict(environment) {
|
|
15
|
+
if (!environment || typeof environment !== 'object') {
|
|
16
|
+
return { ok: false, problems: [`environment "${PRODUCTION_ENVIRONMENT}" was not found`], detail: 'not found' };
|
|
17
|
+
}
|
|
18
|
+
const rules = Array.isArray(environment.protection_rules) ? environment.protection_rules : [];
|
|
19
|
+
const branchPolicies = rules.filter((rule) => rule?.type === 'branch_policy').length;
|
|
20
|
+
const reviewerRules = rules.filter((rule) => rule?.type === 'required_reviewers');
|
|
21
|
+
const reviewers = reviewerRules.flatMap((rule) => (Array.isArray(rule.reviewers) ? rule.reviewers : [])
|
|
22
|
+
.map((entry) => entry?.reviewer?.login || entry?.reviewer?.slug || entry?.reviewer?.name || entry?.type || 'unknown'));
|
|
23
|
+
const problems = [];
|
|
24
|
+
if (environment.can_admins_bypass !== false) problems.push('admins can bypass the environment (can_admins_bypass is not false)');
|
|
25
|
+
if (branchPolicies === 0) problems.push('no branch_policy rule (any branch could deploy)');
|
|
26
|
+
if (reviewerRules.length) problems.push(`a required reviewer is back on the environment (${reviewers.join(', ') || 'unnamed'}) — no human approves a release`);
|
|
27
|
+
return {
|
|
28
|
+
ok: problems.length === 0,
|
|
29
|
+
problems,
|
|
30
|
+
detail: `can_admins_bypass=${environment.can_admins_bypass}; branch_policy rules: ${branchPolicies}; required_reviewers rules: ${reviewerRules.length}`
|
|
31
|
+
+ (problems.length ? `\n${problems.join('\n')}` : ''),
|
|
32
|
+
};
|
|
33
|
+
}
|