ruvnet-brain 3.9.134-dev → 4.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +14 -0
- package/README.md +5 -5
- package/bin/install.mjs +382 -36
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/kb/zip-extract.mjs +53 -14
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +14 -22
- package/plugin/.claude-plugin/marketplace.json +14 -0
- package/plugin/.claude-plugin/plugin.json +22 -0
- package/plugin/.codex-plugin/plugin.json +21 -0
- package/plugin/.mcp.json +8 -0
- package/plugin/commands/brain-console.md +16 -0
- package/plugin/commands/configure.md +33 -0
- package/plugin/commands/rvbc.md +79 -0
- package/plugin/commands/rvcb.md +16 -0
- package/plugin/commands/whats-new.md +57 -0
- package/plugin/hooks/codex-hooks.json +160 -0
- package/plugin/hooks/hook-contracts.json +77 -0
- package/plugin/hooks/hooks.json +202 -0
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +56 -6
- package/plugin/scripts/anticipate.sh +534 -0
- package/plugin/scripts/codex-hook-adapter.mjs +96 -0
- package/plugin/scripts/continuation-gate.mjs +267 -0
- package/plugin/scripts/design-wall.sh +137 -0
- package/plugin/scripts/detach.mjs +182 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/gate-receipt.sh +35 -0
- package/plugin/scripts/ground-before-write.sh +199 -0
- package/plugin/scripts/ground-ruvnet.sh +517 -0
- package/plugin/scripts/grounding-stamp.sh +113 -0
- package/plugin/scripts/grounding-substance.mjs +595 -0
- package/plugin/scripts/hijack-ruvnet.sh +81 -0
- package/plugin/scripts/hook-input.mjs +558 -0
- package/plugin/scripts/hook-shim-bash.mjs +55 -0
- package/plugin/scripts/hook-shim.mjs +303 -0
- package/plugin/scripts/host-update.mjs +58 -0
- package/plugin/scripts/kling-preflight.sh +146 -0
- package/plugin/scripts/learn-capture.sh +173 -0
- package/plugin/scripts/learn-flush.mjs +155 -0
- package/plugin/scripts/lesson-hooks.sh +213 -0
- package/plugin/scripts/md-stamp.mjs +219 -0
- package/plugin/scripts/protect-brain-state.sh +84 -0
- package/plugin/scripts/route-dispatch.sh +147 -0
- package/plugin/scripts/routing-outcome-capture.mjs +89 -0
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +477 -0
- package/plugin/scripts/session-start.sh +13 -0
- package/plugin/scripts/signal-watch.mjs +193 -0
- package/plugin/scripts/unprompted-runtime.mjs +377 -0
- package/plugin/scripts/update-apply.mjs +419 -0
- package/plugin/scripts/verify-interface.sh +53 -0
- package/plugin/scripts/version-bump-gate.sh +112 -0
- package/plugin/skills/brain-build/SKILL.md +123 -0
- package/plugin/skills/brain-console/SKILL.md +22 -0
- package/plugin/skills/brain-prompt/SKILL.md +83 -0
- package/plugin/skills/brain-score/SKILL.md +101 -0
- package/plugin/skills/release-proof/SKILL.md +81 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +121 -0
- package/plugin/skills/ruvnet-brain/SKILL.md +234 -0
- package/plugin/skills/rvbc/SKILL.md +23 -0
- package/plugin/skills/savings/SKILL.md +46 -0
- package/plugin/skills/whats-new/SKILL.md +22 -0
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +522 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +250 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +639 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +271 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +180 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2749 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +395 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +508 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +864 -0
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// eval-brain.mjs — the eval flywheel. Ask the frozen held-out questions, and judge the answers by
|
|
3
|
+
// GROUND TRUTH rather than by a model's opinion of them. (ADR-0011 Phase 0.)
|
|
4
|
+
//
|
|
5
|
+
// FIVE STRATA, because a gate that only asks easy questions cannot fail:
|
|
6
|
+
// named — the repo is named in the question pass = grounded AND routed
|
|
7
|
+
// described — capability described, no names pass = grounded AND routed
|
|
8
|
+
// scenario — a real-world situation, no names pass = grounded AND routed
|
|
9
|
+
// adversarial — the correct answer is NOT in this corpus pass = ABSTAINED (top ce < 0, or no hits)
|
|
10
|
+
// provenance — gist-shaped content pass = grounded AND (if the top hit IS a
|
|
11
|
+
// gist chunk, it must carry its GIST STATUS banner — better repo grounding also passes)
|
|
12
|
+
//
|
|
13
|
+
// GATING IS ON THE WILSON LOWER BOUND, never the point estimate. With n=12, routed 10/12 had a 95%
|
|
14
|
+
// CI of [55.2%, 95.3%] and 9/12's upper bound (91.1%) overlapped it completely — the gate could not
|
|
15
|
+
// detect the regression it existed to catch. n=120 gives ≥80% power for a 0.90 -> 0.80 drop (n≈69
|
|
16
|
+
// suffices; computed 2026-07-09).
|
|
17
|
+
//
|
|
18
|
+
// FAIL-CLOSED PROMOTION: `--gate` compares each metric's lower bound against evals/baseline.json and
|
|
19
|
+
// exits 1 on any drop. A missing baseline is a failure too — you cannot promote against nothing.
|
|
20
|
+
// Baselines are only ever written deliberately with `--record`.
|
|
21
|
+
//
|
|
22
|
+
// Why no model judge: an LLM panel scored a ZERO-CITATION answer 98/100 on this repo.
|
|
23
|
+
//
|
|
24
|
+
// node scripts/eval-brain.mjs # run + table
|
|
25
|
+
// node scripts/eval-brain.mjs --gate # run + exit 1 on regression (or missing baseline)
|
|
26
|
+
// node scripts/eval-brain.mjs --record # run + write evals/baseline.json (deliberate)
|
|
27
|
+
// node scripts/eval-brain.mjs --json # machine-readable
|
|
28
|
+
// node scripts/eval-brain.mjs --strata named,adversarial # subset (never gate on a subset)
|
|
29
|
+
|
|
30
|
+
import fs from 'node:fs';
|
|
31
|
+
import os from 'node:os';
|
|
32
|
+
import path from 'node:path';
|
|
33
|
+
import { spawnSync } from 'node:child_process';
|
|
34
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
35
|
+
|
|
36
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
37
|
+
const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
|
|
38
|
+
const HELD_OUT = path.join(ROOT, 'evals', 'held-out.json');
|
|
39
|
+
const BASELINE = path.join(ROOT, 'evals', 'baseline.json');
|
|
40
|
+
|
|
41
|
+
// The cross-encoder emits a relevance logit per (query, passage): strongly negative when unrelated.
|
|
42
|
+
// An adversarial question "passes" when the brain effectively found nothing relevant. 0 is the
|
|
43
|
+
// neutral cut; per-question ce values are recorded so this stays inspectable, and the Wilson-gated
|
|
44
|
+
// baseline makes the stratum a regression detector even if the absolute rate is imperfect.
|
|
45
|
+
export const ABSTAIN_CE = 0;
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Order-independent, tamper-evident hash of the held-out set — rUv's own frozen-eval pattern
|
|
49
|
+
* (ruflo harness-frozen-eval: humanEvalHash / FROZEN_HUMAN_EVAL_HASH). The pinned constant lives in
|
|
50
|
+
* tests/unit/eval-brain-gate.test.mjs; editing ANY question turns that test red, which is what
|
|
51
|
+
* "frozen" means mechanically. Reordering does not (sorting makes the hash order-independent).
|
|
52
|
+
*/
|
|
53
|
+
export async function heldOutHash(questions) {
|
|
54
|
+
const { createHash } = await import('node:crypto');
|
|
55
|
+
const per = questions.map((q) =>
|
|
56
|
+
createHash('sha256').update(JSON.stringify({ id: q.id, stratum: q.stratum, query: q.query, expectRepo: q.expectRepo ?? null })).digest('hex'));
|
|
57
|
+
return createHash('sha256').update(per.sort().join('')).digest('hex');
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** Wilson score interval — the same instrument rUv uses on every SWE-bench number. */
|
|
61
|
+
export function wilson(k, n, z = 1.96) {
|
|
62
|
+
if (!n) return { p: 0, lo: 0, hi: 1 };
|
|
63
|
+
const p = k / n;
|
|
64
|
+
const d = 1 + (z * z) / n;
|
|
65
|
+
const c = (p + (z * z) / (2 * n)) / d;
|
|
66
|
+
const h = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / d;
|
|
67
|
+
return { p, lo: Math.max(0, c - h), hi: Math.min(1, c + h) };
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Grade one answered question by its stratum's rule. Pure — takes the verifier verdict, the top
|
|
72
|
+
* citation (with its ce score), and whether the raw output carried the gist provenance banner.
|
|
73
|
+
*/
|
|
74
|
+
export function gradeQuestion(q, { grounded, citations, bannerPresent }) {
|
|
75
|
+
const top = citations?.[0] ?? null;
|
|
76
|
+
const routed = !!(grounded && q.expectRepo?.length && top && q.expectRepo.includes(top.repo));
|
|
77
|
+
const abstained = !top || (typeof top.ce === 'number' && top.ce < ABSTAIN_CE);
|
|
78
|
+
switch (q.stratum) {
|
|
79
|
+
case 'adversarial':
|
|
80
|
+
return { grounded, routed: null, abstained, pass: abstained };
|
|
81
|
+
case 'provenance':
|
|
82
|
+
// The mechanism under test: IF a gist chunk wins, it must carry its own status banner.
|
|
83
|
+
// A better hit from the real repo is not a failure — it is better grounding.
|
|
84
|
+
return { grounded, routed: null, abstained, pass: !!grounded && (top?.repo !== 'ruv-gists' || bannerPresent) };
|
|
85
|
+
default:
|
|
86
|
+
return { grounded, routed, abstained, pass: !!grounded && routed };
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/** Aggregate graded rows into the four gated metrics, each with its Wilson interval. */
|
|
91
|
+
export function aggregate(rows) {
|
|
92
|
+
const by = (pred) => rows.filter(pred);
|
|
93
|
+
const routedStrata = (r) => ['named', 'described', 'scenario'].includes(r.stratum);
|
|
94
|
+
const groundable = by((r) => r.stratum !== 'adversarial');
|
|
95
|
+
const routed = by(routedStrata);
|
|
96
|
+
const adversarial = by((r) => r.stratum === 'adversarial');
|
|
97
|
+
const provenance = by((r) => r.stratum === 'provenance');
|
|
98
|
+
const metric = (arr, key) => {
|
|
99
|
+
const k = arr.filter((r) => r[key]).length;
|
|
100
|
+
return { k, n: arr.length, ...wilson(k, arr.length) };
|
|
101
|
+
};
|
|
102
|
+
return {
|
|
103
|
+
grounded: metric(groundable, 'grounded'),
|
|
104
|
+
routed: metric(routed, 'pass'),
|
|
105
|
+
abstain: metric(adversarial, 'pass'),
|
|
106
|
+
banner: metric(provenance, 'pass'),
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/** The fail-closed comparison: every metric's lower bound must hold the baseline's lower bound. */
|
|
111
|
+
export function gateAgainst(current, baseline) {
|
|
112
|
+
if (!baseline) return { pass: false, regressions: ['no baseline to promote against — record one deliberately (--record)'] };
|
|
113
|
+
// An old-schema baseline (pre-strata counts, no {k,n,lo}) must FAIL, not pass vacuously — a gate
|
|
114
|
+
// that silently compares against nothing is the decorated-green failure this whole phase kills.
|
|
115
|
+
if (!['grounded', 'routed', 'abstain', 'banner'].some((m) => baseline[m]?.n)) {
|
|
116
|
+
return { pass: false, regressions: ['baseline is from an older schema — re-record deliberately (--record)'] };
|
|
117
|
+
}
|
|
118
|
+
const regressions = [];
|
|
119
|
+
for (const m of ['grounded', 'routed', 'abstain', 'banner']) {
|
|
120
|
+
const cur = current[m];
|
|
121
|
+
const base = baseline[m];
|
|
122
|
+
if (!base || !base.n) continue; // metric absent from an older baseline — cannot regress against nothing
|
|
123
|
+
if (!cur.n) { regressions.push(`${m}: stratum is empty but the baseline has n=${base.n}`); continue; }
|
|
124
|
+
if (cur.lo < base.lo - 1e-9) regressions.push(`${m}: lower bound ${(cur.lo * 100).toFixed(1)}% < baseline ${(base.lo * 100).toFixed(1)}%`);
|
|
125
|
+
}
|
|
126
|
+
return { pass: regressions.length === 0, regressions };
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
async function main() {
|
|
130
|
+
const argv = process.argv.slice(2);
|
|
131
|
+
const GATE = argv.includes('--gate');
|
|
132
|
+
const RECORD = argv.includes('--record');
|
|
133
|
+
const JSON_OUT = argv.includes('--json');
|
|
134
|
+
const strataArg = argv.includes('--strata') ? argv[argv.indexOf('--strata') + 1]?.split(',') : null;
|
|
135
|
+
const die = (msg) => { console.error(`eval-brain: ${msg}`); process.exit(2); };
|
|
136
|
+
|
|
137
|
+
if (!fs.existsSync(path.join(KB, 'forge-ask-all.mjs'))) die(`no brain at ${KB} — run: npx ruvnet-brain`);
|
|
138
|
+
const verifierPath = path.join(KB, 'verify-citation.mjs');
|
|
139
|
+
if (!fs.existsSync(verifierPath)) die('this bundle predates verify-citation.mjs — refusing to score grounding without a way to check it');
|
|
140
|
+
const { verifyGrounding } = await import(pathToFileURL(verifierPath).href);
|
|
141
|
+
|
|
142
|
+
let { questions } = JSON.parse(fs.readFileSync(HELD_OUT, 'utf8'));
|
|
143
|
+
for (const q of questions) if (!q.stratum) die(`question ${q.id} has no stratum`);
|
|
144
|
+
if (strataArg) {
|
|
145
|
+
questions = questions.filter((q) => strataArg.includes(q.stratum));
|
|
146
|
+
if (GATE || RECORD) die('--strata is for local iteration only; never gate or record on a subset');
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
// Each question is an independent subprocess with its own ONNX thread, so the pool parallelizes
|
|
150
|
+
// cleanly across cores — spawnSync would serialize the whole run on the event loop. Measured
|
|
151
|
+
// serial cost was ~13.5s/question ≈ 27 min for 120; at concurrency 6 the wall drops ~6×.
|
|
152
|
+
const { execFile } = await import('node:child_process');
|
|
153
|
+
const ask = (query) => new Promise((resolve) => {
|
|
154
|
+
execFile('node', ['forge-ask-all.mjs', '--dir', KB, '--q', query, '--k', '3'],
|
|
155
|
+
{ cwd: KB, timeout: 240000, env: process.env, maxBuffer: 64 * 1024 * 1024 },
|
|
156
|
+
(err, stdout) => resolve(err ? '' : String(stdout || '')));
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
const CONC = Math.max(1, Number(process.env.EVAL_CONCURRENCY ?? (argv.includes('--concurrency') ? argv[argv.indexOf('--concurrency') + 1] : 6)) || 6);
|
|
160
|
+
const rows = new Array(questions.length);
|
|
161
|
+
let cursor = 0;
|
|
162
|
+
let done = 0;
|
|
163
|
+
// An empty answer means the SUBPROCESS failed (contention, timeout, OOM) — an infrastructure
|
|
164
|
+
// event, not a retrieval verdict. Scoring it as "not grounded" pollutes the quality metric with
|
|
165
|
+
// ops noise: one dead process under 6-wide load dragged grounded's lower bound below baseline and
|
|
166
|
+
// failed a gate that retrieval never failed (caught live, 2026-07-10, question ho-04 → "ce=—").
|
|
167
|
+
// Retry once; if it still returns nothing, the row is an INFRA ERROR and the run is inconclusive.
|
|
168
|
+
const infraErrors = [];
|
|
169
|
+
const runOne = async () => {
|
|
170
|
+
while (true) {
|
|
171
|
+
const i = cursor++;
|
|
172
|
+
if (i >= questions.length) return;
|
|
173
|
+
const q = questions[i];
|
|
174
|
+
let out = await ask(q.query);
|
|
175
|
+
if (!out) { await new Promise((r) => setTimeout(r, 2000)); out = await ask(q.query); }
|
|
176
|
+
if (!out) {
|
|
177
|
+
infraErrors.push(q.id);
|
|
178
|
+
done++;
|
|
179
|
+
process.stderr.write(`\r[eval] ${done}/${questions.length} ! ${q.id} (infra) `);
|
|
180
|
+
continue;
|
|
181
|
+
}
|
|
182
|
+
const v = await verifyGrounding(out, KB);
|
|
183
|
+
const graded = gradeQuestion(q, { grounded: v.grounded, citations: v.citations, bannerPresent: /GIST STATUS/.test(out) });
|
|
184
|
+
const top = v.citations?.[0] ?? null;
|
|
185
|
+
rows[i] = {
|
|
186
|
+
id: q.id, stratum: q.stratum, query: q.query,
|
|
187
|
+
citedRepo: top?.repo ?? null, citedPath: top?.fullPath ?? null, ce: top?.ce ?? null,
|
|
188
|
+
...graded,
|
|
189
|
+
};
|
|
190
|
+
done++;
|
|
191
|
+
process.stderr.write(`\r[eval] ${done}/${questions.length} ${graded.pass ? '✓' : '✗'} ${q.id} `);
|
|
192
|
+
}
|
|
193
|
+
};
|
|
194
|
+
await Promise.all(Array.from({ length: Math.min(CONC, questions.length) }, runOne));
|
|
195
|
+
process.stderr.write('\n');
|
|
196
|
+
|
|
197
|
+
if (infraErrors.length) {
|
|
198
|
+
console.error(`[eval-brain] INCONCLUSIVE — ${infraErrors.length} infrastructure failure(s) after retry: ${infraErrors.join(', ')}`);
|
|
199
|
+
console.error(' A dead subprocess is not a retrieval verdict. Re-run (consider EVAL_CONCURRENCY=3 on a loaded machine).');
|
|
200
|
+
process.exit(2); // never a quality verdict, never a recorded baseline
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
const score = aggregate(rows.filter(Boolean));
|
|
204
|
+
|
|
205
|
+
if (JSON_OUT) {
|
|
206
|
+
console.log(JSON.stringify({ score, rows }, null, 2));
|
|
207
|
+
} else {
|
|
208
|
+
console.log(`\n# eval-brain — frozen held-out set (${rows.length} questions)\n`);
|
|
209
|
+
console.log('| metric | k/n | rate | 95% Wilson |');
|
|
210
|
+
console.log('|---|---|---|---|');
|
|
211
|
+
for (const [m, s] of Object.entries(score)) {
|
|
212
|
+
if (!s.n) continue;
|
|
213
|
+
console.log(`| ${m} | ${s.k}/${s.n} | ${(s.p * 100).toFixed(1)}% | [${(s.lo * 100).toFixed(1)}%, ${(s.hi * 100).toFixed(1)}%] |`);
|
|
214
|
+
}
|
|
215
|
+
const fails = rows.filter((r) => !r.pass);
|
|
216
|
+
if (fails.length) {
|
|
217
|
+
console.log(`\nFailures (${fails.length}):`);
|
|
218
|
+
for (const f of fails) console.log(` ✗ ${f.id} [${f.stratum}] → ${f.citedRepo ?? '—'} ce=${f.ce ?? '—'} ${f.query.slice(0, 64)}`);
|
|
219
|
+
}
|
|
220
|
+
console.log('\nGrounded = the cited passage exists on disk. Routed = the owning repo answered.');
|
|
221
|
+
console.log('Abstain = the brain declined an out-of-corpus question. Banner = a winning gist chunk carried its provenance.');
|
|
222
|
+
console.log('No model graded anything here.\n');
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
if (RECORD) {
|
|
226
|
+
fs.mkdirSync(path.dirname(BASELINE), { recursive: true });
|
|
227
|
+
fs.writeFileSync(BASELINE, JSON.stringify({ recorded: new Date().toISOString(), n: rows.length, score }, null, 2) + '\n');
|
|
228
|
+
console.error(`[eval-brain] baseline recorded (${rows.length} questions)`);
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
if (GATE) {
|
|
232
|
+
const baseline = fs.existsSync(BASELINE) ? JSON.parse(fs.readFileSync(BASELINE, 'utf8')).score : null;
|
|
233
|
+
const g = gateAgainst(score, baseline);
|
|
234
|
+
if (!g.pass) {
|
|
235
|
+
console.error(`[eval-brain] FAIL (fail-closed): ${g.regressions.join('; ')}`);
|
|
236
|
+
process.exit(1);
|
|
237
|
+
}
|
|
238
|
+
console.error('[eval-brain] PASS: every metric holds its baseline Wilson lower bound');
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
243
|
+
await main();
|
|
244
|
+
}
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// fix-metaharness-memretrieve.mjs — repair (and guard) a real bug in Ruflo's metaharness plugin.
|
|
3
|
+
//
|
|
4
|
+
// THE BUG
|
|
5
|
+
// -------
|
|
6
|
+
// `audit-list.mjs` and `audit-trend.mjs` read a stored audit record with:
|
|
7
|
+
// npx @claude-flow/cli memory retrieve --namespace <ns> --key <key>
|
|
8
|
+
// ...WITHOUT `--format json`, then greedily `JSON.parse` whatever `{...}` they can scrape from
|
|
9
|
+
// human-formatted stdout. But `memory retrieve --format json` returns an ENVELOPE:
|
|
10
|
+
// { id, key, namespace, content: "<the audit record, JSON-stringified>", ... }
|
|
11
|
+
// so the record lives in `.content` (a string) and must be unwrapped. Without both changes,
|
|
12
|
+
// memRetrieve() returns null for EVERY key, and `metaharness_audit_list` /
|
|
13
|
+
// `metaharness_drift_from_history` silently report `records: []` while `totalInNamespace > 0`.
|
|
14
|
+
// (Symptom: "0 audit records" even right after a successful oia-audit persisted one.)
|
|
15
|
+
//
|
|
16
|
+
// WHY THIS SCRIPT EXISTS
|
|
17
|
+
// ----------------------
|
|
18
|
+
// Both files live inside the GLOBAL npm package `@claude-flow/cli`, so `npm update -g` (or any
|
|
19
|
+
// reinstall) silently reverts the patch. This has already happened twice. Rather than hand-edit
|
|
20
|
+
// a third time, this script re-applies it idempotently AND can verify it in CI / a doctor check.
|
|
21
|
+
//
|
|
22
|
+
// node scripts/fix-metaharness-memretrieve.mjs --check # exit 1 if reverted (guard)
|
|
23
|
+
// node scripts/fix-metaharness-memretrieve.mjs --apply # patch in place (idempotent)
|
|
24
|
+
//
|
|
25
|
+
// The durable fix is upstream; this keeps the local install honest until that lands.
|
|
26
|
+
|
|
27
|
+
import fs from 'node:fs';
|
|
28
|
+
import path from 'node:path';
|
|
29
|
+
import os from 'node:os';
|
|
30
|
+
import { pathToFileURL } from 'node:url';
|
|
31
|
+
|
|
32
|
+
const TARGET_DIR =
|
|
33
|
+
process.env.METAHARNESS_SCRIPTS_DIR ||
|
|
34
|
+
path.join(os.homedir(), '.npm-global/lib/node_modules/@claude-flow/cli/plugins/ruflo-metaharness/scripts');
|
|
35
|
+
|
|
36
|
+
export const FILES = ['audit-list.mjs', 'audit-trend.mjs'];
|
|
37
|
+
const SENTINEL = 'outer.content'; // present only when the fix is applied
|
|
38
|
+
|
|
39
|
+
const FIXED = `function memRetrieve(key) {
|
|
40
|
+
const r = spawnSync('npx', [
|
|
41
|
+
CLI_PKG, 'memory', 'retrieve',
|
|
42
|
+
'--namespace', NS, '--key', key, '--format', 'json',
|
|
43
|
+
], { stdio: ['ignore', 'pipe', 'pipe'], encoding: 'utf-8', shell: process.platform === 'win32' });
|
|
44
|
+
if (r.status !== 0) return null;
|
|
45
|
+
const m = /\\{[\\s\\S]*\\}/.exec(r.stdout || '');
|
|
46
|
+
if (!m) return null;
|
|
47
|
+
try {
|
|
48
|
+
const outer = JSON.parse(m[0]);
|
|
49
|
+
// \`memory retrieve --format json\` wraps the record: { id, key, content: "<record JSON>" }.
|
|
50
|
+
// Unwrap \`content\` — parsing the envelope AS the record is the empty-results bug.
|
|
51
|
+
const inner = typeof outer.content === 'string' ? outer.content
|
|
52
|
+
: typeof outer.value === 'string' ? outer.value
|
|
53
|
+
: null;
|
|
54
|
+
if (inner) { try { return JSON.parse(inner); } catch { return null; } }
|
|
55
|
+
return (outer.startedAt || outer.composite) ? outer : null;
|
|
56
|
+
} catch { return null; }
|
|
57
|
+
}`;
|
|
58
|
+
|
|
59
|
+
// Match the whole memRetrieve function, up to the first closing brace at column 0.
|
|
60
|
+
const FN_RE = /function memRetrieve\(key\) \{[\s\S]*?\n\}/;
|
|
61
|
+
|
|
62
|
+
export function statusOf(file, dir = TARGET_DIR) {
|
|
63
|
+
const p = path.join(dir, file);
|
|
64
|
+
if (!fs.existsSync(p)) return { file, p, state: 'missing' };
|
|
65
|
+
const src = fs.readFileSync(p, 'utf-8');
|
|
66
|
+
if (src.includes(SENTINEL)) return { file, p, state: 'fixed', src };
|
|
67
|
+
if (FN_RE.test(src)) return { file, p, state: 'reverted', src };
|
|
68
|
+
return { file, p, state: 'unrecognized', src };
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export function apply(dir = TARGET_DIR, log = console.log) {
|
|
72
|
+
let changed = 0;
|
|
73
|
+
for (const file of FILES) {
|
|
74
|
+
const s = statusOf(file, dir);
|
|
75
|
+
if (s.state === 'missing') { log(` – ${file}: not installed here — nothing to patch`); continue; }
|
|
76
|
+
if (s.state === 'fixed') { log(` ✓ ${file}: already fixed`); continue; }
|
|
77
|
+
if (s.state === 'unrecognized') { log(` ⚠ ${file}: memRetrieve() not recognized — upstream changed shape; skipping`); continue; }
|
|
78
|
+
fs.writeFileSync(s.p, s.src.replace(FN_RE, FIXED), 'utf-8');
|
|
79
|
+
log(` ✓ ${file}: PATCHED (envelope unwrap + --format json)`);
|
|
80
|
+
changed++;
|
|
81
|
+
}
|
|
82
|
+
log(changed ? `\napplied to ${changed} file(s).` : '\nnothing to do — already healthy.');
|
|
83
|
+
return 0;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
// A guard that cries wolf gets ignored. `missing` means the metaharness plugin simply isn't
|
|
87
|
+
// installed on this machine (the common case in CI) — that is NOT a failure. Only a file that
|
|
88
|
+
// EXISTS and has lost the fix is a failure, because that is the exact state an `npm update -g`
|
|
89
|
+
// leaves behind, and the symptom is silent (`records: []`, never an error).
|
|
90
|
+
export function check(dir = TARGET_DIR, log = console.log) {
|
|
91
|
+
let reverted = 0;
|
|
92
|
+
let present = 0;
|
|
93
|
+
for (const file of FILES) {
|
|
94
|
+
const s = statusOf(file, dir);
|
|
95
|
+
if (s.state === 'missing') { log(` – ${file}: n/a (metaharness plugin not installed)`); continue; }
|
|
96
|
+
present++;
|
|
97
|
+
if (s.state === 'reverted') reverted++;
|
|
98
|
+
log(` ${s.state === 'fixed' ? '✓' : '✗'} ${file}: ${s.state}`);
|
|
99
|
+
}
|
|
100
|
+
if (!present) {
|
|
101
|
+
log('\n– metaharness plugin not installed — guard not applicable (pass).');
|
|
102
|
+
return 0;
|
|
103
|
+
}
|
|
104
|
+
if (reverted) {
|
|
105
|
+
log(`\n✗ ${reverted} file(s) reverted — an npm update wiped the fix.`);
|
|
106
|
+
log(' Repair: node scripts/fix-metaharness-memretrieve.mjs --apply');
|
|
107
|
+
return 1;
|
|
108
|
+
}
|
|
109
|
+
log('\n✓ metaharness memRetrieve fix intact.');
|
|
110
|
+
return 0;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// Only act when run directly, so tests can import the pure functions above.
|
|
114
|
+
// Compare in URL space (pathToFileURL), not by decoding a URL into a path: `new URL(...).pathname`
|
|
115
|
+
// yields "/D:/..." on Windows and never matches, so `--apply` would silently no-op there.
|
|
116
|
+
const invokedDirectly = process.argv[1] && pathToFileURL(path.resolve(process.argv[1])).href === import.meta.url;
|
|
117
|
+
if (invokedDirectly) {
|
|
118
|
+
const mode = process.argv.includes('--check') ? 'check' : 'apply';
|
|
119
|
+
console.log(`metaharness memRetrieve ${mode} — ${TARGET_DIR}\n`);
|
|
120
|
+
process.exit(mode === 'check' ? check() : apply());
|
|
121
|
+
}
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
// full-hints.mjs — SHARED per-repo depth configuration for every builder entrypoint.
|
|
2
|
+
//
|
|
3
|
+
// Why this file exists: FULL_HINTS originally lived inline in self-update.mjs only. Any rebuild
|
|
4
|
+
// that went through scripts/ingest-repo.mjs (which never passed --full) silently downgraded a
|
|
5
|
+
// repo's full-body source indexing to doc-comment-only snippets — 2026-07-10 depth-restore run
|
|
6
|
+
// reproduced exactly that: ruvector rebuilt with 18,491 passages and 0 full bodies. One map,
|
|
7
|
+
// imported by BOTH self-update.mjs and ingest-repo.mjs, kills that class of bug.
|
|
8
|
+
//
|
|
9
|
+
// FULL_HINTS: comma-separated path prefixes (relative to repo root) whose source files are
|
|
10
|
+
// indexed with FULL bodies by kb/forge-build.mjs --full. See self-update.mjs history
|
|
11
|
+
// (fc9a178, f17bbf7) for how these were empirically derived from the shipped bundle.
|
|
12
|
+
//
|
|
13
|
+
// KEEP_DIRS: comma-separated directory NAMES that forge-build.mjs must NOT skip for this repo
|
|
14
|
+
// (kb/forge-build.mjs --keep). forge-build's SKIP_DIRS globally excludes 'v2' (noise in most
|
|
15
|
+
// repos), but open-claude-code's entire source lives in v2/src and RuView's active tree is
|
|
16
|
+
// v2/crates — without --keep v2 those repos index almost nothing (open-claude-code: 68 passages
|
|
17
|
+
// total) and their FULL_HINTS can never match.
|
|
18
|
+
|
|
19
|
+
export const FULL_HINTS = {
|
|
20
|
+
'ruflo': 'v3/@claude-flow,v3/mcp,ruflo/src',
|
|
21
|
+
'synthlang': 'proxy/src/cli/synthlang,proxy/src/app/synthlang',
|
|
22
|
+
'agentdb': 'src',
|
|
23
|
+
'rulake': 'crates',
|
|
24
|
+
'daa': 'crates/daa-ai,crates/daa-chain,crates/daa-economy,crates/daa-rules,daa-ai/src,daa-chain/src,daa-cli/src,daa-compute/benches,daa-compute/build.rs,daa-compute/src,daa-compute/tests,daa-economy/src,daa-mcp/src,daa-orchestrator/daa-napi,daa-orchestrator/src,daa-orchestrator/tests,daa-rules/src,daa-sdk/crates,examples/agents,examples/basic-crypto.ts,examples/decentralized-task-scheduler.ts,examples/federated-learning.ts,examples/full-stack-agent.ts,examples/orchestrator.ts,examples/performance-benchmark.ts,prime-rust/crates,prime-rust/prime-napi,prime-rust/tests,src/main.rs,src/security',
|
|
25
|
+
'qudag': 'benchmarks/benches,benchmarks/cli,benchmarks/dark_addressing,benchmarks/lib.rs,benchmarks/optimized_benchmarks.rs,benchmarks/src,cli-standalone/src,cli-standalone/tests,core/crypto,core/dag,core/health.rs,core/monitoring,core/network,core/optimized,core/protocol,core/swarm,core/vault,examples/bitchat,examples/crypto,examples/dark_addressing_example.rs,examples/dht_discovery_example.rs,examples/nat_traversal_example.rs,examples/onion_routing_example.rs,examples/peer_management_example.rs,examples/persistence_example.rs,examples/shadow_address_example.rs,examples/traffic_obfuscation_example.rs,qudag-exchange/cli,qudag-exchange/core,qudag-exchange/crates,qudag-exchange/src,qudag-exchange/test_core_fee_model.rs,qudag-exchange/test_fee_model.rs,qudag-exchange/tests,qudag-mcp/benches,qudag-mcp/examples,qudag-mcp/src,qudag-mcp/tests,qudag-testnet/configs,qudag-wasm/final-test.mjs,qudag-wasm/simple-test.mjs,qudag-wasm/src,qudag-wasm/test-nodejs.mjs,qudag-wasm/test-setup.ts,qudag-wasm/tests,qudag-wasm/vitest.config.ts,qudag-wasm/vitest.workspace.ts,qudag-wasm/working-features-test.mjs,tools/cli,tools/simulator,tools/swarm-test,vault-standalone/examples,vault-standalone/src,vault-standalone/tests',
|
|
26
|
+
'ruvector': 'crates/agentic-robotics-core,crates/agentic-robotics-mcp,crates/agentic-robotics-node,crates/agentic-robotics-rt,crates/cognitum-gate-kernel,crates/cognitum-gate-tilezero,crates/emergent-time,crates/emergent-time-wasm,crates/hailort-sys,crates/mcp-brain,crates/mcp-brain-server,crates/mcp-gate,crates/micro-hnsw-wasm,crates/neural-trader-coherence,crates/neural-trader-core,crates/neural-trader-replay,crates/neural-trader-wasm,crates/photonlayer-bench,crates/photonlayer-cli,crates/photonlayer-core,crates/photonlayer-ruvector,crates/photonlayer-wasm,crates/prime-radiant,crates/ruos-thermal,crates/ruvector-acorn,crates/ruvector-acorn-wasm,crates/ruvector-agent-memory,crates/ruvector-attention,crates/ruvector-attention-cli,crates/ruvector-attention-node,crates/ruvector-attention-wasm,crates/ruvector-attn-mincut,crates/ruvector-bench,crates/ruvector-bet4-ivf-bench,crates/ruvector-capgated,crates/ruvector-cli,crates/ruvector-cluster,crates/ruvector-cnn,crates/ruvector-cnn-wasm,crates/ruvector-coherence,crates/ruvector-coherence-hnsw,crates/ruvector-collections,crates/ruvector-consciousness,crates/ruvector-core,crates/ruvector-crv,crates/ruvector-dag,crates/ruvector-dag-wasm,crates/ruvector-decompiler,crates/ruvector-delta-core,crates/ruvector-delta-graph,crates/ruvector-delta-index,crates/ruvector-delta-wasm,crates/ruvector-diskann,crates/ruvector-diskann-node,crates/ruvector-dither,crates/ruvector-economy-wasm,crates/ruvector-exotic-wasm,crates/ruvector-filter,crates/ruvector-gnn,crates/ruvector-gnn-node,crates/ruvector-gnn-rerank,crates/ruvector-gnn-wasm,crates/ruvector-graph,crates/ruvector-graph-condense,crates/ruvector-graph-node,crates/ruvector-graph-wasm,crates/ruvector-hailo,crates/ruvector-hailo-cluster,crates/ruvector-hnsw-repair,crates/ruvector-hybrid,crates/ruvector-kalshi,crates/ruvector-learning-wasm,crates/ruvector-lsm-ann,crates/ruvector-math,crates/ruvector-math-wasm,crates/ruvector-matryoshka,crates/ruvector-maxsim,crates/ruvector-metrics,crates/ruvector-mincut,crates/ruvector-mincut-node,crates/ruvector-mincut-wasm,crates/ruvector-mmwave,crates/ruvector-nervous-system,crates/ruvector-node,crates/ruvector-perception,crates/ruvector-postgres,crates/ruvector-pq-search,crates/ruvector-profiler,crates/ruvector-proof-gate,crates/ruvector-rabitq,crates/ruvector-rabitq-wasm,crates/ruvector-raft,crates/ruvector-rairs,crates/ruvector-replication,crates/ruvector-robotics,crates/ruvector-router-cli,crates/ruvector-router-core,crates/ruvector-router-ffi,crates/ruvector-router-wasm,crates/ruvector-rulake,crates/ruvector-server,crates/ruvector-snapshot,crates/ruvector-solver,crates/ruvector-solver-node,crates/ruvector-solver-wasm,crates/ruvector-sota-bench,crates/ruvector-spann,crates/ruvector-sparsifier,crates/ruvector-tiny-dancer-node,crates/ruvector-verified,crates/ruvector-verified-wasm,crates/ruvector-wasm,crates/ruvix,crates/ruvllm,crates/ruvllm-cli,crates/ruvllm-wasm,crates/ruvllm_sparse_attention,crates/rvAgent,crates/rvf,crates/rvlite,crates/rvm,crates/sona,crates/sonic-ct,crates/sonic-ct-wasm,crates/thermorust,crates/timesfm',
|
|
27
|
+
'ruv-fann': 'cuda-wasm/.eslintrc.js,cuda-wasm/benches,cuda-wasm/build.rs,cuda-wasm/cli,cuda-wasm/cuda-examples,cuda-wasm/demo,cuda-wasm/examples,cuda-wasm/jest.config.js,cuda-wasm/scripts,cuda-wasm/src,cuda-wasm/tests,examples/basic_usage.rs,examples/cuda_wasm_neural_integration.rs,examples/final_performance_demo.rs,examples/gpu_sweet_spot_benchmark.rs,examples/gpu_training_test.rs,examples/test_adam.rs,examples/test_gpu_detection.rs,examples/test_optimizers_simple.rs,examples/xor.rs,neuro-divergent/src,neuro-divergent/tests,opencv-rust/opencv-core,opencv-rust/opencv-sdk,opencv-rust/opencv-sys,opencv-rust/opencv-wasm,opencv-rust/tests,ruv-swarm/benches,ruv-swarm/benchmarking,ruv-swarm/crates,ruv-swarm/examples,ruv-swarm/ml-training,ruv-swarm/models,ruv-swarm/npm,ruv-swarm/test,ruv-swarm/test-simd-fix.mjs,ruv-swarm/test-wasm.js,ruv-swarm/tests,ruv-swarm/vitest.config.js,src/activation.rs,src/cascade.rs,src/connection.rs,src/errors.rs,src/integration.rs,src/io,src/layer.rs,src/lib.rs,src/memory_manager.rs,src/mock_types.rs,src/network.rs,src/network_gpu.rs,src/neuron.rs,src/simd,src/tests,src/training,src/webgpu',
|
|
28
|
+
'agentic-flow': 'agentic-flow/.claude,agentic-flow/Python,agentic-flow/add_two_numbers.py,agentic-flow/app,agentic-flow/benchmark,agentic-flow/examples,agentic-flow/path,agentic-flow/scripts,agentic-flow/src,agentic-flow/tests,agentic-flow/validation,agentic-flow/vitest.config.ts,crates/agentic-flow-quic,examples/batch-query.js,examples/batch-store.js,examples/billing-example.ts,examples/cached-query.js,examples/climate-prediction,examples/connection-pool.js,examples/deepseek-direct-api.js,examples/nova-medicina,examples/perf-monitor.js,examples/quic-server-coordinator.js,examples/quic-swarm-coordination.js,examples/reasoningbank-benchmark.js,examples/reasoningbank-learning-demo.js,examples/reasoningbank-optimize.js,examples/research-swarm,examples/verification-example.ts,packages/agent-booster,packages/agentdb-onnx,packages/agentic-jujutsu,packages/agentic-llm,reasoningbank/examples,reasoningbank/tests,src/App.tsx,src/api,src/cli,src/components,src/consent,src/controller,src/controllers,src/main.tsx,src/mcp,src/middleware,src/notifications,src/pages,src/providers,src/routing,src/security,src/services,src/transport,src/types,src/utils,src/verification',
|
|
29
|
+
'metaharness': 'crates/kernel,crates/kernel-napi,crates/kernel-wasm,crates/poker-darwin,crates/template-catalog,packages/aws-finops,packages/bench,packages/create-agent-harness,packages/darwin-mode,packages/harness,packages/host-claude-code,packages/host-codex,packages/host-copilot,packages/host-github-actions,packages/host-hermes,packages/host-openclaw,packages/host-opencode,packages/host-pi-dev,packages/host-rvm,packages/jujutsu,packages/kernel-js,packages/projects,packages/redblue,packages/router,packages/sdk,packages/vertical-base,packages/vertical-trading,packages/weight-eft',
|
|
30
|
+
'safla': 'benchmarks/__init__.py,benchmarks/cli_benchmarks.py,benchmarks/core.py,benchmarks/database.py,benchmarks/safla_benchmarks.py,benchmarks/utils.py,examples/01_basic_setup.py,examples/02_simple_memory.py,examples/03_basic_safety.py,examples/05_delta_evaluation.py,examples/12_ai_assistant.py,examples/15_enterprise_integration.py,examples/config_examples.py,examples/hybrid_memory_demo.py,examples/mcp_auth_client.py,examples/mcp_usage,examples/safety_validation_demo.py,safla/__init__.py,safla/__main__.py,safla/api,safla/auth,safla/cli.py,safla/cli_implementations.py,safla/cli_interactive.py,safla/cli_main.py,safla/cli_manager.py,safla/core,safla/exceptions.py,safla/installer.py,safla/integrations,safla/mcp,safla/mcp_stdio_server.py,safla/middleware,safla/security,safla/utils,safla/validation,safla_mcp_enhanced.py,safla_mcp_server.py,safla_mcp_simple.py,scripts/advanced_optimization_engine.py,scripts/agent_swarm_optimizer.py,scripts/build.py,scripts/comprehensive_capability_test.py,scripts/continuous_optimization_engine.py,scripts/debug_enhanced_server.py,scripts/demo_jwt_mcp_client.py,scripts/final_capability_verification.py,scripts/final_system_test.py,scripts/gpu_optimization_benchmark.py,scripts/install.py,scripts/minimal_security_test.py,scripts/quick_capability_test.py,scripts/remote_gpu_benchmarker.py,scripts/save_extreme_optimizations.py,scripts/save_optimized_models.py,scripts/system_status_report.py,scripts/verify_system.py',
|
|
31
|
+
'ruview': 'firmware/esp32-csi-node,firmware/esp32-hello-world,v2/crates',
|
|
32
|
+
'open-claude-code': 'v2/src',
|
|
33
|
+
'ruv-dev': 'bin/index.js,src/cli,src/core,src/index.js,src/utils',
|
|
34
|
+
'agenticow': 'bin/agenticow.js,examples/_shared.mjs,examples/ab-at-scale.mjs,examples/ab-branches.mjs,examples/checkpointing.mjs,examples/compliance-lineage.mjs,examples/git-workflow.mjs,examples/memory-evolution.mjs,examples/multi-persona-consensus.mjs,examples/multi-tenant-saas.mjs,examples/parallel-agents.mjs,examples/parallel-selves.mjs,examples/personalization.mjs,examples/promotion-pipeline.mjs,examples/red-team-sandbox.mjs,examples/rollback-quarantine.mjs,examples/simulated-org.mjs,examples/time-travel-debug.mjs,src/index.d.ts,src/index.js',
|
|
35
|
+
// cognitum-cogs (cognitum-one org): thin, scattered full-body coverage — superset of every
|
|
36
|
+
// top-level dir with ANY full-body content today (see self-update.mjs history for rationale).
|
|
37
|
+
'cognitum-cogs': 'benches,benchmarks,cognitum-sim,crates,examples,scripts,shared,src,tests',
|
|
38
|
+
// cognitum-support: 100% docs content, zero full-body entries — no --full prefix needed.
|
|
39
|
+
|
|
40
|
+
// ---- Cognitum One flagship-depth sweep, 2026-07-18 (Stuart: "same absolute crisp deep dive as
|
|
41
|
+
// the other repos — all ADRs, walk all the Rust crates, every markdown"). seed and v0-appliance
|
|
42
|
+
// were indexed SHALLOW until today — the same one-size-under-indexes-code-rich-repos failure the
|
|
43
|
+
// v0.5.0 depth audit fixed for qudag/ruv-fann. Prefixes are generic-but-superset (crates,src,…):
|
|
44
|
+
// forge-build skips prefixes that don't exist, so a superset is safe where a repo lacks a dir. ----
|
|
45
|
+
'cognitum-seed': 'crates,src,firmware,scripts,tests,examples,benches,docs/adr',
|
|
46
|
+
'cognitum-v0-appliance': 'crates,src,scripts,tests,examples,deploy,docs/adr',
|
|
47
|
+
'cognitum-open-design': 'src,app,apps,packages,electron,scripts,tests,server',
|
|
48
|
+
// cognitum-claude-plugin: REMOVED 2026-07-24. It declared 'src,scripts,plugin,plugins,mcp,tests'
|
|
49
|
+
// and only `plugins/` exists — but the whole repo, measured, is 233 .md + 108 .json + 1 .yml and
|
|
50
|
+
// ZERO source files, so it can never produce a full-body SOURCE passage. corpus-qa's S2 check
|
|
51
|
+
// ("declares --full but has 0 full bodies — silent depth loss") was therefore unsatisfiable by
|
|
52
|
+
// construction, and it FAILED THE NIGHTLY PUBLISH: self-update aborts fail-closed on any failed
|
|
53
|
+
// repo build, so no GitHub Release was cut while npm advanced. That is how the release channel
|
|
54
|
+
// reached v3.9.56 while npm was at 3.9.57 — a correct gate firing on a false declaration.
|
|
55
|
+
// It belongs with the docs-content group below: default depth IS the right depth here.
|
|
56
|
+
'cognitum-spoton': 'src,crates,scripts,tests,harness',
|
|
57
|
+
// cognitum-platform-docs / cognitum-meta-llm-docs / cognitum-meta-proxy-dist: docs/dist content —
|
|
58
|
+
// default depth is the right depth; no full-body prefixes needed.
|
|
59
|
+
|
|
60
|
+
// ---- tier-1 corpus expansion, 2026-07-10 (ADR-0011 Phase 5). Prefixes derived by inspecting
|
|
61
|
+
// each fresh shallow clone's top-level layout; junk/vendored/build dirs deliberately absent. ----
|
|
62
|
+
'midstream': 'src,crates,examples,benches,wasm,wasm-bindings,integrations,lean-agentic-js,xtask,AIMDS,fuzz,scripts,tests',
|
|
63
|
+
'rudevolution': 'src,examples,benches,scripts,npm,dashboard,tests',
|
|
64
|
+
'marketing': 'src,tests',
|
|
65
|
+
// flow-nexus: sdk/ is all markdown (platform is closed-source); the runnable JS lives in tutorials/.
|
|
66
|
+
'flow-nexus': 'sdk,tutorials',
|
|
67
|
+
'symbolic-scribe': 'src,harness,scripts,wasm',
|
|
68
|
+
// sublinear-time-solver: archive/, build-temp/, data/ (26MB datasets) intentionally excluded.
|
|
69
|
+
'sublinear-time-solver': 'src,crates,js,bin,server,benches,benchmarks,examples,integrations,types,validation,optimization,npx,scripts,tests',
|
|
70
|
+
// synaptic-mesh: src/rs VENDORS full copies of QuDAG (90M), daa (92M), ruv-FANN (19M) and
|
|
71
|
+
// src/js vendors claude-flow + ruv-swarm — ALL already covered by their own stores. Full-body
|
|
72
|
+
// only the mesh-specific code; vendored trees still get doc-comment/lead indexing from the walk.
|
|
73
|
+
'synaptic-mesh': 'src/mcp,src/neural,src/rs/neural-mesh,src/rs/synaptic-mesh-cli,src/rs/synaptic-mesh-p2p,src/rs/qudag-core,src/rs/daa-swarm,src/js/synaptic-cli,standalone-crates,tests,examples,scripts',
|
|
74
|
+
'agentic-security': 'src,tests,gui,security_pipeline.py,fix_cycle.py',
|
|
75
|
+
};
|
|
76
|
+
|
|
77
|
+
// Per-repo directory names to EXEMPT from forge-build's SKIP_DIRS walk exclusion.
|
|
78
|
+
export const KEEP_DIRS = {
|
|
79
|
+
'ruview': 'v2', // active tree is v2/crates (862 files) — skipped as noise otherwise
|
|
80
|
+
'open-claude-code': 'v2', // ALL source lives in v2/src — without this the store is ~68 passages
|
|
81
|
+
};
|
|
82
|
+
|
|
83
|
+
// CLI helper so shell drivers can read the config: node scripts/full-hints.mjs <kb-name>
|
|
84
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
85
|
+
const kb = (process.argv[2] || '').toLowerCase();
|
|
86
|
+
console.log(JSON.stringify({ full: FULL_HINTS[kb] || null, keep: KEEP_DIRS[kb] || null }));
|
|
87
|
+
}
|
package/scripts/gate.sh
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# gate.sh — rebuild the concepts/capability layer and run the three pass/fail routing gates.
|
|
3
|
+
# SEC-0010 #1: this gate must be able to FAIL. Each prove.mjs computes pass!=total -> exit 1
|
|
4
|
+
# (scripts/prove.mjs); previously the grep pipe masked that exit code and the script printed
|
|
5
|
+
# "GATES COMPLETE" (exit 0) even at 0%. Now every gate's real exit code is captured via
|
|
6
|
+
# PIPESTATUS and any miss makes the whole gate exit non-zero. Paths are repo-relative, not
|
|
7
|
+
# hardcoded to one machine.
|
|
8
|
+
set -uo pipefail
|
|
9
|
+
cd "$(dirname "$0")/.." || exit 1
|
|
10
|
+
# Model cache: honor an existing KB_MODEL_CACHE; else default to a repo-local dir (no personal path).
|
|
11
|
+
export KB_MODEL_CACHE="${KB_MODEL_CACHE:-$PWD/kb/models-cache}"
|
|
12
|
+
|
|
13
|
+
FAILED=0
|
|
14
|
+
run_gate() { # $1=label $2..=command
|
|
15
|
+
local label="$1"; shift
|
|
16
|
+
echo "== $label =="
|
|
17
|
+
"$@"
|
|
18
|
+
local rc="${PIPESTATUS[0]}"
|
|
19
|
+
if [ "$rc" -ne 0 ]; then echo " ✗ GATE FAILED (exit $rc)"; FAILED=1; else echo " ✓ gate passed"; fi
|
|
20
|
+
echo
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
echo "== rebuild concepts (L2 + primers + capability cards) =="
|
|
24
|
+
node scripts/build-concepts.mjs 2>&1 | tail -2
|
|
25
|
+
( cd kb && node forge-big.mjs both --dir . --name concepts 2>&1 | tail -2 )
|
|
26
|
+
echo
|
|
27
|
+
|
|
28
|
+
run_gate "GATE 1 — described-need battery (newcomer, no repo names) · target >=85%" \
|
|
29
|
+
node scripts/prove.mjs --questions scripts/described-questions.json --k 2 --out DESCRIBED-PROOF
|
|
30
|
+
run_gate "GATE 2 — named/specific battery · must HOLD" \
|
|
31
|
+
node scripts/prove.mjs --questions scripts/proof-questions.json --k 3 --out PROOF
|
|
32
|
+
run_gate "GATE 3 — Helix-context demo · target >=6/8" \
|
|
33
|
+
node scripts/prove.mjs --questions scripts/helix-scenario-questions.json --k 2 --out HELIX-DEMO-NOHELIX
|
|
34
|
+
|
|
35
|
+
if [ "$FAILED" -ne 0 ]; then
|
|
36
|
+
echo "== GATES FAILED — at least one gate missed its threshold. Read DESCRIBED-PROOF.md / PROOF.md / HELIX-DEMO-NOHELIX.md =="
|
|
37
|
+
exit 1
|
|
38
|
+
fi
|
|
39
|
+
echo "== GATES COMPLETE — all passed. Read DESCRIBED-PROOF.md, PROOF.md, HELIX-DEMO-NOHELIX.md =="
|