ruvnet-brain 4.0.1 → 4.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +303 -24
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/commands/whats-new.md +6 -6
- package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +90 -32
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +34 -29
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +503 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/scripts/whats-new.mjs +42 -0
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +98 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/plugin/skills/whats-new/SKILL.md +4 -4
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +531 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/fix-workstream.mjs +291 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +237 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +558 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +326 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +172 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2922 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/protected-release-invocation.mjs +76 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/publication-receipt.mjs +307 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-authority.mjs +93 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +439 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +296 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stabilization-receipt.mjs +108 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +867 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
// remedy-registry.mjs — ONE object per recommendation id, owning detector ⇄ executor ⇄ inverse.
|
|
2
|
+
//
|
|
3
|
+
// WHY THIS EXISTS. Before this file, a recommendation's id, the code that ran it, and the code that
|
|
4
|
+
// reversed it lived in three different places that nothing forced to agree. All three drifted, and
|
|
5
|
+
// every drift was invisible until someone clicked the button:
|
|
6
|
+
//
|
|
7
|
+
// 1. `learning:enable-fleet` was constructed, validated, and offered — with NO executor at all.
|
|
8
|
+
// It fell through apply()'s if/else to `Unknown recommendation id`. The single most important
|
|
9
|
+
// recommendation in the product (ADR-027's North Star case) was a dead button.
|
|
10
|
+
// 2. `repair:memory-index` journalled `kind:'restore-memory-backup'`, and undo() had no branch for
|
|
11
|
+
// it. It hit the default arm and reported "nothing to undo (the change reverses itself
|
|
12
|
+
// automatically)" — while the recommendation had promised "restore the backup taken immediately
|
|
13
|
+
// before the repair." The undo did not exist. The promise was a lie.
|
|
14
|
+
// 3. `repair:memory-index` also satisfies `startsWith('repair:')`, so ONE reordering of an if/else
|
|
15
|
+
// chain silently routed a database repair into a global npm sync. That was caught by review,
|
|
16
|
+
// but only by review — nothing structural prevented it.
|
|
17
|
+
//
|
|
18
|
+
// The common shape: a chain of `if (id.startsWith(...))` cannot be audited, because the set of ids
|
|
19
|
+
// it handles is not a value anything can inspect. So it becomes a value here. Each Remedy owns its
|
|
20
|
+
// id, the executor as DATA (not a spawn), and a DECLARED inverse. `assertRegistryClosure()` then
|
|
21
|
+
// proves, in a test, that every id the builders can construct resolves to exactly one remedy with a
|
|
22
|
+
// real undo handler behind it — so a dead button fails CI instead of failing a user.
|
|
23
|
+
//
|
|
24
|
+
// PURITY: no I/O, no spawn, no fs — same discipline as console-engine.mjs (DDD context 4). A remedy
|
|
25
|
+
// RETURNS a description of what to run; onboarding-console.mjs is the only thing that runs it. That
|
|
26
|
+
// is what lets the closure test check every path without touching the machine.
|
|
27
|
+
|
|
28
|
+
// ── Undo kinds ───────────────────────────────────────────────────────────────────────────────────
|
|
29
|
+
// The set of inverses the console can actually perform. A remedy may not name a kind outside this
|
|
30
|
+
// set, and every kind here MUST have a live branch in onboarding-console.undo(). Both directions are
|
|
31
|
+
// enforced by test, because a missing branch does not throw — it silently returns "nothing to undo",
|
|
32
|
+
// which is the most dangerous possible answer: it reads like success.
|
|
33
|
+
//
|
|
34
|
+
// NONE is a real, declared value, not an absence. "This genuinely has no inverse" and "nobody wrote
|
|
35
|
+
// one" must never look the same, which is exactly the bug that made #2 above invisible.
|
|
36
|
+
export const UNDO_KINDS = Object.freeze({
|
|
37
|
+
NONE: 'none',
|
|
38
|
+
REINSTALL_VERSION: 'reinstall-version',
|
|
39
|
+
RESTORE_BACKUP: 'restore-backup',
|
|
40
|
+
RESTORE_MEMORY_BACKUP: 'restore-memory-backup',
|
|
41
|
+
RESTORE_STORE_BACKUPS: 'restore-store-backups',
|
|
42
|
+
AUTO_REBUILD: 'auto-rebuild',
|
|
43
|
+
// distill-project.mjs's OWN `--restore` (see its header: a tested inverse, not a re-implementation
|
|
44
|
+
// of one). Distinct from RESTORE_MEMORY_BACKUP/RESTORE_STORE_BACKUPS because those restore backups
|
|
45
|
+
// *this server* located and named; this one hands the restore entirely to the same script that took
|
|
46
|
+
// the snapshot, which already knows where its own backups live.
|
|
47
|
+
RESTORE_PROJECT_DISTILL: 'restore-project-distill',
|
|
48
|
+
});
|
|
49
|
+
const K = UNDO_KINDS;
|
|
50
|
+
|
|
51
|
+
// Ids that are exact, reserved words. A parameterized matcher (`repair:<pkg>`) must never capture
|
|
52
|
+
// one of these — see the ambiguity throw in resolveRemedy().
|
|
53
|
+
const RESERVED = new Set(['repair:memory-index', 'purge:shadows']);
|
|
54
|
+
|
|
55
|
+
// ── The registry ─────────────────────────────────────────────────────────────────────────────────
|
|
56
|
+
// match(id) → params object if this remedy owns the id, else null.
|
|
57
|
+
// plan(p) → { script, args } — the executor, as data.
|
|
58
|
+
// inverse(p) → { kind, ...params } — journalled BEFORE the change is made.
|
|
59
|
+
export const REMEDIES = [
|
|
60
|
+
{
|
|
61
|
+
key: 'memory-index',
|
|
62
|
+
autoEligible: true,
|
|
63
|
+
summary: 'REINDEX a corrupt AgentDB store',
|
|
64
|
+
match: (id) => (id === 'repair:memory-index' ? {} : null),
|
|
65
|
+
plan: () => ({ script: 'scripts/health-repair.mjs', args: ['--repair-memory'] }),
|
|
66
|
+
// health-repair.mjs takes an sqlite `.backup` of the store immediately before REINDEX (never a
|
|
67
|
+
// cp — that silently truncates a live WAL database, a standing lesson proven by experiment).
|
|
68
|
+
// The inverse is restoring it. This is the branch whose absence made the promise a lie.
|
|
69
|
+
inverse: () => ({ kind: K.RESTORE_MEMORY_BACKUP }),
|
|
70
|
+
},
|
|
71
|
+
{
|
|
72
|
+
key: 'learning-flush',
|
|
73
|
+
summary: 'drain the capture queue into the learner',
|
|
74
|
+
match: (id) => (id === 'learning:flush' ? {} : null),
|
|
75
|
+
plan: () => ({ script: 'scripts/health-repair.mjs', args: ['--flush-learning'] }),
|
|
76
|
+
// Genuinely additive: it moves already-captured local events into the learner. Declared NONE on
|
|
77
|
+
// purpose, and the human string says what a user would actually do instead.
|
|
78
|
+
inverse: () => ({ kind: K.NONE, human: 'nothing to reverse — this only adds observations the learner already had queued; learned state can be reset separately' }),
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
key: 'learning-train',
|
|
82
|
+
summary: 'run one training cycle',
|
|
83
|
+
match: (id) => (id === 'learning:train' ? {} : null),
|
|
84
|
+
plan: () => ({ script: 'scripts/health-repair.mjs', args: ['--train-learning'] }),
|
|
85
|
+
inverse: () => ({ kind: K.NONE, human: 'nothing to reverse here — learned state is reset with `ruflo hooks intelligence --reset`, which is a separate, deliberate action' }),
|
|
86
|
+
},
|
|
87
|
+
{
|
|
88
|
+
// THE ONE THAT HAD NO EXECUTOR. See ADR-027's North Star case: stores full of memories that
|
|
89
|
+
// teach nothing. The remedy is not ours to invent — memory-doctor.mjs has printed the exact fix
|
|
90
|
+
// since the day it was written ("embedded but never distilled — run: ruflo memory distill run"),
|
|
91
|
+
// and the console simply never said it out loud. This wires that sentence to a button.
|
|
92
|
+
key: 'distill-fleet',
|
|
93
|
+
summary: 'distill embedded-but-never-distilled stores into reusable patterns',
|
|
94
|
+
match: (id) => (id === 'learning:distill-fleet' ? {} : null),
|
|
95
|
+
// needsReceipt: this remedy touches a SET of stores discovered at run time, so the inverse
|
|
96
|
+
// cannot be described up front. The executor writes down exactly which stores it snapshotted
|
|
97
|
+
// and where; the inverse reads that receipt. Without it, "restore the backups" would be a hope
|
|
98
|
+
// rather than an instruction — and a hope is what made the memory-index undo a lie.
|
|
99
|
+
plan: () => ({ script: 'scripts/health-repair.mjs', args: ['--distill-fleet'], needsReceipt: true }),
|
|
100
|
+
// Distillation WRITES (reasoning_patterns, episodes, causal_edges), so it needs a real inverse.
|
|
101
|
+
// health-repair snapshots each store with `ruflo memory backup` (rUv's own WAL-safe, rotated
|
|
102
|
+
// snapshotter) before distilling it; the inverse restores those snapshots.
|
|
103
|
+
inverse: () => ({ kind: K.RESTORE_STORE_BACKUPS }),
|
|
104
|
+
},
|
|
105
|
+
{
|
|
106
|
+
// THE CAPABILITY BRIDGE'S ONLY CURRENT MEMBER. buildCapabilityRecommendations() in
|
|
107
|
+
// console-engine.mjs offers `enable:memory-distillation` only while the capability is OFF; this
|
|
108
|
+
// is the executor behind it. See distill-project.mjs's header for why THIS script and not bare
|
|
109
|
+
// `ruflo memory distill run`: the wrapper snapshots first, fails closed on a receipt-write
|
|
110
|
+
// failure, and its `--restore` is the tested inverse (proven 644→648→644→648, 2026-07-24).
|
|
111
|
+
key: 'enable-memory-distillation',
|
|
112
|
+
autoEligible: true,
|
|
113
|
+
summary: "mine this project's stored memories into reusable patterns (snapshots first; reversible)",
|
|
114
|
+
match: (id) => (id === 'enable:memory-distillation' ? {} : null),
|
|
115
|
+
// This console instance is always scoped to ONE project — the directory it was started in — the
|
|
116
|
+
// same assumption the memory-index/learning-flush/learning-train remedies above already make.
|
|
117
|
+
// `usesServerProject` asks onboarding-console.mjs (impure, process-aware) to supply that directory
|
|
118
|
+
// at call time; this file stays pure and never reads process.cwd() itself (see header).
|
|
119
|
+
plan: () => ({ script: 'scripts/distill-project.mjs', args: [], usesServerProject: true }),
|
|
120
|
+
// `--restore` with no path argument uses distill-project.mjs's OWN newestSnapshot() lookup inside
|
|
121
|
+
// this project's `.swarm/backups` — the exact mechanism its header proves end to end. Re-deriving
|
|
122
|
+
// which snapshot to restore here, instead of asking the tool that took it, is the kind of
|
|
123
|
+
// duplicate implementation this project has already been burned by once (ADR-047's rejected
|
|
124
|
+
// "offered command and promised undo live on different execution paths" bug).
|
|
125
|
+
inverse: () => ({ kind: K.RESTORE_PROJECT_DISTILL }),
|
|
126
|
+
},
|
|
127
|
+
{
|
|
128
|
+
key: 'stack-sync',
|
|
129
|
+
summary: 'install/repair a global package to its target version',
|
|
130
|
+
match: (id) => {
|
|
131
|
+
if (RESERVED.has(id)) return null; // `repair:memory-index` is NOT a package repair
|
|
132
|
+
const m = /^(?:sync|repair):(.+)$/.exec(id);
|
|
133
|
+
return m ? { pkg: m[1] } : null;
|
|
134
|
+
},
|
|
135
|
+
plan: () => ({ script: 'scripts/stack-sync.mjs', args: ['--sync'] }),
|
|
136
|
+
// The inverse of a version bump is the version that was on disk a moment ago — which only the
|
|
137
|
+
// caller can read, so it is filled in at journal time. Declaring it here is what makes the
|
|
138
|
+
// closure test able to check that undo() can honour it.
|
|
139
|
+
inverse: ({ pkg }) => ({ kind: K.REINSTALL_VERSION, pkg }),
|
|
140
|
+
},
|
|
141
|
+
{
|
|
142
|
+
key: 'purge-shadows',
|
|
143
|
+
summary: 'delete stale duplicate copies from the npx cache',
|
|
144
|
+
match: (id) => (id === 'purge:shadows' ? {} : null),
|
|
145
|
+
plan: () => ({ script: 'scripts/stack-sync.mjs', args: ['--sync'] }),
|
|
146
|
+
inverse: () => ({ kind: K.AUTO_REBUILD, human: 'the temporary cache re-fills itself on next use; no manual step needed' }),
|
|
147
|
+
},
|
|
148
|
+
{
|
|
149
|
+
key: 'reconcile-project',
|
|
150
|
+
autoEligible: true,
|
|
151
|
+
summary: 'rewire a project from npx to the global binary',
|
|
152
|
+
match: (id) => {
|
|
153
|
+
const m = /^reconcile:(.+)$/.exec(id);
|
|
154
|
+
return m ? { project: m[1] } : null;
|
|
155
|
+
},
|
|
156
|
+
plan: ({ project }) => ({ script: 'scripts/reconcile-project.mjs', args: ['--apply', '--project', project], resolveProject: true }),
|
|
157
|
+
inverse: ({ project }) => ({ kind: K.RESTORE_BACKUP, project }),
|
|
158
|
+
},
|
|
159
|
+
];
|
|
160
|
+
|
|
161
|
+
/**
|
|
162
|
+
* Resolve an id to EXACTLY ONE remedy.
|
|
163
|
+
*
|
|
164
|
+
* Ambiguity throws rather than picking a winner. Silently preferring the first match is precisely
|
|
165
|
+
* how `repair:memory-index` once routed into a global npm sync while telling the user their database
|
|
166
|
+
* had been repaired — a wrong action reported as the right one. A throw is a loud developer error;
|
|
167
|
+
* a silent misroute is a user's data.
|
|
168
|
+
*/
|
|
169
|
+
export function resolveRemedy(id) {
|
|
170
|
+
const hits = [];
|
|
171
|
+
for (const r of REMEDIES) {
|
|
172
|
+
const params = r.match(id);
|
|
173
|
+
if (params) hits.push({ remedy: r, params });
|
|
174
|
+
}
|
|
175
|
+
if (hits.length > 1) {
|
|
176
|
+
throw new Error(`Remedy id "${id}" is ambiguous — claimed by: ${hits.map((h) => h.remedy.key).join(', ')}. Exactly one remedy must own an id.`);
|
|
177
|
+
}
|
|
178
|
+
return hits[0] ?? null;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/** The full plan for an id: what to run, and what reverses it. Null when nothing owns the id. */
|
|
182
|
+
export function planFor(id) {
|
|
183
|
+
const hit = resolveRemedy(id);
|
|
184
|
+
if (!hit) return null;
|
|
185
|
+
const { remedy, params } = hit;
|
|
186
|
+
return {
|
|
187
|
+
key: remedy.key,
|
|
188
|
+
summary: remedy.summary,
|
|
189
|
+
autoEligible: remedy.autoEligible === true,
|
|
190
|
+
exec: remedy.plan(params),
|
|
191
|
+
undo: remedy.inverse(params),
|
|
192
|
+
params,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
/**
|
|
197
|
+
* THE CLOSURE PROOF. Every id that can be OFFERED must be runnable and reversible.
|
|
198
|
+
*
|
|
199
|
+
* Takes the ids the recommendation builders actually constructed (never a hand-typed list — a
|
|
200
|
+
* hand-typed list drifts from the builders, which is the whole failure mode) plus the undo kinds
|
|
201
|
+
* onboarding-console.undo() implements, and returns every gap it finds.
|
|
202
|
+
*
|
|
203
|
+
* @param {string[]} offeredIds ids from buildHealth/Stack/WiringRecommendations
|
|
204
|
+
* @param {string[]} handledUndoKinds kinds undo() has a real branch for
|
|
205
|
+
* @returns {{orphanIds:string[], ambiguousIds:string[], unhandledUndoKinds:string[], deadKinds:string[]}}
|
|
206
|
+
*/
|
|
207
|
+
export function assertRegistryClosure(offeredIds = [], handledUndoKinds = []) {
|
|
208
|
+
const orphanIds = [];
|
|
209
|
+
const ambiguousIds = [];
|
|
210
|
+
const unhandledUndoKinds = [];
|
|
211
|
+
const handled = new Set(handledUndoKinds);
|
|
212
|
+
const usedKinds = new Set();
|
|
213
|
+
|
|
214
|
+
for (const id of offeredIds) {
|
|
215
|
+
let plan = null;
|
|
216
|
+
try { plan = planFor(id); } catch { ambiguousIds.push(id); continue; }
|
|
217
|
+
if (!plan) { orphanIds.push(id); continue; } // offered with no executor — the dead-button bug
|
|
218
|
+
usedKinds.add(plan.undo.kind);
|
|
219
|
+
// NONE is self-handling by definition, but it must still be DECLARED (see UNDO_KINDS).
|
|
220
|
+
if (plan.undo.kind !== K.NONE && !handled.has(plan.undo.kind)) unhandledUndoKinds.push(`${id} → ${plan.undo.kind}`);
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
// The other direction: a kind the registry declares that undo() cannot perform is a broken promise
|
|
224
|
+
// waiting to happen, even if no builder currently emits that id.
|
|
225
|
+
const declared = new Set();
|
|
226
|
+
for (const r of REMEDIES) {
|
|
227
|
+
try { declared.add(r.inverse(r.match(sampleIdFor(r)) || {}).kind); } catch { /* sampling is best-effort */ }
|
|
228
|
+
}
|
|
229
|
+
const deadKinds = [...declared].filter((k) => k !== K.NONE && !handled.has(k));
|
|
230
|
+
|
|
231
|
+
return { orphanIds, ambiguousIds, unhandledUndoKinds, deadKinds };
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
/** A representative id for each remedy, so closure can sample parameterized matchers too. */
|
|
235
|
+
export function sampleIdFor(remedy) {
|
|
236
|
+
switch (remedy.key) {
|
|
237
|
+
case 'memory-index': return 'repair:memory-index';
|
|
238
|
+
case 'learning-flush': return 'learning:flush';
|
|
239
|
+
case 'learning-train': return 'learning:train';
|
|
240
|
+
case 'distill-fleet': return 'learning:distill-fleet';
|
|
241
|
+
case 'enable-memory-distillation': return 'enable:memory-distillation';
|
|
242
|
+
case 'stack-sync': return 'sync:ruflo';
|
|
243
|
+
case 'purge-shadows': return 'purge:shadows';
|
|
244
|
+
case 'reconcile-project': return 'reconcile:example';
|
|
245
|
+
default: return `__unknown:${remedy.key}`;
|
|
246
|
+
}
|
|
247
|
+
}
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// rerank-cap-eval.mjs — does bounding the cross-encoder pool change the ANSWERS?
|
|
3
|
+
//
|
|
4
|
+
// The cross-encoder is 84.7% of a query's wall and reads 605 (query, passage) pairs for an
|
|
5
|
+
// all-repos question. Capping that pool is the only lever that moves the number. But a cap
|
|
6
|
+
// re-orders results, and a naive one was MEASURED to lose the right answer — so no cap ships
|
|
7
|
+
// without a before/after on a real question set.
|
|
8
|
+
//
|
|
9
|
+
// TWO PHASES, because the measurement is expensive and the policy search is not:
|
|
10
|
+
//
|
|
11
|
+
// --collect Runs the FROZEN held-out set (evals/held-out.json — the same corpus
|
|
12
|
+
// scripts/eval-brain.mjs gates on) UNCAPPED, recording every scored candidate:
|
|
13
|
+
// repo, path, lane, within-lane depth, pool position, and its cross-encoder score.
|
|
14
|
+
// One ~8-minute query per question. This is the whole cost of the experiment.
|
|
15
|
+
//
|
|
16
|
+
// --report Replays capRerankPool + selectResults — the SHIPPING functions, imported, not
|
|
17
|
+
// re-implemented — against those recorded scores, for every candidate budget. The
|
|
18
|
+
// cross-encoder is deterministic per (query, passage) pair and pairs are scored
|
|
19
|
+
// independently, so a replay of a subset is EXACT: it is the same arithmetic on the
|
|
20
|
+
// same numbers, not a model of it.
|
|
21
|
+
//
|
|
22
|
+
// Reported metrics are the ones a wrong answer would move, graded by ground truth (never a model
|
|
23
|
+
// judge — an LLM panel once scored a zero-citation answer 98/100 on this repo):
|
|
24
|
+
// pairs — cross-encoder pairs actually scored. The load-independent primary evidence.
|
|
25
|
+
// top1-same — the winning document is byte-identical to the uncapped winner.
|
|
26
|
+
// kept — the uncapped top-k documents still present in the capped top-k.
|
|
27
|
+
// routed — top-1 lands in an expected repo (eval-brain's own metric, same Wilson bound).
|
|
28
|
+
// abstain — adversarial questions still decline (top ce < 0).
|
|
29
|
+
// banner — a winning gist chunk still carries its provenance banner.
|
|
30
|
+
//
|
|
31
|
+
// node scripts/rerank-cap-eval.mjs --collect [--conc 3] [--only ho-01,ho-02]
|
|
32
|
+
// node scripts/rerank-cap-eval.mjs --report [--budgets 605,272,136,69]
|
|
33
|
+
|
|
34
|
+
import fs from 'node:fs';
|
|
35
|
+
import os from 'node:os';
|
|
36
|
+
import path from 'node:path';
|
|
37
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
38
|
+
import { execFile } from 'node:child_process';
|
|
39
|
+
import { gradeQuestion, aggregate } from './eval-brain.mjs';
|
|
40
|
+
|
|
41
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
42
|
+
const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
|
|
43
|
+
// The code under test is THIS checkout's, run against whatever corpus KB points at.
|
|
44
|
+
const CODE = path.join(ROOT, 'kb');
|
|
45
|
+
const argv = process.argv.slice(2);
|
|
46
|
+
const arg = (f, d) => { const i = argv.indexOf(f); return i >= 0 && argv[i + 1] ? argv[i + 1] : d; };
|
|
47
|
+
const TRACES = arg('--traces', path.join(os.tmpdir(), 'ruvnet-brain-ce-cap-traces'));
|
|
48
|
+
|
|
49
|
+
// The frozen 120, plus probes for the paths a cap could silently break. These are NOT scored
|
|
50
|
+
// against expectRepo (they are not part of the frozen set and must never contaminate it) — they
|
|
51
|
+
// are watched for one thing only: does the cap change the winner?
|
|
52
|
+
const PROBES = [
|
|
53
|
+
{ id: 'px-mcp-policy', query: 'which file enforces the MCP tool policy in ruvector', why: 'the case a naive cap was measured to lose (mcp-policy.js -> an ADR)' },
|
|
54
|
+
{ id: 'px-adr-085', query: 'ADR-085', why: 'bare ADR number — per-repo collision disclosure reads the whole pool' },
|
|
55
|
+
{ id: 'px-pkg-rvf', query: 'what is @ruvector/rvf', why: 'exact @scope/name — exercises the RESCUE lane the cap must never drop' },
|
|
56
|
+
{ id: 'px-meetings', query: 'how many commits and contributors does the project have', why: 'answer is BM25-only in a transcript store; dense buries it past rank 40' },
|
|
57
|
+
];
|
|
58
|
+
|
|
59
|
+
// Probes first, then the frozen set DEALT ROUND-ROBIN ACROSS STRATA. The held-out file is grouped
|
|
60
|
+
// by stratum, and at ~8 minutes a question a run can be interrupted — grouped order would make any
|
|
61
|
+
// prefix a biased sample (all 'described', no 'adversarial'), which is the kind of partial evidence
|
|
62
|
+
// that reads as a result and is not one. Round-robin makes every prefix stratified.
|
|
63
|
+
function loadQuestions() {
|
|
64
|
+
const { questions } = JSON.parse(fs.readFileSync(path.join(ROOT, 'evals', 'held-out.json'), 'utf8'));
|
|
65
|
+
const byStratum = new Map();
|
|
66
|
+
for (const q of questions) (byStratum.get(q.stratum) ?? byStratum.set(q.stratum, []).get(q.stratum)).push(q);
|
|
67
|
+
const lanes = [...byStratum.values()];
|
|
68
|
+
const dealt = [];
|
|
69
|
+
for (let i = 0; dealt.length < questions.length; i++) for (const lane of lanes) if (lane[i]) dealt.push(lane[i]);
|
|
70
|
+
return [...PROBES.map((p) => ({ ...p, stratum: 'probe' })), ...dealt];
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// ── collect ─────────────────────────────────────────────────────────────────────────────────────
|
|
74
|
+
async function collect() {
|
|
75
|
+
const only = arg('--only', '') ? new Set(arg('--only', '').split(',')) : null;
|
|
76
|
+
const questions = loadQuestions().filter((q) => !only || only.has(q.id));
|
|
77
|
+
const CONC = Math.max(1, parseInt(arg('--conc', '3'), 10) || 3);
|
|
78
|
+
fs.mkdirSync(TRACES, { recursive: true });
|
|
79
|
+
|
|
80
|
+
const run = (q) => new Promise((resolve) => {
|
|
81
|
+
const trace = path.join(TRACES, `${q.id}.jsonl`);
|
|
82
|
+
// A partial trace from a killed run must never be read as a complete one: write to .part and
|
|
83
|
+
// rename only on a clean exit, so --report can never score a truncated pool as a real pool.
|
|
84
|
+
const part = `${trace}.part`;
|
|
85
|
+
try { fs.rmSync(part, { force: true }); } catch { /* fresh anyway */ }
|
|
86
|
+
const t0 = Date.now();
|
|
87
|
+
execFile('node', ['forge-ask-all.mjs', '--dir', KB, '--q', q.query, '--k', '3'], {
|
|
88
|
+
cwd: CODE,
|
|
89
|
+
timeout: 1800000,
|
|
90
|
+
maxBuffer: 256 * 1024 * 1024,
|
|
91
|
+
// --cap 0 (the default) is the uncapped baseline. A NON-zero --cap re-collects the same
|
|
92
|
+
// questions with the cap actually engaged, which is the only way to check the replay against
|
|
93
|
+
// reality: capping changes which pairs share a batch, and batch composition was MEASURED to
|
|
94
|
+
// move scores by up to 0.26 logits, so a replayed cap is an approximation of a real one.
|
|
95
|
+
// --cascade K does the same for ADR-058's two-stage cascade. Both are collected FOR REAL for
|
|
96
|
+
// exactly that reason: the cascade re-batches its survivors too, so its scores are its own.
|
|
97
|
+
env: {
|
|
98
|
+
...process.env,
|
|
99
|
+
KB_CE_TRACE: part,
|
|
100
|
+
KB_CE_MAX_PAIRS: arg('--cap', '0'),
|
|
101
|
+
KB_CE_CASCADE_K: arg('--cascade', '0'),
|
|
102
|
+
KB_CE_CASCADE_TOKENS: arg('--tokens', '192'),
|
|
103
|
+
},
|
|
104
|
+
}, (err) => {
|
|
105
|
+
const ms = Date.now() - t0;
|
|
106
|
+
if (!err && fs.existsSync(part) && fs.statSync(part).size > 0) {
|
|
107
|
+
const rec = JSON.parse(fs.readFileSync(part, 'utf8').trim().split('\n')[0]);
|
|
108
|
+
fs.writeFileSync(trace, JSON.stringify({ id: q.id, stratum: q.stratum, expectRepo: q.expectRepo ?? null, ms, ...rec }) + '\n');
|
|
109
|
+
fs.rmSync(part, { force: true });
|
|
110
|
+
resolve({ id: q.id, ok: true, ms, pooled: rec.pooledAll });
|
|
111
|
+
} else {
|
|
112
|
+
try { fs.rmSync(part, { force: true }); } catch { /* nothing to clean */ }
|
|
113
|
+
resolve({ id: q.id, ok: false, ms, err: String(err?.message || 'no trace written').slice(0, 120) });
|
|
114
|
+
}
|
|
115
|
+
});
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
let cursor = 0, done = 0;
|
|
119
|
+
const t0 = Date.now();
|
|
120
|
+
const worker = async () => {
|
|
121
|
+
while (cursor < questions.length) {
|
|
122
|
+
const q = questions[cursor++];
|
|
123
|
+
if (fs.existsSync(path.join(TRACES, `${q.id}.jsonl`))) { done++; continue; } // resumable: an 8h run must survive a restart
|
|
124
|
+
const r = await run(q);
|
|
125
|
+
done++;
|
|
126
|
+
const la = os.loadavg()[0].toFixed(0);
|
|
127
|
+
process.stderr.write(`[cap-eval] ${done}/${questions.length} ${r.ok ? 'ok' : 'FAIL'} ${r.id} ${(r.ms / 1000).toFixed(0)}s pooled=${r.pooled ?? '-'} load=${la} elapsed=${((Date.now() - t0) / 60000).toFixed(0)}m${r.ok ? '' : ` :: ${r.err}`}\n`);
|
|
128
|
+
}
|
|
129
|
+
};
|
|
130
|
+
await Promise.all(Array.from({ length: Math.min(CONC, questions.length) }, worker));
|
|
131
|
+
console.error(`[cap-eval] traces in ${TRACES}`);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// ── report ──────────────────────────────────────────────────────────────────────────────────────
|
|
135
|
+
async function report() {
|
|
136
|
+
const { capRerankPool, selectResults } = await import(pathToFileURL(path.join(CODE, 'forge-ask-all.mjs')).href);
|
|
137
|
+
if (!fs.existsSync(TRACES)) { console.error(`no traces at ${TRACES} — run --collect first`); process.exit(2); }
|
|
138
|
+
const files = fs.readdirSync(TRACES).filter((f) => f.endsWith('.jsonl'));
|
|
139
|
+
if (!files.length) { console.error(`no traces at ${TRACES} — run --collect first`); process.exit(2); }
|
|
140
|
+
const traces = files.map((f) => JSON.parse(fs.readFileSync(path.join(TRACES, f), 'utf8').trim().split('\n')[0]));
|
|
141
|
+
|
|
142
|
+
const budgets = (arg('--budgets', '') ? arg('--budgets', '').split(',').map(Number)
|
|
143
|
+
: [0, 408, 272, 204, 170, 136, 102, 69, 48, 24]).sort((a, b) => b - a);
|
|
144
|
+
|
|
145
|
+
// Rebuild each trace's pool in its ORIGINAL fan-out order — the cap's tie-break depends on it.
|
|
146
|
+
const pools = traces.map((t) => ({
|
|
147
|
+
...t,
|
|
148
|
+
pool: [...t.cands].sort((a, b) => a.poolIdx - b.poolIdx)
|
|
149
|
+
.map((c) => ({ ...c, ceScore: c.ce, fullText: '', text: '', _lane: c.lane, _srcRank: c.rank, _poolIdx: c.poolIdx })),
|
|
150
|
+
}));
|
|
151
|
+
|
|
152
|
+
// The exact-package boost reads a candidate's BODY, which a trace deliberately does not carry
|
|
153
|
+
// (605 whole documents x 120 questions). It fires only on an `@scope/name` query, and it fires
|
|
154
|
+
// identically on any candidate that survives the cap, so it cannot flip a comparison between two
|
|
155
|
+
// survivors — but say so out loud rather than let a reader assume the replay is total.
|
|
156
|
+
const bodyDependent = pools.filter((p) => /@[a-z0-9][a-z0-9._-]*\/[a-z0-9._-]+/i.test(p.query)).map((p) => p.id);
|
|
157
|
+
|
|
158
|
+
const idOf = (r) => `${r.repo}/${r.path}`;
|
|
159
|
+
|
|
160
|
+
// CASCADE — the alternative policy, kept because it was measured and lost. Score every store's
|
|
161
|
+
// best passage first (~1 pair per store), then spend the rest of the budget only on the R stores
|
|
162
|
+
// whose best passage scored highest. It is the obvious "let the cross-encoder decide where to
|
|
163
|
+
// dig" design; the table below is the reason it is not what ships.
|
|
164
|
+
const cascadeKeep = (pool, R) => {
|
|
165
|
+
const best = new Map();
|
|
166
|
+
for (const c of pool) if (c._srcRank === 0 && (!best.has(c.repo) || c.ceScore > best.get(c.repo))) best.set(c.repo, c.ceScore);
|
|
167
|
+
const top = new Set([...best.entries()].sort((a, b) => b[1] - a[1]).slice(0, R).map(([r]) => r));
|
|
168
|
+
return pool.filter((c) => c._srcRank === 0 || c._lane === 'rescue' || top.has(c.repo));
|
|
169
|
+
};
|
|
170
|
+
|
|
171
|
+
const runKept = (p, kept) => {
|
|
172
|
+
const ranked = [...kept].sort((a, b) => (b.ceScore ?? -Infinity) - (a.ceScore ?? -Infinity));
|
|
173
|
+
const { results } = selectResults({ query: p.query, ranked, k: p.k });
|
|
174
|
+
return { pairs: kept.length, results };
|
|
175
|
+
};
|
|
176
|
+
const runPolicy = (p, limit) => runKept(p, capRerankPool(p.pool, { limit }).kept);
|
|
177
|
+
|
|
178
|
+
const base = new Map(pools.map((p) => [p.id, runPolicy(p, 0)]));
|
|
179
|
+
|
|
180
|
+
// ── ROOT CAUSE ────────────────────────────────────────────────────────────────────────────────
|
|
181
|
+
// How deep in its OWN store's vector ranking does the winning document sit? This one histogram
|
|
182
|
+
// decides whether any pre-score cap can be safe. If winners clustered at depth 0, a tiny budget
|
|
183
|
+
// would be free. They do not: the cross-encoder routinely promotes a passage its own store
|
|
184
|
+
// ranked 4th or 7th, which means depth carries little information about who wins, which means
|
|
185
|
+
// every pair a depth cut removes is a real chance of removing the answer.
|
|
186
|
+
const depthTop1 = {}, depthTopK = {};
|
|
187
|
+
for (const p of pools) {
|
|
188
|
+
const r = base.get(p.id).results;
|
|
189
|
+
if (!r.length) continue;
|
|
190
|
+
depthTop1[r[0]._srcRank] = (depthTop1[r[0]._srcRank] || 0) + 1;
|
|
191
|
+
for (const x of r) depthTopK[x._srcRank] = (depthTopK[x._srcRank] || 0) + 1;
|
|
192
|
+
}
|
|
193
|
+
const hist = (h) => Object.keys(h).map(Number).sort((a, b) => a - b).map((d) => `depth ${d}: ${h[d]}`).join(' | ');
|
|
194
|
+
|
|
195
|
+
console.log(`\n# cross-encoder pool cap — quality vs pair count`);
|
|
196
|
+
console.log(`corpus: ${pools.length} questions (${pools.filter((p) => p.stratum !== 'probe').length} frozen held-out + ${pools.filter((p) => p.stratum === 'probe').length} probes)`);
|
|
197
|
+
console.log(`uncapped pool: min ${Math.min(...pools.map((p) => p.pooledAll))}, median ${median(pools.map((p) => p.pooledAll))}, max ${Math.max(...pools.map((p) => p.pooledAll))} pairs`);
|
|
198
|
+
console.log(`body-dependent boost not replayed for: ${bodyDependent.length ? bodyDependent.join(', ') : '(none — no @scope/name query in the set)'}`);
|
|
199
|
+
console.log(`\n## where the answer actually lives, in its own store's vector ranking`);
|
|
200
|
+
console.log(` winning document : ${hist(depthTop1)}`);
|
|
201
|
+
console.log(` every top-k hit : ${hist(depthTopK)}`);
|
|
202
|
+
console.log(` (a budget of B pairs across S stores reaches depth B/S. Winners spread across depth`);
|
|
203
|
+
console.log(` means a depth cut drops answers roughly in proportion to what it saves.)\n`);
|
|
204
|
+
console.log('| policy | pairs (median) | top1-same | top-k kept | routed k/n (Wilson lo) | abstain | banner |');
|
|
205
|
+
console.log('|---|---|---|---|---|---|---|');
|
|
206
|
+
|
|
207
|
+
// Every policy the run compares: the shipping depth cap at a range of budgets, plus cascade.
|
|
208
|
+
const policies = [
|
|
209
|
+
...budgets.map((limit) => ({ label: limit === 0 ? 'uncapped' : `depth B=${limit}`, key: `depth-${limit}`, keep: (p) => capRerankPool(p.pool, { limit }).kept })),
|
|
210
|
+
...[4, 8, 12, 16, 24, 32].map((R) => ({ label: `cascade R=${R}`, key: `cascade-${R}`, keep: (p) => cascadeKeep(p.pool, R) })),
|
|
211
|
+
];
|
|
212
|
+
|
|
213
|
+
const detail = {};
|
|
214
|
+
for (const policy of policies) {
|
|
215
|
+
const rows = [];
|
|
216
|
+
let top1Same = 0, keptNum = 0, keptDen = 0;
|
|
217
|
+
const changed = [];
|
|
218
|
+
for (const p of pools) {
|
|
219
|
+
const cur = runKept(p, policy.keep(p));
|
|
220
|
+
const b = base.get(p.id);
|
|
221
|
+
const same = idOf2(cur.results[0]) === idOf2(b.results[0]);
|
|
222
|
+
if (same) top1Same++; else changed.push(`${p.id}: ${idOf2(b.results[0])} -> ${idOf2(cur.results[0])}`);
|
|
223
|
+
const curSet = new Set(cur.results.map(idOf));
|
|
224
|
+
for (const r of b.results) { keptDen++; if (curSet.has(idOf(r))) keptNum++; }
|
|
225
|
+
if (p.stratum !== 'probe') {
|
|
226
|
+
const top = cur.results[0] ?? null;
|
|
227
|
+
// Same grading rule as scripts/eval-brain.mjs, fed from the replayed result set. `grounded`
|
|
228
|
+
// is structurally true here: every candidate is a passage the retriever actually returned.
|
|
229
|
+
rows.push({
|
|
230
|
+
id: p.id, stratum: p.stratum,
|
|
231
|
+
...gradeQuestion({ stratum: p.stratum, expectRepo: p.expectRepo },
|
|
232
|
+
{ grounded: cur.results.length > 0,
|
|
233
|
+
citations: top ? [{ repo: top.repo, fullPath: idOf(top), ce: top.ceScore }] : [],
|
|
234
|
+
bannerPresent: cur.results.some((r) => r.gist) }),
|
|
235
|
+
});
|
|
236
|
+
}
|
|
237
|
+
detail[policy.key] ??= {};
|
|
238
|
+
detail[policy.key][p.id] = { pairs: cur.pairs, top1: idOf2(cur.results[0]) };
|
|
239
|
+
}
|
|
240
|
+
const agg = aggregate(rows);
|
|
241
|
+
const pairs = median(pools.map((p) => policy.keep(p).length));
|
|
242
|
+
console.log(`| ${policy.label} | ${pairs} | ${top1Same}/${pools.length} (${pct(top1Same / pools.length)}) | ${keptNum}/${keptDen} (${pct(keptNum / keptDen)}) | ${agg.routed.k}/${agg.routed.n} (${pct(agg.routed.lo)}) | ${agg.abstain.k}/${agg.abstain.n} | ${agg.banner.k}/${agg.banner.n} |`);
|
|
243
|
+
if (changed.length && changed.length <= 12) detail[`changed@${policy.key}`] = changed;
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
console.log('\n## winners that changed, per policy (empty = the policy changed no answer)');
|
|
247
|
+
for (const policy of policies) {
|
|
248
|
+
if (policy.key === 'depth-0') continue;
|
|
249
|
+
const c = detail[`changed@${policy.key}`];
|
|
250
|
+
console.log(`\n### ${policy.label}`);
|
|
251
|
+
if (!c) console.log(' (too many to list — see the top1-same column)');
|
|
252
|
+
else if (!c.length) console.log(' none');
|
|
253
|
+
else for (const line of c) console.log(` ${line}`);
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
if (argv.includes('--json')) fs.writeFileSync(path.join(TRACES, 'report.json'), JSON.stringify(detail, null, 2));
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
const idOf2 = (r) => (r ? `${r.repo}/${r.path}` : '(no result)');
|
|
260
|
+
const pct = (x) => `${(x * 100).toFixed(1)}%`;
|
|
261
|
+
const median = (a) => { const s = [...a].sort((x, y) => x - y); return s.length % 2 ? s[(s.length - 1) / 2] : Math.round((s[s.length / 2 - 1] + s[s.length / 2]) / 2); };
|
|
262
|
+
|
|
263
|
+
if (argv.includes('--collect')) await collect();
|
|
264
|
+
else if (argv.includes('--report')) await report();
|
|
265
|
+
else { console.error('usage: rerank-cap-eval.mjs --collect | --report'); process.exit(2); }
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// rerank-cap-warm-ab.mjs — the paired, WARM before/after for the cross-encoder pool cap.
|
|
3
|
+
//
|
|
4
|
+
// Why this exists rather than timing the CLI: a cold `forge-ask-all.mjs` spends ~53s loading two
|
|
5
|
+
// ONNX models before it scores anything, and the cap cannot touch that. Timing cold runs would
|
|
6
|
+
// dilute the effect being measured by roughly 3x and would also compare runs taken hours apart on a
|
|
7
|
+
// machine whose load moved underneath them. This harness loads the models ONCE and then runs each
|
|
8
|
+
// question twice in the same process — uncapped and capped — so the only difference between the two
|
|
9
|
+
// numbers is the thing under test.
|
|
10
|
+
//
|
|
11
|
+
// PAIRED AND ORDER-ALTERNATED: question i runs uncapped-then-capped on even i and capped-then-
|
|
12
|
+
// uncapped on odd i, so any residual warm-up or thermal drift cannot systematically favour one arm.
|
|
13
|
+
// Both arms' ANSWERS are recorded, not just their times: a cap that is fast and wrong is a failure,
|
|
14
|
+
// and this is the file that would catch it.
|
|
15
|
+
//
|
|
16
|
+
// TWO POLICIES SHARE THIS HARNESS, because a number is only comparable to another number taken
|
|
17
|
+
// the same way. --cap is ADR-057's flat pool cap (select by vector distance, one full read each).
|
|
18
|
+
// --cascade is ADR-058's two-stage cascade (read every pooled pair at a truncated length, then
|
|
19
|
+
// re-read the top K in full). Same questions, same pairing, same warm process, same table — so
|
|
20
|
+
// "-30.4%" and whatever the cascade measures can be put side by side honestly.
|
|
21
|
+
//
|
|
22
|
+
// node scripts/rerank-cap-warm-ab.mjs --cap 408 [--n 24] [--out result.json]
|
|
23
|
+
// node scripts/rerank-cap-warm-ab.mjs --cascade 64 [--n 24] [--tokens 192]
|
|
24
|
+
|
|
25
|
+
import fs from 'node:fs';
|
|
26
|
+
import os from 'node:os';
|
|
27
|
+
import path from 'node:path';
|
|
28
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
29
|
+
|
|
30
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
31
|
+
const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
|
|
32
|
+
const argv = process.argv.slice(2);
|
|
33
|
+
const arg = (f, d) => { const i = argv.indexOf(f); return i >= 0 && argv[i + 1] ? argv[i + 1] : d; };
|
|
34
|
+
// Exactly one arm is under test. --cascade wins if both are given, and says so rather than
|
|
35
|
+
// silently measuring a policy the operator did not ask for.
|
|
36
|
+
const CASCADE = arg('--cascade', '');
|
|
37
|
+
const CAP = arg('--cap', CASCADE ? '0' : '408');
|
|
38
|
+
const TOKENS = arg('--tokens', '192');
|
|
39
|
+
const MODE = CASCADE ? 'cascade' : 'cap';
|
|
40
|
+
const LABEL = CASCADE ? `cascade K=${CASCADE} @${TOKENS}tok` : `capped B=${CAP}`;
|
|
41
|
+
const N = parseInt(arg('--n', '24'), 10);
|
|
42
|
+
const OUT = arg('--out', path.join(os.tmpdir(), `ce-${MODE}-warm-ab-${CASCADE || CAP}.json`));
|
|
43
|
+
|
|
44
|
+
const { searchAll } = await import(pathToFileURL(path.join(ROOT, 'kb', 'forge-ask-all.mjs')).href);
|
|
45
|
+
const { gradeQuestion, aggregate } = await import(pathToFileURL(path.join(ROOT, 'scripts', 'eval-brain.mjs')).href);
|
|
46
|
+
|
|
47
|
+
// Stratified subset of the frozen held-out set, dealt round-robin so every stratum is represented
|
|
48
|
+
// even at small n — a prefix of a grouped file would be all 'described' and no 'adversarial'.
|
|
49
|
+
const { questions } = JSON.parse(fs.readFileSync(path.join(ROOT, 'evals', 'held-out.json'), 'utf8'));
|
|
50
|
+
const byStratum = new Map();
|
|
51
|
+
for (const q of questions) (byStratum.get(q.stratum) ?? byStratum.set(q.stratum, []).get(q.stratum)).push(q);
|
|
52
|
+
const lanes = [...byStratum.values()];
|
|
53
|
+
const dealt = [];
|
|
54
|
+
for (let i = 0; dealt.length < questions.length; i++) for (const lane of lanes) if (lane[i]) dealt.push(lane[i]);
|
|
55
|
+
const set = dealt.slice(0, N);
|
|
56
|
+
|
|
57
|
+
const idOf = (r) => (r ? `${r.repo}/${r.path}` : '(none)');
|
|
58
|
+
// `on` selects the arm under test; `off` is always the untouched uncapped, uncascaded path. Both
|
|
59
|
+
// env knobs are set on EVERY call rather than only when engaged — a leftover value from the
|
|
60
|
+
// previous call is exactly how an A/B measures the same arm twice and reports a 0% delta.
|
|
61
|
+
async function once(query, on) {
|
|
62
|
+
process.env.KB_CE_MAX_PAIRS = String(on && MODE === 'cap' ? CAP : 0);
|
|
63
|
+
process.env.KB_CE_CASCADE_K = String(on && MODE === 'cascade' ? CASCADE : 0);
|
|
64
|
+
process.env.KB_CE_CASCADE_TOKENS = String(TOKENS);
|
|
65
|
+
const t0 = Date.now();
|
|
66
|
+
const out = await searchAll({ dir: KB, query, k: 3 });
|
|
67
|
+
return {
|
|
68
|
+
ms: Date.now() - t0, pairs: out.pooled, pooledAll: out.pooledAll,
|
|
69
|
+
prefiltered: out.prefiltered ?? 0, prefilterMs: out.prefilterMs ?? 0,
|
|
70
|
+
results: out.results.map((r) => ({ id: idOf(r), repo: r.repo, ce: r.ceScore, gist: !!r.gist })),
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// Warm-up: the first query of a process pays both model loads. It is thrown away deliberately —
|
|
75
|
+
// including it would credit the cap with a saving it did not produce.
|
|
76
|
+
process.stderr.write(`[warm-ab] mode=${MODE} (${LABEL}) — loading models (first query is discarded)...\n`);
|
|
77
|
+
const w0 = Date.now();
|
|
78
|
+
await once('what is ruvector', false);
|
|
79
|
+
process.stderr.write(`[warm-ab] warm after ${((Date.now() - w0) / 1000).toFixed(1)}s\n`);
|
|
80
|
+
|
|
81
|
+
const rows = [];
|
|
82
|
+
for (let i = 0; i < set.length; i++) {
|
|
83
|
+
const q = set[i];
|
|
84
|
+
const onFirst = i % 2 === 1;
|
|
85
|
+
const a = await once(q.query, onFirst);
|
|
86
|
+
const b = await once(q.query, !onFirst);
|
|
87
|
+
const [off, on] = onFirst ? [b, a] : [a, b];
|
|
88
|
+
rows.push({ id: q.id, stratum: q.stratum, expectRepo: q.expectRepo ?? null, onFirst, off, on });
|
|
89
|
+
process.stderr.write(`[warm-ab] ${i + 1}/${set.length} ${q.id} off=${(off.ms / 1000).toFixed(1)}s/${off.pairs}p on=${(on.ms / 1000).toFixed(1)}s/${on.pairs}p top1${idOf2(off) === idOf2(on) ? '=same' : ' CHANGED'}\n`);
|
|
90
|
+
}
|
|
91
|
+
function idOf2(x) { return x.results[0]?.id ?? '(none)'; }
|
|
92
|
+
|
|
93
|
+
fs.writeFileSync(OUT, JSON.stringify({ mode: MODE, cap: CAP, cascade: CASCADE, tokens: TOKENS, kb: KB, n: set.length, load: os.loadavg(), rows }, null, 2));
|
|
94
|
+
|
|
95
|
+
// ── the table ───────────────────────────────────────────────────────────────────────────────────
|
|
96
|
+
const med = (a) => { const s = [...a].sort((x, y) => x - y); return s.length % 2 ? s[(s.length - 1) / 2] : Math.round((s[s.length / 2 - 1] + s[s.length / 2]) / 2); };
|
|
97
|
+
const pct = (x) => `${(x * 100).toFixed(1)}%`;
|
|
98
|
+
const grade = (arm) => aggregate(rows.map((r) => {
|
|
99
|
+
const top = r[arm].results[0] ?? null;
|
|
100
|
+
return { stratum: r.stratum, ...gradeQuestion({ stratum: r.stratum, expectRepo: r.expectRepo },
|
|
101
|
+
{ grounded: r[arm].results.length > 0,
|
|
102
|
+
citations: top ? [{ repo: top.repo, fullPath: top.id, ce: top.ce }] : [],
|
|
103
|
+
bannerPresent: r[arm].results.some((x) => x.gist) }) };
|
|
104
|
+
}));
|
|
105
|
+
const top1Same = rows.filter((r) => idOf2(r.off) === idOf2(r.on)).length;
|
|
106
|
+
let kn = 0, kd = 0;
|
|
107
|
+
for (const r of rows) { const s = new Set(r.on.results.map((x) => x.id)); for (const x of r.off.results) { kd++; if (s.has(x.id)) kn++; } }
|
|
108
|
+
const gOff = grade('off'), gOn = grade('on');
|
|
109
|
+
|
|
110
|
+
console.log(`\n# warm A/B — ${MODE === 'cascade' ? `cross-encoder CASCADE KB_CE_CASCADE_K=${CASCADE} KB_CE_CASCADE_TOKENS=${TOKENS}` : `cross-encoder pool cap KB_CE_MAX_PAIRS=${CAP}`}`);
|
|
111
|
+
console.log(`${rows.length} questions from the frozen held-out set, paired, order-alternated, one warm process. load1=${os.loadavg()[0].toFixed(1)} on ${os.cpus().length} cores.\n`);
|
|
112
|
+
console.log('| | full reads (median) | warm wall median | warm wall mean | routed | abstain | banner |');
|
|
113
|
+
console.log('|---|---|---|---|---|---|---|');
|
|
114
|
+
const fmt = (arm, g) => `| ${med(rows.map((r) => r[arm].pairs))} | ${(med(rows.map((r) => r[arm].ms)) / 1000).toFixed(2)}s | ${(rows.reduce((a, r) => a + r[arm].ms, 0) / rows.length / 1000).toFixed(2)}s | ${g.routed.k}/${g.routed.n} | ${g.abstain.k}/${g.abstain.n} | ${g.banner.k}/${g.banner.n} |`;
|
|
115
|
+
console.log(`| baseline, no policy (before) ${fmt('off', gOff)}`);
|
|
116
|
+
console.log(`| ${LABEL} (after) ${fmt('on', gOn)}`);
|
|
117
|
+
if (MODE === 'cascade') {
|
|
118
|
+
// Stage 1 is real work and must be visible, or the table reads as if 64 pairs were the whole cost.
|
|
119
|
+
console.log(`\nstage-1 prefilter: ${med(rows.map((r) => r.on.prefiltered))} pairs read at ${TOKENS} tokens, median ${(med(rows.map((r) => r.on.prefilterMs)) / 1000).toFixed(2)}s of the after-time above`);
|
|
120
|
+
}
|
|
121
|
+
const dMed = 1 - med(rows.map((r) => r.on.ms)) / med(rows.map((r) => r.off.ms));
|
|
122
|
+
console.log(`\nwall-time change (median, paired): ${dMed >= 0 ? '-' : '+'}${pct(Math.abs(dMed))}`);
|
|
123
|
+
console.log(`top-1 cited path identical : ${top1Same}/${rows.length} (${pct(top1Same / rows.length)})`);
|
|
124
|
+
console.log(`top-3 cited paths retained : ${kn}/${kd} (${pct(kn / kd)})`);
|
|
125
|
+
console.log('\n## every question whose top-1 changed');
|
|
126
|
+
const changed = rows.filter((r) => idOf2(r.off) !== idOf2(r.on));
|
|
127
|
+
if (!changed.length) console.log(' none');
|
|
128
|
+
for (const r of changed) console.log(` ${r.id} [${r.stratum}] expect=${(r.expectRepo || ['-']).join('|')}\n before: ${idOf2(r.off)} (ce ${r.off.results[0]?.ce?.toFixed(3)})\n after : ${idOf2(r.on)} (ce ${r.on.results[0]?.ce?.toFixed(3)})`);
|
|
129
|
+
console.log(`\nraw: ${OUT}`);
|