ruvnet-brain 4.0.1 → 4.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/.claude-plugin/marketplace.json +1 -0
  2. package/README.md +4 -4
  3. package/bin/install.mjs +303 -24
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  25. package/package.json +8 -22
  26. package/plugin/.claude-plugin/marketplace.json +1 -0
  27. package/plugin/.claude-plugin/plugin.json +2 -3
  28. package/plugin/.codex-plugin/plugin.json +1 -1
  29. package/plugin/commands/brain-console.md +2 -2
  30. package/plugin/commands/configure.md +3 -2
  31. package/plugin/commands/rvbc.md +4 -3
  32. package/plugin/commands/rvcb.md +2 -2
  33. package/plugin/commands/whats-new.md +6 -6
  34. package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
  35. package/plugin/hooks/hooks.json +1 -2
  36. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  37. package/plugin/mcp/server.mjs +90 -32
  38. package/plugin/scripts/detach.mjs +14 -0
  39. package/plugin/scripts/first-session-worker.mjs +38 -0
  40. package/plugin/scripts/ground-ruvnet.sh +16 -6
  41. package/plugin/scripts/hook-shim.mjs +34 -29
  42. package/plugin/scripts/learn-capture.sh +22 -3
  43. package/plugin/scripts/learn-flush.mjs +21 -4
  44. package/plugin/scripts/runtime-preferences.mjs +269 -0
  45. package/plugin/scripts/session-start-core.mjs +503 -0
  46. package/plugin/scripts/session-start.sh +3 -858
  47. package/plugin/scripts/whats-new.mjs +42 -0
  48. package/plugin/skills/brain-console/SKILL.md +4 -2
  49. package/plugin/skills/release-proof/SKILL.md +98 -0
  50. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  51. package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
  52. package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
  53. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
  54. package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
  55. package/plugin/skills/rvbc/SKILL.md +9 -6
  56. package/plugin/skills/whats-new/SKILL.md +4 -4
  57. package/scripts/adr-backfill.mjs +107 -0
  58. package/scripts/advocacy-outcomes.mjs +808 -0
  59. package/scripts/agentdb-context.mjs +216 -0
  60. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  61. package/scripts/ascii-drift.mjs +236 -0
  62. package/scripts/behavioral-l1-l4.mjs +210 -0
  63. package/scripts/brain-capability-check.mjs +72 -0
  64. package/scripts/brain-grade-groundtruth.mjs +100 -0
  65. package/scripts/brain-latency-50.mjs +227 -0
  66. package/scripts/brain-novice-50.mjs +189 -0
  67. package/scripts/brain-stamp.mjs +94 -0
  68. package/scripts/brain-state.mjs +212 -0
  69. package/scripts/build-bundle.mjs +531 -0
  70. package/scripts/build-concepts.mjs +132 -0
  71. package/scripts/build-l2.mjs +71 -0
  72. package/scripts/build-primer.mjs +73 -0
  73. package/scripts/build-symbols.mjs +68 -0
  74. package/scripts/calibrate-router.mjs +97 -0
  75. package/scripts/capability-audit.mjs +321 -0
  76. package/scripts/capability-registry.mjs +876 -0
  77. package/scripts/check-indexation.mjs +108 -0
  78. package/scripts/check-legibility.mjs +189 -0
  79. package/scripts/ci/build-fixture-kb.mjs +67 -0
  80. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  81. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  82. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  83. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  84. package/scripts/ci/stranger-scenario.mjs +228 -0
  85. package/scripts/ci/stranger-timeout.mjs +25 -0
  86. package/scripts/ci-verdict.mjs +29 -0
  87. package/scripts/claims-verify.mjs +710 -0
  88. package/scripts/clear-claude-tmp.sh +31 -0
  89. package/scripts/console-engine.mjs +434 -0
  90. package/scripts/console-engine.test.mjs +125 -0
  91. package/scripts/corpus-qa.mjs +250 -0
  92. package/scripts/correction-detect-embed.mjs +346 -0
  93. package/scripts/correction-detect-measure.mjs +270 -0
  94. package/scripts/correction-detect.mjs +686 -0
  95. package/scripts/count-chunks.mjs +54 -0
  96. package/scripts/described-questions.json +30 -0
  97. package/scripts/design-grade.mjs +58 -0
  98. package/scripts/dev-plugin-link.sh +105 -0
  99. package/scripts/distill-project.mjs +200 -0
  100. package/scripts/doc-currency.mjs +801 -0
  101. package/scripts/eval-brain.mjs +244 -0
  102. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  103. package/scripts/fix-workstream.mjs +291 -0
  104. package/scripts/full-hints.mjs +87 -0
  105. package/scripts/gate.sh +39 -0
  106. package/scripts/gates.mjs +146 -0
  107. package/scripts/gen-console-images.mjs +54 -0
  108. package/scripts/gen-images.mjs +47 -0
  109. package/scripts/git-clone-refresh.mjs +52 -0
  110. package/scripts/git-hooks/pre-push +126 -0
  111. package/scripts/goal-match.mjs +398 -0
  112. package/scripts/goldie-research.mjs +223 -0
  113. package/scripts/goldie-weekly.sh +67 -0
  114. package/scripts/health-repair.mjs +237 -0
  115. package/scripts/helix-scenario-questions.json +10 -0
  116. package/scripts/ingest-gists.mjs +230 -0
  117. package/scripts/ingest-meeting.mjs +115 -0
  118. package/scripts/ingest-repo.mjs +79 -0
  119. package/scripts/install-npx-witness.sh +49 -0
  120. package/scripts/issue-fix.mjs +558 -0
  121. package/scripts/issue-watch.mjs +276 -0
  122. package/scripts/issue4-close-note.md +31 -0
  123. package/scripts/key-canary.mjs +91 -0
  124. package/scripts/latency-to-surface.mjs +233 -0
  125. package/scripts/learning-enable.mjs +380 -0
  126. package/scripts/learning-replay.mjs +1570 -0
  127. package/scripts/learnings.mjs +62 -0
  128. package/scripts/lesson-gate.mjs +680 -0
  129. package/scripts/lesson-lifecycle.mjs +449 -0
  130. package/scripts/lesson-promote.mjs +262 -0
  131. package/scripts/lesson-ratify.mjs +98 -0
  132. package/scripts/lesson-seed.mjs +252 -0
  133. package/scripts/lesson-store.mjs +447 -0
  134. package/scripts/loop-checkpoint.mjs +86 -0
  135. package/scripts/memdb-health.sh +14 -0
  136. package/scripts/memory-doctor.mjs +326 -0
  137. package/scripts/model-catalog.mjs +79 -0
  138. package/scripts/nightly-controller.mjs +66 -0
  139. package/scripts/nightly-gists.sh +72 -0
  140. package/scripts/nightly-wrapper.sh +172 -0
  141. package/scripts/notify.sh +12 -0
  142. package/scripts/npx-witness.sh +56 -0
  143. package/scripts/onboarding-console.mjs +2922 -0
  144. package/scripts/private-fence.mjs +69 -0
  145. package/scripts/proactivity-metrics.mjs +118 -0
  146. package/scripts/proof-questions.json +56 -0
  147. package/scripts/protected-release-invocation.mjs +76 -0
  148. package/scripts/prove.mjs +95 -0
  149. package/scripts/proxy/claude-proxied.sh +57 -0
  150. package/scripts/proxy/proxy-revert.sh +59 -0
  151. package/scripts/proxy/proxy-up.sh +60 -0
  152. package/scripts/proxy/proxy-verify.mjs +142 -0
  153. package/scripts/publication-receipt.mjs +307 -0
  154. package/scripts/published-surface-probe.mjs +241 -0
  155. package/scripts/qe/card-lane-gate.mjs +162 -0
  156. package/scripts/qe/session-start-gate.mjs +229 -0
  157. package/scripts/qe/ux-suite.mjs +323 -0
  158. package/scripts/reconcile-project.mjs +0 -0
  159. package/scripts/record-lesson.mjs +113 -0
  160. package/scripts/refresh-model-catalog.mjs +99 -0
  161. package/scripts/release-authority.mjs +93 -0
  162. package/scripts/release-proof.mjs +9 -0
  163. package/scripts/release-vector.mjs +281 -0
  164. package/scripts/release.mjs +439 -0
  165. package/scripts/remedy-registry.mjs +247 -0
  166. package/scripts/rerank-cap-eval.mjs +265 -0
  167. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  168. package/scripts/route-cheap.mjs +20 -15
  169. package/scripts/router-utilization.mjs +182 -0
  170. package/scripts/routing-flywheel.mjs +596 -0
  171. package/scripts/rvf-generation.mjs +104 -0
  172. package/scripts/rvf-index-audit.mjs +138 -0
  173. package/scripts/self-update.mjs +296 -0
  174. package/scripts/selfcheck.mjs +7 -1
  175. package/scripts/sign-bundle.mjs +69 -0
  176. package/scripts/signal-watch.mjs +171 -0
  177. package/scripts/stabilization-receipt.mjs +108 -0
  178. package/scripts/stack-sync.mjs +469 -0
  179. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  180. package/scripts/stamp-sweep.mjs +144 -0
  181. package/scripts/status-honesty.mjs +102 -0
  182. package/scripts/sync-version.mjs +217 -0
  183. package/scripts/token-report.mjs +102 -0
  184. package/scripts/top100-benchmark.mjs +479 -0
  185. package/scripts/top100-corpus.mjs +112 -0
  186. package/scripts/top100-semantic-assertions.mjs +449 -0
  187. package/scripts/update-apply.mjs +9 -0
  188. package/scripts/upgrade-notice.mjs +14 -0
  189. package/scripts/verify-bundle.mjs +51 -0
  190. package/scripts/verify-channels.mjs +184 -0
  191. package/scripts/verify-model-catalog.mjs +104 -0
  192. package/scripts/verify-nightly-close-issue4.sh +31 -0
  193. package/scripts/version.mjs +40 -0
  194. package/scripts/wired-check.mjs +867 -0
  195. package/plugin/scripts/finalize-token-meter.mjs +0 -25
@@ -0,0 +1,247 @@
1
+ // remedy-registry.mjs — ONE object per recommendation id, owning detector ⇄ executor ⇄ inverse.
2
+ //
3
+ // WHY THIS EXISTS. Before this file, a recommendation's id, the code that ran it, and the code that
4
+ // reversed it lived in three different places that nothing forced to agree. All three drifted, and
5
+ // every drift was invisible until someone clicked the button:
6
+ //
7
+ // 1. `learning:enable-fleet` was constructed, validated, and offered — with NO executor at all.
8
+ // It fell through apply()'s if/else to `Unknown recommendation id`. The single most important
9
+ // recommendation in the product (ADR-027's North Star case) was a dead button.
10
+ // 2. `repair:memory-index` journalled `kind:'restore-memory-backup'`, and undo() had no branch for
11
+ // it. It hit the default arm and reported "nothing to undo (the change reverses itself
12
+ // automatically)" — while the recommendation had promised "restore the backup taken immediately
13
+ // before the repair." The undo did not exist. The promise was a lie.
14
+ // 3. `repair:memory-index` also satisfies `startsWith('repair:')`, so ONE reordering of an if/else
15
+ // chain silently routed a database repair into a global npm sync. That was caught by review,
16
+ // but only by review — nothing structural prevented it.
17
+ //
18
+ // The common shape: a chain of `if (id.startsWith(...))` cannot be audited, because the set of ids
19
+ // it handles is not a value anything can inspect. So it becomes a value here. Each Remedy owns its
20
+ // id, the executor as DATA (not a spawn), and a DECLARED inverse. `assertRegistryClosure()` then
21
+ // proves, in a test, that every id the builders can construct resolves to exactly one remedy with a
22
+ // real undo handler behind it — so a dead button fails CI instead of failing a user.
23
+ //
24
+ // PURITY: no I/O, no spawn, no fs — same discipline as console-engine.mjs (DDD context 4). A remedy
25
+ // RETURNS a description of what to run; onboarding-console.mjs is the only thing that runs it. That
26
+ // is what lets the closure test check every path without touching the machine.
27
+
28
+ // ── Undo kinds ───────────────────────────────────────────────────────────────────────────────────
29
+ // The set of inverses the console can actually perform. A remedy may not name a kind outside this
30
+ // set, and every kind here MUST have a live branch in onboarding-console.undo(). Both directions are
31
+ // enforced by test, because a missing branch does not throw — it silently returns "nothing to undo",
32
+ // which is the most dangerous possible answer: it reads like success.
33
+ //
34
+ // NONE is a real, declared value, not an absence. "This genuinely has no inverse" and "nobody wrote
35
+ // one" must never look the same, which is exactly the bug that made #2 above invisible.
36
+ export const UNDO_KINDS = Object.freeze({
37
+ NONE: 'none',
38
+ REINSTALL_VERSION: 'reinstall-version',
39
+ RESTORE_BACKUP: 'restore-backup',
40
+ RESTORE_MEMORY_BACKUP: 'restore-memory-backup',
41
+ RESTORE_STORE_BACKUPS: 'restore-store-backups',
42
+ AUTO_REBUILD: 'auto-rebuild',
43
+ // distill-project.mjs's OWN `--restore` (see its header: a tested inverse, not a re-implementation
44
+ // of one). Distinct from RESTORE_MEMORY_BACKUP/RESTORE_STORE_BACKUPS because those restore backups
45
+ // *this server* located and named; this one hands the restore entirely to the same script that took
46
+ // the snapshot, which already knows where its own backups live.
47
+ RESTORE_PROJECT_DISTILL: 'restore-project-distill',
48
+ });
49
+ const K = UNDO_KINDS;
50
+
51
+ // Ids that are exact, reserved words. A parameterized matcher (`repair:<pkg>`) must never capture
52
+ // one of these — see the ambiguity throw in resolveRemedy().
53
+ const RESERVED = new Set(['repair:memory-index', 'purge:shadows']);
54
+
55
+ // ── The registry ─────────────────────────────────────────────────────────────────────────────────
56
+ // match(id) → params object if this remedy owns the id, else null.
57
+ // plan(p) → { script, args } — the executor, as data.
58
+ // inverse(p) → { kind, ...params } — journalled BEFORE the change is made.
59
+ export const REMEDIES = [
60
+ {
61
+ key: 'memory-index',
62
+ autoEligible: true,
63
+ summary: 'REINDEX a corrupt AgentDB store',
64
+ match: (id) => (id === 'repair:memory-index' ? {} : null),
65
+ plan: () => ({ script: 'scripts/health-repair.mjs', args: ['--repair-memory'] }),
66
+ // health-repair.mjs takes an sqlite `.backup` of the store immediately before REINDEX (never a
67
+ // cp — that silently truncates a live WAL database, a standing lesson proven by experiment).
68
+ // The inverse is restoring it. This is the branch whose absence made the promise a lie.
69
+ inverse: () => ({ kind: K.RESTORE_MEMORY_BACKUP }),
70
+ },
71
+ {
72
+ key: 'learning-flush',
73
+ summary: 'drain the capture queue into the learner',
74
+ match: (id) => (id === 'learning:flush' ? {} : null),
75
+ plan: () => ({ script: 'scripts/health-repair.mjs', args: ['--flush-learning'] }),
76
+ // Genuinely additive: it moves already-captured local events into the learner. Declared NONE on
77
+ // purpose, and the human string says what a user would actually do instead.
78
+ inverse: () => ({ kind: K.NONE, human: 'nothing to reverse — this only adds observations the learner already had queued; learned state can be reset separately' }),
79
+ },
80
+ {
81
+ key: 'learning-train',
82
+ summary: 'run one training cycle',
83
+ match: (id) => (id === 'learning:train' ? {} : null),
84
+ plan: () => ({ script: 'scripts/health-repair.mjs', args: ['--train-learning'] }),
85
+ inverse: () => ({ kind: K.NONE, human: 'nothing to reverse here — learned state is reset with `ruflo hooks intelligence --reset`, which is a separate, deliberate action' }),
86
+ },
87
+ {
88
+ // THE ONE THAT HAD NO EXECUTOR. See ADR-027's North Star case: stores full of memories that
89
+ // teach nothing. The remedy is not ours to invent — memory-doctor.mjs has printed the exact fix
90
+ // since the day it was written ("embedded but never distilled — run: ruflo memory distill run"),
91
+ // and the console simply never said it out loud. This wires that sentence to a button.
92
+ key: 'distill-fleet',
93
+ summary: 'distill embedded-but-never-distilled stores into reusable patterns',
94
+ match: (id) => (id === 'learning:distill-fleet' ? {} : null),
95
+ // needsReceipt: this remedy touches a SET of stores discovered at run time, so the inverse
96
+ // cannot be described up front. The executor writes down exactly which stores it snapshotted
97
+ // and where; the inverse reads that receipt. Without it, "restore the backups" would be a hope
98
+ // rather than an instruction — and a hope is what made the memory-index undo a lie.
99
+ plan: () => ({ script: 'scripts/health-repair.mjs', args: ['--distill-fleet'], needsReceipt: true }),
100
+ // Distillation WRITES (reasoning_patterns, episodes, causal_edges), so it needs a real inverse.
101
+ // health-repair snapshots each store with `ruflo memory backup` (rUv's own WAL-safe, rotated
102
+ // snapshotter) before distilling it; the inverse restores those snapshots.
103
+ inverse: () => ({ kind: K.RESTORE_STORE_BACKUPS }),
104
+ },
105
+ {
106
+ // THE CAPABILITY BRIDGE'S ONLY CURRENT MEMBER. buildCapabilityRecommendations() in
107
+ // console-engine.mjs offers `enable:memory-distillation` only while the capability is OFF; this
108
+ // is the executor behind it. See distill-project.mjs's header for why THIS script and not bare
109
+ // `ruflo memory distill run`: the wrapper snapshots first, fails closed on a receipt-write
110
+ // failure, and its `--restore` is the tested inverse (proven 644→648→644→648, 2026-07-24).
111
+ key: 'enable-memory-distillation',
112
+ autoEligible: true,
113
+ summary: "mine this project's stored memories into reusable patterns (snapshots first; reversible)",
114
+ match: (id) => (id === 'enable:memory-distillation' ? {} : null),
115
+ // This console instance is always scoped to ONE project — the directory it was started in — the
116
+ // same assumption the memory-index/learning-flush/learning-train remedies above already make.
117
+ // `usesServerProject` asks onboarding-console.mjs (impure, process-aware) to supply that directory
118
+ // at call time; this file stays pure and never reads process.cwd() itself (see header).
119
+ plan: () => ({ script: 'scripts/distill-project.mjs', args: [], usesServerProject: true }),
120
+ // `--restore` with no path argument uses distill-project.mjs's OWN newestSnapshot() lookup inside
121
+ // this project's `.swarm/backups` — the exact mechanism its header proves end to end. Re-deriving
122
+ // which snapshot to restore here, instead of asking the tool that took it, is the kind of
123
+ // duplicate implementation this project has already been burned by once (ADR-047's rejected
124
+ // "offered command and promised undo live on different execution paths" bug).
125
+ inverse: () => ({ kind: K.RESTORE_PROJECT_DISTILL }),
126
+ },
127
+ {
128
+ key: 'stack-sync',
129
+ summary: 'install/repair a global package to its target version',
130
+ match: (id) => {
131
+ if (RESERVED.has(id)) return null; // `repair:memory-index` is NOT a package repair
132
+ const m = /^(?:sync|repair):(.+)$/.exec(id);
133
+ return m ? { pkg: m[1] } : null;
134
+ },
135
+ plan: () => ({ script: 'scripts/stack-sync.mjs', args: ['--sync'] }),
136
+ // The inverse of a version bump is the version that was on disk a moment ago — which only the
137
+ // caller can read, so it is filled in at journal time. Declaring it here is what makes the
138
+ // closure test able to check that undo() can honour it.
139
+ inverse: ({ pkg }) => ({ kind: K.REINSTALL_VERSION, pkg }),
140
+ },
141
+ {
142
+ key: 'purge-shadows',
143
+ summary: 'delete stale duplicate copies from the npx cache',
144
+ match: (id) => (id === 'purge:shadows' ? {} : null),
145
+ plan: () => ({ script: 'scripts/stack-sync.mjs', args: ['--sync'] }),
146
+ inverse: () => ({ kind: K.AUTO_REBUILD, human: 'the temporary cache re-fills itself on next use; no manual step needed' }),
147
+ },
148
+ {
149
+ key: 'reconcile-project',
150
+ autoEligible: true,
151
+ summary: 'rewire a project from npx to the global binary',
152
+ match: (id) => {
153
+ const m = /^reconcile:(.+)$/.exec(id);
154
+ return m ? { project: m[1] } : null;
155
+ },
156
+ plan: ({ project }) => ({ script: 'scripts/reconcile-project.mjs', args: ['--apply', '--project', project], resolveProject: true }),
157
+ inverse: ({ project }) => ({ kind: K.RESTORE_BACKUP, project }),
158
+ },
159
+ ];
160
+
161
+ /**
162
+ * Resolve an id to EXACTLY ONE remedy.
163
+ *
164
+ * Ambiguity throws rather than picking a winner. Silently preferring the first match is precisely
165
+ * how `repair:memory-index` once routed into a global npm sync while telling the user their database
166
+ * had been repaired — a wrong action reported as the right one. A throw is a loud developer error;
167
+ * a silent misroute is a user's data.
168
+ */
169
+ export function resolveRemedy(id) {
170
+ const hits = [];
171
+ for (const r of REMEDIES) {
172
+ const params = r.match(id);
173
+ if (params) hits.push({ remedy: r, params });
174
+ }
175
+ if (hits.length > 1) {
176
+ throw new Error(`Remedy id "${id}" is ambiguous — claimed by: ${hits.map((h) => h.remedy.key).join(', ')}. Exactly one remedy must own an id.`);
177
+ }
178
+ return hits[0] ?? null;
179
+ }
180
+
181
+ /** The full plan for an id: what to run, and what reverses it. Null when nothing owns the id. */
182
+ export function planFor(id) {
183
+ const hit = resolveRemedy(id);
184
+ if (!hit) return null;
185
+ const { remedy, params } = hit;
186
+ return {
187
+ key: remedy.key,
188
+ summary: remedy.summary,
189
+ autoEligible: remedy.autoEligible === true,
190
+ exec: remedy.plan(params),
191
+ undo: remedy.inverse(params),
192
+ params,
193
+ };
194
+ }
195
+
196
+ /**
197
+ * THE CLOSURE PROOF. Every id that can be OFFERED must be runnable and reversible.
198
+ *
199
+ * Takes the ids the recommendation builders actually constructed (never a hand-typed list — a
200
+ * hand-typed list drifts from the builders, which is the whole failure mode) plus the undo kinds
201
+ * onboarding-console.undo() implements, and returns every gap it finds.
202
+ *
203
+ * @param {string[]} offeredIds ids from buildHealth/Stack/WiringRecommendations
204
+ * @param {string[]} handledUndoKinds kinds undo() has a real branch for
205
+ * @returns {{orphanIds:string[], ambiguousIds:string[], unhandledUndoKinds:string[], deadKinds:string[]}}
206
+ */
207
+ export function assertRegistryClosure(offeredIds = [], handledUndoKinds = []) {
208
+ const orphanIds = [];
209
+ const ambiguousIds = [];
210
+ const unhandledUndoKinds = [];
211
+ const handled = new Set(handledUndoKinds);
212
+ const usedKinds = new Set();
213
+
214
+ for (const id of offeredIds) {
215
+ let plan = null;
216
+ try { plan = planFor(id); } catch { ambiguousIds.push(id); continue; }
217
+ if (!plan) { orphanIds.push(id); continue; } // offered with no executor — the dead-button bug
218
+ usedKinds.add(plan.undo.kind);
219
+ // NONE is self-handling by definition, but it must still be DECLARED (see UNDO_KINDS).
220
+ if (plan.undo.kind !== K.NONE && !handled.has(plan.undo.kind)) unhandledUndoKinds.push(`${id} → ${plan.undo.kind}`);
221
+ }
222
+
223
+ // The other direction: a kind the registry declares that undo() cannot perform is a broken promise
224
+ // waiting to happen, even if no builder currently emits that id.
225
+ const declared = new Set();
226
+ for (const r of REMEDIES) {
227
+ try { declared.add(r.inverse(r.match(sampleIdFor(r)) || {}).kind); } catch { /* sampling is best-effort */ }
228
+ }
229
+ const deadKinds = [...declared].filter((k) => k !== K.NONE && !handled.has(k));
230
+
231
+ return { orphanIds, ambiguousIds, unhandledUndoKinds, deadKinds };
232
+ }
233
+
234
+ /** A representative id for each remedy, so closure can sample parameterized matchers too. */
235
+ export function sampleIdFor(remedy) {
236
+ switch (remedy.key) {
237
+ case 'memory-index': return 'repair:memory-index';
238
+ case 'learning-flush': return 'learning:flush';
239
+ case 'learning-train': return 'learning:train';
240
+ case 'distill-fleet': return 'learning:distill-fleet';
241
+ case 'enable-memory-distillation': return 'enable:memory-distillation';
242
+ case 'stack-sync': return 'sync:ruflo';
243
+ case 'purge-shadows': return 'purge:shadows';
244
+ case 'reconcile-project': return 'reconcile:example';
245
+ default: return `__unknown:${remedy.key}`;
246
+ }
247
+ }
@@ -0,0 +1,265 @@
1
+ #!/usr/bin/env node
2
+ // rerank-cap-eval.mjs — does bounding the cross-encoder pool change the ANSWERS?
3
+ //
4
+ // The cross-encoder is 84.7% of a query's wall and reads 605 (query, passage) pairs for an
5
+ // all-repos question. Capping that pool is the only lever that moves the number. But a cap
6
+ // re-orders results, and a naive one was MEASURED to lose the right answer — so no cap ships
7
+ // without a before/after on a real question set.
8
+ //
9
+ // TWO PHASES, because the measurement is expensive and the policy search is not:
10
+ //
11
+ // --collect Runs the FROZEN held-out set (evals/held-out.json — the same corpus
12
+ // scripts/eval-brain.mjs gates on) UNCAPPED, recording every scored candidate:
13
+ // repo, path, lane, within-lane depth, pool position, and its cross-encoder score.
14
+ // One ~8-minute query per question. This is the whole cost of the experiment.
15
+ //
16
+ // --report Replays capRerankPool + selectResults — the SHIPPING functions, imported, not
17
+ // re-implemented — against those recorded scores, for every candidate budget. The
18
+ // cross-encoder is deterministic per (query, passage) pair and pairs are scored
19
+ // independently, so a replay of a subset is EXACT: it is the same arithmetic on the
20
+ // same numbers, not a model of it.
21
+ //
22
+ // Reported metrics are the ones a wrong answer would move, graded by ground truth (never a model
23
+ // judge — an LLM panel once scored a zero-citation answer 98/100 on this repo):
24
+ // pairs — cross-encoder pairs actually scored. The load-independent primary evidence.
25
+ // top1-same — the winning document is byte-identical to the uncapped winner.
26
+ // kept — the uncapped top-k documents still present in the capped top-k.
27
+ // routed — top-1 lands in an expected repo (eval-brain's own metric, same Wilson bound).
28
+ // abstain — adversarial questions still decline (top ce < 0).
29
+ // banner — a winning gist chunk still carries its provenance banner.
30
+ //
31
+ // node scripts/rerank-cap-eval.mjs --collect [--conc 3] [--only ho-01,ho-02]
32
+ // node scripts/rerank-cap-eval.mjs --report [--budgets 605,272,136,69]
33
+
34
+ import fs from 'node:fs';
35
+ import os from 'node:os';
36
+ import path from 'node:path';
37
+ import { fileURLToPath, pathToFileURL } from 'node:url';
38
+ import { execFile } from 'node:child_process';
39
+ import { gradeQuestion, aggregate } from './eval-brain.mjs';
40
+
41
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
42
+ const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
43
+ // The code under test is THIS checkout's, run against whatever corpus KB points at.
44
+ const CODE = path.join(ROOT, 'kb');
45
+ const argv = process.argv.slice(2);
46
+ const arg = (f, d) => { const i = argv.indexOf(f); return i >= 0 && argv[i + 1] ? argv[i + 1] : d; };
47
+ const TRACES = arg('--traces', path.join(os.tmpdir(), 'ruvnet-brain-ce-cap-traces'));
48
+
49
+ // The frozen 120, plus probes for the paths a cap could silently break. These are NOT scored
50
+ // against expectRepo (they are not part of the frozen set and must never contaminate it) — they
51
+ // are watched for one thing only: does the cap change the winner?
52
+ const PROBES = [
53
+ { id: 'px-mcp-policy', query: 'which file enforces the MCP tool policy in ruvector', why: 'the case a naive cap was measured to lose (mcp-policy.js -> an ADR)' },
54
+ { id: 'px-adr-085', query: 'ADR-085', why: 'bare ADR number — per-repo collision disclosure reads the whole pool' },
55
+ { id: 'px-pkg-rvf', query: 'what is @ruvector/rvf', why: 'exact @scope/name — exercises the RESCUE lane the cap must never drop' },
56
+ { id: 'px-meetings', query: 'how many commits and contributors does the project have', why: 'answer is BM25-only in a transcript store; dense buries it past rank 40' },
57
+ ];
58
+
59
+ // Probes first, then the frozen set DEALT ROUND-ROBIN ACROSS STRATA. The held-out file is grouped
60
+ // by stratum, and at ~8 minutes a question a run can be interrupted — grouped order would make any
61
+ // prefix a biased sample (all 'described', no 'adversarial'), which is the kind of partial evidence
62
+ // that reads as a result and is not one. Round-robin makes every prefix stratified.
63
+ function loadQuestions() {
64
+ const { questions } = JSON.parse(fs.readFileSync(path.join(ROOT, 'evals', 'held-out.json'), 'utf8'));
65
+ const byStratum = new Map();
66
+ for (const q of questions) (byStratum.get(q.stratum) ?? byStratum.set(q.stratum, []).get(q.stratum)).push(q);
67
+ const lanes = [...byStratum.values()];
68
+ const dealt = [];
69
+ for (let i = 0; dealt.length < questions.length; i++) for (const lane of lanes) if (lane[i]) dealt.push(lane[i]);
70
+ return [...PROBES.map((p) => ({ ...p, stratum: 'probe' })), ...dealt];
71
+ }
72
+
73
+ // ── collect ─────────────────────────────────────────────────────────────────────────────────────
74
+ async function collect() {
75
+ const only = arg('--only', '') ? new Set(arg('--only', '').split(',')) : null;
76
+ const questions = loadQuestions().filter((q) => !only || only.has(q.id));
77
+ const CONC = Math.max(1, parseInt(arg('--conc', '3'), 10) || 3);
78
+ fs.mkdirSync(TRACES, { recursive: true });
79
+
80
+ const run = (q) => new Promise((resolve) => {
81
+ const trace = path.join(TRACES, `${q.id}.jsonl`);
82
+ // A partial trace from a killed run must never be read as a complete one: write to .part and
83
+ // rename only on a clean exit, so --report can never score a truncated pool as a real pool.
84
+ const part = `${trace}.part`;
85
+ try { fs.rmSync(part, { force: true }); } catch { /* fresh anyway */ }
86
+ const t0 = Date.now();
87
+ execFile('node', ['forge-ask-all.mjs', '--dir', KB, '--q', q.query, '--k', '3'], {
88
+ cwd: CODE,
89
+ timeout: 1800000,
90
+ maxBuffer: 256 * 1024 * 1024,
91
+ // --cap 0 (the default) is the uncapped baseline. A NON-zero --cap re-collects the same
92
+ // questions with the cap actually engaged, which is the only way to check the replay against
93
+ // reality: capping changes which pairs share a batch, and batch composition was MEASURED to
94
+ // move scores by up to 0.26 logits, so a replayed cap is an approximation of a real one.
95
+ // --cascade K does the same for ADR-058's two-stage cascade. Both are collected FOR REAL for
96
+ // exactly that reason: the cascade re-batches its survivors too, so its scores are its own.
97
+ env: {
98
+ ...process.env,
99
+ KB_CE_TRACE: part,
100
+ KB_CE_MAX_PAIRS: arg('--cap', '0'),
101
+ KB_CE_CASCADE_K: arg('--cascade', '0'),
102
+ KB_CE_CASCADE_TOKENS: arg('--tokens', '192'),
103
+ },
104
+ }, (err) => {
105
+ const ms = Date.now() - t0;
106
+ if (!err && fs.existsSync(part) && fs.statSync(part).size > 0) {
107
+ const rec = JSON.parse(fs.readFileSync(part, 'utf8').trim().split('\n')[0]);
108
+ fs.writeFileSync(trace, JSON.stringify({ id: q.id, stratum: q.stratum, expectRepo: q.expectRepo ?? null, ms, ...rec }) + '\n');
109
+ fs.rmSync(part, { force: true });
110
+ resolve({ id: q.id, ok: true, ms, pooled: rec.pooledAll });
111
+ } else {
112
+ try { fs.rmSync(part, { force: true }); } catch { /* nothing to clean */ }
113
+ resolve({ id: q.id, ok: false, ms, err: String(err?.message || 'no trace written').slice(0, 120) });
114
+ }
115
+ });
116
+ });
117
+
118
+ let cursor = 0, done = 0;
119
+ const t0 = Date.now();
120
+ const worker = async () => {
121
+ while (cursor < questions.length) {
122
+ const q = questions[cursor++];
123
+ if (fs.existsSync(path.join(TRACES, `${q.id}.jsonl`))) { done++; continue; } // resumable: an 8h run must survive a restart
124
+ const r = await run(q);
125
+ done++;
126
+ const la = os.loadavg()[0].toFixed(0);
127
+ process.stderr.write(`[cap-eval] ${done}/${questions.length} ${r.ok ? 'ok' : 'FAIL'} ${r.id} ${(r.ms / 1000).toFixed(0)}s pooled=${r.pooled ?? '-'} load=${la} elapsed=${((Date.now() - t0) / 60000).toFixed(0)}m${r.ok ? '' : ` :: ${r.err}`}\n`);
128
+ }
129
+ };
130
+ await Promise.all(Array.from({ length: Math.min(CONC, questions.length) }, worker));
131
+ console.error(`[cap-eval] traces in ${TRACES}`);
132
+ }
133
+
134
+ // ── report ──────────────────────────────────────────────────────────────────────────────────────
135
+ async function report() {
136
+ const { capRerankPool, selectResults } = await import(pathToFileURL(path.join(CODE, 'forge-ask-all.mjs')).href);
137
+ if (!fs.existsSync(TRACES)) { console.error(`no traces at ${TRACES} — run --collect first`); process.exit(2); }
138
+ const files = fs.readdirSync(TRACES).filter((f) => f.endsWith('.jsonl'));
139
+ if (!files.length) { console.error(`no traces at ${TRACES} — run --collect first`); process.exit(2); }
140
+ const traces = files.map((f) => JSON.parse(fs.readFileSync(path.join(TRACES, f), 'utf8').trim().split('\n')[0]));
141
+
142
+ const budgets = (arg('--budgets', '') ? arg('--budgets', '').split(',').map(Number)
143
+ : [0, 408, 272, 204, 170, 136, 102, 69, 48, 24]).sort((a, b) => b - a);
144
+
145
+ // Rebuild each trace's pool in its ORIGINAL fan-out order — the cap's tie-break depends on it.
146
+ const pools = traces.map((t) => ({
147
+ ...t,
148
+ pool: [...t.cands].sort((a, b) => a.poolIdx - b.poolIdx)
149
+ .map((c) => ({ ...c, ceScore: c.ce, fullText: '', text: '', _lane: c.lane, _srcRank: c.rank, _poolIdx: c.poolIdx })),
150
+ }));
151
+
152
+ // The exact-package boost reads a candidate's BODY, which a trace deliberately does not carry
153
+ // (605 whole documents x 120 questions). It fires only on an `@scope/name` query, and it fires
154
+ // identically on any candidate that survives the cap, so it cannot flip a comparison between two
155
+ // survivors — but say so out loud rather than let a reader assume the replay is total.
156
+ const bodyDependent = pools.filter((p) => /@[a-z0-9][a-z0-9._-]*\/[a-z0-9._-]+/i.test(p.query)).map((p) => p.id);
157
+
158
+ const idOf = (r) => `${r.repo}/${r.path}`;
159
+
160
+ // CASCADE — the alternative policy, kept because it was measured and lost. Score every store's
161
+ // best passage first (~1 pair per store), then spend the rest of the budget only on the R stores
162
+ // whose best passage scored highest. It is the obvious "let the cross-encoder decide where to
163
+ // dig" design; the table below is the reason it is not what ships.
164
+ const cascadeKeep = (pool, R) => {
165
+ const best = new Map();
166
+ for (const c of pool) if (c._srcRank === 0 && (!best.has(c.repo) || c.ceScore > best.get(c.repo))) best.set(c.repo, c.ceScore);
167
+ const top = new Set([...best.entries()].sort((a, b) => b[1] - a[1]).slice(0, R).map(([r]) => r));
168
+ return pool.filter((c) => c._srcRank === 0 || c._lane === 'rescue' || top.has(c.repo));
169
+ };
170
+
171
+ const runKept = (p, kept) => {
172
+ const ranked = [...kept].sort((a, b) => (b.ceScore ?? -Infinity) - (a.ceScore ?? -Infinity));
173
+ const { results } = selectResults({ query: p.query, ranked, k: p.k });
174
+ return { pairs: kept.length, results };
175
+ };
176
+ const runPolicy = (p, limit) => runKept(p, capRerankPool(p.pool, { limit }).kept);
177
+
178
+ const base = new Map(pools.map((p) => [p.id, runPolicy(p, 0)]));
179
+
180
+ // ── ROOT CAUSE ────────────────────────────────────────────────────────────────────────────────
181
+ // How deep in its OWN store's vector ranking does the winning document sit? This one histogram
182
+ // decides whether any pre-score cap can be safe. If winners clustered at depth 0, a tiny budget
183
+ // would be free. They do not: the cross-encoder routinely promotes a passage its own store
184
+ // ranked 4th or 7th, which means depth carries little information about who wins, which means
185
+ // every pair a depth cut removes is a real chance of removing the answer.
186
+ const depthTop1 = {}, depthTopK = {};
187
+ for (const p of pools) {
188
+ const r = base.get(p.id).results;
189
+ if (!r.length) continue;
190
+ depthTop1[r[0]._srcRank] = (depthTop1[r[0]._srcRank] || 0) + 1;
191
+ for (const x of r) depthTopK[x._srcRank] = (depthTopK[x._srcRank] || 0) + 1;
192
+ }
193
+ const hist = (h) => Object.keys(h).map(Number).sort((a, b) => a - b).map((d) => `depth ${d}: ${h[d]}`).join(' | ');
194
+
195
+ console.log(`\n# cross-encoder pool cap — quality vs pair count`);
196
+ console.log(`corpus: ${pools.length} questions (${pools.filter((p) => p.stratum !== 'probe').length} frozen held-out + ${pools.filter((p) => p.stratum === 'probe').length} probes)`);
197
+ console.log(`uncapped pool: min ${Math.min(...pools.map((p) => p.pooledAll))}, median ${median(pools.map((p) => p.pooledAll))}, max ${Math.max(...pools.map((p) => p.pooledAll))} pairs`);
198
+ console.log(`body-dependent boost not replayed for: ${bodyDependent.length ? bodyDependent.join(', ') : '(none — no @scope/name query in the set)'}`);
199
+ console.log(`\n## where the answer actually lives, in its own store's vector ranking`);
200
+ console.log(` winning document : ${hist(depthTop1)}`);
201
+ console.log(` every top-k hit : ${hist(depthTopK)}`);
202
+ console.log(` (a budget of B pairs across S stores reaches depth B/S. Winners spread across depth`);
203
+ console.log(` means a depth cut drops answers roughly in proportion to what it saves.)\n`);
204
+ console.log('| policy | pairs (median) | top1-same | top-k kept | routed k/n (Wilson lo) | abstain | banner |');
205
+ console.log('|---|---|---|---|---|---|---|');
206
+
207
+ // Every policy the run compares: the shipping depth cap at a range of budgets, plus cascade.
208
+ const policies = [
209
+ ...budgets.map((limit) => ({ label: limit === 0 ? 'uncapped' : `depth B=${limit}`, key: `depth-${limit}`, keep: (p) => capRerankPool(p.pool, { limit }).kept })),
210
+ ...[4, 8, 12, 16, 24, 32].map((R) => ({ label: `cascade R=${R}`, key: `cascade-${R}`, keep: (p) => cascadeKeep(p.pool, R) })),
211
+ ];
212
+
213
+ const detail = {};
214
+ for (const policy of policies) {
215
+ const rows = [];
216
+ let top1Same = 0, keptNum = 0, keptDen = 0;
217
+ const changed = [];
218
+ for (const p of pools) {
219
+ const cur = runKept(p, policy.keep(p));
220
+ const b = base.get(p.id);
221
+ const same = idOf2(cur.results[0]) === idOf2(b.results[0]);
222
+ if (same) top1Same++; else changed.push(`${p.id}: ${idOf2(b.results[0])} -> ${idOf2(cur.results[0])}`);
223
+ const curSet = new Set(cur.results.map(idOf));
224
+ for (const r of b.results) { keptDen++; if (curSet.has(idOf(r))) keptNum++; }
225
+ if (p.stratum !== 'probe') {
226
+ const top = cur.results[0] ?? null;
227
+ // Same grading rule as scripts/eval-brain.mjs, fed from the replayed result set. `grounded`
228
+ // is structurally true here: every candidate is a passage the retriever actually returned.
229
+ rows.push({
230
+ id: p.id, stratum: p.stratum,
231
+ ...gradeQuestion({ stratum: p.stratum, expectRepo: p.expectRepo },
232
+ { grounded: cur.results.length > 0,
233
+ citations: top ? [{ repo: top.repo, fullPath: idOf(top), ce: top.ceScore }] : [],
234
+ bannerPresent: cur.results.some((r) => r.gist) }),
235
+ });
236
+ }
237
+ detail[policy.key] ??= {};
238
+ detail[policy.key][p.id] = { pairs: cur.pairs, top1: idOf2(cur.results[0]) };
239
+ }
240
+ const agg = aggregate(rows);
241
+ const pairs = median(pools.map((p) => policy.keep(p).length));
242
+ console.log(`| ${policy.label} | ${pairs} | ${top1Same}/${pools.length} (${pct(top1Same / pools.length)}) | ${keptNum}/${keptDen} (${pct(keptNum / keptDen)}) | ${agg.routed.k}/${agg.routed.n} (${pct(agg.routed.lo)}) | ${agg.abstain.k}/${agg.abstain.n} | ${agg.banner.k}/${agg.banner.n} |`);
243
+ if (changed.length && changed.length <= 12) detail[`changed@${policy.key}`] = changed;
244
+ }
245
+
246
+ console.log('\n## winners that changed, per policy (empty = the policy changed no answer)');
247
+ for (const policy of policies) {
248
+ if (policy.key === 'depth-0') continue;
249
+ const c = detail[`changed@${policy.key}`];
250
+ console.log(`\n### ${policy.label}`);
251
+ if (!c) console.log(' (too many to list — see the top1-same column)');
252
+ else if (!c.length) console.log(' none');
253
+ else for (const line of c) console.log(` ${line}`);
254
+ }
255
+
256
+ if (argv.includes('--json')) fs.writeFileSync(path.join(TRACES, 'report.json'), JSON.stringify(detail, null, 2));
257
+ }
258
+
259
+ const idOf2 = (r) => (r ? `${r.repo}/${r.path}` : '(no result)');
260
+ const pct = (x) => `${(x * 100).toFixed(1)}%`;
261
+ const median = (a) => { const s = [...a].sort((x, y) => x - y); return s.length % 2 ? s[(s.length - 1) / 2] : Math.round((s[s.length / 2 - 1] + s[s.length / 2]) / 2); };
262
+
263
+ if (argv.includes('--collect')) await collect();
264
+ else if (argv.includes('--report')) await report();
265
+ else { console.error('usage: rerank-cap-eval.mjs --collect | --report'); process.exit(2); }
@@ -0,0 +1,129 @@
1
+ #!/usr/bin/env node
2
+ // rerank-cap-warm-ab.mjs — the paired, WARM before/after for the cross-encoder pool cap.
3
+ //
4
+ // Why this exists rather than timing the CLI: a cold `forge-ask-all.mjs` spends ~53s loading two
5
+ // ONNX models before it scores anything, and the cap cannot touch that. Timing cold runs would
6
+ // dilute the effect being measured by roughly 3x and would also compare runs taken hours apart on a
7
+ // machine whose load moved underneath them. This harness loads the models ONCE and then runs each
8
+ // question twice in the same process — uncapped and capped — so the only difference between the two
9
+ // numbers is the thing under test.
10
+ //
11
+ // PAIRED AND ORDER-ALTERNATED: question i runs uncapped-then-capped on even i and capped-then-
12
+ // uncapped on odd i, so any residual warm-up or thermal drift cannot systematically favour one arm.
13
+ // Both arms' ANSWERS are recorded, not just their times: a cap that is fast and wrong is a failure,
14
+ // and this is the file that would catch it.
15
+ //
16
+ // TWO POLICIES SHARE THIS HARNESS, because a number is only comparable to another number taken
17
+ // the same way. --cap is ADR-057's flat pool cap (select by vector distance, one full read each).
18
+ // --cascade is ADR-058's two-stage cascade (read every pooled pair at a truncated length, then
19
+ // re-read the top K in full). Same questions, same pairing, same warm process, same table — so
20
+ // "-30.4%" and whatever the cascade measures can be put side by side honestly.
21
+ //
22
+ // node scripts/rerank-cap-warm-ab.mjs --cap 408 [--n 24] [--out result.json]
23
+ // node scripts/rerank-cap-warm-ab.mjs --cascade 64 [--n 24] [--tokens 192]
24
+
25
+ import fs from 'node:fs';
26
+ import os from 'node:os';
27
+ import path from 'node:path';
28
+ import { fileURLToPath, pathToFileURL } from 'node:url';
29
+
30
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
31
+ const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
32
+ const argv = process.argv.slice(2);
33
+ const arg = (f, d) => { const i = argv.indexOf(f); return i >= 0 && argv[i + 1] ? argv[i + 1] : d; };
34
+ // Exactly one arm is under test. --cascade wins if both are given, and says so rather than
35
+ // silently measuring a policy the operator did not ask for.
36
+ const CASCADE = arg('--cascade', '');
37
+ const CAP = arg('--cap', CASCADE ? '0' : '408');
38
+ const TOKENS = arg('--tokens', '192');
39
+ const MODE = CASCADE ? 'cascade' : 'cap';
40
+ const LABEL = CASCADE ? `cascade K=${CASCADE} @${TOKENS}tok` : `capped B=${CAP}`;
41
+ const N = parseInt(arg('--n', '24'), 10);
42
+ const OUT = arg('--out', path.join(os.tmpdir(), `ce-${MODE}-warm-ab-${CASCADE || CAP}.json`));
43
+
44
+ const { searchAll } = await import(pathToFileURL(path.join(ROOT, 'kb', 'forge-ask-all.mjs')).href);
45
+ const { gradeQuestion, aggregate } = await import(pathToFileURL(path.join(ROOT, 'scripts', 'eval-brain.mjs')).href);
46
+
47
+ // Stratified subset of the frozen held-out set, dealt round-robin so every stratum is represented
48
+ // even at small n — a prefix of a grouped file would be all 'described' and no 'adversarial'.
49
+ const { questions } = JSON.parse(fs.readFileSync(path.join(ROOT, 'evals', 'held-out.json'), 'utf8'));
50
+ const byStratum = new Map();
51
+ for (const q of questions) (byStratum.get(q.stratum) ?? byStratum.set(q.stratum, []).get(q.stratum)).push(q);
52
+ const lanes = [...byStratum.values()];
53
+ const dealt = [];
54
+ for (let i = 0; dealt.length < questions.length; i++) for (const lane of lanes) if (lane[i]) dealt.push(lane[i]);
55
+ const set = dealt.slice(0, N);
56
+
57
+ const idOf = (r) => (r ? `${r.repo}/${r.path}` : '(none)');
58
+ // `on` selects the arm under test; `off` is always the untouched uncapped, uncascaded path. Both
59
+ // env knobs are set on EVERY call rather than only when engaged — a leftover value from the
60
+ // previous call is exactly how an A/B measures the same arm twice and reports a 0% delta.
61
+ async function once(query, on) {
62
+ process.env.KB_CE_MAX_PAIRS = String(on && MODE === 'cap' ? CAP : 0);
63
+ process.env.KB_CE_CASCADE_K = String(on && MODE === 'cascade' ? CASCADE : 0);
64
+ process.env.KB_CE_CASCADE_TOKENS = String(TOKENS);
65
+ const t0 = Date.now();
66
+ const out = await searchAll({ dir: KB, query, k: 3 });
67
+ return {
68
+ ms: Date.now() - t0, pairs: out.pooled, pooledAll: out.pooledAll,
69
+ prefiltered: out.prefiltered ?? 0, prefilterMs: out.prefilterMs ?? 0,
70
+ results: out.results.map((r) => ({ id: idOf(r), repo: r.repo, ce: r.ceScore, gist: !!r.gist })),
71
+ };
72
+ }
73
+
74
+ // Warm-up: the first query of a process pays both model loads. It is thrown away deliberately —
75
+ // including it would credit the cap with a saving it did not produce.
76
+ process.stderr.write(`[warm-ab] mode=${MODE} (${LABEL}) — loading models (first query is discarded)...\n`);
77
+ const w0 = Date.now();
78
+ await once('what is ruvector', false);
79
+ process.stderr.write(`[warm-ab] warm after ${((Date.now() - w0) / 1000).toFixed(1)}s\n`);
80
+
81
+ const rows = [];
82
+ for (let i = 0; i < set.length; i++) {
83
+ const q = set[i];
84
+ const onFirst = i % 2 === 1;
85
+ const a = await once(q.query, onFirst);
86
+ const b = await once(q.query, !onFirst);
87
+ const [off, on] = onFirst ? [b, a] : [a, b];
88
+ rows.push({ id: q.id, stratum: q.stratum, expectRepo: q.expectRepo ?? null, onFirst, off, on });
89
+ process.stderr.write(`[warm-ab] ${i + 1}/${set.length} ${q.id} off=${(off.ms / 1000).toFixed(1)}s/${off.pairs}p on=${(on.ms / 1000).toFixed(1)}s/${on.pairs}p top1${idOf2(off) === idOf2(on) ? '=same' : ' CHANGED'}\n`);
90
+ }
91
+ function idOf2(x) { return x.results[0]?.id ?? '(none)'; }
92
+
93
+ fs.writeFileSync(OUT, JSON.stringify({ mode: MODE, cap: CAP, cascade: CASCADE, tokens: TOKENS, kb: KB, n: set.length, load: os.loadavg(), rows }, null, 2));
94
+
95
+ // ── the table ───────────────────────────────────────────────────────────────────────────────────
96
+ const med = (a) => { const s = [...a].sort((x, y) => x - y); return s.length % 2 ? s[(s.length - 1) / 2] : Math.round((s[s.length / 2 - 1] + s[s.length / 2]) / 2); };
97
+ const pct = (x) => `${(x * 100).toFixed(1)}%`;
98
+ const grade = (arm) => aggregate(rows.map((r) => {
99
+ const top = r[arm].results[0] ?? null;
100
+ return { stratum: r.stratum, ...gradeQuestion({ stratum: r.stratum, expectRepo: r.expectRepo },
101
+ { grounded: r[arm].results.length > 0,
102
+ citations: top ? [{ repo: top.repo, fullPath: top.id, ce: top.ce }] : [],
103
+ bannerPresent: r[arm].results.some((x) => x.gist) }) };
104
+ }));
105
+ const top1Same = rows.filter((r) => idOf2(r.off) === idOf2(r.on)).length;
106
+ let kn = 0, kd = 0;
107
+ for (const r of rows) { const s = new Set(r.on.results.map((x) => x.id)); for (const x of r.off.results) { kd++; if (s.has(x.id)) kn++; } }
108
+ const gOff = grade('off'), gOn = grade('on');
109
+
110
+ console.log(`\n# warm A/B — ${MODE === 'cascade' ? `cross-encoder CASCADE KB_CE_CASCADE_K=${CASCADE} KB_CE_CASCADE_TOKENS=${TOKENS}` : `cross-encoder pool cap KB_CE_MAX_PAIRS=${CAP}`}`);
111
+ console.log(`${rows.length} questions from the frozen held-out set, paired, order-alternated, one warm process. load1=${os.loadavg()[0].toFixed(1)} on ${os.cpus().length} cores.\n`);
112
+ console.log('| | full reads (median) | warm wall median | warm wall mean | routed | abstain | banner |');
113
+ console.log('|---|---|---|---|---|---|---|');
114
+ const fmt = (arm, g) => `| ${med(rows.map((r) => r[arm].pairs))} | ${(med(rows.map((r) => r[arm].ms)) / 1000).toFixed(2)}s | ${(rows.reduce((a, r) => a + r[arm].ms, 0) / rows.length / 1000).toFixed(2)}s | ${g.routed.k}/${g.routed.n} | ${g.abstain.k}/${g.abstain.n} | ${g.banner.k}/${g.banner.n} |`;
115
+ console.log(`| baseline, no policy (before) ${fmt('off', gOff)}`);
116
+ console.log(`| ${LABEL} (after) ${fmt('on', gOn)}`);
117
+ if (MODE === 'cascade') {
118
+ // Stage 1 is real work and must be visible, or the table reads as if 64 pairs were the whole cost.
119
+ console.log(`\nstage-1 prefilter: ${med(rows.map((r) => r.on.prefiltered))} pairs read at ${TOKENS} tokens, median ${(med(rows.map((r) => r.on.prefilterMs)) / 1000).toFixed(2)}s of the after-time above`);
120
+ }
121
+ const dMed = 1 - med(rows.map((r) => r.on.ms)) / med(rows.map((r) => r.off.ms));
122
+ console.log(`\nwall-time change (median, paired): ${dMed >= 0 ? '-' : '+'}${pct(Math.abs(dMed))}`);
123
+ console.log(`top-1 cited path identical : ${top1Same}/${rows.length} (${pct(top1Same / rows.length)})`);
124
+ console.log(`top-3 cited paths retained : ${kn}/${kd} (${pct(kn / kd)})`);
125
+ console.log('\n## every question whose top-1 changed');
126
+ const changed = rows.filter((r) => idOf2(r.off) !== idOf2(r.on));
127
+ if (!changed.length) console.log(' none');
128
+ for (const r of changed) console.log(` ${r.id} [${r.stratum}] expect=${(r.expectRepo || ['-']).join('|')}\n before: ${idOf2(r.off)} (ce ${r.off.results[0]?.ce?.toFixed(3)})\n after : ${idOf2(r.on)} (ce ${r.on.results[0]?.ce?.toFixed(3)})`);
129
+ console.log(`\nraw: ${OUT}`);