ruvnet-brain 4.0.1 → 4.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/.claude-plugin/marketplace.json +1 -0
  2. package/README.md +4 -4
  3. package/bin/install.mjs +303 -24
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  25. package/package.json +8 -22
  26. package/plugin/.claude-plugin/marketplace.json +1 -0
  27. package/plugin/.claude-plugin/plugin.json +2 -3
  28. package/plugin/.codex-plugin/plugin.json +1 -1
  29. package/plugin/commands/brain-console.md +2 -2
  30. package/plugin/commands/configure.md +3 -2
  31. package/plugin/commands/rvbc.md +4 -3
  32. package/plugin/commands/rvcb.md +2 -2
  33. package/plugin/commands/whats-new.md +6 -6
  34. package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
  35. package/plugin/hooks/hooks.json +1 -2
  36. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  37. package/plugin/mcp/server.mjs +90 -32
  38. package/plugin/scripts/detach.mjs +14 -0
  39. package/plugin/scripts/first-session-worker.mjs +38 -0
  40. package/plugin/scripts/ground-ruvnet.sh +16 -6
  41. package/plugin/scripts/hook-shim.mjs +34 -29
  42. package/plugin/scripts/learn-capture.sh +22 -3
  43. package/plugin/scripts/learn-flush.mjs +21 -4
  44. package/plugin/scripts/runtime-preferences.mjs +269 -0
  45. package/plugin/scripts/session-start-core.mjs +503 -0
  46. package/plugin/scripts/session-start.sh +3 -858
  47. package/plugin/scripts/whats-new.mjs +42 -0
  48. package/plugin/skills/brain-console/SKILL.md +4 -2
  49. package/plugin/skills/release-proof/SKILL.md +98 -0
  50. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  51. package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
  52. package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
  53. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
  54. package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
  55. package/plugin/skills/rvbc/SKILL.md +9 -6
  56. package/plugin/skills/whats-new/SKILL.md +4 -4
  57. package/scripts/adr-backfill.mjs +107 -0
  58. package/scripts/advocacy-outcomes.mjs +808 -0
  59. package/scripts/agentdb-context.mjs +216 -0
  60. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  61. package/scripts/ascii-drift.mjs +236 -0
  62. package/scripts/behavioral-l1-l4.mjs +210 -0
  63. package/scripts/brain-capability-check.mjs +72 -0
  64. package/scripts/brain-grade-groundtruth.mjs +100 -0
  65. package/scripts/brain-latency-50.mjs +227 -0
  66. package/scripts/brain-novice-50.mjs +189 -0
  67. package/scripts/brain-stamp.mjs +94 -0
  68. package/scripts/brain-state.mjs +212 -0
  69. package/scripts/build-bundle.mjs +531 -0
  70. package/scripts/build-concepts.mjs +132 -0
  71. package/scripts/build-l2.mjs +71 -0
  72. package/scripts/build-primer.mjs +73 -0
  73. package/scripts/build-symbols.mjs +68 -0
  74. package/scripts/calibrate-router.mjs +97 -0
  75. package/scripts/capability-audit.mjs +321 -0
  76. package/scripts/capability-registry.mjs +876 -0
  77. package/scripts/check-indexation.mjs +108 -0
  78. package/scripts/check-legibility.mjs +189 -0
  79. package/scripts/ci/build-fixture-kb.mjs +67 -0
  80. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  81. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  82. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  83. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  84. package/scripts/ci/stranger-scenario.mjs +228 -0
  85. package/scripts/ci/stranger-timeout.mjs +25 -0
  86. package/scripts/ci-verdict.mjs +29 -0
  87. package/scripts/claims-verify.mjs +710 -0
  88. package/scripts/clear-claude-tmp.sh +31 -0
  89. package/scripts/console-engine.mjs +434 -0
  90. package/scripts/console-engine.test.mjs +125 -0
  91. package/scripts/corpus-qa.mjs +250 -0
  92. package/scripts/correction-detect-embed.mjs +346 -0
  93. package/scripts/correction-detect-measure.mjs +270 -0
  94. package/scripts/correction-detect.mjs +686 -0
  95. package/scripts/count-chunks.mjs +54 -0
  96. package/scripts/described-questions.json +30 -0
  97. package/scripts/design-grade.mjs +58 -0
  98. package/scripts/dev-plugin-link.sh +105 -0
  99. package/scripts/distill-project.mjs +200 -0
  100. package/scripts/doc-currency.mjs +801 -0
  101. package/scripts/eval-brain.mjs +244 -0
  102. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  103. package/scripts/fix-workstream.mjs +291 -0
  104. package/scripts/full-hints.mjs +87 -0
  105. package/scripts/gate.sh +39 -0
  106. package/scripts/gates.mjs +146 -0
  107. package/scripts/gen-console-images.mjs +54 -0
  108. package/scripts/gen-images.mjs +47 -0
  109. package/scripts/git-clone-refresh.mjs +52 -0
  110. package/scripts/git-hooks/pre-push +126 -0
  111. package/scripts/goal-match.mjs +398 -0
  112. package/scripts/goldie-research.mjs +223 -0
  113. package/scripts/goldie-weekly.sh +67 -0
  114. package/scripts/health-repair.mjs +237 -0
  115. package/scripts/helix-scenario-questions.json +10 -0
  116. package/scripts/ingest-gists.mjs +230 -0
  117. package/scripts/ingest-meeting.mjs +115 -0
  118. package/scripts/ingest-repo.mjs +79 -0
  119. package/scripts/install-npx-witness.sh +49 -0
  120. package/scripts/issue-fix.mjs +558 -0
  121. package/scripts/issue-watch.mjs +276 -0
  122. package/scripts/issue4-close-note.md +31 -0
  123. package/scripts/key-canary.mjs +91 -0
  124. package/scripts/latency-to-surface.mjs +233 -0
  125. package/scripts/learning-enable.mjs +380 -0
  126. package/scripts/learning-replay.mjs +1570 -0
  127. package/scripts/learnings.mjs +62 -0
  128. package/scripts/lesson-gate.mjs +680 -0
  129. package/scripts/lesson-lifecycle.mjs +449 -0
  130. package/scripts/lesson-promote.mjs +262 -0
  131. package/scripts/lesson-ratify.mjs +98 -0
  132. package/scripts/lesson-seed.mjs +252 -0
  133. package/scripts/lesson-store.mjs +447 -0
  134. package/scripts/loop-checkpoint.mjs +86 -0
  135. package/scripts/memdb-health.sh +14 -0
  136. package/scripts/memory-doctor.mjs +326 -0
  137. package/scripts/model-catalog.mjs +79 -0
  138. package/scripts/nightly-controller.mjs +66 -0
  139. package/scripts/nightly-gists.sh +72 -0
  140. package/scripts/nightly-wrapper.sh +172 -0
  141. package/scripts/notify.sh +12 -0
  142. package/scripts/npx-witness.sh +56 -0
  143. package/scripts/onboarding-console.mjs +2922 -0
  144. package/scripts/private-fence.mjs +69 -0
  145. package/scripts/proactivity-metrics.mjs +118 -0
  146. package/scripts/proof-questions.json +56 -0
  147. package/scripts/protected-release-invocation.mjs +76 -0
  148. package/scripts/prove.mjs +95 -0
  149. package/scripts/proxy/claude-proxied.sh +57 -0
  150. package/scripts/proxy/proxy-revert.sh +59 -0
  151. package/scripts/proxy/proxy-up.sh +60 -0
  152. package/scripts/proxy/proxy-verify.mjs +142 -0
  153. package/scripts/publication-receipt.mjs +307 -0
  154. package/scripts/published-surface-probe.mjs +241 -0
  155. package/scripts/qe/card-lane-gate.mjs +162 -0
  156. package/scripts/qe/session-start-gate.mjs +229 -0
  157. package/scripts/qe/ux-suite.mjs +323 -0
  158. package/scripts/reconcile-project.mjs +0 -0
  159. package/scripts/record-lesson.mjs +113 -0
  160. package/scripts/refresh-model-catalog.mjs +99 -0
  161. package/scripts/release-authority.mjs +93 -0
  162. package/scripts/release-proof.mjs +9 -0
  163. package/scripts/release-vector.mjs +281 -0
  164. package/scripts/release.mjs +439 -0
  165. package/scripts/remedy-registry.mjs +247 -0
  166. package/scripts/rerank-cap-eval.mjs +265 -0
  167. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  168. package/scripts/route-cheap.mjs +20 -15
  169. package/scripts/router-utilization.mjs +182 -0
  170. package/scripts/routing-flywheel.mjs +596 -0
  171. package/scripts/rvf-generation.mjs +104 -0
  172. package/scripts/rvf-index-audit.mjs +138 -0
  173. package/scripts/self-update.mjs +296 -0
  174. package/scripts/selfcheck.mjs +7 -1
  175. package/scripts/sign-bundle.mjs +69 -0
  176. package/scripts/signal-watch.mjs +171 -0
  177. package/scripts/stabilization-receipt.mjs +108 -0
  178. package/scripts/stack-sync.mjs +469 -0
  179. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  180. package/scripts/stamp-sweep.mjs +144 -0
  181. package/scripts/status-honesty.mjs +102 -0
  182. package/scripts/sync-version.mjs +217 -0
  183. package/scripts/token-report.mjs +102 -0
  184. package/scripts/top100-benchmark.mjs +479 -0
  185. package/scripts/top100-corpus.mjs +112 -0
  186. package/scripts/top100-semantic-assertions.mjs +449 -0
  187. package/scripts/update-apply.mjs +9 -0
  188. package/scripts/upgrade-notice.mjs +14 -0
  189. package/scripts/verify-bundle.mjs +51 -0
  190. package/scripts/verify-channels.mjs +184 -0
  191. package/scripts/verify-model-catalog.mjs +104 -0
  192. package/scripts/verify-nightly-close-issue4.sh +31 -0
  193. package/scripts/version.mjs +40 -0
  194. package/scripts/wired-check.mjs +867 -0
  195. package/plugin/scripts/finalize-token-meter.mjs +0 -25
@@ -0,0 +1,250 @@
1
+ #!/usr/bin/env node
2
+ // corpus-qa.mjs — permanent machine gate: "everything embeds correctly and everything gets read
3
+ // correctly." Born from the 2026-07-10 depth-restore failure, where a rebuilt ruvector store had
4
+ // 18,491 passages and 0 full bodies and nothing noticed until a human grepped for '(full body):'.
5
+ //
6
+ // For EVERY .rvf store in the kb dir. Canonical repository stores are .big bge-768; legacy
7
+ // MiniLM-384 stores remain discoverable during migration, and big-only stores may share the
8
+ // canonical unsuffixed passages sidecar to avoid storing the same source text twice.
9
+ //
10
+ // STRUCTURAL (cheap, always):
11
+ // S1 <name>.passages.jsonl exists and has > 0 rows
12
+ // S2 full-body passage count > 0 whenever scripts/full-hints.mjs FULL_HINTS names the store
13
+ // (the exact failure class this gate exists to kill — hints defined, bodies zeroed = FAIL)
14
+ // S3 .rvf totalVectors === passages rows (no missing/extra rows), via RvfDatabase.openReadonly
15
+ // S4 <name>.rvf.embed.json exists (a store the read path can't embed queries for is unreadable)
16
+ //
17
+ // ROUND-TRIP (heavy, skipped with --structural):
18
+ // R1 sample 3 passages DETERMINISTICALLY (FNV-1a of "store.variant:k" — reproducible, no RNG),
19
+ // re-embed each passage exactly as the pipeline indexed it (small: "title — path\ntext",
20
+ // mean pooling; big: raw text, cls pooling — read from the store's own embed.json),
21
+ // query THAT store, and require the sampled row itself (same id, same path, or identical
22
+ // text — overlapping chunks of one doc may legitimately outrank each other) in top-3,
23
+ // OR within NEAR_DUP_EPS cosine distance of the best hit inside top-10. The epsilon arm
24
+ // exists because near-duplicate corpus rows (e.g. FACT's timestamped benchmark_report
25
+ // JSONs, ~95% identical text) crowd the podium while quantized batch-vs-single embed
26
+ // drift (~0.035 measured) exceeds the true gap between near-dups — the row IS stored and
27
+ // readable (fact.big id=722: rank 6, Δ0.007 behind rank 1), so that's a photo-finish,
28
+ // not a broken store. Photo-finish passes are still surfaced as a `note` so the
29
+ // near-dup-noise signal feeds the dedup backlog instead of being hidden. A row absent
30
+ // from top-10 (or far from the leader) remains a hard FAIL — missing/zero vectors and
31
+ // broken read paths cannot hide behind the epsilon.
32
+ // Proves embed-write AND read-path in one check. Retrieval QUALITY (real questions) stays
33
+ // forge-guard/prove's job; this gate proves the machinery, not the answers.
34
+ //
35
+ // Usage:
36
+ // node scripts/corpus-qa.mjs # whole corpus, structural + round-trip, serial
37
+ // node scripts/corpus-qa.mjs --store ruvector # one store (both variants)
38
+ // node scripts/corpus-qa.mjs --structural # cheap checks only
39
+ // [--dir <kb-dir>] [--samples N] # fixture/test hooks
40
+ //
41
+ // Output: one table row per store-variant, PASS/FAIL + reasons; skipped store classes are printed
42
+ // with why (nothing is skipped silently). Exit 1 if ANY row fails — self-update.mjs runs this per
43
+ // rebuilt store and aborts before publish on failure.
44
+
45
+ import fs from 'node:fs';
46
+ import path from 'node:path';
47
+ import readline from 'node:readline';
48
+ import { fileURLToPath } from 'node:url';
49
+ import { FULL_HINTS } from './full-hints.mjs';
50
+
51
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
52
+ const KB = path.join(ROOT, 'kb');
53
+
54
+ // resolve-deps lives in the real kb/ regardless of --dir (fixtures point --dir elsewhere).
55
+ const { loadRvf, loadTransformers, configureModel, chooseModelCache, closeReadonlyRvf } =
56
+ await import(path.join(KB, 'resolve-deps.mjs')).then((m) => m);
57
+ const { materializeModelRevision } =
58
+ await import(path.join(KB, 'model-requirements.mjs')).then((m) => m);
59
+
60
+ const FULL_BODY_MARK = '(full body):';
61
+ // R1 photo-finish epsilon: measured quantized batch-vs-single embed drift is ~0.035 cosine
62
+ // distance (fact.big id=722 replay); near-dup siblings sit within ~0.005 of each other. 0.02
63
+ // forgives the drift-scale tie WITHOUT forgiving genuinely different rows.
64
+ const NEAR_DUP_EPS = 0.02;
65
+
66
+ function fnv1a(s) {
67
+ let h = 0x811c9dc5;
68
+ for (let i = 0; i < s.length; i++) { h ^= s.charCodeAt(i); h = Math.imul(h, 0x01000193) >>> 0; }
69
+ return h >>> 0;
70
+ }
71
+
72
+ function readPassages(file) {
73
+ return new Promise((resolve, reject) => {
74
+ const rows = [];
75
+ const rl = readline.createInterface({ input: fs.createReadStream(file), crlfDelay: Infinity });
76
+ rl.on('line', (l) => { const s = l.trim(); if (!s) return; try { rows.push(JSON.parse(s)); } catch { /* counted structurally via row parse below */ } });
77
+ rl.on('close', () => resolve(rows));
78
+ rl.on('error', reject);
79
+ });
80
+ }
81
+
82
+ // Deterministic k distinct sample indices for a store-variant. Same store => same samples, always.
83
+ export function sampleIndices(storeKey, n, k) {
84
+ const idx = new Set();
85
+ for (let salt = 0; idx.size < Math.min(k, n) && salt < 50 * k; salt++) idx.add(fnv1a(`${storeKey}:${salt}`) % n);
86
+ return [...idx].slice(0, Math.min(k, n));
87
+ }
88
+
89
+ // ---- embedder cache: one pipeline per model, shared across stores. Caches the in-flight PROMISE
90
+ // (not just the resolved pipe) so concurrent qaStore() calls that both miss on the same model don't
91
+ // each trigger their own T.pipeline() load — first caller wins, the rest await its promise.
92
+ const pipelines = new Map();
93
+ function getPipeline(embedConf) {
94
+ const { model, revision } = embedConf;
95
+ const key = `${model}@${revision || 'unversioned'}`;
96
+ if (pipelines.has(key)) return pipelines.get(key);
97
+ const p = (async () => {
98
+ const { T } = await loadTransformers();
99
+ const cache = chooseModelCache();
100
+ materializeModelRevision(cache, model, revision);
101
+ configureModel(T, cache);
102
+ return T.pipeline('feature-extraction', model, {
103
+ quantized: true,
104
+ ...(revision ? { revision } : {}),
105
+ });
106
+ })();
107
+ pipelines.set(key, p);
108
+ return p;
109
+ }
110
+
111
+ /**
112
+ * QA one store-variant. Returns { store, variant, passages, fullBodies, vectors, roundtrip, fails, notes }.
113
+ * `fails` is [] on PASS; every entry is a specific reason string (receipts, not adjectives).
114
+ * `notes` are non-fatal signals (e.g. near-dup crowding) that should reach the dedup backlog.
115
+ */
116
+ export async function qaStore(dir, store, variant, { roundtrip = true, samples = 3 } = {}) {
117
+ const base = variant === 'big' ? `${store}.big` : store;
118
+ const rvfPath = path.join(dir, `${base}.rvf`);
119
+ const variantPassages = path.join(dir, `${base}.passages.jsonl`);
120
+ const canonicalPassages = path.join(dir, `${store}.passages.jsonl`);
121
+ const passagesPath = variant === 'big' && !fs.existsSync(variantPassages)
122
+ ? canonicalPassages
123
+ : variantPassages;
124
+ const embedPath = `${rvfPath}.embed.json`;
125
+ const res = { store, variant, passages: 0, fullBodies: 0, vectors: null, roundtrip: 'skipped', fails: [], notes: [] };
126
+
127
+ // S1: passages sidecar
128
+ if (!fs.existsSync(passagesPath)) { res.fails.push(`S1 missing ${path.basename(passagesPath)}`); return res; }
129
+ const rows = await readPassages(passagesPath);
130
+ res.passages = rows.length;
131
+ if (rows.length === 0) { res.fails.push('S1 passages file has 0 rows'); return res; }
132
+
133
+ // S2: full-body floor wherever hints exist (the 2026-07-10 failure class)
134
+ res.fullBodies = rows.filter((r) => typeof r.text === 'string' && r.text.includes(FULL_BODY_MARK)).length;
135
+ if (FULL_HINTS[store] && res.fullBodies === 0) {
136
+ res.fails.push(`S2 FULL_HINTS defines --full for "${store}" but store has 0 full-body passages (silent depth loss)`);
137
+ }
138
+
139
+ // S3: vector count parity
140
+ let db = null;
141
+ try {
142
+ const { mod } = loadRvf();
143
+ db = await mod.RvfDatabase.openReadonly(rvfPath);
144
+ const st = await db.status();
145
+ res.vectors = st.totalVectors;
146
+ if (st.totalVectors !== rows.length) res.fails.push(`S3 vectors=${st.totalVectors} != passages=${rows.length}`);
147
+ } catch (e) {
148
+ res.fails.push(`S3 cannot open ${path.basename(rvfPath)}: ${e.message.split('\n')[0]}`);
149
+ }
150
+
151
+ // S4: query-side embed config
152
+ let embedConf = null;
153
+ if (!fs.existsSync(embedPath)) res.fails.push(`S4 missing ${path.basename(embedPath)} (read path cannot embed queries)`);
154
+ else embedConf = JSON.parse(fs.readFileSync(embedPath, 'utf8'));
155
+
156
+ // R1: deterministic self-retrieval round trip
157
+ if (roundtrip && db && embedConf && !res.fails.some((f) => f.startsWith('S3'))) {
158
+ try {
159
+ const picks = sampleIndices(`${store}.${variant}`, rows.length, samples);
160
+ const byId = new Map(rows.map((r) => [String(r.id), r]));
161
+ let hit = 0;
162
+ const misses = [];
163
+ const pipe = await getPipeline(embedConf);
164
+ for (const i of picks) {
165
+ const r = rows[i];
166
+ // Re-embed EXACTLY what the pipeline indexed for this variant (forge-build/forge-big):
167
+ const text = variant === 'big'
168
+ ? r.text
169
+ : `${r.title} — ${r.path}\n${r.text}`.slice(0, 4300);
170
+ const out = await pipe([text], { pooling: embedConf.pooling || 'mean', normalize: true });
171
+ if (out.dims[1] !== embedConf.dimensions) throw new Error(`embed dim ${out.dims[1]} != ${embedConf.dimensions}`);
172
+ const top = await db.query(Array.from(out.data), 10);
173
+ const matches = (t) => String(t.id) === String(r.id)
174
+ || byId.get(String(t.id))?.path === r.path
175
+ || byId.get(String(t.id))?.text === r.text;
176
+ const rank = top.findIndex(matches); // -1 = absent from top-10
177
+ if (rank >= 0 && rank < 3) hit++;
178
+ else if (rank >= 0 && top[rank].distance - top[0].distance <= NEAR_DUP_EPS) {
179
+ hit++; // photo-finish behind near-duplicates: stored + readable; surface the crowd as a note
180
+ res.notes.push(`near-dup crowd: id=${r.id} rank ${rank + 1}, Δ${(top[rank].distance - top[0].distance).toFixed(4)} behind #1 (${r.path})`);
181
+ } else {
182
+ misses.push(`id=${r.id} ${rank < 0 ? 'ABSENT from top-10' : `rank ${rank + 1}, Δ${(top[rank].distance - top[0].distance).toFixed(4)}`} ${r.path}`);
183
+ }
184
+ }
185
+ res.roundtrip = `${hit}/${picks.length}`;
186
+ if (hit < picks.length) res.fails.push(`R1 self-retrieval missed: ${misses.join('; ')}`);
187
+ } catch (e) {
188
+ res.roundtrip = 'error';
189
+ res.fails.push(`R1 round-trip error: ${e.message.split('\n')[0]}`);
190
+ }
191
+ }
192
+ await closeReadonlyRvf(db);
193
+ return res;
194
+ }
195
+
196
+ /** Discover store-variants in a dir. Returns { stores: [{store, variant}], skipped: [{file, why}] }. */
197
+ export function discoverStores(dir) {
198
+ const stores = [];
199
+ const skipped = [];
200
+ for (const f of fs.readdirSync(dir).sort()) {
201
+ if (!f.endsWith('.rvf')) continue; // idmaps/embed.json/etc. are per-store sidecars, not stores
202
+ const name = f.slice(0, -'.rvf'.length);
203
+ if (name.endsWith('.big')) stores.push({ store: name.slice(0, -'.big'.length), variant: 'big' });
204
+ else stores.push({ store: name, variant: 'small' });
205
+ }
206
+ return { stores, skipped };
207
+ }
208
+
209
+ // ---------------- CLI ----------------
210
+ if (import.meta.url === `file://${process.argv[1]}`) {
211
+ const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
212
+ const has = (f) => process.argv.includes(f);
213
+ const DIR = path.resolve(arg('--dir', KB));
214
+ const ONLY = arg('--store', null);
215
+ const STRUCTURAL_ONLY = has('--structural');
216
+ const SAMPLES = parseInt(arg('--samples', '3'), 10) || 3;
217
+ const CONC = Math.max(1, parseInt(arg('--concurrency', '4'), 10) || 4);
218
+
219
+ const { stores } = discoverStores(DIR);
220
+ const todo = ONLY ? stores.filter((s) => s.store === ONLY) : stores;
221
+ if (ONLY && todo.length === 0) { console.error(`[corpus-qa] no store named "${ONLY}" in ${DIR}`); process.exit(2); }
222
+ console.log(`[corpus-qa] ${todo.length} store-variant(s) in ${DIR}${ONLY ? ` (store=${ONLY})` : ''}${STRUCTURAL_ONLY ? ' [structural only]' : ' [structural + round-trip]'} — concurrency ${CONC}, one process`);
223
+
224
+ // qaStore() shares the in-process pipelines cache (getPipeline above) and each store-variant
225
+ // touches its own .rvf/.passages.jsonl files, so concurrent runs don't step on each other.
226
+ const results = new Array(todo.length);
227
+ let cursor = 0;
228
+ const runOne = async () => {
229
+ while (true) {
230
+ const i = cursor++;
231
+ if (i >= todo.length) return;
232
+ const { store, variant } = todo[i];
233
+ results[i] = await qaStore(DIR, store, variant, { roundtrip: !STRUCTURAL_ONLY, samples: SAMPLES });
234
+ }
235
+ };
236
+ await Promise.all(Array.from({ length: Math.min(CONC, todo.length) }, runOne));
237
+
238
+ const pad = (s, n) => String(s).padEnd(n);
239
+ console.log('\n' + pad('store', 26) + pad('variant', 8) + pad('passages', 10) + pad('full-b', 8) + pad('vectors', 9) + pad('roundtrip', 11) + 'verdict');
240
+ let failed = 0;
241
+ for (const r of results) {
242
+ const verdict = r.fails.length ? 'FAIL' : 'PASS';
243
+ if (r.fails.length) failed++;
244
+ console.log(pad(r.store, 26) + pad(r.variant, 8) + pad(r.passages, 10) + pad(r.fullBodies, 8) + pad(r.vectors ?? '?', 9) + pad(r.roundtrip, 11) + verdict);
245
+ for (const f of r.fails) console.log(' ↳ ' + f);
246
+ for (const n of r.notes) console.log(' · note: ' + n);
247
+ }
248
+ console.log(`\n[corpus-qa] ${results.length - failed}/${results.length} store-variants PASS${failed ? ` — ${failed} FAILED` : ''}`);
249
+ process.exit(failed ? 1 : 0);
250
+ }
@@ -0,0 +1,346 @@
1
+ #!/usr/bin/env node
2
+ // correction-detect-embed.mjs — an HONEST measurement of whether a DIFFERENT PRIMITIVE (a local
3
+ // embedding k-NN classifier) clears ADR-033 §2's floor (≥90% precision on ≥100 detections) where
4
+ // the lexical regex detector (scripts/correction-detect.mjs) did not.
5
+ //
6
+ // This is a MEASUREMENT SCRIPT, not a shipped feature. It does not wire into any hook, gate, or
7
+ // store. It answers one question for an owner decision: does swapping the primitive from regex to
8
+ // embedding similarity change the verdict on THIS corpus? See the accompanying report for the
9
+ // answer; this file is how the numbers in that report were produced, reproducibly.
10
+ //
11
+ // ─────────────────────────────────────────────────────────────────────────────────────────────────
12
+ // THE PRIMITIVE: k-NEAREST-NEIGHBOUR OVER LOCAL MiniLM EMBEDDINGS — NOT A TRAINED CLASSIFIER
13
+ //
14
+ // Grounded against how this repo already does embeddings (kb/forge-build.mjs, kb/resolve-deps.mjs):
15
+ // the SAME model (`Xenova/all-MiniLM-L6-v2`, 384-dim, mean-pooled, L2-normalized, pinned to the
16
+ // same HuggingFace revision the KB build uses) via the SAME local ONNX runtime
17
+ // (`@xenova/transformers`, resolved through `kb/resolve-deps.mjs`'s `loadTransformers()` /
18
+ // `configureModel()` — no network call when the model is already cached, per CLAUDE.md Rule 1: RVF/
19
+ // local-ONNX first, never an external embedding API). This script imports that resolver directly
20
+ // rather than re-implementing model loading, cache resolution, or the network-hang guard a second
21
+ // time — those are already solved once, correctly, in `kb/resolve-deps.mjs`.
22
+ //
23
+ // The classifier itself is deliberately the simplest thing that could work (Karpathy: minimum code,
24
+ // no speculative abstraction): embed each candidate as `PRECEDING ACTION: <summary> \n USER:
25
+ // <utterance>` (utterance ± the preceding-action context the task asked for), embed a small labelled
26
+ // reference set drawn ONLY from the TUNE split, and classify a holdout candidate by a similarity-
27
+ // weighted vote of its k nearest TUNE neighbours. k and the similarity floor are chosen by
28
+ // leave-one-out cross-validation ON THE TUNE SET ONLY, then frozen before touching holdout — the
29
+ // same no-leakage discipline `correction-detect-measure.mjs` uses for its file-level split.
30
+ //
31
+ // ─────────────────────────────────────────────────────────────────────────────────────────────────
32
+ // WHERE THE GROUND TRUTH COMES FROM
33
+ //
34
+ // `scripts/correction-detect-measure.mjs --dump-pool` already builds a candidate pool (Signal-1:
35
+ // adjacent to a preceding assistant action) and this task's own grounding pass narrowed it with the
36
+ // SAME loose superset lexical net that script's `broad-pool.mjs` companion applies (deliberately
37
+ // looser than the regex's own signals — a safety margin so a genuine correction is not excluded from
38
+ // the labelling pool just because it doesn't use the regex's exact vocabulary). That pool — 271
39
+ // candidates, 112 tune / 159 holdout, spanning ALL 1,328 transcripts — was hand-labelled by this
40
+ // task's author (true / borderline / false) against ADR-033's actual four-signal definition, NOT
41
+ // against what either detector happens to fire on. That hand-labelling is the same self-graded
42
+ // caveat every number in `correction-detect.mjs`'s own header already carries (Verification #6 in
43
+ // ADR-033: "not independently graded") — repeated here rather than hidden.
44
+ //
45
+ // The labelled pool contains real (if redacted-of-secrets) transcript text and is NOT committed to
46
+ // this repo, for the same reason `correction-detect-measure.mjs` never writes transcript text into
47
+ // the repo. Point `--labels` / `--tune-pool` / `--holdout-pool` at that data (default: this
48
+ // project's scratchpad locations used to build it) to reproduce the run. A small, hand-picked,
49
+ // already-public-in-spirit subset (utterances that already became named standing orders in this
50
+ // project's own committed memory index) ships as `tests/fixtures/correction-embed-sample.jsonl` so
51
+ // `tests/unit/correction-detect-embed.test.mjs` can run the classifier's MECHANICS without any
52
+ // private corpus present.
53
+ //
54
+ // ─────────────────────────────────────────────────────────────────────────────────────────────────
55
+ // ─────────────────────────────────────────────────────────────────────────────────────────────────
56
+ // MEASURED, 2026-07-23 — same 271-item hand-labelled pool (112 tune / 159 holdout, all 1,328
57
+ // transcripts) the total-genuine-correction count in the report was taken from. Reproduce with the
58
+ // `measure` command below, pointed at that pool + labels.json (see this file's own USAGE).
59
+ //
60
+ // tune-set ground truth: 14 true / 98 false (of 112)
61
+ // holdout ground truth: 23 true / 6 borderline / 130 false (of 159)
62
+ //
63
+ // Leave-one-out search on TUNE ONLY (no holdout leakage) picked k=1, minSim=0.6 (tune-LOO
64
+ // precision 66.7% / recall 42.9% — already far below the regex's tune-side 100%/4-of-4).
65
+ //
66
+ // Applied to the SAME 159-item holdout the regex's own 4 detections came from:
67
+ // flagged positive: 8
68
+ // true positives: 2
69
+ // false positives: 6 (0 of the 6 borderline-labelled rows were flagged)
70
+ // PRECISION: 25.0% (2/8)
71
+ // RECALL (of 29 true+borderline): 6.9%
72
+ //
73
+ // With the frozen default operating point below (k=5, minSim=0.3, chosen the same way but on an
74
+ // earlier tune-only search) instead of a fresh --tune-k run: 7 flagged, 2 true positives, 5 false
75
+ // — precision 28.6%, recall 6.9%. Same order of magnitude either way: roughly a QUARTER of what
76
+ // this primitive flags on the real holdout is a genuine correction, versus the regex's own
77
+ // 50-100% (n=4) on the same pool.
78
+ //
79
+ // Extended to the FULL Signal-1 holdout population (784 candidates, `--wide-pool`, not just the
80
+ // 159-item loose-net subset a human would ever be asked to label): 12 flagged, of which 5 fall
81
+ // OUTSIDE the labelled subset. Hand-reviewed those 5 for this measurement: all 5 are false
82
+ // (a repo-naming brainstorm, two raw image-paste captions, a status question, and a delegation
83
+ // statement) — so at realistic operational scale, precision is 2/12 ≈ 16.7%, recall unchanged.
84
+ //
85
+ // CONCLUSION: this primitive does NOT beat the regex on precision on this corpus, and comes
86
+ // nowhere near ADR-033 §2's ≥90% floor at any N tried. See the report this task produced for the
87
+ // full discussion (why: MiniLM sentence embeddings here separate by TOPIC — "this utterance is
88
+ // about AgentDB/versions/the console" — not by the PRAGMATIC property ADR-033 actually needs
89
+ // ("is this utterance correcting the agent's behaviour"). Two utterances about the same topic,
90
+ // one a bug report and one a correction, land close together in embedding space; two genuine
91
+ // corrections about DIFFERENT topics (READMEs vs. scoring vs. version pinning) often do not.
92
+ // This is the same conclusion ADR-033 reached about lexical overlap, now shown to also hold for
93
+ // semantic (embedding) overlap on this corpus — it is not a lexical-vs-semantic gap, it is a
94
+ // topic-vs-pragmatics gap that neither primitive, as tried, closes.
95
+ //
96
+ // ─────────────────────────────────────────────────────────────────────────────────────────────────
97
+ // USAGE
98
+ // node scripts/correction-detect-embed.mjs measure \
99
+ // --tune-pool <broad-tune.jsonl> --holdout-pool <broad-holdout.jsonl> --labels <labels.json> \
100
+ // [--wide-pool <cand-holdout.jsonl>] [--k 5] [--min-sim 0.35] [--tune-k]
101
+ //
102
+ // --wide-pool, optional: the FULL Signal-1 holdout pool (784 candidates, not just the 159-item
103
+ // loose-net subset) — runs the frozen classifier over it and reports how many candidates OUTSIDE
104
+ // the labelled subset it flags positive, honestly marked UNLABELLED rather than guessed at.
105
+ //
106
+ // --tune-k: run leave-one-out CV over a small (k, minSim) grid on the tune set only, print the
107
+ // chosen operating point, then use it. Without this flag the script uses the value already found
108
+ // this way and recorded in DEFAULT_K / DEFAULT_MIN_SIM below (reproducible without re-searching).
109
+
110
+ import fs from 'node:fs';
111
+ import path from 'node:path';
112
+ import { fileURLToPath } from 'node:url';
113
+
114
+ const __dirname = path.dirname(fileURLToPath(import.meta.url));
115
+ const KB_DIR = path.join(__dirname, '..', 'kb');
116
+
117
+ // Pinned to the exact commit forge-build.mjs pins to, so this script's vectors are byte-identical
118
+ // to the ones the shipped KB would produce for the same text (see forge-build.mjs's own comment on
119
+ // MINILM_REVISION for why: address-by-SHA, not floating `main`).
120
+ const MINILM_REVISION = '751bff37182d3f1213fa05d7196b954e230abad9';
121
+
122
+ // Found by `--tune-k` leave-one-out search over k in [1,3,5,7,9] and minSim in [0.0,0.15,...,0.6],
123
+ // maximizing tune-set F1 (ties broken toward higher minSim, i.e. more conservative — precision is
124
+ // the safety property here, per ADR-033 §2, so a tie goes to the pickier operating point). Re-run
125
+ // `--tune-k` to reproduce; this corpus is small (112 tune rows) so the search is seconds, not a
126
+ // separate offline step, but the chosen point is frozen here so a bare `measure` run is deterministic
127
+ // without depending on the search being re-run identically.
128
+ export const DEFAULT_K = 5;
129
+ export const DEFAULT_MIN_SIM = 0.3;
130
+
131
+ /** Resolve the local MiniLM embedder through this repo's OWN resolver — no re-implementation, no
132
+ * external API call (CLAUDE.md Rule 1). Fails loudly (via loadTransformers' own network guard) if
133
+ * neither a project node_modules nor an env override can find @xenova/transformers. */
134
+ async function loadEmbedder() {
135
+ const { loadTransformers, configureModel } = await import(path.join(KB_DIR, 'resolve-deps.mjs'));
136
+ const { T, modelCache, via } = await loadTransformers();
137
+ const { haveLocalModel } = configureModel(T, modelCache);
138
+ const embed = await T.pipeline('feature-extraction', 'Xenova/all-MiniLM-L6-v2', {
139
+ quantized: true, revision: MINILM_REVISION,
140
+ });
141
+ return { embed, via, modelCache, haveLocalModel };
142
+ }
143
+
144
+ /** Utterance ± preceding-action context, exactly as the task specified — the same two fields
145
+ * `detectCorrection()` consumes (Signal 1's adjacency evidence), just embedded instead of regexed. */
146
+ export function candidateText(row) {
147
+ const prior = row.precedingAssistantAction || {};
148
+ const action = prior.summary || prior.tool || '';
149
+ return action
150
+ ? `PRECEDING ACTION: ${String(action).slice(0, 200)}\nUSER: ${row.promptText}`
151
+ : `USER: ${row.promptText}`;
152
+ }
153
+
154
+ async function embedBatch(embed, texts, batchSize = 16) {
155
+ const vectors = [];
156
+ for (let i = 0; i < texts.length; i += batchSize) {
157
+ const batch = texts.slice(i, i + batchSize);
158
+ const out = await embed(batch, { pooling: 'mean', normalize: true });
159
+ const dim = out.dims[1];
160
+ for (let j = 0; j < batch.length; j++) vectors.push(Array.from(out.data.slice(j * dim, (j + 1) * dim)));
161
+ }
162
+ return vectors;
163
+ }
164
+
165
+ /** Vectors are already L2-normalized (normalize:true above), so dot product IS cosine similarity —
166
+ * same convention kb/forge-build.mjs uses for its RVF store (metric:'cosine' over normalized vecs). */
167
+ function dot(a, b) {
168
+ let s = 0;
169
+ for (let i = 0; i < a.length; i++) s += a[i] * b[i];
170
+ return s;
171
+ }
172
+
173
+ /**
174
+ * The classifier: similarity-weighted k-NN vote against a labelled reference set. Deliberately NOT
175
+ * a trained model (no gradient descent, no held weights beyond the reference vectors themselves) —
176
+ * this keeps every verdict traceable to "which labelled examples it resembles, and how much" rather
177
+ * than to opaque learned parameters. It is still less legible than the regex (see LEGIBILITY note
178
+ * at the bottom of this file), but it is the most legible embedding-based option available: the
179
+ * `neighbors` returned alongside every verdict ARE the explanation, not a post-hoc rationalization.
180
+ */
181
+ export function classify(vec, refs, { k = DEFAULT_K, minSim = DEFAULT_MIN_SIM } = {}) {
182
+ const scored = refs.map((r) => ({ ...r, sim: dot(vec, r.vector) })).sort((a, b) => b.sim - a.sim);
183
+ const top = scored.slice(0, k).filter((r) => r.sim >= minSim);
184
+ if (!top.length) return { isCorrection: false, score: 0, neighbors: scored.slice(0, 3) };
185
+ const posWeight = top.filter((r) => r.label === 'true').reduce((s, r) => s + r.sim, 0);
186
+ const negWeight = top.filter((r) => r.label !== 'true').reduce((s, r) => s + r.sim, 0);
187
+ return { isCorrection: posWeight > negWeight, score: posWeight - negWeight, neighbors: top.slice(0, 3) };
188
+ }
189
+
190
+ /** Leave-one-out CV over a small grid, TUNE SET ONLY — never touches holdout. Maximizes F1; ties
191
+ * broken toward the higher minSim (more conservative), per ADR-033's precision-over-recall stance. */
192
+ function looSearch(tuneVecs, tuneLabels) {
193
+ const refs = tuneVecs.map((vector, i) => ({ vector, label: tuneLabels[i] }));
194
+ let best = null;
195
+ for (const k of [1, 3, 5, 7, 9]) {
196
+ for (const minSim of [0, 0.1, 0.15, 0.2, 0.25, 0.3, 0.35, 0.4, 0.45, 0.5, 0.55, 0.6]) {
197
+ let tp = 0, fp = 0, fn = 0;
198
+ for (let i = 0; i < refs.length; i++) {
199
+ const others = refs.slice(0, i).concat(refs.slice(i + 1));
200
+ const { isCorrection } = classify(refs[i].vector, others, { k, minSim });
201
+ const truth = refs[i].label === 'true';
202
+ if (isCorrection && truth) tp++;
203
+ else if (isCorrection && !truth) fp++;
204
+ else if (!isCorrection && truth) fn++;
205
+ }
206
+ const precision = tp + fp ? tp / (tp + fp) : 0;
207
+ const recall = tp + fn ? tp / (tp + fn) : 0;
208
+ const f1 = precision + recall ? (2 * precision * recall) / (precision + recall) : 0;
209
+ const cand = { k, minSim, tp, fp, fn, precision, recall, f1 };
210
+ if (!best || f1 > best.f1 || (f1 === best.f1 && minSim > best.minSim)) best = cand;
211
+ }
212
+ }
213
+ return best;
214
+ }
215
+
216
+ // ── I/O helpers ──────────────────────────────────────────────────────────────────────────────────
217
+
218
+ function readJsonl(file) {
219
+ return fs.readFileSync(file, 'utf8').split('\n').filter(Boolean).map((l) => JSON.parse(l));
220
+ }
221
+
222
+ function loadLabels(file) {
223
+ const rows = JSON.parse(fs.readFileSync(file, 'utf8'));
224
+ const map = new Map();
225
+ for (const r of rows) map.set(`${r.file}#${r.turnIndex}`, r.label);
226
+ return map;
227
+ }
228
+
229
+ function flag(name, fallback = null) {
230
+ const i = process.argv.indexOf(name);
231
+ return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback;
232
+ }
233
+
234
+ // ── Measurement ──────────────────────────────────────────────────────────────────────────────────
235
+
236
+ async function measure() {
237
+ const tunePoolFile = flag('--tune-pool');
238
+ const holdoutPoolFile = flag('--holdout-pool');
239
+ const labelsFile = flag('--labels');
240
+ const widePoolFile = flag('--wide-pool');
241
+ const doSearch = process.argv.includes('--tune-k');
242
+
243
+ if (!tunePoolFile || !holdoutPoolFile || !labelsFile) {
244
+ console.error('Usage: node scripts/correction-detect-embed.mjs measure --tune-pool <jsonl> '
245
+ + '--holdout-pool <jsonl> --labels <labels.json> [--wide-pool <jsonl>] [--tune-k]');
246
+ console.error('These point at real (private) transcript-derived data — see this file\'s header '
247
+ + 'for how to regenerate them; nothing of that shape is committed to this repo.');
248
+ process.exit(1);
249
+ }
250
+
251
+ const labels = loadLabels(labelsFile);
252
+ const tuneRows = readJsonl(tunePoolFile).map((r) => ({ ...r, label: labels.get(`${r.file}#${r.turnIndex}`) || 'false' }));
253
+ const holdoutRows = readJsonl(holdoutPoolFile).map((r) => ({ ...r, label: labels.get(`${r.file}#${r.turnIndex}`) || 'false' }));
254
+
255
+ console.error(`[embed] tune=${tuneRows.length} (true=${tuneRows.filter(r=>r.label==='true').length}) `
256
+ + `holdout=${holdoutRows.length} (true=${holdoutRows.filter(r=>r.label==='true').length}, `
257
+ + `borderline=${holdoutRows.filter(r=>r.label==='borderline').length})`);
258
+
259
+ const { embed, via, haveLocalModel, modelCache } = await loadEmbedder();
260
+ console.error(`[embed] transformers via: ${via} | model: ${haveLocalModel ? 'local cache' : 'REMOTE DOWNLOAD'} (${modelCache})`);
261
+
262
+ const tuneVecs = await embedBatch(embed, tuneRows.map(candidateText));
263
+ const holdoutVecs = await embedBatch(embed, holdoutRows.map(candidateText));
264
+
265
+ let opPoint = { k: DEFAULT_K, minSim: DEFAULT_MIN_SIM };
266
+ if (doSearch) {
267
+ const found = looSearch(tuneVecs, tuneRows.map((r) => r.label));
268
+ console.error(`[embed] --tune-k search (leave-one-out, tune only): k=${found.k} minSim=${found.minSim} `
269
+ + `tune-LOO precision=${(100*found.precision).toFixed(1)}% recall=${(100*found.recall).toFixed(1)}% f1=${found.f1.toFixed(3)}`);
270
+ opPoint = { k: found.k, minSim: found.minSim };
271
+ } else {
272
+ console.error(`[embed] using frozen operating point k=${opPoint.k} minSim=${opPoint.minSim} (rerun with --tune-k to re-search)`);
273
+ }
274
+
275
+ const refs = tuneVecs.map((vector, i) => ({ vector, label: tuneRows[i].label, text: tuneRows[i].promptText }));
276
+
277
+ // ── Primary measurement: SAME 159-item labelled holdout population the regex's own examples
278
+ // came from (broad-holdout.jsonl) — apples-to-apples with correction-detect.mjs's own numbers. ──
279
+ let tp = 0, fp = 0, fpBorderline = 0, fn = 0, tn = 0;
280
+ const positives = [];
281
+ for (let i = 0; i < holdoutRows.length; i++) {
282
+ const { isCorrection, score, neighbors } = classify(holdoutVecs[i], refs, opPoint);
283
+ const row = holdoutRows[i];
284
+ if (isCorrection) {
285
+ positives.push({ ...row, score, neighbors: neighbors.map((n) => ({ label: n.label, sim: n.sim.toFixed(3), text: n.text.slice(0, 100) })) });
286
+ if (row.label === 'true') tp++;
287
+ else if (row.label === 'borderline') { fpBorderline++; }
288
+ else fp++;
289
+ } else {
290
+ if (row.label === 'true' || row.label === 'borderline') fn++;
291
+ else tn++;
292
+ }
293
+ }
294
+ const totalFp = fp + fpBorderline; // strict: borderline counts against precision, same as the regex's own header treats its 2 holdout borderlines
295
+ const precisionStrict = tp + totalFp ? tp / (tp + totalFp) : 0;
296
+ const precisionLenient = (tp + fpBorderline) + fp ? (tp + fpBorderline) / (tp + fpBorderline + fp) : 0;
297
+ const recallStrict = tp + fn ? tp / (tp + fn) : 0; // fn includes borderlines missed, conservative
298
+
299
+ console.log(`\n=== EMBEDDING CLASSIFIER — holdout (n=${holdoutRows.length}, same pool the regex's hand-labelled examples came from) ===`);
300
+ console.log(`positives (flagged): ${positives.length}`);
301
+ console.log(` true positives: ${tp}`);
302
+ console.log(` borderline positives: ${fpBorderline}`);
303
+ console.log(` false positives: ${fp}`);
304
+ console.log(` false negatives (missed true+borderline): ${fn}`);
305
+ console.log(`precision (strict, borderline counts against): ${(100*precisionStrict).toFixed(1)}% (${tp}/${tp+totalFp})`);
306
+ console.log(`precision (lenient, borderline counts for): ${(100*precisionLenient).toFixed(1)}% (${tp+fpBorderline}/${tp+fpBorderline+fp})`);
307
+ console.log(`recall (of ${tp+fn+ (holdoutRows.filter(r=>r.label==='true'||r.label==='borderline').length - (tp+fn))} known true/borderline): ${(100*recallStrict).toFixed(1)}%`);
308
+
309
+ console.log(`\n--- flagged positives, with nearest tune neighbours (the "legibility" a verdict can offer) ---`);
310
+ for (const p of positives) {
311
+ console.log(`[${p.label.toUpperCase()}] score=${p.score.toFixed(3)} file=${p.file} turn=${p.turnIndex}`);
312
+ console.log(` UTTERANCE: ${p.promptText.slice(0, 160).replace(/\n/g,' ')}`);
313
+ for (const n of p.neighbors) console.log(` neighbor(${n.label}, sim=${n.sim}): ${n.text.replace(/\n/g,' ')}`);
314
+ }
315
+
316
+ // ── Secondary, optional: the FULL Signal-1 holdout population (not just the loose-net subset) —
317
+ // the real deployment-scale test. Anything flagged OUTSIDE the labelled 159 is marked UNLABELLED,
318
+ // never silently assumed either way. ──
319
+ if (widePoolFile) {
320
+ const wideRows = readJsonl(widePoolFile);
321
+ const labelledKeys = new Set(holdoutRows.map((r) => `${r.file}#${r.turnIndex}`));
322
+ const wideVecs = await embedBatch(embed, wideRows.map(candidateText));
323
+ let wideFlagged = 0, outsideSubset = 0;
324
+ const outsideHits = [];
325
+ for (let i = 0; i < wideRows.length; i++) {
326
+ const { isCorrection, score } = classify(wideVecs[i], refs, opPoint);
327
+ if (!isCorrection) continue;
328
+ wideFlagged++;
329
+ const k = `${wideRows[i].file}#${wideRows[i].turnIndex}`;
330
+ if (!labelledKeys.has(k)) { outsideSubset++; outsideHits.push({ ...wideRows[i], score }); }
331
+ }
332
+ console.log(`\n=== WIDE HOLDOUT (n=${wideRows.length}, full Signal-1 pool, not just the loose-net subset) ===`);
333
+ console.log(`flagged: ${wideFlagged} (of which ${outsideSubset} fall OUTSIDE the 159-item labelled subset — UNLABELLED, hand-review needed, not counted in precision/recall above)`);
334
+ for (const h of outsideHits.slice(0, 20)) {
335
+ console.log(` UNLABELLED file=${h.file} turn=${h.turnIndex} score=${h.score.toFixed(3)}: ${h.promptText.slice(0,160).replace(/\n/g,' ')}`);
336
+ }
337
+ }
338
+ }
339
+
340
+ const invokedDirectly = process.argv[1]
341
+ && path.resolve(process.argv[1]).endsWith(`correction-detect-embed${path.extname(process.argv[1])}`);
342
+ if (invokedDirectly) {
343
+ const cmd = process.argv[2];
344
+ if (cmd === 'measure') measure().catch((e) => { console.error(e); process.exit(1); });
345
+ else { console.error('Usage: node scripts/correction-detect-embed.mjs measure ...'); process.exit(1); }
346
+ }