ruvnet-brain 3.9.134-dev → 4.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +14 -0
- package/README.md +5 -5
- package/bin/install.mjs +382 -36
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/kb/zip-extract.mjs +53 -14
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +14 -22
- package/plugin/.claude-plugin/marketplace.json +14 -0
- package/plugin/.claude-plugin/plugin.json +22 -0
- package/plugin/.codex-plugin/plugin.json +21 -0
- package/plugin/.mcp.json +8 -0
- package/plugin/commands/brain-console.md +16 -0
- package/plugin/commands/configure.md +33 -0
- package/plugin/commands/rvbc.md +79 -0
- package/plugin/commands/rvcb.md +16 -0
- package/plugin/commands/whats-new.md +57 -0
- package/plugin/hooks/codex-hooks.json +160 -0
- package/plugin/hooks/hook-contracts.json +77 -0
- package/plugin/hooks/hooks.json +202 -0
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +56 -6
- package/plugin/scripts/anticipate.sh +534 -0
- package/plugin/scripts/codex-hook-adapter.mjs +96 -0
- package/plugin/scripts/continuation-gate.mjs +267 -0
- package/plugin/scripts/design-wall.sh +137 -0
- package/plugin/scripts/detach.mjs +182 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/gate-receipt.sh +35 -0
- package/plugin/scripts/ground-before-write.sh +199 -0
- package/plugin/scripts/ground-ruvnet.sh +517 -0
- package/plugin/scripts/grounding-stamp.sh +113 -0
- package/plugin/scripts/grounding-substance.mjs +595 -0
- package/plugin/scripts/hijack-ruvnet.sh +81 -0
- package/plugin/scripts/hook-input.mjs +558 -0
- package/plugin/scripts/hook-shim-bash.mjs +55 -0
- package/plugin/scripts/hook-shim.mjs +303 -0
- package/plugin/scripts/host-update.mjs +58 -0
- package/plugin/scripts/kling-preflight.sh +146 -0
- package/plugin/scripts/learn-capture.sh +173 -0
- package/plugin/scripts/learn-flush.mjs +155 -0
- package/plugin/scripts/lesson-hooks.sh +213 -0
- package/plugin/scripts/md-stamp.mjs +219 -0
- package/plugin/scripts/protect-brain-state.sh +84 -0
- package/plugin/scripts/route-dispatch.sh +147 -0
- package/plugin/scripts/routing-outcome-capture.mjs +89 -0
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +477 -0
- package/plugin/scripts/session-start.sh +13 -0
- package/plugin/scripts/signal-watch.mjs +193 -0
- package/plugin/scripts/unprompted-runtime.mjs +377 -0
- package/plugin/scripts/update-apply.mjs +419 -0
- package/plugin/scripts/verify-interface.sh +53 -0
- package/plugin/scripts/version-bump-gate.sh +112 -0
- package/plugin/skills/brain-build/SKILL.md +123 -0
- package/plugin/skills/brain-console/SKILL.md +22 -0
- package/plugin/skills/brain-prompt/SKILL.md +83 -0
- package/plugin/skills/brain-score/SKILL.md +101 -0
- package/plugin/skills/release-proof/SKILL.md +81 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +121 -0
- package/plugin/skills/ruvnet-brain/SKILL.md +234 -0
- package/plugin/skills/rvbc/SKILL.md +23 -0
- package/plugin/skills/savings/SKILL.md +46 -0
- package/plugin/skills/whats-new/SKILL.md +22 -0
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +522 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +250 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +639 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +271 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +180 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2749 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +395 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +508 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +864 -0
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// corpus-qa.mjs — permanent machine gate: "everything embeds correctly and everything gets read
|
|
3
|
+
// correctly." Born from the 2026-07-10 depth-restore failure, where a rebuilt ruvector store had
|
|
4
|
+
// 18,491 passages and 0 full bodies and nothing noticed until a human grepped for '(full body):'.
|
|
5
|
+
//
|
|
6
|
+
// For EVERY .rvf store in the kb dir. Canonical repository stores are .big bge-768; legacy
|
|
7
|
+
// MiniLM-384 stores remain discoverable during migration, and big-only stores may share the
|
|
8
|
+
// canonical unsuffixed passages sidecar to avoid storing the same source text twice.
|
|
9
|
+
//
|
|
10
|
+
// STRUCTURAL (cheap, always):
|
|
11
|
+
// S1 <name>.passages.jsonl exists and has > 0 rows
|
|
12
|
+
// S2 full-body passage count > 0 whenever scripts/full-hints.mjs FULL_HINTS names the store
|
|
13
|
+
// (the exact failure class this gate exists to kill — hints defined, bodies zeroed = FAIL)
|
|
14
|
+
// S3 .rvf totalVectors === passages rows (no missing/extra rows), via RvfDatabase.openReadonly
|
|
15
|
+
// S4 <name>.rvf.embed.json exists (a store the read path can't embed queries for is unreadable)
|
|
16
|
+
//
|
|
17
|
+
// ROUND-TRIP (heavy, skipped with --structural):
|
|
18
|
+
// R1 sample 3 passages DETERMINISTICALLY (FNV-1a of "store.variant:k" — reproducible, no RNG),
|
|
19
|
+
// re-embed each passage exactly as the pipeline indexed it (small: "title — path\ntext",
|
|
20
|
+
// mean pooling; big: raw text, cls pooling — read from the store's own embed.json),
|
|
21
|
+
// query THAT store, and require the sampled row itself (same id, same path, or identical
|
|
22
|
+
// text — overlapping chunks of one doc may legitimately outrank each other) in top-3,
|
|
23
|
+
// OR within NEAR_DUP_EPS cosine distance of the best hit inside top-10. The epsilon arm
|
|
24
|
+
// exists because near-duplicate corpus rows (e.g. FACT's timestamped benchmark_report
|
|
25
|
+
// JSONs, ~95% identical text) crowd the podium while quantized batch-vs-single embed
|
|
26
|
+
// drift (~0.035 measured) exceeds the true gap between near-dups — the row IS stored and
|
|
27
|
+
// readable (fact.big id=722: rank 6, Δ0.007 behind rank 1), so that's a photo-finish,
|
|
28
|
+
// not a broken store. Photo-finish passes are still surfaced as a `note` so the
|
|
29
|
+
// near-dup-noise signal feeds the dedup backlog instead of being hidden. A row absent
|
|
30
|
+
// from top-10 (or far from the leader) remains a hard FAIL — missing/zero vectors and
|
|
31
|
+
// broken read paths cannot hide behind the epsilon.
|
|
32
|
+
// Proves embed-write AND read-path in one check. Retrieval QUALITY (real questions) stays
|
|
33
|
+
// forge-guard/prove's job; this gate proves the machinery, not the answers.
|
|
34
|
+
//
|
|
35
|
+
// Usage:
|
|
36
|
+
// node scripts/corpus-qa.mjs # whole corpus, structural + round-trip, serial
|
|
37
|
+
// node scripts/corpus-qa.mjs --store ruvector # one store (both variants)
|
|
38
|
+
// node scripts/corpus-qa.mjs --structural # cheap checks only
|
|
39
|
+
// [--dir <kb-dir>] [--samples N] # fixture/test hooks
|
|
40
|
+
//
|
|
41
|
+
// Output: one table row per store-variant, PASS/FAIL + reasons; skipped store classes are printed
|
|
42
|
+
// with why (nothing is skipped silently). Exit 1 if ANY row fails — self-update.mjs runs this per
|
|
43
|
+
// rebuilt store and aborts before publish on failure.
|
|
44
|
+
|
|
45
|
+
import fs from 'node:fs';
|
|
46
|
+
import path from 'node:path';
|
|
47
|
+
import readline from 'node:readline';
|
|
48
|
+
import { fileURLToPath } from 'node:url';
|
|
49
|
+
import { FULL_HINTS } from './full-hints.mjs';
|
|
50
|
+
|
|
51
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
52
|
+
const KB = path.join(ROOT, 'kb');
|
|
53
|
+
|
|
54
|
+
// resolve-deps lives in the real kb/ regardless of --dir (fixtures point --dir elsewhere).
|
|
55
|
+
const { loadRvf, loadTransformers, configureModel, chooseModelCache, closeReadonlyRvf } =
|
|
56
|
+
await import(path.join(KB, 'resolve-deps.mjs')).then((m) => m);
|
|
57
|
+
const { materializeModelRevision } =
|
|
58
|
+
await import(path.join(KB, 'model-requirements.mjs')).then((m) => m);
|
|
59
|
+
|
|
60
|
+
const FULL_BODY_MARK = '(full body):';
|
|
61
|
+
// R1 photo-finish epsilon: measured quantized batch-vs-single embed drift is ~0.035 cosine
|
|
62
|
+
// distance (fact.big id=722 replay); near-dup siblings sit within ~0.005 of each other. 0.02
|
|
63
|
+
// forgives the drift-scale tie WITHOUT forgiving genuinely different rows.
|
|
64
|
+
const NEAR_DUP_EPS = 0.02;
|
|
65
|
+
|
|
66
|
+
function fnv1a(s) {
|
|
67
|
+
let h = 0x811c9dc5;
|
|
68
|
+
for (let i = 0; i < s.length; i++) { h ^= s.charCodeAt(i); h = Math.imul(h, 0x01000193) >>> 0; }
|
|
69
|
+
return h >>> 0;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function readPassages(file) {
|
|
73
|
+
return new Promise((resolve, reject) => {
|
|
74
|
+
const rows = [];
|
|
75
|
+
const rl = readline.createInterface({ input: fs.createReadStream(file), crlfDelay: Infinity });
|
|
76
|
+
rl.on('line', (l) => { const s = l.trim(); if (!s) return; try { rows.push(JSON.parse(s)); } catch { /* counted structurally via row parse below */ } });
|
|
77
|
+
rl.on('close', () => resolve(rows));
|
|
78
|
+
rl.on('error', reject);
|
|
79
|
+
});
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// Deterministic k distinct sample indices for a store-variant. Same store => same samples, always.
|
|
83
|
+
export function sampleIndices(storeKey, n, k) {
|
|
84
|
+
const idx = new Set();
|
|
85
|
+
for (let salt = 0; idx.size < Math.min(k, n) && salt < 50 * k; salt++) idx.add(fnv1a(`${storeKey}:${salt}`) % n);
|
|
86
|
+
return [...idx].slice(0, Math.min(k, n));
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// ---- embedder cache: one pipeline per model, shared across stores. Caches the in-flight PROMISE
|
|
90
|
+
// (not just the resolved pipe) so concurrent qaStore() calls that both miss on the same model don't
|
|
91
|
+
// each trigger their own T.pipeline() load — first caller wins, the rest await its promise.
|
|
92
|
+
const pipelines = new Map();
|
|
93
|
+
function getPipeline(embedConf) {
|
|
94
|
+
const { model, revision } = embedConf;
|
|
95
|
+
const key = `${model}@${revision || 'unversioned'}`;
|
|
96
|
+
if (pipelines.has(key)) return pipelines.get(key);
|
|
97
|
+
const p = (async () => {
|
|
98
|
+
const { T } = await loadTransformers();
|
|
99
|
+
const cache = chooseModelCache();
|
|
100
|
+
materializeModelRevision(cache, model, revision);
|
|
101
|
+
configureModel(T, cache);
|
|
102
|
+
return T.pipeline('feature-extraction', model, {
|
|
103
|
+
quantized: true,
|
|
104
|
+
...(revision ? { revision } : {}),
|
|
105
|
+
});
|
|
106
|
+
})();
|
|
107
|
+
pipelines.set(key, p);
|
|
108
|
+
return p;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* QA one store-variant. Returns { store, variant, passages, fullBodies, vectors, roundtrip, fails, notes }.
|
|
113
|
+
* `fails` is [] on PASS; every entry is a specific reason string (receipts, not adjectives).
|
|
114
|
+
* `notes` are non-fatal signals (e.g. near-dup crowding) that should reach the dedup backlog.
|
|
115
|
+
*/
|
|
116
|
+
export async function qaStore(dir, store, variant, { roundtrip = true, samples = 3 } = {}) {
|
|
117
|
+
const base = variant === 'big' ? `${store}.big` : store;
|
|
118
|
+
const rvfPath = path.join(dir, `${base}.rvf`);
|
|
119
|
+
const variantPassages = path.join(dir, `${base}.passages.jsonl`);
|
|
120
|
+
const canonicalPassages = path.join(dir, `${store}.passages.jsonl`);
|
|
121
|
+
const passagesPath = variant === 'big' && !fs.existsSync(variantPassages)
|
|
122
|
+
? canonicalPassages
|
|
123
|
+
: variantPassages;
|
|
124
|
+
const embedPath = `${rvfPath}.embed.json`;
|
|
125
|
+
const res = { store, variant, passages: 0, fullBodies: 0, vectors: null, roundtrip: 'skipped', fails: [], notes: [] };
|
|
126
|
+
|
|
127
|
+
// S1: passages sidecar
|
|
128
|
+
if (!fs.existsSync(passagesPath)) { res.fails.push(`S1 missing ${path.basename(passagesPath)}`); return res; }
|
|
129
|
+
const rows = await readPassages(passagesPath);
|
|
130
|
+
res.passages = rows.length;
|
|
131
|
+
if (rows.length === 0) { res.fails.push('S1 passages file has 0 rows'); return res; }
|
|
132
|
+
|
|
133
|
+
// S2: full-body floor wherever hints exist (the 2026-07-10 failure class)
|
|
134
|
+
res.fullBodies = rows.filter((r) => typeof r.text === 'string' && r.text.includes(FULL_BODY_MARK)).length;
|
|
135
|
+
if (FULL_HINTS[store] && res.fullBodies === 0) {
|
|
136
|
+
res.fails.push(`S2 FULL_HINTS defines --full for "${store}" but store has 0 full-body passages (silent depth loss)`);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
// S3: vector count parity
|
|
140
|
+
let db = null;
|
|
141
|
+
try {
|
|
142
|
+
const { mod } = loadRvf();
|
|
143
|
+
db = await mod.RvfDatabase.openReadonly(rvfPath);
|
|
144
|
+
const st = await db.status();
|
|
145
|
+
res.vectors = st.totalVectors;
|
|
146
|
+
if (st.totalVectors !== rows.length) res.fails.push(`S3 vectors=${st.totalVectors} != passages=${rows.length}`);
|
|
147
|
+
} catch (e) {
|
|
148
|
+
res.fails.push(`S3 cannot open ${path.basename(rvfPath)}: ${e.message.split('\n')[0]}`);
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
// S4: query-side embed config
|
|
152
|
+
let embedConf = null;
|
|
153
|
+
if (!fs.existsSync(embedPath)) res.fails.push(`S4 missing ${path.basename(embedPath)} (read path cannot embed queries)`);
|
|
154
|
+
else embedConf = JSON.parse(fs.readFileSync(embedPath, 'utf8'));
|
|
155
|
+
|
|
156
|
+
// R1: deterministic self-retrieval round trip
|
|
157
|
+
if (roundtrip && db && embedConf && !res.fails.some((f) => f.startsWith('S3'))) {
|
|
158
|
+
try {
|
|
159
|
+
const picks = sampleIndices(`${store}.${variant}`, rows.length, samples);
|
|
160
|
+
const byId = new Map(rows.map((r) => [String(r.id), r]));
|
|
161
|
+
let hit = 0;
|
|
162
|
+
const misses = [];
|
|
163
|
+
const pipe = await getPipeline(embedConf);
|
|
164
|
+
for (const i of picks) {
|
|
165
|
+
const r = rows[i];
|
|
166
|
+
// Re-embed EXACTLY what the pipeline indexed for this variant (forge-build/forge-big):
|
|
167
|
+
const text = variant === 'big'
|
|
168
|
+
? r.text
|
|
169
|
+
: `${r.title} — ${r.path}\n${r.text}`.slice(0, 4300);
|
|
170
|
+
const out = await pipe([text], { pooling: embedConf.pooling || 'mean', normalize: true });
|
|
171
|
+
if (out.dims[1] !== embedConf.dimensions) throw new Error(`embed dim ${out.dims[1]} != ${embedConf.dimensions}`);
|
|
172
|
+
const top = await db.query(Array.from(out.data), 10);
|
|
173
|
+
const matches = (t) => String(t.id) === String(r.id)
|
|
174
|
+
|| byId.get(String(t.id))?.path === r.path
|
|
175
|
+
|| byId.get(String(t.id))?.text === r.text;
|
|
176
|
+
const rank = top.findIndex(matches); // -1 = absent from top-10
|
|
177
|
+
if (rank >= 0 && rank < 3) hit++;
|
|
178
|
+
else if (rank >= 0 && top[rank].distance - top[0].distance <= NEAR_DUP_EPS) {
|
|
179
|
+
hit++; // photo-finish behind near-duplicates: stored + readable; surface the crowd as a note
|
|
180
|
+
res.notes.push(`near-dup crowd: id=${r.id} rank ${rank + 1}, Δ${(top[rank].distance - top[0].distance).toFixed(4)} behind #1 (${r.path})`);
|
|
181
|
+
} else {
|
|
182
|
+
misses.push(`id=${r.id} ${rank < 0 ? 'ABSENT from top-10' : `rank ${rank + 1}, Δ${(top[rank].distance - top[0].distance).toFixed(4)}`} ${r.path}`);
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
res.roundtrip = `${hit}/${picks.length}`;
|
|
186
|
+
if (hit < picks.length) res.fails.push(`R1 self-retrieval missed: ${misses.join('; ')}`);
|
|
187
|
+
} catch (e) {
|
|
188
|
+
res.roundtrip = 'error';
|
|
189
|
+
res.fails.push(`R1 round-trip error: ${e.message.split('\n')[0]}`);
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
await closeReadonlyRvf(db);
|
|
193
|
+
return res;
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
/** Discover store-variants in a dir. Returns { stores: [{store, variant}], skipped: [{file, why}] }. */
|
|
197
|
+
export function discoverStores(dir) {
|
|
198
|
+
const stores = [];
|
|
199
|
+
const skipped = [];
|
|
200
|
+
for (const f of fs.readdirSync(dir).sort()) {
|
|
201
|
+
if (!f.endsWith('.rvf')) continue; // idmaps/embed.json/etc. are per-store sidecars, not stores
|
|
202
|
+
const name = f.slice(0, -'.rvf'.length);
|
|
203
|
+
if (name.endsWith('.big')) stores.push({ store: name.slice(0, -'.big'.length), variant: 'big' });
|
|
204
|
+
else stores.push({ store: name, variant: 'small' });
|
|
205
|
+
}
|
|
206
|
+
return { stores, skipped };
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
// ---------------- CLI ----------------
|
|
210
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
211
|
+
const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
|
|
212
|
+
const has = (f) => process.argv.includes(f);
|
|
213
|
+
const DIR = path.resolve(arg('--dir', KB));
|
|
214
|
+
const ONLY = arg('--store', null);
|
|
215
|
+
const STRUCTURAL_ONLY = has('--structural');
|
|
216
|
+
const SAMPLES = parseInt(arg('--samples', '3'), 10) || 3;
|
|
217
|
+
const CONC = Math.max(1, parseInt(arg('--concurrency', '4'), 10) || 4);
|
|
218
|
+
|
|
219
|
+
const { stores } = discoverStores(DIR);
|
|
220
|
+
const todo = ONLY ? stores.filter((s) => s.store === ONLY) : stores;
|
|
221
|
+
if (ONLY && todo.length === 0) { console.error(`[corpus-qa] no store named "${ONLY}" in ${DIR}`); process.exit(2); }
|
|
222
|
+
console.log(`[corpus-qa] ${todo.length} store-variant(s) in ${DIR}${ONLY ? ` (store=${ONLY})` : ''}${STRUCTURAL_ONLY ? ' [structural only]' : ' [structural + round-trip]'} — concurrency ${CONC}, one process`);
|
|
223
|
+
|
|
224
|
+
// qaStore() shares the in-process pipelines cache (getPipeline above) and each store-variant
|
|
225
|
+
// touches its own .rvf/.passages.jsonl files, so concurrent runs don't step on each other.
|
|
226
|
+
const results = new Array(todo.length);
|
|
227
|
+
let cursor = 0;
|
|
228
|
+
const runOne = async () => {
|
|
229
|
+
while (true) {
|
|
230
|
+
const i = cursor++;
|
|
231
|
+
if (i >= todo.length) return;
|
|
232
|
+
const { store, variant } = todo[i];
|
|
233
|
+
results[i] = await qaStore(DIR, store, variant, { roundtrip: !STRUCTURAL_ONLY, samples: SAMPLES });
|
|
234
|
+
}
|
|
235
|
+
};
|
|
236
|
+
await Promise.all(Array.from({ length: Math.min(CONC, todo.length) }, runOne));
|
|
237
|
+
|
|
238
|
+
const pad = (s, n) => String(s).padEnd(n);
|
|
239
|
+
console.log('\n' + pad('store', 26) + pad('variant', 8) + pad('passages', 10) + pad('full-b', 8) + pad('vectors', 9) + pad('roundtrip', 11) + 'verdict');
|
|
240
|
+
let failed = 0;
|
|
241
|
+
for (const r of results) {
|
|
242
|
+
const verdict = r.fails.length ? 'FAIL' : 'PASS';
|
|
243
|
+
if (r.fails.length) failed++;
|
|
244
|
+
console.log(pad(r.store, 26) + pad(r.variant, 8) + pad(r.passages, 10) + pad(r.fullBodies, 8) + pad(r.vectors ?? '?', 9) + pad(r.roundtrip, 11) + verdict);
|
|
245
|
+
for (const f of r.fails) console.log(' ↳ ' + f);
|
|
246
|
+
for (const n of r.notes) console.log(' · note: ' + n);
|
|
247
|
+
}
|
|
248
|
+
console.log(`\n[corpus-qa] ${results.length - failed}/${results.length} store-variants PASS${failed ? ` — ${failed} FAILED` : ''}`);
|
|
249
|
+
process.exit(failed ? 1 : 0);
|
|
250
|
+
}
|
|
@@ -0,0 +1,346 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// correction-detect-embed.mjs — an HONEST measurement of whether a DIFFERENT PRIMITIVE (a local
|
|
3
|
+
// embedding k-NN classifier) clears ADR-033 §2's floor (≥90% precision on ≥100 detections) where
|
|
4
|
+
// the lexical regex detector (scripts/correction-detect.mjs) did not.
|
|
5
|
+
//
|
|
6
|
+
// This is a MEASUREMENT SCRIPT, not a shipped feature. It does not wire into any hook, gate, or
|
|
7
|
+
// store. It answers one question for an owner decision: does swapping the primitive from regex to
|
|
8
|
+
// embedding similarity change the verdict on THIS corpus? See the accompanying report for the
|
|
9
|
+
// answer; this file is how the numbers in that report were produced, reproducibly.
|
|
10
|
+
//
|
|
11
|
+
// ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
12
|
+
// THE PRIMITIVE: k-NEAREST-NEIGHBOUR OVER LOCAL MiniLM EMBEDDINGS — NOT A TRAINED CLASSIFIER
|
|
13
|
+
//
|
|
14
|
+
// Grounded against how this repo already does embeddings (kb/forge-build.mjs, kb/resolve-deps.mjs):
|
|
15
|
+
// the SAME model (`Xenova/all-MiniLM-L6-v2`, 384-dim, mean-pooled, L2-normalized, pinned to the
|
|
16
|
+
// same HuggingFace revision the KB build uses) via the SAME local ONNX runtime
|
|
17
|
+
// (`@xenova/transformers`, resolved through `kb/resolve-deps.mjs`'s `loadTransformers()` /
|
|
18
|
+
// `configureModel()` — no network call when the model is already cached, per CLAUDE.md Rule 1: RVF/
|
|
19
|
+
// local-ONNX first, never an external embedding API). This script imports that resolver directly
|
|
20
|
+
// rather than re-implementing model loading, cache resolution, or the network-hang guard a second
|
|
21
|
+
// time — those are already solved once, correctly, in `kb/resolve-deps.mjs`.
|
|
22
|
+
//
|
|
23
|
+
// The classifier itself is deliberately the simplest thing that could work (Karpathy: minimum code,
|
|
24
|
+
// no speculative abstraction): embed each candidate as `PRECEDING ACTION: <summary> \n USER:
|
|
25
|
+
// <utterance>` (utterance ± the preceding-action context the task asked for), embed a small labelled
|
|
26
|
+
// reference set drawn ONLY from the TUNE split, and classify a holdout candidate by a similarity-
|
|
27
|
+
// weighted vote of its k nearest TUNE neighbours. k and the similarity floor are chosen by
|
|
28
|
+
// leave-one-out cross-validation ON THE TUNE SET ONLY, then frozen before touching holdout — the
|
|
29
|
+
// same no-leakage discipline `correction-detect-measure.mjs` uses for its file-level split.
|
|
30
|
+
//
|
|
31
|
+
// ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
32
|
+
// WHERE THE GROUND TRUTH COMES FROM
|
|
33
|
+
//
|
|
34
|
+
// `scripts/correction-detect-measure.mjs --dump-pool` already builds a candidate pool (Signal-1:
|
|
35
|
+
// adjacent to a preceding assistant action) and this task's own grounding pass narrowed it with the
|
|
36
|
+
// SAME loose superset lexical net that script's `broad-pool.mjs` companion applies (deliberately
|
|
37
|
+
// looser than the regex's own signals — a safety margin so a genuine correction is not excluded from
|
|
38
|
+
// the labelling pool just because it doesn't use the regex's exact vocabulary). That pool — 271
|
|
39
|
+
// candidates, 112 tune / 159 holdout, spanning ALL 1,328 transcripts — was hand-labelled by this
|
|
40
|
+
// task's author (true / borderline / false) against ADR-033's actual four-signal definition, NOT
|
|
41
|
+
// against what either detector happens to fire on. That hand-labelling is the same self-graded
|
|
42
|
+
// caveat every number in `correction-detect.mjs`'s own header already carries (Verification #6 in
|
|
43
|
+
// ADR-033: "not independently graded") — repeated here rather than hidden.
|
|
44
|
+
//
|
|
45
|
+
// The labelled pool contains real (if redacted-of-secrets) transcript text and is NOT committed to
|
|
46
|
+
// this repo, for the same reason `correction-detect-measure.mjs` never writes transcript text into
|
|
47
|
+
// the repo. Point `--labels` / `--tune-pool` / `--holdout-pool` at that data (default: this
|
|
48
|
+
// project's scratchpad locations used to build it) to reproduce the run. A small, hand-picked,
|
|
49
|
+
// already-public-in-spirit subset (utterances that already became named standing orders in this
|
|
50
|
+
// project's own committed memory index) ships as `tests/fixtures/correction-embed-sample.jsonl` so
|
|
51
|
+
// `tests/unit/correction-detect-embed.test.mjs` can run the classifier's MECHANICS without any
|
|
52
|
+
// private corpus present.
|
|
53
|
+
//
|
|
54
|
+
// ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
55
|
+
// ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
56
|
+
// MEASURED, 2026-07-23 — same 271-item hand-labelled pool (112 tune / 159 holdout, all 1,328
|
|
57
|
+
// transcripts) the total-genuine-correction count in the report was taken from. Reproduce with the
|
|
58
|
+
// `measure` command below, pointed at that pool + labels.json (see this file's own USAGE).
|
|
59
|
+
//
|
|
60
|
+
// tune-set ground truth: 14 true / 98 false (of 112)
|
|
61
|
+
// holdout ground truth: 23 true / 6 borderline / 130 false (of 159)
|
|
62
|
+
//
|
|
63
|
+
// Leave-one-out search on TUNE ONLY (no holdout leakage) picked k=1, minSim=0.6 (tune-LOO
|
|
64
|
+
// precision 66.7% / recall 42.9% — already far below the regex's tune-side 100%/4-of-4).
|
|
65
|
+
//
|
|
66
|
+
// Applied to the SAME 159-item holdout the regex's own 4 detections came from:
|
|
67
|
+
// flagged positive: 8
|
|
68
|
+
// true positives: 2
|
|
69
|
+
// false positives: 6 (0 of the 6 borderline-labelled rows were flagged)
|
|
70
|
+
// PRECISION: 25.0% (2/8)
|
|
71
|
+
// RECALL (of 29 true+borderline): 6.9%
|
|
72
|
+
//
|
|
73
|
+
// With the frozen default operating point below (k=5, minSim=0.3, chosen the same way but on an
|
|
74
|
+
// earlier tune-only search) instead of a fresh --tune-k run: 7 flagged, 2 true positives, 5 false
|
|
75
|
+
// — precision 28.6%, recall 6.9%. Same order of magnitude either way: roughly a QUARTER of what
|
|
76
|
+
// this primitive flags on the real holdout is a genuine correction, versus the regex's own
|
|
77
|
+
// 50-100% (n=4) on the same pool.
|
|
78
|
+
//
|
|
79
|
+
// Extended to the FULL Signal-1 holdout population (784 candidates, `--wide-pool`, not just the
|
|
80
|
+
// 159-item loose-net subset a human would ever be asked to label): 12 flagged, of which 5 fall
|
|
81
|
+
// OUTSIDE the labelled subset. Hand-reviewed those 5 for this measurement: all 5 are false
|
|
82
|
+
// (a repo-naming brainstorm, two raw image-paste captions, a status question, and a delegation
|
|
83
|
+
// statement) — so at realistic operational scale, precision is 2/12 ≈ 16.7%, recall unchanged.
|
|
84
|
+
//
|
|
85
|
+
// CONCLUSION: this primitive does NOT beat the regex on precision on this corpus, and comes
|
|
86
|
+
// nowhere near ADR-033 §2's ≥90% floor at any N tried. See the report this task produced for the
|
|
87
|
+
// full discussion (why: MiniLM sentence embeddings here separate by TOPIC — "this utterance is
|
|
88
|
+
// about AgentDB/versions/the console" — not by the PRAGMATIC property ADR-033 actually needs
|
|
89
|
+
// ("is this utterance correcting the agent's behaviour"). Two utterances about the same topic,
|
|
90
|
+
// one a bug report and one a correction, land close together in embedding space; two genuine
|
|
91
|
+
// corrections about DIFFERENT topics (READMEs vs. scoring vs. version pinning) often do not.
|
|
92
|
+
// This is the same conclusion ADR-033 reached about lexical overlap, now shown to also hold for
|
|
93
|
+
// semantic (embedding) overlap on this corpus — it is not a lexical-vs-semantic gap, it is a
|
|
94
|
+
// topic-vs-pragmatics gap that neither primitive, as tried, closes.
|
|
95
|
+
//
|
|
96
|
+
// ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
97
|
+
// USAGE
|
|
98
|
+
// node scripts/correction-detect-embed.mjs measure \
|
|
99
|
+
// --tune-pool <broad-tune.jsonl> --holdout-pool <broad-holdout.jsonl> --labels <labels.json> \
|
|
100
|
+
// [--wide-pool <cand-holdout.jsonl>] [--k 5] [--min-sim 0.35] [--tune-k]
|
|
101
|
+
//
|
|
102
|
+
// --wide-pool, optional: the FULL Signal-1 holdout pool (784 candidates, not just the 159-item
|
|
103
|
+
// loose-net subset) — runs the frozen classifier over it and reports how many candidates OUTSIDE
|
|
104
|
+
// the labelled subset it flags positive, honestly marked UNLABELLED rather than guessed at.
|
|
105
|
+
//
|
|
106
|
+
// --tune-k: run leave-one-out CV over a small (k, minSim) grid on the tune set only, print the
|
|
107
|
+
// chosen operating point, then use it. Without this flag the script uses the value already found
|
|
108
|
+
// this way and recorded in DEFAULT_K / DEFAULT_MIN_SIM below (reproducible without re-searching).
|
|
109
|
+
|
|
110
|
+
import fs from 'node:fs';
|
|
111
|
+
import path from 'node:path';
|
|
112
|
+
import { fileURLToPath } from 'node:url';
|
|
113
|
+
|
|
114
|
+
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
115
|
+
const KB_DIR = path.join(__dirname, '..', 'kb');
|
|
116
|
+
|
|
117
|
+
// Pinned to the exact commit forge-build.mjs pins to, so this script's vectors are byte-identical
|
|
118
|
+
// to the ones the shipped KB would produce for the same text (see forge-build.mjs's own comment on
|
|
119
|
+
// MINILM_REVISION for why: address-by-SHA, not floating `main`).
|
|
120
|
+
const MINILM_REVISION = '751bff37182d3f1213fa05d7196b954e230abad9';
|
|
121
|
+
|
|
122
|
+
// Found by `--tune-k` leave-one-out search over k in [1,3,5,7,9] and minSim in [0.0,0.15,...,0.6],
|
|
123
|
+
// maximizing tune-set F1 (ties broken toward higher minSim, i.e. more conservative — precision is
|
|
124
|
+
// the safety property here, per ADR-033 §2, so a tie goes to the pickier operating point). Re-run
|
|
125
|
+
// `--tune-k` to reproduce; this corpus is small (112 tune rows) so the search is seconds, not a
|
|
126
|
+
// separate offline step, but the chosen point is frozen here so a bare `measure` run is deterministic
|
|
127
|
+
// without depending on the search being re-run identically.
|
|
128
|
+
export const DEFAULT_K = 5;
|
|
129
|
+
export const DEFAULT_MIN_SIM = 0.3;
|
|
130
|
+
|
|
131
|
+
/** Resolve the local MiniLM embedder through this repo's OWN resolver — no re-implementation, no
|
|
132
|
+
* external API call (CLAUDE.md Rule 1). Fails loudly (via loadTransformers' own network guard) if
|
|
133
|
+
* neither a project node_modules nor an env override can find @xenova/transformers. */
|
|
134
|
+
async function loadEmbedder() {
|
|
135
|
+
const { loadTransformers, configureModel } = await import(path.join(KB_DIR, 'resolve-deps.mjs'));
|
|
136
|
+
const { T, modelCache, via } = await loadTransformers();
|
|
137
|
+
const { haveLocalModel } = configureModel(T, modelCache);
|
|
138
|
+
const embed = await T.pipeline('feature-extraction', 'Xenova/all-MiniLM-L6-v2', {
|
|
139
|
+
quantized: true, revision: MINILM_REVISION,
|
|
140
|
+
});
|
|
141
|
+
return { embed, via, modelCache, haveLocalModel };
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/** Utterance ± preceding-action context, exactly as the task specified — the same two fields
|
|
145
|
+
* `detectCorrection()` consumes (Signal 1's adjacency evidence), just embedded instead of regexed. */
|
|
146
|
+
export function candidateText(row) {
|
|
147
|
+
const prior = row.precedingAssistantAction || {};
|
|
148
|
+
const action = prior.summary || prior.tool || '';
|
|
149
|
+
return action
|
|
150
|
+
? `PRECEDING ACTION: ${String(action).slice(0, 200)}\nUSER: ${row.promptText}`
|
|
151
|
+
: `USER: ${row.promptText}`;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
async function embedBatch(embed, texts, batchSize = 16) {
|
|
155
|
+
const vectors = [];
|
|
156
|
+
for (let i = 0; i < texts.length; i += batchSize) {
|
|
157
|
+
const batch = texts.slice(i, i + batchSize);
|
|
158
|
+
const out = await embed(batch, { pooling: 'mean', normalize: true });
|
|
159
|
+
const dim = out.dims[1];
|
|
160
|
+
for (let j = 0; j < batch.length; j++) vectors.push(Array.from(out.data.slice(j * dim, (j + 1) * dim)));
|
|
161
|
+
}
|
|
162
|
+
return vectors;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/** Vectors are already L2-normalized (normalize:true above), so dot product IS cosine similarity —
|
|
166
|
+
* same convention kb/forge-build.mjs uses for its RVF store (metric:'cosine' over normalized vecs). */
|
|
167
|
+
function dot(a, b) {
|
|
168
|
+
let s = 0;
|
|
169
|
+
for (let i = 0; i < a.length; i++) s += a[i] * b[i];
|
|
170
|
+
return s;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* The classifier: similarity-weighted k-NN vote against a labelled reference set. Deliberately NOT
|
|
175
|
+
* a trained model (no gradient descent, no held weights beyond the reference vectors themselves) —
|
|
176
|
+
* this keeps every verdict traceable to "which labelled examples it resembles, and how much" rather
|
|
177
|
+
* than to opaque learned parameters. It is still less legible than the regex (see LEGIBILITY note
|
|
178
|
+
* at the bottom of this file), but it is the most legible embedding-based option available: the
|
|
179
|
+
* `neighbors` returned alongside every verdict ARE the explanation, not a post-hoc rationalization.
|
|
180
|
+
*/
|
|
181
|
+
export function classify(vec, refs, { k = DEFAULT_K, minSim = DEFAULT_MIN_SIM } = {}) {
|
|
182
|
+
const scored = refs.map((r) => ({ ...r, sim: dot(vec, r.vector) })).sort((a, b) => b.sim - a.sim);
|
|
183
|
+
const top = scored.slice(0, k).filter((r) => r.sim >= minSim);
|
|
184
|
+
if (!top.length) return { isCorrection: false, score: 0, neighbors: scored.slice(0, 3) };
|
|
185
|
+
const posWeight = top.filter((r) => r.label === 'true').reduce((s, r) => s + r.sim, 0);
|
|
186
|
+
const negWeight = top.filter((r) => r.label !== 'true').reduce((s, r) => s + r.sim, 0);
|
|
187
|
+
return { isCorrection: posWeight > negWeight, score: posWeight - negWeight, neighbors: top.slice(0, 3) };
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/** Leave-one-out CV over a small grid, TUNE SET ONLY — never touches holdout. Maximizes F1; ties
|
|
191
|
+
* broken toward the higher minSim (more conservative), per ADR-033's precision-over-recall stance. */
|
|
192
|
+
function looSearch(tuneVecs, tuneLabels) {
|
|
193
|
+
const refs = tuneVecs.map((vector, i) => ({ vector, label: tuneLabels[i] }));
|
|
194
|
+
let best = null;
|
|
195
|
+
for (const k of [1, 3, 5, 7, 9]) {
|
|
196
|
+
for (const minSim of [0, 0.1, 0.15, 0.2, 0.25, 0.3, 0.35, 0.4, 0.45, 0.5, 0.55, 0.6]) {
|
|
197
|
+
let tp = 0, fp = 0, fn = 0;
|
|
198
|
+
for (let i = 0; i < refs.length; i++) {
|
|
199
|
+
const others = refs.slice(0, i).concat(refs.slice(i + 1));
|
|
200
|
+
const { isCorrection } = classify(refs[i].vector, others, { k, minSim });
|
|
201
|
+
const truth = refs[i].label === 'true';
|
|
202
|
+
if (isCorrection && truth) tp++;
|
|
203
|
+
else if (isCorrection && !truth) fp++;
|
|
204
|
+
else if (!isCorrection && truth) fn++;
|
|
205
|
+
}
|
|
206
|
+
const precision = tp + fp ? tp / (tp + fp) : 0;
|
|
207
|
+
const recall = tp + fn ? tp / (tp + fn) : 0;
|
|
208
|
+
const f1 = precision + recall ? (2 * precision * recall) / (precision + recall) : 0;
|
|
209
|
+
const cand = { k, minSim, tp, fp, fn, precision, recall, f1 };
|
|
210
|
+
if (!best || f1 > best.f1 || (f1 === best.f1 && minSim > best.minSim)) best = cand;
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
return best;
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
// ── I/O helpers ──────────────────────────────────────────────────────────────────────────────────
|
|
217
|
+
|
|
218
|
+
function readJsonl(file) {
|
|
219
|
+
return fs.readFileSync(file, 'utf8').split('\n').filter(Boolean).map((l) => JSON.parse(l));
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
function loadLabels(file) {
|
|
223
|
+
const rows = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
224
|
+
const map = new Map();
|
|
225
|
+
for (const r of rows) map.set(`${r.file}#${r.turnIndex}`, r.label);
|
|
226
|
+
return map;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
function flag(name, fallback = null) {
|
|
230
|
+
const i = process.argv.indexOf(name);
|
|
231
|
+
return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback;
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
// ── Measurement ──────────────────────────────────────────────────────────────────────────────────
|
|
235
|
+
|
|
236
|
+
async function measure() {
|
|
237
|
+
const tunePoolFile = flag('--tune-pool');
|
|
238
|
+
const holdoutPoolFile = flag('--holdout-pool');
|
|
239
|
+
const labelsFile = flag('--labels');
|
|
240
|
+
const widePoolFile = flag('--wide-pool');
|
|
241
|
+
const doSearch = process.argv.includes('--tune-k');
|
|
242
|
+
|
|
243
|
+
if (!tunePoolFile || !holdoutPoolFile || !labelsFile) {
|
|
244
|
+
console.error('Usage: node scripts/correction-detect-embed.mjs measure --tune-pool <jsonl> '
|
|
245
|
+
+ '--holdout-pool <jsonl> --labels <labels.json> [--wide-pool <jsonl>] [--tune-k]');
|
|
246
|
+
console.error('These point at real (private) transcript-derived data — see this file\'s header '
|
|
247
|
+
+ 'for how to regenerate them; nothing of that shape is committed to this repo.');
|
|
248
|
+
process.exit(1);
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
const labels = loadLabels(labelsFile);
|
|
252
|
+
const tuneRows = readJsonl(tunePoolFile).map((r) => ({ ...r, label: labels.get(`${r.file}#${r.turnIndex}`) || 'false' }));
|
|
253
|
+
const holdoutRows = readJsonl(holdoutPoolFile).map((r) => ({ ...r, label: labels.get(`${r.file}#${r.turnIndex}`) || 'false' }));
|
|
254
|
+
|
|
255
|
+
console.error(`[embed] tune=${tuneRows.length} (true=${tuneRows.filter(r=>r.label==='true').length}) `
|
|
256
|
+
+ `holdout=${holdoutRows.length} (true=${holdoutRows.filter(r=>r.label==='true').length}, `
|
|
257
|
+
+ `borderline=${holdoutRows.filter(r=>r.label==='borderline').length})`);
|
|
258
|
+
|
|
259
|
+
const { embed, via, haveLocalModel, modelCache } = await loadEmbedder();
|
|
260
|
+
console.error(`[embed] transformers via: ${via} | model: ${haveLocalModel ? 'local cache' : 'REMOTE DOWNLOAD'} (${modelCache})`);
|
|
261
|
+
|
|
262
|
+
const tuneVecs = await embedBatch(embed, tuneRows.map(candidateText));
|
|
263
|
+
const holdoutVecs = await embedBatch(embed, holdoutRows.map(candidateText));
|
|
264
|
+
|
|
265
|
+
let opPoint = { k: DEFAULT_K, minSim: DEFAULT_MIN_SIM };
|
|
266
|
+
if (doSearch) {
|
|
267
|
+
const found = looSearch(tuneVecs, tuneRows.map((r) => r.label));
|
|
268
|
+
console.error(`[embed] --tune-k search (leave-one-out, tune only): k=${found.k} minSim=${found.minSim} `
|
|
269
|
+
+ `tune-LOO precision=${(100*found.precision).toFixed(1)}% recall=${(100*found.recall).toFixed(1)}% f1=${found.f1.toFixed(3)}`);
|
|
270
|
+
opPoint = { k: found.k, minSim: found.minSim };
|
|
271
|
+
} else {
|
|
272
|
+
console.error(`[embed] using frozen operating point k=${opPoint.k} minSim=${opPoint.minSim} (rerun with --tune-k to re-search)`);
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
const refs = tuneVecs.map((vector, i) => ({ vector, label: tuneRows[i].label, text: tuneRows[i].promptText }));
|
|
276
|
+
|
|
277
|
+
// ── Primary measurement: SAME 159-item labelled holdout population the regex's own examples
|
|
278
|
+
// came from (broad-holdout.jsonl) — apples-to-apples with correction-detect.mjs's own numbers. ──
|
|
279
|
+
let tp = 0, fp = 0, fpBorderline = 0, fn = 0, tn = 0;
|
|
280
|
+
const positives = [];
|
|
281
|
+
for (let i = 0; i < holdoutRows.length; i++) {
|
|
282
|
+
const { isCorrection, score, neighbors } = classify(holdoutVecs[i], refs, opPoint);
|
|
283
|
+
const row = holdoutRows[i];
|
|
284
|
+
if (isCorrection) {
|
|
285
|
+
positives.push({ ...row, score, neighbors: neighbors.map((n) => ({ label: n.label, sim: n.sim.toFixed(3), text: n.text.slice(0, 100) })) });
|
|
286
|
+
if (row.label === 'true') tp++;
|
|
287
|
+
else if (row.label === 'borderline') { fpBorderline++; }
|
|
288
|
+
else fp++;
|
|
289
|
+
} else {
|
|
290
|
+
if (row.label === 'true' || row.label === 'borderline') fn++;
|
|
291
|
+
else tn++;
|
|
292
|
+
}
|
|
293
|
+
}
|
|
294
|
+
const totalFp = fp + fpBorderline; // strict: borderline counts against precision, same as the regex's own header treats its 2 holdout borderlines
|
|
295
|
+
const precisionStrict = tp + totalFp ? tp / (tp + totalFp) : 0;
|
|
296
|
+
const precisionLenient = (tp + fpBorderline) + fp ? (tp + fpBorderline) / (tp + fpBorderline + fp) : 0;
|
|
297
|
+
const recallStrict = tp + fn ? tp / (tp + fn) : 0; // fn includes borderlines missed, conservative
|
|
298
|
+
|
|
299
|
+
console.log(`\n=== EMBEDDING CLASSIFIER — holdout (n=${holdoutRows.length}, same pool the regex's hand-labelled examples came from) ===`);
|
|
300
|
+
console.log(`positives (flagged): ${positives.length}`);
|
|
301
|
+
console.log(` true positives: ${tp}`);
|
|
302
|
+
console.log(` borderline positives: ${fpBorderline}`);
|
|
303
|
+
console.log(` false positives: ${fp}`);
|
|
304
|
+
console.log(` false negatives (missed true+borderline): ${fn}`);
|
|
305
|
+
console.log(`precision (strict, borderline counts against): ${(100*precisionStrict).toFixed(1)}% (${tp}/${tp+totalFp})`);
|
|
306
|
+
console.log(`precision (lenient, borderline counts for): ${(100*precisionLenient).toFixed(1)}% (${tp+fpBorderline}/${tp+fpBorderline+fp})`);
|
|
307
|
+
console.log(`recall (of ${tp+fn+ (holdoutRows.filter(r=>r.label==='true'||r.label==='borderline').length - (tp+fn))} known true/borderline): ${(100*recallStrict).toFixed(1)}%`);
|
|
308
|
+
|
|
309
|
+
console.log(`\n--- flagged positives, with nearest tune neighbours (the "legibility" a verdict can offer) ---`);
|
|
310
|
+
for (const p of positives) {
|
|
311
|
+
console.log(`[${p.label.toUpperCase()}] score=${p.score.toFixed(3)} file=${p.file} turn=${p.turnIndex}`);
|
|
312
|
+
console.log(` UTTERANCE: ${p.promptText.slice(0, 160).replace(/\n/g,' ')}`);
|
|
313
|
+
for (const n of p.neighbors) console.log(` neighbor(${n.label}, sim=${n.sim}): ${n.text.replace(/\n/g,' ')}`);
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
// ── Secondary, optional: the FULL Signal-1 holdout population (not just the loose-net subset) —
|
|
317
|
+
// the real deployment-scale test. Anything flagged OUTSIDE the labelled 159 is marked UNLABELLED,
|
|
318
|
+
// never silently assumed either way. ──
|
|
319
|
+
if (widePoolFile) {
|
|
320
|
+
const wideRows = readJsonl(widePoolFile);
|
|
321
|
+
const labelledKeys = new Set(holdoutRows.map((r) => `${r.file}#${r.turnIndex}`));
|
|
322
|
+
const wideVecs = await embedBatch(embed, wideRows.map(candidateText));
|
|
323
|
+
let wideFlagged = 0, outsideSubset = 0;
|
|
324
|
+
const outsideHits = [];
|
|
325
|
+
for (let i = 0; i < wideRows.length; i++) {
|
|
326
|
+
const { isCorrection, score } = classify(wideVecs[i], refs, opPoint);
|
|
327
|
+
if (!isCorrection) continue;
|
|
328
|
+
wideFlagged++;
|
|
329
|
+
const k = `${wideRows[i].file}#${wideRows[i].turnIndex}`;
|
|
330
|
+
if (!labelledKeys.has(k)) { outsideSubset++; outsideHits.push({ ...wideRows[i], score }); }
|
|
331
|
+
}
|
|
332
|
+
console.log(`\n=== WIDE HOLDOUT (n=${wideRows.length}, full Signal-1 pool, not just the loose-net subset) ===`);
|
|
333
|
+
console.log(`flagged: ${wideFlagged} (of which ${outsideSubset} fall OUTSIDE the 159-item labelled subset — UNLABELLED, hand-review needed, not counted in precision/recall above)`);
|
|
334
|
+
for (const h of outsideHits.slice(0, 20)) {
|
|
335
|
+
console.log(` UNLABELLED file=${h.file} turn=${h.turnIndex} score=${h.score.toFixed(3)}: ${h.promptText.slice(0,160).replace(/\n/g,' ')}`);
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
const invokedDirectly = process.argv[1]
|
|
341
|
+
&& path.resolve(process.argv[1]).endsWith(`correction-detect-embed${path.extname(process.argv[1])}`);
|
|
342
|
+
if (invokedDirectly) {
|
|
343
|
+
const cmd = process.argv[2];
|
|
344
|
+
if (cmd === 'measure') measure().catch((e) => { console.error(e); process.exit(1); });
|
|
345
|
+
else { console.error('Usage: node scripts/correction-detect-embed.mjs measure ...'); process.exit(1); }
|
|
346
|
+
}
|