@monoes/monomindcli 2.7.12 → 2.7.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/mastermind/createorg.md +2 -0
- package/.claude/commands/mastermind/okf-import.md +6 -1
- package/.claude/helpers/handlers/route-handler.cjs +40 -7
- package/.claude/helpers/handlers/session-restore-handler.cjs +24 -6
- package/.claude/skills/mastermind-createorg/SKILL.md +11 -4
- package/README.md +12 -6
- package/dist/src/capabilities/index.d.ts.map +1 -1
- package/dist/src/capabilities/index.js +17 -0
- package/dist/src/capabilities/index.js.map +1 -1
- package/dist/src/capabilities/types.d.ts +0 -11
- package/dist/src/capabilities/types.d.ts.map +1 -1
- package/dist/src/commands/doc.d.ts.map +1 -1
- package/dist/src/commands/doc.js +251 -8
- package/dist/src/commands/doc.js.map +1 -1
- package/dist/src/commands/doctor-project-checks.d.ts +25 -3
- package/dist/src/commands/doctor-project-checks.d.ts.map +1 -1
- package/dist/src/commands/doctor-project-checks.js +111 -8
- package/dist/src/commands/doctor-project-checks.js.map +1 -1
- package/dist/src/commands/doctor.d.ts.map +1 -1
- package/dist/src/commands/doctor.js +18 -2
- package/dist/src/commands/doctor.js.map +1 -1
- package/dist/src/commands/init.d.ts.map +1 -1
- package/dist/src/commands/init.js +32 -3
- package/dist/src/commands/init.js.map +1 -1
- package/dist/src/commands/monograph.d.ts.map +1 -1
- package/dist/src/commands/monograph.js +26 -3
- package/dist/src/commands/monograph.js.map +1 -1
- package/dist/src/commands/org.d.ts +15 -0
- package/dist/src/commands/org.d.ts.map +1 -1
- package/dist/src/commands/org.js +251 -6
- package/dist/src/commands/org.js.map +1 -1
- package/dist/src/init/claudemd-generator.d.ts.map +1 -1
- package/dist/src/init/claudemd-generator.js +4 -1
- package/dist/src/init/claudemd-generator.js.map +1 -1
- package/dist/src/knowledge/document-pipeline.d.ts +77 -4
- package/dist/src/knowledge/document-pipeline.d.ts.map +1 -1
- package/dist/src/knowledge/document-pipeline.js +261 -15
- package/dist/src/knowledge/document-pipeline.js.map +1 -1
- package/dist/src/knowledge/eval/corpus.d.ts +56 -0
- package/dist/src/knowledge/eval/corpus.d.ts.map +1 -0
- package/dist/src/knowledge/eval/corpus.js +126 -0
- package/dist/src/knowledge/eval/corpus.js.map +1 -0
- package/dist/src/knowledge/eval/golden-set.d.ts +62 -0
- package/dist/src/knowledge/eval/golden-set.d.ts.map +1 -0
- package/dist/src/knowledge/eval/golden-set.js +331 -0
- package/dist/src/knowledge/eval/golden-set.js.map +1 -0
- package/dist/src/knowledge/eval/harness.d.ts +221 -0
- package/dist/src/knowledge/eval/harness.d.ts.map +1 -0
- package/dist/src/knowledge/eval/harness.js +610 -0
- package/dist/src/knowledge/eval/harness.js.map +1 -0
- package/dist/src/knowledge/eval/metrics.d.ts +120 -0
- package/dist/src/knowledge/eval/metrics.d.ts.map +1 -0
- package/dist/src/knowledge/eval/metrics.js +243 -0
- package/dist/src/knowledge/eval/metrics.js.map +1 -0
- package/dist/src/knowledge/eval/model-presence.d.ts +48 -0
- package/dist/src/knowledge/eval/model-presence.d.ts.map +1 -0
- package/dist/src/knowledge/eval/model-presence.js +128 -0
- package/dist/src/knowledge/eval/model-presence.js.map +1 -0
- package/dist/src/knowledge/eval/network-guard.d.ts +46 -0
- package/dist/src/knowledge/eval/network-guard.d.ts.map +1 -0
- package/dist/src/knowledge/eval/network-guard.js +112 -0
- package/dist/src/knowledge/eval/network-guard.js.map +1 -0
- package/dist/src/knowledge/eval/retrievers.d.ts +65 -0
- package/dist/src/knowledge/eval/retrievers.d.ts.map +1 -0
- package/dist/src/knowledge/eval/retrievers.js +180 -0
- package/dist/src/knowledge/eval/retrievers.js.map +1 -0
- package/dist/src/knowledge/eval/signals.d.ts +111 -0
- package/dist/src/knowledge/eval/signals.d.ts.map +1 -0
- package/dist/src/knowledge/eval/signals.js +232 -0
- package/dist/src/knowledge/eval/signals.js.map +1 -0
- package/dist/src/mcp-tools/agent-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/agent-tools.js +7 -34
- package/dist/src/mcp-tools/agent-tools.js.map +1 -1
- package/dist/src/mcp-tools/knowledge-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/knowledge-tools.js +88 -6
- package/dist/src/mcp-tools/knowledge-tools.js.map +1 -1
- package/dist/src/mcp-tools/quality/coverage-analysis/prioritize-gaps.d.ts +44 -144
- package/dist/src/mcp-tools/quality/coverage-analysis/prioritize-gaps.d.ts.map +1 -1
- package/dist/src/mcp-tools/quality/security-compliance/detect-secrets.d.ts +30 -38
- package/dist/src/mcp-tools/quality/security-compliance/detect-secrets.d.ts.map +1 -1
- package/dist/src/mcp-tools/task-tools.d.ts +1 -0
- package/dist/src/mcp-tools/task-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/task-tools.js +76 -111
- package/dist/src/mcp-tools/task-tools.js.map +1 -1
- package/dist/src/mcp-tools/terminal-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/terminal-tools.js +3 -28
- package/dist/src/mcp-tools/terminal-tools.js.map +1 -1
- package/dist/src/memory/bm25-index.d.ts +137 -0
- package/dist/src/memory/bm25-index.d.ts.map +1 -0
- package/dist/src/memory/bm25-index.js +219 -0
- package/dist/src/memory/bm25-index.js.map +1 -0
- package/dist/src/memory/embedding-operations.d.ts.map +1 -1
- package/dist/src/memory/embedding-operations.js +8 -7
- package/dist/src/memory/embedding-operations.js.map +1 -1
- package/dist/src/memory/ewc-consolidation.d.ts.map +1 -1
- package/dist/src/memory/ewc-consolidation.js +2 -1
- package/dist/src/memory/ewc-consolidation.js.map +1 -1
- package/dist/src/memory/hnsw-operations.d.ts.map +1 -1
- package/dist/src/memory/hnsw-operations.js +4 -3
- package/dist/src/memory/hnsw-operations.js.map +1 -1
- package/dist/src/memory/intelligence.d.ts.map +1 -1
- package/dist/src/memory/intelligence.js +2 -1
- package/dist/src/memory/intelligence.js.map +1 -1
- package/dist/src/memory/memory-bridge.d.ts +34 -0
- package/dist/src/memory/memory-bridge.d.ts.map +1 -1
- package/dist/src/memory/memory-bridge.js +184 -26
- package/dist/src/memory/memory-bridge.js.map +1 -1
- package/dist/src/memory/memory-initializer.d.ts.map +1 -1
- package/dist/src/memory/memory-initializer.js +4 -3
- package/dist/src/memory/memory-initializer.js.map +1 -1
- package/dist/src/memory/memory-schema.d.ts +1 -1
- package/dist/src/memory/memory-schema.js +1 -1
- package/dist/src/memory/text-tokens.d.ts +10 -0
- package/dist/src/memory/text-tokens.d.ts.map +1 -0
- package/dist/src/memory/text-tokens.js +37 -0
- package/dist/src/memory/text-tokens.js.map +1 -0
- package/dist/src/orgrt/daemon.d.ts +16 -1
- package/dist/src/orgrt/daemon.d.ts.map +1 -1
- package/dist/src/orgrt/daemon.js +65 -9
- package/dist/src/orgrt/daemon.js.map +1 -1
- package/dist/src/orgrt/scheduler.d.ts +14 -1
- package/dist/src/orgrt/scheduler.d.ts.map +1 -1
- package/dist/src/orgrt/scheduler.js +51 -5
- package/dist/src/orgrt/scheduler.js.map +1 -1
- package/dist/src/orgrt/session.js +12 -0
- package/dist/src/orgrt/session.js.map +1 -1
- package/dist/src/orgrt/types.d.ts +61 -831
- package/dist/src/orgrt/types.d.ts.map +1 -1
- package/dist/src/orgrt/types.js +7 -1
- package/dist/src/orgrt/types.js.map +1 -1
- package/dist/src/services/crash-reporter.d.ts.map +1 -1
- package/dist/src/services/crash-reporter.js +7 -0
- package/dist/src/services/crash-reporter.js.map +1 -1
- package/dist/src/ui/dashboard.html +57 -25
- package/dist/src/ui/orgs.html +30 -13
- package/dist/src/ui/routes-monograph.mjs +929 -0
- package/dist/src/ui/routes-org.mjs +2513 -0
- package/dist/src/ui/server.mjs +726 -4117
- package/dist/src/utils/json-file.d.ts +11 -0
- package/dist/src/utils/json-file.d.ts.map +1 -1
- package/dist/src/utils/json-file.js +27 -0
- package/dist/src/utils/json-file.js.map +1 -1
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/package.json +8 -6
|
@@ -0,0 +1,610 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `monomind doc eval` — the Second Brain scoreboard.
|
|
3
|
+
*
|
|
4
|
+
* This harness is the arbiter of the retrieval work. Its job is to be HOSTILE
|
|
5
|
+
* to its own result. Everything it does that could flatter the numbers is
|
|
6
|
+
* either disabled or reported:
|
|
7
|
+
*
|
|
8
|
+
* - Vacuous-eval assert: k must be a small fraction of the corpus. A retrieval
|
|
9
|
+
* window near corpus size makes recall 1.0 by construction. Hard failure.
|
|
10
|
+
* - Weak baselines: the same golden set is run through a seeded random picker
|
|
11
|
+
* and a plain BM25 scorer. The GAP is the signal; a high random score is the
|
|
12
|
+
* signature of a vacuous eval, and a high BM25 score means the set is too easy.
|
|
13
|
+
* - Anti-triviality: pairs whose query is near-verbatim in the target are
|
|
14
|
+
* dropped and counted, because those measure string matching.
|
|
15
|
+
* - Short-return instrumentation: a query that gets back fewer than k results
|
|
16
|
+
* cannot support an @k metric; those are counted and reported.
|
|
17
|
+
* - Network guard: the network is BLOCKED during the query phase, not assumed
|
|
18
|
+
* absent. Any attempt is recorded with its stack.
|
|
19
|
+
* - Live-doc pinning: the eval store is rebuilt from the corpus with exactly
|
|
20
|
+
* one ingest per document, so it holds no superseded versions.
|
|
21
|
+
*
|
|
22
|
+
* @module v1/cli/knowledge/eval/harness
|
|
23
|
+
*/
|
|
24
|
+
import * as fs from 'node:fs';
|
|
25
|
+
import * as os from 'node:os';
|
|
26
|
+
import * as path from 'node:path';
|
|
27
|
+
import { buildCorpus, readDoc, resolveRepoRoot } from './corpus.js';
|
|
28
|
+
import { GOLDEN_SET, pairsForSplit, SPLIT_SCHEME } from './golden-set.js';
|
|
29
|
+
import { aggregate, assessTriviality, buildIdf, dedupeByDoc, idfOverlap, scoreQuery, terciles, } from './metrics.js';
|
|
30
|
+
import { Bm25Retriever, FnRetriever, RandomRetriever, RrfRetriever } from './retrievers.js';
|
|
31
|
+
import { installNetworkGuard } from './network-guard.js';
|
|
32
|
+
import { scoreSignals } from './signals.js';
|
|
33
|
+
import { assertModelProvisioned } from './model-presence.js';
|
|
34
|
+
/** k must be at most this share of the corpus, else the eval is vacuous. */
|
|
35
|
+
export const MAX_K_CORPUS_RATIO = 0.05;
|
|
36
|
+
function detectDbDriver() {
|
|
37
|
+
try {
|
|
38
|
+
const req = createRequire(import.meta.url);
|
|
39
|
+
req.resolve('better-sqlite3');
|
|
40
|
+
try {
|
|
41
|
+
req('better-sqlite3');
|
|
42
|
+
return 'better-sqlite3';
|
|
43
|
+
}
|
|
44
|
+
catch {
|
|
45
|
+
return 'sql.js (better-sqlite3 present but failed to load)';
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
catch {
|
|
49
|
+
return 'sql.js (WASM fallback)';
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
import { createRequire } from 'node:module';
|
|
53
|
+
async function buildChunks(docs) {
|
|
54
|
+
let chunker = null;
|
|
55
|
+
try {
|
|
56
|
+
const mem = await import('@monoes/memory');
|
|
57
|
+
if (typeof mem.chunkDocument === 'function')
|
|
58
|
+
chunker = mem.chunkDocument;
|
|
59
|
+
}
|
|
60
|
+
catch { /* fall through to whole-document chunks */ }
|
|
61
|
+
const out = [];
|
|
62
|
+
for (const d of docs) {
|
|
63
|
+
const text = readDoc(d);
|
|
64
|
+
if (!chunker) {
|
|
65
|
+
out.push({ docId: d.id, chunkIndex: 0, text });
|
|
66
|
+
continue;
|
|
67
|
+
}
|
|
68
|
+
const chunks = await chunker(d.id, text);
|
|
69
|
+
const list = Array.isArray(chunks) ? chunks : [];
|
|
70
|
+
if (list.length === 0) {
|
|
71
|
+
out.push({ docId: d.id, chunkIndex: 0, text });
|
|
72
|
+
continue;
|
|
73
|
+
}
|
|
74
|
+
for (const c of list)
|
|
75
|
+
out.push({ docId: d.id, chunkIndex: c.chunkIndex ?? 0, text: c.text ?? '' });
|
|
76
|
+
}
|
|
77
|
+
return out;
|
|
78
|
+
}
|
|
79
|
+
export async function runEval(opts) {
|
|
80
|
+
const k = opts.k ?? 10;
|
|
81
|
+
const split = opts.split ?? 'dev';
|
|
82
|
+
const sealed = split === 'test';
|
|
83
|
+
const progress = opts.onProgress ?? (() => { });
|
|
84
|
+
const t0 = Date.now();
|
|
85
|
+
// Telemetry off for the duration. Condition (a) of the ruled carve-out: a
|
|
86
|
+
// "0 attempts" verdict must be a fact about retrieval, not a coincidence.
|
|
87
|
+
const prevCrash = process.env.MONOMIND_CRASH_REPORTING;
|
|
88
|
+
process.env.MONOMIND_CRASH_REPORTING = 'off';
|
|
89
|
+
// FIRST, before corpus construction and before ANY dynamic import. The weights
|
|
90
|
+
// must already be on disk. The eval never fetches: doing so would be a network
|
|
91
|
+
// call at query time and would make the run non-reproducible. This assert sat
|
|
92
|
+
// after the chunk mirror in its first version, and the chunk mirror's own
|
|
93
|
+
// import of @monoes/memory was what triggered an 89MB download — so the guard
|
|
94
|
+
// ran after the event it existed to prevent.
|
|
95
|
+
const modelPresence = assertModelProvisioned([
|
|
96
|
+
resolveRepoRoot(opts.repoRoot),
|
|
97
|
+
path.resolve(new URL('../../../..', import.meta.url).pathname),
|
|
98
|
+
process.cwd(),
|
|
99
|
+
]);
|
|
100
|
+
// ── 1. Corpus ────────────────────────────────────────────────────
|
|
101
|
+
const repoRoot = resolveRepoRoot(opts.repoRoot);
|
|
102
|
+
const corpus = buildCorpus(repoRoot);
|
|
103
|
+
if (corpus.appleDoubleCount > 0) {
|
|
104
|
+
throw new Error(`[doc eval] ${corpus.appleDoubleCount} AppleDouble "._" resource-fork files are in the eval corpus. ` +
|
|
105
|
+
`These are binary junk that reads as markdown and pads the document count without being real. Corpus rejected.`);
|
|
106
|
+
}
|
|
107
|
+
const ratio = corpus.contentUnits === 0 ? 1 : k / corpus.contentUnits;
|
|
108
|
+
// Vacuous-eval assert. A retrieval window that approaches corpus size makes
|
|
109
|
+
// recall 1.0 by construction — the single most common published error in
|
|
110
|
+
// this field. Hard failure, never a warning.
|
|
111
|
+
if (ratio > MAX_K_CORPUS_RATIO) {
|
|
112
|
+
throw new Error(`[doc eval] VACUOUS EVAL REFUSED: k=${k} against a ${corpus.contentUnits}-document corpus ` +
|
|
113
|
+
`is ${(ratio * 100).toFixed(1)}% of the corpus (limit ${(MAX_K_CORPUS_RATIO * 100).toFixed(0)}%). ` +
|
|
114
|
+
`At this ratio recall approaches 1.0 by construction and measures nothing. ` +
|
|
115
|
+
`Grow the corpus or lower k.`);
|
|
116
|
+
}
|
|
117
|
+
progress(`corpus: ${corpus.docs.length} files -> ${corpus.contentUnits} distinct documents ` +
|
|
118
|
+
`(${corpus.duplicateGroups} byte-identical groups collapsed, hash ${corpus.corpusHash})`);
|
|
119
|
+
const byId = new Map(corpus.docs.map(d => [d.id, d]));
|
|
120
|
+
/** Map any document path onto its content-unit representative. */
|
|
121
|
+
const canon = (id) => corpus.canonicalOf.get(id) ?? id;
|
|
122
|
+
// ── 2. Golden-set validation + anti-triviality ───────────────────
|
|
123
|
+
const scored = [];
|
|
124
|
+
const dropped = [];
|
|
125
|
+
const docTextCache = new Map();
|
|
126
|
+
const textOf = (id) => {
|
|
127
|
+
let t = docTextCache.get(id);
|
|
128
|
+
if (t === undefined) {
|
|
129
|
+
t = readDoc(byId.get(id));
|
|
130
|
+
docTextCache.set(id, t);
|
|
131
|
+
}
|
|
132
|
+
return t;
|
|
133
|
+
};
|
|
134
|
+
const candidatePairs = pairsForSplit(split);
|
|
135
|
+
for (const pair of candidatePairs) {
|
|
136
|
+
const unknown = pair.relevant.filter(r => !byId.has(r));
|
|
137
|
+
if (unknown.length > 0) {
|
|
138
|
+
// Never a silent skip: a golden set pointing at documents the corpus does
|
|
139
|
+
// not contain is a broken set, and a broken set produces a fake number.
|
|
140
|
+
throw new Error(`[doc eval] golden pair "${pair.id}" references documents not in the corpus: ${unknown.join(', ')}`);
|
|
141
|
+
}
|
|
142
|
+
let worst = { trivial: false, reason: '', maxContiguousRun: 0, overlapRatio: 0 };
|
|
143
|
+
for (const r of pair.relevant) {
|
|
144
|
+
const t = assessTriviality(pair.query, textOf(r));
|
|
145
|
+
if (t.maxContiguousRun > worst.maxContiguousRun) {
|
|
146
|
+
worst = { trivial: t.trivial, reason: t.reason ?? '', maxContiguousRun: t.maxContiguousRun, overlapRatio: t.overlapRatio };
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
if (worst.trivial) {
|
|
150
|
+
dropped.push({ id: pair.id, reason: worst.reason, maxContiguousRun: worst.maxContiguousRun, overlapRatio: worst.overlapRatio });
|
|
151
|
+
}
|
|
152
|
+
else {
|
|
153
|
+
scored.push(pair);
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
progress(`golden set: ${scored.length} scored, ${dropped.length} dropped as trivially solvable`);
|
|
157
|
+
if (scored.length === 0)
|
|
158
|
+
throw new Error('[doc eval] no golden pairs survived the triviality filter');
|
|
159
|
+
// ── 3. Isolated eval store (live documents only) ─────────────────
|
|
160
|
+
const storeRoot = opts.storeRoot ?? path.join(repoRoot, '.monomind', 'eval');
|
|
161
|
+
const storeDir = path.join(storeRoot, `store-${corpus.corpusHash}`);
|
|
162
|
+
const prevGlobal = process.env.MONOMIND_GLOBAL_BRAIN_DIR;
|
|
163
|
+
process.env.MONOMIND_GLOBAL_BRAIN_DIR = storeDir;
|
|
164
|
+
let ingestMs = 0;
|
|
165
|
+
let report;
|
|
166
|
+
try {
|
|
167
|
+
const pipeline = await import('../document-pipeline.js');
|
|
168
|
+
const stampPath = path.join(storeDir, 'eval-stamp.json');
|
|
169
|
+
const fresh = !opts.rebuild && fs.existsSync(stampPath)
|
|
170
|
+
&& JSON.parse(fs.readFileSync(stampPath, 'utf8')).corpusHash === corpus.corpusHash;
|
|
171
|
+
if (opts.rebuild && fs.existsSync(storeDir))
|
|
172
|
+
fs.rmSync(storeDir, { recursive: true, force: true });
|
|
173
|
+
fs.mkdirSync(storeDir, { recursive: true });
|
|
174
|
+
if (!fresh) {
|
|
175
|
+
const ti = Date.now();
|
|
176
|
+
let n = 0;
|
|
177
|
+
for (const d of corpus.docs) {
|
|
178
|
+
// scope 'global' routes to MONOMIND_GLOBAL_BRAIN_DIR — an isolated
|
|
179
|
+
// store that never touches the user's project or personal brain.
|
|
180
|
+
await pipeline.ingestDocument(d.absPath, 'global', storeDir);
|
|
181
|
+
if (++n % 50 === 0)
|
|
182
|
+
progress(`ingested ${n}/${corpus.docs.length}`);
|
|
183
|
+
}
|
|
184
|
+
ingestMs = Date.now() - ti;
|
|
185
|
+
fs.writeFileSync(stampPath, JSON.stringify({ corpusHash: corpus.corpusHash, docs: corpus.docs.length, builtAt: new Date().toISOString() }, null, 2));
|
|
186
|
+
progress(`ingest complete in ${(ingestMs / 1000).toFixed(1)}s`);
|
|
187
|
+
}
|
|
188
|
+
else {
|
|
189
|
+
progress('reusing existing eval store (corpus hash unchanged)');
|
|
190
|
+
}
|
|
191
|
+
// Row count of the isolated store. If this exceeds the chunk count, a
|
|
192
|
+
// superseded version leaked in and the live-doc pinning claim is false.
|
|
193
|
+
let evalStoreRows = -1;
|
|
194
|
+
try {
|
|
195
|
+
const bridge = await import('../../memory/memory-bridge.js');
|
|
196
|
+
const listed = await bridge.bridgeListEntries({ namespace: 'knowledge:global', limit: 1_000_000, dbPath: '@global' });
|
|
197
|
+
if (listed?.success && Array.isArray(listed.entries))
|
|
198
|
+
evalStoreRows = listed.entries.length;
|
|
199
|
+
}
|
|
200
|
+
catch { /* diagnostic only */ }
|
|
201
|
+
// ── 4. Chunk mirror for the weak baselines ─────────────────────
|
|
202
|
+
// Canonical documents only. The store keys chunks by CONTENT hash, so
|
|
203
|
+
// byte-identical files collapse there too — mirroring that here keeps the
|
|
204
|
+
// `evalStoreRows === corpusChunks` cross-check meaningful instead of
|
|
205
|
+
// permanently red, and stops duplicates skewing BM25 document frequencies.
|
|
206
|
+
const canonicalDocs = corpus.docs.filter(d => corpus.canonicalOf.get(d.id) === d.id);
|
|
207
|
+
const chunks = await buildChunks(canonicalDocs);
|
|
208
|
+
progress(`chunk mirror: ${chunks.length} chunks`);
|
|
209
|
+
// ── 5. IDF overlap characterisation ────────────────────────────
|
|
210
|
+
const idf = buildIdf(corpus.docs.map(d => textOf(d.id)));
|
|
211
|
+
const overlapOf = (p) => Math.max(...p.relevant.map(r => idfOverlap(idf, p.query, textOf(r))));
|
|
212
|
+
// ── 6. Retrievers ──────────────────────────────────────────────
|
|
213
|
+
const denseRetriever = new FnRetriever('dense-only (gte-modernbert-base)', 'The current shipping stack: searchKnowledge over the local vector store', async (query, limit) => {
|
|
214
|
+
const hits = await pipeline.searchKnowledge(query, {
|
|
215
|
+
limit, minScore: 0.0, store: 'global', rootDir: storeDir, includeSuperseded: false,
|
|
216
|
+
skipRerank: true, // isolate dense-only baseline from the reranker
|
|
217
|
+
});
|
|
218
|
+
return hits.map(h => ({
|
|
219
|
+
docId: path.relative(repoRoot, h.filePath),
|
|
220
|
+
chunkIndex: h.chunkIndex,
|
|
221
|
+
score: h.similarity,
|
|
222
|
+
}));
|
|
223
|
+
});
|
|
224
|
+
const bm25Retriever = new Bm25Retriever(chunks);
|
|
225
|
+
const retrievers = [
|
|
226
|
+
denseRetriever,
|
|
227
|
+
bm25Retriever,
|
|
228
|
+
new RandomRetriever(chunks),
|
|
229
|
+
];
|
|
230
|
+
// RRF fusion sweep: equal-weight, k ∈ {10, 20, 40, 60, 100}.
|
|
231
|
+
// Null hypothesis row — expected to fail the low-overlap gate.
|
|
232
|
+
const RRF_K_SWEEP = [10, 20, 40, 60, 100];
|
|
233
|
+
for (const rrfK of RRF_K_SWEEP) {
|
|
234
|
+
retrievers.push(new RrfRetriever([denseRetriever, bm25Retriever], rrfK));
|
|
235
|
+
}
|
|
236
|
+
// ── 6b. Reranked retriever (ettin-32m cross-encoder) ──────────
|
|
237
|
+
// Pre-load the reranker BEFORE the network guard goes up, so the model
|
|
238
|
+
// download happens while we still have connectivity.
|
|
239
|
+
let rerankerLoaded = false;
|
|
240
|
+
if (process.env.MONOMIND_RERANKER !== '0') {
|
|
241
|
+
try {
|
|
242
|
+
const bridge = await import('../../memory/memory-bridge.js');
|
|
243
|
+
await bridge.loadReranker();
|
|
244
|
+
rerankerLoaded = true;
|
|
245
|
+
progress('reranker loaded: cross-encoder/ettin-reranker-32m-v1');
|
|
246
|
+
}
|
|
247
|
+
catch (e) {
|
|
248
|
+
progress(`reranker failed to load — skipping reranked retriever: ${e}`);
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
if (rerankerLoaded) {
|
|
252
|
+
// The reranked retriever uses the same searchKnowledge path but with
|
|
253
|
+
// the reranker active (it was pre-loaded above). The dense-only
|
|
254
|
+
// retriever is kept WITHOUT reranking (skipRerank) for comparison.
|
|
255
|
+
const rerankedRetriever = new FnRetriever('dense+rerank (ettin-32m)', 'Dense retrieval + cross-encoder reranking via ettin-reranker-32m-v1', async (query, limit) => {
|
|
256
|
+
// searchKnowledge flows through bridgeSearchEntries which auto-reranks
|
|
257
|
+
// when the reranker is loaded. Over-retrieval happens inside.
|
|
258
|
+
const hits = await pipeline.searchKnowledge(query, {
|
|
259
|
+
limit, minScore: 0.0, store: 'global', rootDir: storeDir, includeSuperseded: false,
|
|
260
|
+
});
|
|
261
|
+
return hits.map(h => ({
|
|
262
|
+
docId: path.relative(repoRoot, h.filePath),
|
|
263
|
+
chunkIndex: h.chunkIndex,
|
|
264
|
+
score: h.similarity,
|
|
265
|
+
}));
|
|
266
|
+
});
|
|
267
|
+
retrievers.push(rerankedRetriever);
|
|
268
|
+
}
|
|
269
|
+
// ── 7. Query phase, network blocked ────────────────────────────
|
|
270
|
+
progress(`model provisioned: ${(modelPresence.bytes / 1e6).toFixed(0)}MB at ${modelPresence.resolvedPath}`);
|
|
271
|
+
const guard = installNetworkGuard();
|
|
272
|
+
// The search-path probe runs INSIDE the guarded window. It used to run
|
|
273
|
+
// outside it, which is exactly how a model download escaped the guard and
|
|
274
|
+
// still reported "0 attempts". If this says "keyword" we are not measuring
|
|
275
|
+
// semantic retrieval at all and the scoreboard must be read differently.
|
|
276
|
+
let searchMethodProbe = 'unknown';
|
|
277
|
+
const results = {};
|
|
278
|
+
const te = Date.now();
|
|
279
|
+
try {
|
|
280
|
+
try {
|
|
281
|
+
const bridge = await import('../../memory/memory-bridge.js');
|
|
282
|
+
const probe = await bridge.bridgeSearchEntries({
|
|
283
|
+
query: 'how are hooks dispatched', namespace: 'knowledge:global', limit: 3, threshold: 0.05, dbPath: '@global',
|
|
284
|
+
});
|
|
285
|
+
searchMethodProbe = String(probe?.searchMethod ?? 'unknown');
|
|
286
|
+
}
|
|
287
|
+
catch { /* probe is diagnostic; a blocked fetch here is recorded by the guard */ }
|
|
288
|
+
for (const r of retrievers) {
|
|
289
|
+
const outcomes = [];
|
|
290
|
+
let shortReturns = 0;
|
|
291
|
+
for (const pair of scored) {
|
|
292
|
+
const tq = Date.now();
|
|
293
|
+
// Over-fetch at the chunk level: k documents need more than k chunks
|
|
294
|
+
// when several chunks of one document rank highly.
|
|
295
|
+
const raw = await r.search(pair.query, k * 5);
|
|
296
|
+
const latencyMs = Date.now() - tq;
|
|
297
|
+
// Collapse to content units BEFORE ranking is cut off, so a
|
|
298
|
+
// byte-identical twin never consumes a top-k slot twice.
|
|
299
|
+
const ranked = dedupeByDoc(raw.map(h => ({ ...h, docId: canon(h.docId) })), k);
|
|
300
|
+
if (ranked.length < k)
|
|
301
|
+
shortReturns++;
|
|
302
|
+
outcomes.push(scoreQuery({
|
|
303
|
+
queryId: pair.id, query: pair.query, relevant: pair.relevant.map(canon),
|
|
304
|
+
ranked, latencyMs, overlap: overlapOf(pair),
|
|
305
|
+
}));
|
|
306
|
+
}
|
|
307
|
+
const agg = aggregate(outcomes);
|
|
308
|
+
const terc = terciles(outcomes);
|
|
309
|
+
results[r.name] = {
|
|
310
|
+
name: r.name,
|
|
311
|
+
description: r.description,
|
|
312
|
+
scoreboard: agg,
|
|
313
|
+
terciles: terc,
|
|
314
|
+
// Sealed split: aggregates only. Withholding this is the whole point.
|
|
315
|
+
outcomes: sealed ? [] : outcomes,
|
|
316
|
+
shortReturns,
|
|
317
|
+
shortReturnRate: shortReturns / outcomes.length,
|
|
318
|
+
};
|
|
319
|
+
progress(`${r.name}: Recall@5 ${results[r.name].scoreboard.recallAt5.toFixed(3)}`);
|
|
320
|
+
}
|
|
321
|
+
}
|
|
322
|
+
finally {
|
|
323
|
+
guard.release();
|
|
324
|
+
}
|
|
325
|
+
const evalMs = Date.now() - te;
|
|
326
|
+
const allOverlaps = scored.map(overlapOf).sort((a, b) => a - b);
|
|
327
|
+
const q = (p) => allOverlaps[Math.min(allOverlaps.length - 1, Math.floor(p * allOverlaps.length))] ?? 0;
|
|
328
|
+
const denseName = denseRetriever.name;
|
|
329
|
+
const dense = results[denseName];
|
|
330
|
+
const bm25 = results['bm25-only'];
|
|
331
|
+
const rand = results['random'];
|
|
332
|
+
report = {
|
|
333
|
+
schemaVersion: 1,
|
|
334
|
+
generatedAt: new Date().toISOString(),
|
|
335
|
+
method: {
|
|
336
|
+
goldenSetVersion: 'v1 (2026-07-28)',
|
|
337
|
+
split,
|
|
338
|
+
testExposureCount: null,
|
|
339
|
+
stopConditionEvaluable: sealed,
|
|
340
|
+
corpusHash: corpus.corpusHash,
|
|
341
|
+
corpusFiles: corpus.docs.length,
|
|
342
|
+
corpusDocs: corpus.contentUnits,
|
|
343
|
+
duplicateGroupsCollapsed: corpus.duplicateGroups,
|
|
344
|
+
appleDoubleCount: corpus.appleDoubleCount,
|
|
345
|
+
corpusChunks: chunks.length,
|
|
346
|
+
evalStoreRows,
|
|
347
|
+
corpusPinning: 'git-tracked files at HEAD, content-addressed: corpusHash = sha256 over the sorted (path, sha256) pairs. ' +
|
|
348
|
+
'The eval NEVER reads the live project or personal store — it builds a dedicated store, one ingest per document, ' +
|
|
349
|
+
'so it holds zero superseded versions and cannot drift while a session re-ingests generated artefacts. ' +
|
|
350
|
+
'Untracked generated files (e.g. GRAPH_REPORT.md) are absent from the corpus by construction.',
|
|
351
|
+
storeProfile: 'fresh',
|
|
352
|
+
representativeness: 'REPRODUCIBLE BUT NOT REPRESENTATIVE: this corpus is a clean git-HEAD snapshot with no version history, ' +
|
|
353
|
+
'no ingest churn and no dangling entries pointing at deleted files. A real user store has all three. ' +
|
|
354
|
+
'Numbers here are an upper bound on live behaviour and will diverge from it.',
|
|
355
|
+
topK: k,
|
|
356
|
+
kCorpusRatio: ratio,
|
|
357
|
+
pairsAuthored: candidatePairs.length,
|
|
358
|
+
pairsAuthoredTotal: GOLDEN_SET.length,
|
|
359
|
+
pairsScored: scored.length,
|
|
360
|
+
pairsDroppedTrivial: dropped.length,
|
|
361
|
+
relevancePinnedToLiveDocs: true,
|
|
362
|
+
embeddingModel: 'Alibaba-NLP/gte-modernbert-base (768d, q8, local)',
|
|
363
|
+
dbDriver: detectDbDriver(),
|
|
364
|
+
searchMethodProbe,
|
|
365
|
+
modelPresence,
|
|
366
|
+
provisioningIntact: (modelPresence.present && guard.attempts.length === 0 && guard.unpatched.length === 0) ? 1 : 0,
|
|
367
|
+
includesGlobalBrain: false,
|
|
368
|
+
hardware: {
|
|
369
|
+
platform: process.platform, arch: process.arch,
|
|
370
|
+
cpus: os.cpus().length, cpuModel: os.cpus()[0]?.model ?? 'unknown',
|
|
371
|
+
nodeVersion: process.version,
|
|
372
|
+
},
|
|
373
|
+
},
|
|
374
|
+
networkFree: {
|
|
375
|
+
verdict: guard.attempts.length > 0 ? 'violated' : guard.unpatched.length > 0 ? 'partial' : 'proven-blocked',
|
|
376
|
+
method: 'fetch/http/https/net/tls/dns replaced with throwing stubs for the whole query phase; every attempt recorded with its stack. Does not cover sockets opened inside a native addon — see lsof corroboration in the baseline report.',
|
|
377
|
+
attempts: guard.attempts,
|
|
378
|
+
unpatched: guard.unpatched,
|
|
379
|
+
telemetryCarveOut: 'Clause 4 scope = the retrieval path (whatever a query requires or triggers to return ' +
|
|
380
|
+
'results). Crash reporting and the update checker are carved out, and are DISABLED for ' +
|
|
381
|
+
'the duration of this run (MONOMIND_CRASH_REPORTING=off), so a zero here describes ' +
|
|
382
|
+
'retrieval rather than the absence of a crash. Both remain user-disableable in normal use.',
|
|
383
|
+
},
|
|
384
|
+
droppedPairs: sealed ? dropped.map(d => ({ ...d, id: '<sealed>' })) : dropped,
|
|
385
|
+
overlap: {
|
|
386
|
+
p25: q(0.25), p50: q(0.5), p75: q(0.75),
|
|
387
|
+
tercileCutLow: dense.terciles.cutLow, tercileCutHigh: dense.terciles.cutHigh,
|
|
388
|
+
},
|
|
389
|
+
results,
|
|
390
|
+
headline: {
|
|
391
|
+
retriever: denseName,
|
|
392
|
+
recallAt1: dense.scoreboard.recallAt1,
|
|
393
|
+
recallAt5: dense.scoreboard.recallAt5,
|
|
394
|
+
recallAt10: dense.scoreboard.recallAt10,
|
|
395
|
+
mrrAt10: dense.scoreboard.mrrAt10,
|
|
396
|
+
lowOverlapRecallAt5: dense.terciles.low.recallAt5,
|
|
397
|
+
bm25FloorRecallAt5: bm25?.scoreboard.recallAt5 ?? 0,
|
|
398
|
+
randomFloorRecallAt5: rand?.scoreboard.recallAt5 ?? 0,
|
|
399
|
+
gapOverBm25: dense.scoreboard.recallAt5 - (bm25?.scoreboard.recallAt5 ?? 0),
|
|
400
|
+
},
|
|
401
|
+
regressionSuite: [],
|
|
402
|
+
corpusComposition: (() => {
|
|
403
|
+
const byTopLevel = {};
|
|
404
|
+
const byExtension = {};
|
|
405
|
+
for (const d of corpus.docs) {
|
|
406
|
+
if (corpus.canonicalOf.get(d.id) !== d.id)
|
|
407
|
+
continue;
|
|
408
|
+
const top = d.id.includes('/') ? d.id.split('/')[0] : '<root>';
|
|
409
|
+
byTopLevel[top] = (byTopLevel[top] ?? 0) + 1;
|
|
410
|
+
const ext = path.extname(d.id).toLowerCase() || '<none>';
|
|
411
|
+
byExtension[ext] = (byExtension[ext] ?? 0) + 1;
|
|
412
|
+
}
|
|
413
|
+
return { byTopLevel, byExtension };
|
|
414
|
+
})(),
|
|
415
|
+
timings: { ingestMs, evalMs },
|
|
416
|
+
};
|
|
417
|
+
// Re-score every prior item's pre-registered signal against THIS row.
|
|
418
|
+
report.regressionSuite = scoreSignals(report, 'fresh', dense.scoreboard.hitRateAt5Ci95, { corpusHash: corpus.corpusHash, goldenSetVersion: report.method.goldenSetVersion, splitScheme: SPLIT_SCHEME });
|
|
419
|
+
// Exposure ledger. A sealed set run forty times with tuning in between is
|
|
420
|
+
// no longer sealed; the count is the only way anyone finds out.
|
|
421
|
+
if (sealed) {
|
|
422
|
+
const ledger = path.join(storeRoot, 'test-exposure-ledger.jsonl');
|
|
423
|
+
fs.mkdirSync(storeRoot, { recursive: true });
|
|
424
|
+
const prior = fs.existsSync(ledger)
|
|
425
|
+
? fs.readFileSync(ledger, 'utf8').split('\n').filter(Boolean).length
|
|
426
|
+
: 0;
|
|
427
|
+
report.method.testExposureCount = prior + 1;
|
|
428
|
+
fs.appendFileSync(ledger, JSON.stringify({
|
|
429
|
+
at: report.generatedAt,
|
|
430
|
+
exposure: prior + 1,
|
|
431
|
+
corpusHash: corpus.corpusHash,
|
|
432
|
+
goldenSetVersion: report.method.goldenSetVersion,
|
|
433
|
+
topK: k,
|
|
434
|
+
recallAt5: report.headline.recallAt5,
|
|
435
|
+
mrrAt10: report.headline.mrrAt10,
|
|
436
|
+
}) + '\n');
|
|
437
|
+
}
|
|
438
|
+
}
|
|
439
|
+
finally {
|
|
440
|
+
if (prevGlobal === undefined)
|
|
441
|
+
delete process.env.MONOMIND_GLOBAL_BRAIN_DIR;
|
|
442
|
+
else
|
|
443
|
+
process.env.MONOMIND_GLOBAL_BRAIN_DIR = prevGlobal;
|
|
444
|
+
if (prevCrash === undefined)
|
|
445
|
+
delete process.env.MONOMIND_CRASH_REPORTING;
|
|
446
|
+
else
|
|
447
|
+
process.env.MONOMIND_CRASH_REPORTING = prevCrash;
|
|
448
|
+
}
|
|
449
|
+
void t0;
|
|
450
|
+
return report;
|
|
451
|
+
}
|
|
452
|
+
// ── Human-readable rendering ────────────────────────────────────────
|
|
453
|
+
function pct(x) { return (x * 100).toFixed(1) + '%'; }
|
|
454
|
+
function f3(x) { return x.toFixed(3); }
|
|
455
|
+
export function renderReport(r) {
|
|
456
|
+
const L = [];
|
|
457
|
+
const m = r.method;
|
|
458
|
+
L.push('');
|
|
459
|
+
L.push('Second Brain retrieval scoreboard');
|
|
460
|
+
L.push('='.repeat(72));
|
|
461
|
+
L.push(`corpus ${m.corpusDocs} distinct documents from ${m.corpusFiles} files / ${m.corpusChunks} chunks (hash ${m.corpusHash})`);
|
|
462
|
+
L.push(` ${m.duplicateGroupsCollapsed} byte-identical groups collapsed to one unit each; AppleDouble "._" files: ${m.appleDoubleCount} (asserted zero)`);
|
|
463
|
+
L.push(`eval store ${m.evalStoreRows < 0 ? 'unknown' : m.evalStoreRows + ' rows'}${m.evalStoreRows >= 0 && m.evalStoreRows !== m.corpusChunks ? ' <- MISMATCH vs chunk count, superseded rows may have leaked in' : ''}`);
|
|
464
|
+
L.push(`split ${m.split.toUpperCase()}${m.split === 'test' ? ' (SEALED — aggregates only, no per-query output)' : m.split === 'dev' ? ' (tunable; cannot satisfy the stop condition)' : ' (diagnostic; cannot satisfy the stop condition)'}`);
|
|
465
|
+
if (m.testExposureCount !== null)
|
|
466
|
+
L.push(`exposure TEST has now been run ${m.testExposureCount} time(s)`);
|
|
467
|
+
L.push(`golden set ${m.pairsScored} scored of ${m.pairsAuthored} in this split (${m.pairsAuthoredTotal} authored overall; ${m.pairsDroppedTrivial} dropped as trivially solvable)`);
|
|
468
|
+
L.push(`top_k ${m.topK} = ${(m.kCorpusRatio * 100).toFixed(2)}% of corpus`);
|
|
469
|
+
L.push(`embeddings ${m.embeddingModel}`);
|
|
470
|
+
L.push(`db driver ${m.dbDriver} search path probe: ${m.searchMethodProbe}`);
|
|
471
|
+
L.push(`model weights ${m.modelPresence.present ? 'PRESENT before any query' : 'ABSENT'} ` +
|
|
472
|
+
`(${(m.modelPresence.bytes / 1e6).toFixed(0)}MB, ${m.modelPresence.provenance})`);
|
|
473
|
+
L.push(`relevance pinned to LIVE documents only (store rebuilt, no superseded versions)`);
|
|
474
|
+
L.push(`hardware ${m.hardware.cpuModel} x${m.hardware.cpus}, ${m.hardware.platform}/${m.hardware.arch}, node ${m.hardware.nodeVersion}`);
|
|
475
|
+
L.push(`store profile ${m.storeProfile} (rows with a different profile are NOT comparable)`);
|
|
476
|
+
L.push(`caveat ${m.representativeness}`);
|
|
477
|
+
L.push(`carve-out ${r.networkFree.telemetryCarveOut}`);
|
|
478
|
+
L.push(`network ${r.networkFree.verdict.toUpperCase()} (${r.networkFree.attempts.length} attempts blocked during query phase` +
|
|
479
|
+
(r.networkFree.unpatched.length ? `; UNPATCHED: ${r.networkFree.unpatched.join(', ')}` : '') + ')');
|
|
480
|
+
L.push('');
|
|
481
|
+
const rows = Object.values(r.results);
|
|
482
|
+
const w = Math.max(...rows.map(x => x.name.length), 10);
|
|
483
|
+
const head = ['retriever'.padEnd(w), 'R@1'.padStart(7), 'R@5'.padStart(7), 'R@10'.padStart(7), 'MRR@10'.padStart(7), 'p50ms'.padStart(7), 'p95ms'.padStart(7), 'short'.padStart(7)];
|
|
484
|
+
L.push(head.join(' '));
|
|
485
|
+
L.push('-'.repeat(head.join(' ').length));
|
|
486
|
+
for (const row of rows) {
|
|
487
|
+
const s = row.scoreboard;
|
|
488
|
+
L.push([
|
|
489
|
+
row.name.padEnd(w),
|
|
490
|
+
f3(s.recallAt1).padStart(7), f3(s.recallAt5).padStart(7), f3(s.recallAt10).padStart(7),
|
|
491
|
+
f3(s.mrrAt10).padStart(7),
|
|
492
|
+
String(s.latencyMsP50).padStart(7), String(s.latencyMsP95).padStart(7),
|
|
493
|
+
pct(row.shortReturnRate).padStart(7),
|
|
494
|
+
].join(' '));
|
|
495
|
+
}
|
|
496
|
+
L.push('');
|
|
497
|
+
L.push('Recall@5 by IDF-weighted query/document overlap tercile');
|
|
498
|
+
L.push(` (tercile cuts: low < ${f3(r.overlap.tercileCutLow)} <= mid < ${f3(r.overlap.tercileCutHigh)} <= high)`);
|
|
499
|
+
L.push(['retriever'.padEnd(w), 'low'.padStart(7), 'mid'.padStart(7), 'high'.padStart(7)].join(' '));
|
|
500
|
+
L.push('-'.repeat(w + 24));
|
|
501
|
+
for (const row of rows) {
|
|
502
|
+
L.push([
|
|
503
|
+
row.name.padEnd(w),
|
|
504
|
+
f3(row.terciles.low.recallAt5).padStart(7),
|
|
505
|
+
f3(row.terciles.mid.recallAt5).padStart(7),
|
|
506
|
+
f3(row.terciles.high.recallAt5).padStart(7),
|
|
507
|
+
].join(' '));
|
|
508
|
+
}
|
|
509
|
+
L.push('');
|
|
510
|
+
L.push('Reading this scoreboard');
|
|
511
|
+
L.push(` gap over BM25-only ${f3(r.headline.gapOverBm25)} <- the real signal. A small gap means the`);
|
|
512
|
+
L.push(' golden set is too easy, not that the stack is good.');
|
|
513
|
+
L.push(` random floor Recall@5 ${f3(r.headline.randomFloorRecallAt5)} <- anything but ~0 means a vacuous eval.`);
|
|
514
|
+
L.push(` low-overlap Recall@5 ${f3(r.headline.lowOverlapRecallAt5)} <- the closest proxy to real-world queries.`);
|
|
515
|
+
const ci = rows[0]?.scoreboard.hitRateAt5Ci95 ?? 0;
|
|
516
|
+
L.push(` 95% CI half-width ${f3(ci)} <- a delta smaller than this is noise, not improvement.`);
|
|
517
|
+
L.push('');
|
|
518
|
+
if (r.regressionSuite.length > 0) {
|
|
519
|
+
L.push('Regression suite — every prior item\'s pre-registered signal, re-scored on this row');
|
|
520
|
+
for (const sig of r.regressionSuite) {
|
|
521
|
+
const cur = sig.currentValue === null ? ' n/a' : f3(sig.currentValue);
|
|
522
|
+
const ref = sig.shipValue ?? sig.baselineValue;
|
|
523
|
+
L.push(` [${sig.verdict.padEnd(9)}] item ${sig.item.padEnd(3)} ${sig.id}`);
|
|
524
|
+
L.push(` now ${cur}` + (ref !== null && ref !== undefined ? ` vs ${f3(ref)} at ship/baseline` : '') +
|
|
525
|
+
(sig.nullVerdict ? ` null-verdict: ${sig.nullVerdict}` : ''));
|
|
526
|
+
L.push(` ${sig.note}`);
|
|
527
|
+
}
|
|
528
|
+
const decayed = r.regressionSuite.filter(x => x.verdict === 'DECAYED');
|
|
529
|
+
if (decayed.length > 0) {
|
|
530
|
+
L.push(` !! ${decayed.length} PRE-REGISTERED SIGNAL(S) HAVE DECAYED — a win recorded on an`);
|
|
531
|
+
L.push(' earlier row no longer holds. This is the only evidence that justifies a revert.');
|
|
532
|
+
}
|
|
533
|
+
L.push('');
|
|
534
|
+
}
|
|
535
|
+
L.push('Corpus composition (distinct documents by top-level directory)');
|
|
536
|
+
L.push(' ' + Object.entries(r.corpusComposition.byTopLevel)
|
|
537
|
+
.sort((a, b) => b[1] - a[1]).slice(0, 8)
|
|
538
|
+
.map(([k, v]) => `${k} ${v}`).join(' '));
|
|
539
|
+
L.push('');
|
|
540
|
+
L.push(`Stop condition: Recall@5 >= 0.900 and MRR@10 >= 0.800 on >= 500 documents.`);
|
|
541
|
+
const s = r.results[r.headline.retriever].scoreboard;
|
|
542
|
+
const met = s.recallAt5 >= 0.9 && s.mrrAt10 >= 0.8 && m.corpusDocs >= 500;
|
|
543
|
+
if (!m.stopConditionEvaluable) {
|
|
544
|
+
L.push(` currently: Recall@5 ${f3(s.recallAt5)}, MRR@10 ${f3(s.mrrAt10)}, corpus ${m.corpusDocs}`);
|
|
545
|
+
L.push(` NOT EVALUABLE on the ${m.split} split — the stop condition may only be checked on TEST.`);
|
|
546
|
+
}
|
|
547
|
+
else {
|
|
548
|
+
L.push(` currently: Recall@5 ${f3(s.recallAt5)}, MRR@10 ${f3(s.mrrAt10)}, corpus ${m.corpusDocs} -> ${met ? 'MET' : 'NOT MET'}`);
|
|
549
|
+
}
|
|
550
|
+
L.push('');
|
|
551
|
+
return L.join('\n');
|
|
552
|
+
}
|
|
553
|
+
/**
|
|
554
|
+
* @param bandCuts overlap thresholds; defaults match the v1 TEST terciles so a
|
|
555
|
+
* candidate is judged against the distribution we are trying
|
|
556
|
+
* to move, not against the one it would itself create.
|
|
557
|
+
*/
|
|
558
|
+
export async function screenCandidates(repoRootIn, candidates, bandCuts = { low: 0.247, high: 0.455 }) {
|
|
559
|
+
const repoRoot = resolveRepoRoot(repoRootIn);
|
|
560
|
+
const corpus = buildCorpus(repoRoot);
|
|
561
|
+
const byId = new Map(corpus.docs.map(d => [d.id, d]));
|
|
562
|
+
const cache = new Map();
|
|
563
|
+
const textOf = (id) => {
|
|
564
|
+
let t = cache.get(id);
|
|
565
|
+
if (t === undefined) {
|
|
566
|
+
t = readDoc(byId.get(id));
|
|
567
|
+
cache.set(id, t);
|
|
568
|
+
}
|
|
569
|
+
return t;
|
|
570
|
+
};
|
|
571
|
+
const idf = buildIdf(corpus.docs.map(d => textOf(d.id)));
|
|
572
|
+
const seen = new Set();
|
|
573
|
+
const out = [];
|
|
574
|
+
for (const c of candidates) {
|
|
575
|
+
const missing = c.relevant.filter(r => !byId.has(r));
|
|
576
|
+
if (missing.length > 0) {
|
|
577
|
+
out.push({ ...c, idfOverlap: 0, maxContiguousRun: 0, band: 'low', accepted: false,
|
|
578
|
+
reason: `target not in corpus: ${missing.join(', ')}` });
|
|
579
|
+
continue;
|
|
580
|
+
}
|
|
581
|
+
if (seen.has(c.id)) {
|
|
582
|
+
out.push({ ...c, idfOverlap: 0, maxContiguousRun: 0, band: 'low', accepted: false, reason: 'duplicate id' });
|
|
583
|
+
continue;
|
|
584
|
+
}
|
|
585
|
+
seen.add(c.id);
|
|
586
|
+
const overlap = Math.max(...c.relevant.map(r => idfOverlap(idf, c.query, textOf(r))));
|
|
587
|
+
const run = Math.max(...c.relevant.map(r => assessTriviality(c.query, textOf(r)).maxContiguousRun));
|
|
588
|
+
const trivial = c.relevant.some(r => assessTriviality(c.query, textOf(r)).trivial);
|
|
589
|
+
const band = overlap < bandCuts.low ? 'low' : overlap < bandCuts.high ? 'mid' : 'high';
|
|
590
|
+
out.push({
|
|
591
|
+
...c, idfOverlap: overlap, maxContiguousRun: run, band,
|
|
592
|
+
accepted: !trivial,
|
|
593
|
+
...(trivial ? { reason: `trivially solvable: ${run}-token verbatim run from the query appears in the target` } : {}),
|
|
594
|
+
});
|
|
595
|
+
}
|
|
596
|
+
const acc = out.filter(c => c.accepted);
|
|
597
|
+
return {
|
|
598
|
+
corpusHash: corpus.corpusHash,
|
|
599
|
+
total: out.length,
|
|
600
|
+
accepted: acc.length,
|
|
601
|
+
rejected: out.length - acc.length,
|
|
602
|
+
bands: {
|
|
603
|
+
low: acc.filter(c => c.band === 'low').length,
|
|
604
|
+
mid: acc.filter(c => c.band === 'mid').length,
|
|
605
|
+
high: acc.filter(c => c.band === 'high').length,
|
|
606
|
+
},
|
|
607
|
+
candidates: out,
|
|
608
|
+
};
|
|
609
|
+
}
|
|
610
|
+
//# sourceMappingURL=harness.js.map
|