@gamaze/hicortex 0.23.1 → 0.23.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/assets/dashboard.html +25 -13
  2. package/dist/dashboard.d.ts +18 -7
  3. package/dist/dashboard.js +42 -10
  4. package/dist/dedup.js +2 -2
  5. package/dist/index.js +17 -2
  6. package/dist/learnings-identity.js +20 -1
  7. package/dist/mcp-server.js +6 -2
  8. package/dist/nightly.js +1 -1
  9. package/hermes-plugin/hicortex/provider.py +23 -0
  10. package/opencode-plugin/hicortex/index.ts +24 -1
  11. package/package.json +2 -1
  12. package/pi-extension/hicortex/index.ts +24 -1
  13. package/server.json +2 -2
  14. package/dist/eval/decay-eval.d.ts +0 -111
  15. package/dist/eval/decay-eval.js +0 -214
  16. package/dist/eval/dups.d.ts +0 -100
  17. package/dist/eval/dups.js +0 -174
  18. package/dist/eval/eval-clock.d.ts +0 -32
  19. package/dist/eval/eval-clock.js +0 -47
  20. package/dist/eval/eval-db.d.ts +0 -25
  21. package/dist/eval/eval-db.js +0 -67
  22. package/dist/eval/graph-eval.d.ts +0 -89
  23. package/dist/eval/graph-eval.js +0 -246
  24. package/dist/eval/importance-eval.d.ts +0 -85
  25. package/dist/eval/importance-eval.js +0 -286
  26. package/dist/eval/planted-eval.d.ts +0 -30
  27. package/dist/eval/planted-eval.js +0 -122
  28. package/dist/eval/planted-fixtures.d.ts +0 -107
  29. package/dist/eval/planted-fixtures.js +0 -283
  30. package/dist/eval/planted-harness.d.ts +0 -183
  31. package/dist/eval/planted-harness.js +0 -651
  32. package/dist/eval/ranking-battery.d.ts +0 -125
  33. package/dist/eval/ranking-battery.js +0 -289
  34. package/dist/eval/ranking-eval.d.ts +0 -61
  35. package/dist/eval/ranking-eval.js +0 -554
  36. package/dist/eval/ranking-fixtures.d.ts +0 -117
  37. package/dist/eval/ranking-fixtures.js +0 -485
  38. package/dist/eval/recall-sweep.d.ts +0 -87
  39. package/dist/eval/recall-sweep.js +0 -1030
  40. package/dist/eval/reflection-census.d.ts +0 -19
  41. package/dist/eval/reflection-census.js +0 -25
  42. package/dist/eval/relevance-eval.d.ts +0 -178
  43. package/dist/eval/relevance-eval.js +0 -2240
  44. package/dist/eval/run-eval.d.ts +0 -20
  45. package/dist/eval/run-eval.js +0 -299
@@ -1,2240 +0,0 @@
1
- #!/usr/bin/env node
2
- "use strict";
3
- /**
4
- * Real-query relevance + SNIPPET eval — recall QUALITY on real agent prompts.
5
- *
6
- * v2 (spec `specs/2026-08-02-relevance-eval.md`) extends the v1 selection-only
7
- * eval with the SNIPPET layer: v1 asked "did retrieve() surface the right
8
- * memories?" (judge sees up to 2000 chars). Production shows the agent only a
9
- * ~100-char one-liner (`recall-index.ts#memoryTitle`), so a memory can be
10
- * genuinely relevant while its rendered line is useless — v1 scored that as a
11
- * win. v2 grades BOTH: `full_verdict` (selection quality, judge sees full
12
- * content) and `line_verdict` (snippet quality, judge sees ONLY the rendered
13
- * production one-liner — imported from `recall-index.ts`, never reimplemented).
14
- *
15
- * v2 additions (spec §4, §5, §6, §6b) layered onto the v1 base (prompt
16
- * sampling, readonly snapshot handling, embed-once + neverCalledEmbed,
17
- * lenient JSON parse, distribution/CI reporting):
18
- * 1. Dual verdict per surfaced memory — TWO separate, blind judge calls.
19
- * 2. Snippet-length sweep (100/200/300/title+1st-sentence) on a fixed
20
- * 40-prompt subset (8 per source).
21
- * 3. Similarity-floor + retrieval-source analysis (near-free — logged, not
22
- * re-judged).
23
- * 4. Token-cost estimate (char/4) per K and per snippet-length variant.
24
- * 5. ~20 rendered ACTUAL production blocks dumped into the report.
25
- * 6. Redundancy — one set-level judge call per (prompt × mode) over the
26
- * production 6.
27
- * 7. Rate limiting + resumability (§6b, MANDATORY): serial calls,
28
- * `--judge-delay-ms` (default 2000), exponential backoff with jitter on
29
- * 429/5xx/timeout (5→10→20→40→80s, max 5 retries, respects
30
- * `Retry-After`), checkpoint-per-call to a `.jsonl` sidecar, `--resume`,
31
- * progress logging, 10%-error-rate abort, `--max-calls` budget guard
32
- * (default 900).
33
- *
34
- * Prompt corpus (spec §2, owner decision §11.1): EVEN split, 20 prompts per
35
- * source × 5 sources — Hermes (lenny, raider, nano) + CC (the DevOps
36
- * `infrastructure` project, the `aironic-marine` project). Saved to
37
- * `data/prompts.json`, stable/reused verbatim once a valid v2 set exists.
38
- *
39
- * Judge: GLM-5.2 via z.ai — the INSTRUMENT only. It never picks candidates;
40
- * retrieve() (LLM-free) does. A dedicated raw HTTP caller (NOT `LlmClient`) is
41
- * used here on purpose: `LlmClient.completeReflect` bakes in a
42
- * nightly-tolerant retry policy (30s/60s/120s, unlimited rate-limit patience)
43
- * that conflicts with §6b's specific real-time batch policy (5/10/20/40/80s +
44
- * jitter, 5 retries, a hard call budget). Implemented directly here rather
45
- * than adding a second retry mode to `llm.ts` (out of scope for this eval,
46
- * and another agent is concurrently working elsewhere in this repo).
47
- *
48
- * Honesty invariants (non-negotiable — mirror recall-sweep.ts + spec §7):
49
- * - Snapshot opened READONLY via openSnapshot — never initDb.
50
- * - noStrengthen: true on every retrieve() call.
51
- * - Real bge-small-en-v1.5 embedder, embed-once + queryEmbedding reuse.
52
- * - neverCalledEmbed self-check ABORTS the run if retrieve() ignores
53
- * queryEmbedding (would invalidate every measured number).
54
- * - GLM-5.2 is the JUDGE only, real prompts, no synthetic queries.
55
- * - Production renderer (`formatIndexLine`/`memoryTitle`) imported from
56
- * `recall-index.ts`, never reimplemented.
57
- * - judge_error batches/units excluded from every denominator, reported
58
- * separately.
59
- *
60
- * Run:
61
- * npm run eval:relevance -- <snapshot.db> [prompts.json] [report.md] \
62
- * [--judge-delay-ms=2000] [--max-calls=900] [--resume] \
63
- * [--verdicts-json=path] [--verdicts-jsonl=path] \
64
- * [--now=<ISO>] [--runs=N]
65
- *
66
- * #458 clock pin + judge variance protocol:
67
- * - `--now=<ISO>` pins the clock the retrieve sweep scores against (the
68
- * retrieve() `now` seam) — before/after runs become wall-clock-independent
69
- * (default: live clock, the pre-#458 behavior).
70
- * - `--runs=N` (default 1) runs judge phases 1+2 N times over the SAME
71
- * selections (one retrieve sweep) and reports the run-to-run noise floor
72
- * (median + spread + selection-identical flip counts, §15b). Phases 3+4
73
- * stay single-shot; the shared --max-calls budget spans all runs and the
74
- * default scales ×N unless --max-calls was passed explicitly.
75
- */
76
- var __importDefault = (this && this.__importDefault) || function (mod) {
77
- return (mod && mod.__esModule) ? mod : { "default": mod };
78
- };
79
- Object.defineProperty(exports, "__esModule", { value: true });
80
- exports.parseArgs = parseArgs;
81
- exports.scaleMaxCalls = scaleMaxCalls;
82
- exports.median = median;
83
- exports.computeJudgeVariance = computeJudgeVariance;
84
- const node_fs_1 = require("node:fs");
85
- const node_path_1 = require("node:path");
86
- const node_os_1 = require("node:os");
87
- const better_sqlite3_1 = __importDefault(require("better-sqlite3"));
88
- const eval_db_js_1 = require("./eval-db.js");
89
- const eval_clock_js_1 = require("./eval-clock.js");
90
- const embedder_js_1 = require("../embedder.js");
91
- const retrieval_js_1 = require("../retrieval.js");
92
- const recall_index_js_1 = require("../recall-index.js");
93
- // ---------------------------------------------------------------------------
94
- // Constants
95
- // ---------------------------------------------------------------------------
96
- const RESULT_K = 8; // top-k replayed per prompt — the MEASUREMENT WINDOW. Two
97
- // slots PAST the production menu (PROD_MENU_K=6) on purpose: the bubble
98
- // (Q7-Q8) is cap-tuning data.
99
- // All three mirror the shipped defaults in recall-index.ts (kept in sync
100
- // manually — CLAUDE.md). Updated 2026-08-03 alongside the recall-quality PR.
101
- const PROD_MENU_K = 5; // the shipped `recallMaxItems` default (was 6).
102
- // K is a reporting/labelling param here (precision@4/@6/@8 all derive from
103
- // the same stored verdict rows), so prior runs stay fully comparable.
104
- const PROD_MIN_SIMILARITY = 0.62; // the shipped `recallMinSimilarity` floor (was 0.55).
105
- const PROD_TITLE_CHARS = 100; // the shipped `recallTitleChars` default (was 150).
106
- // Methodology note: this controls what the judge SEES in the baseline
107
- // line_verdict pass, so eval #4's snippet_failure_rate is NOT directly
108
- // comparable to #3's 24.3% (judged at 150). Cross-run comparability survives
109
- // because SWEEP_VARIANTS still measures 150 on the 40-prompt subset — do NOT
110
- // prune 150 from SWEEP_VARIANTS without recording that it breaks the #3↔#4
111
- // comparison (lower N, wider CI, but comparable).
112
- const MIN_PROMPT_CHARS = 20; // drop trivial/short prompts (<20 chars)
113
- const MAX_MEM_CHARS_IN_JUDGE = 2000; // full_verdict per-memory cap.
114
- const ZAI_BASE_URL = "https://api.z.ai/api/anthropic";
115
- const ZAI_MODEL = "glm-5.2";
116
- const N_PER_SOURCE = 20; // spec §2, §11.1: even 20/source.
117
- const SWEEP_N_PER_SOURCE = 8; // spec §4.2: fixed 40-prompt subset (8/source).
118
- const DEFAULT_JUDGE_DELAY_MS = 2000; // spec §6b: 1.5-2s between calls.
119
- const BACKOFF_SCHEDULE_MS = [5_000, 10_000, 20_000, 40_000, 80_000]; // spec §6b.
120
- const ERROR_RATE_ABORT_THRESHOLD = 0.10; // spec §6b.
121
- const ERROR_RATE_MIN_SAMPLE = 20; // don't abort on noise from a tiny sample.
122
- const DEFAULT_MAX_CALLS = 900; // spec §6b budget guard.
123
- const DEFAULT_CC_PROJECTS_DIR = (0, node_path_1.join)((0, node_os_1.homedir)(), ".claude", "projects");
124
- const PROMPTS_SAVE_PATH = (0, node_path_1.join)(process.cwd(), "data", "prompts.json");
125
- const DEFAULT_REPORT_PATH = (0, node_path_1.join)(process.cwd(), "data", "relevance-eval-report.md");
126
- const VERDICTS_JSON_PATH = (0, node_path_1.join)(process.cwd(), "data", "relevance-verdicts.json");
127
- const VERDICTS_JSONL_PATH = (0, node_path_1.join)(process.cwd(), "data", "relevance-verdicts.jsonl");
128
- /** Spec §2 — the 5 sources, in stable corpus order. */
129
- const SOURCE_TAGS = [
130
- "hermes-lenny",
131
- "hermes-raider",
132
- "hermes-nano",
133
- "cc-infrastructure",
134
- "cc-marine",
135
- ];
136
- /** Required Hermes profile state DBs (flat /tmp layout, read-only, owner drop). */
137
- const HERMES_PROFILE_DBS = {
138
- lenny: "/tmp/hermes-lenny-state.db",
139
- raider: "/tmp/hermes-raider-state.db",
140
- nano: "/tmp/hermes-nano-state.db",
141
- };
142
- /** Profile → declared mission domains (corpus-derived, owner directive — see
143
- * v1 header history). Simulates the per-agent config not yet set on bedrock;
144
- * the scope-ON sweep measures what #203 domain affinity WOULD do. */
145
- const AGENT_MISSION_DOMAINS = {
146
- lenny: ["Health", "Ventures"],
147
- nano: ["Finances", "Ventures"],
148
- raider: ["Ventures"],
149
- };
150
- /** CC project dirs named in spec §2. "aironic-marine" matches BOTH the
151
- * AironicVentures and GAV checkouts — collected together as ONE source
152
- * ("cc-marine"), since the spec says "the `*aironic-marine*` DIRS" (plural). */
153
- const CC_INFRA_DIR = (0, node_path_1.join)(DEFAULT_CC_PROJECTS_DIR, "-Users-mattias-Development-DevOps-infrastructure");
154
- const CC_MARINE_DIRS = [
155
- (0, node_path_1.join)(DEFAULT_CC_PROJECTS_DIR, "-Users-mattias-Development-AironicVentures-aironic-marine"),
156
- (0, node_path_1.join)(DEFAULT_CC_PROJECTS_DIR, "-Users-mattias-Development-GAV-aironic-marine"),
157
- ];
158
- /** Hermes session `source` values that are NOT primary conversations (cron =
159
- * scheduled tasks, automated scans) — their "user" turns are boilerplate
160
- * ("Run daily scan") and would inject synthetic-feeling prompts, undermining
161
- * spec §1's "real prompts, not synthetic" requirement. Excluded, same as v1. */
162
- const NON_PRIMARY_HERMES_SOURCES = new Set(["cron"]);
163
- const SWEEP_VARIANTS = ["100", "150", "title1sent"];
164
- const SWEEP_KS = [4, 6, 8]; // K-sweep (supporting table, §5.1).
165
- // ---------------------------------------------------------------------------
166
- // Prompt sampling — spec §2: 20/source × 5 sources (3 Hermes + 2 CC)
167
- // ---------------------------------------------------------------------------
168
- function normalizeForDedupe(s) {
169
- return s.toLowerCase().replace(/\s+/g, " ").trim();
170
- }
171
- /** Pull user prompts from ONE Hermes profile state.db (flat /tmp path).
172
- * Read-only; missing/locked/schema-drifted DB returns []. */
173
- function collectHermesPrompts(dbPath, profile, sourceTag) {
174
- const out = [];
175
- const missionDomains = AGENT_MISSION_DOMAINS[profile] ?? ["Work"];
176
- let db;
177
- try {
178
- db = new better_sqlite3_1.default(dbPath, { readonly: true, fileMustExist: true });
179
- }
180
- catch {
181
- return [];
182
- }
183
- try {
184
- const sessions = db
185
- .prepare("SELECT id FROM sessions WHERE ended_at IS NOT NULL AND (source IS NULL OR source NOT IN ('cron'))")
186
- .all();
187
- const stmt = db.prepare("SELECT content FROM messages WHERE session_id = ? AND role = 'user' AND content IS NOT NULL ORDER BY id");
188
- for (const s of sessions) {
189
- const rows = stmt.all(s.id);
190
- for (const r of rows) {
191
- const text = (r.content ?? "").trim();
192
- if (text.length < MIN_PROMPT_CHARS)
193
- continue;
194
- out.push({ prompt: text, agent: profile, sourceTag, missionDomains, project: null, source: dbPath });
195
- }
196
- }
197
- }
198
- catch {
199
- return [];
200
- }
201
- finally {
202
- try {
203
- db.close();
204
- }
205
- catch {
206
- /* already closed */
207
- }
208
- }
209
- return out;
210
- }
211
- /** Decode a CC project directory name to a project name (last dash-split
212
- * token) — mirrors transcript-reader.ts's decodeProjectDirName. */
213
- function decodeProjectName(dirName) {
214
- const parts = dirName.split("-").filter(Boolean);
215
- return parts.length === 0 ? dirName : parts[parts.length - 1];
216
- }
217
- function extractUserText(content) {
218
- if (typeof content === "string")
219
- return content;
220
- if (!Array.isArray(content))
221
- return "";
222
- const texts = [];
223
- for (const block of content) {
224
- if (typeof block !== "object" || block === null)
225
- continue;
226
- if (block.type === "text") {
227
- texts.push(String(block.text ?? ""));
228
- }
229
- }
230
- return texts.join("\n").trim();
231
- }
232
- /** Collect every user prompt from ONE CC project dir's .jsonl session files. */
233
- function collectCcPromptsFromDir(projectDir, sourceTag) {
234
- const out = [];
235
- let files;
236
- try {
237
- if (!(0, node_fs_1.statSync)(projectDir).isDirectory())
238
- return [];
239
- files = (0, node_fs_1.readdirSync)(projectDir);
240
- }
241
- catch {
242
- return [];
243
- }
244
- const project = decodeProjectName(projectDir.split("/").filter(Boolean).pop() ?? projectDir);
245
- for (const file of files) {
246
- if (!file.endsWith(".jsonl"))
247
- continue;
248
- const filePath = (0, node_path_1.join)(projectDir, file);
249
- let raw;
250
- try {
251
- raw = (0, node_fs_1.readFileSync)(filePath, "utf-8");
252
- }
253
- catch {
254
- continue;
255
- }
256
- for (const line of raw.split("\n")) {
257
- if (!line.trim())
258
- continue;
259
- let entry;
260
- try {
261
- entry = JSON.parse(line);
262
- }
263
- catch {
264
- continue;
265
- }
266
- if (typeof entry !== "object" || entry === null)
267
- continue;
268
- const e = entry;
269
- const nested = e.message;
270
- const role = String(e.type ?? nested?.role ?? "");
271
- if (role !== "user")
272
- continue;
273
- const text = extractUserText(e.content ?? nested?.content).trim();
274
- if (text.length < MIN_PROMPT_CHARS)
275
- continue;
276
- out.push({
277
- prompt: text,
278
- agent: `cc-${project}`,
279
- sourceTag,
280
- missionDomains: [],
281
- project,
282
- source: filePath,
283
- });
284
- }
285
- }
286
- return out;
287
- }
288
- /** Dedupe (normalized text) + sort length DESC (prefer substantial prompts,
289
- * matching v1's heuristic) + take the first `n`. Logs a shortfall loudly —
290
- * never pads with synthetic content. */
291
- function pickTopN(all, n, sourceTag) {
292
- const seen = new Set();
293
- const deduped = [];
294
- for (const p of all) {
295
- const key = normalizeForDedupe(p.prompt);
296
- if (seen.has(key))
297
- continue;
298
- seen.add(key);
299
- deduped.push(p);
300
- }
301
- deduped.sort((a, b) => b.prompt.length - a.prompt.length);
302
- const picked = deduped.slice(0, n);
303
- if (picked.length < n) {
304
- console.warn(`[relevance-eval] WARNING: source "${sourceTag}" yielded only ${picked.length}/${n} prompts ` +
305
- `after dedupe — corpus will be short by ${n - picked.length}. Not padded with synthetic content.`);
306
- }
307
- return picked;
308
- }
309
- /** Fail loudly if a required Hermes profile DB is missing — this is a
310
- * precondition (task instructions), not a soft corpus shortfall. */
311
- function verifyHermesPreconditions() {
312
- const missing = Object.keys(HERMES_PROFILE_DBS).filter((p) => !(0, node_fs_1.existsSync)(HERMES_PROFILE_DBS[p]));
313
- if (missing.length > 0) {
314
- const lines = missing.map((p) => ` ssh agents@bedrock 'sqlite3 ~/.hermes/profiles/${p}/state.db ".backup /tmp/hermes-${p}-state.db"' && ` +
315
- `scp agents@bedrock:/tmp/hermes-${p}-state.db ${HERMES_PROFILE_DBS[p]}`);
316
- throw new Error(`relevance-eval: missing required Hermes state DB(s) for: ${missing.join(", ")}.\n` +
317
- `Create via a consistent SQLite backup:\n${lines.join("\n")}`);
318
- }
319
- }
320
- /** Build the v2 corpus fresh: 20 prompts/source × 5 sources (spec §2). */
321
- function buildV2Corpus() {
322
- verifyHermesPreconditions();
323
- const buckets = [];
324
- for (const [profile, dbPath] of Object.entries(HERMES_PROFILE_DBS)) {
325
- const sourceTag = `hermes-${profile}`;
326
- const all = collectHermesPrompts(dbPath, profile, sourceTag);
327
- buckets.push(...pickTopN(all, N_PER_SOURCE, sourceTag));
328
- }
329
- const infraAll = collectCcPromptsFromDir(CC_INFRA_DIR, "cc-infrastructure");
330
- buckets.push(...pickTopN(infraAll, N_PER_SOURCE, "cc-infrastructure"));
331
- const marineAll = CC_MARINE_DIRS.flatMap((d) => collectCcPromptsFromDir(d, "cc-marine"));
332
- buckets.push(...pickTopN(marineAll, N_PER_SOURCE, "cc-marine"));
333
- if (buckets.length === 0) {
334
- throw new Error("relevance-eval: corpus construction yielded 0 prompts across all 5 sources — aborting.");
335
- }
336
- return buckets;
337
- }
338
- /** A saved prompts.json satisfies the v2 corpus contract iff every one of the
339
- * 5 sources is present (count may be < 20/source if a source was exhausted —
340
- * see pickTopN's logged shortfall — but ALL 5 sources must be represented,
341
- * never silently missing one). */
342
- function isValidV2PromptSet(prompts) {
343
- if (prompts.length === 0)
344
- return false;
345
- const present = new Set(prompts.map((p) => p.sourceTag));
346
- return SOURCE_TAGS.every((t) => present.has(t));
347
- }
348
- function loadPromptsJson(path) {
349
- const raw = (0, node_fs_1.readFileSync)(path, "utf-8");
350
- const parsed = JSON.parse(raw);
351
- if (!Array.isArray(parsed))
352
- throw new Error(`relevance-eval: ${path} is not a JSON array`);
353
- return parsed.map((p, i) => {
354
- if (typeof p !== "object" || p === null)
355
- throw new Error(`relevance-eval: ${path}[${i}] is not an object`);
356
- const o = p;
357
- const prompt = o.prompt;
358
- if (typeof prompt !== "string")
359
- throw new Error(`relevance-eval: ${path}[${i}] missing string prompt`);
360
- const project = typeof o.project === "string" ? o.project : null;
361
- const missionDomainsRaw = o.missionDomains;
362
- const missionDomains = Array.isArray(missionDomainsRaw)
363
- ? missionDomainsRaw.filter((d) => typeof d === "string")
364
- : [];
365
- const agent = typeof o.agent === "string" ? o.agent : project !== null ? `cc-${project}` : "unknown";
366
- const source = typeof o.source === "string" ? o.source : "(saved set)";
367
- const sourceTag = SOURCE_TAGS.includes(String(o.sourceTag))
368
- ? o.sourceTag
369
- : project !== null
370
- ? `cc-${project}` // best-effort backfill for a pre-v2 saved set
371
- : `hermes-${agent}`;
372
- return { prompt, agent, sourceTag, missionDomains, project, source };
373
- });
374
- }
375
- function savePromptsJson(path, prompts) {
376
- (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(path), { recursive: true });
377
- (0, node_fs_1.writeFileSync)(path, JSON.stringify(prompts, null, 2), "utf-8");
378
- }
379
- /** For each source (in SOURCE_TAGS order), the first SWEEP_N_PER_SOURCE
380
- * prompts (by their position in the saved corpus) — the fixed 40-prompt
381
- * length-sweep subset (spec §4.2). Deterministic given the saved order. */
382
- function computeSweepSubsetIndices(prompts) {
383
- const bySource = new Map();
384
- prompts.forEach((p, idx) => {
385
- const arr = bySource.get(p.sourceTag) ?? [];
386
- arr.push(idx);
387
- bySource.set(p.sourceTag, arr);
388
- });
389
- const out = [];
390
- for (const tag of SOURCE_TAGS) {
391
- out.push(...(bySource.get(tag) ?? []).slice(0, SWEEP_N_PER_SOURCE));
392
- }
393
- return out;
394
- }
395
- // ---------------------------------------------------------------------------
396
- // Retrieve sweep (scope OFF vs ON) — LLM-free system-under-test
397
- // ---------------------------------------------------------------------------
398
- async function runRetrieveSweep(db, prompts, opts) {
399
- const embedCache = new Map();
400
- for (const p of prompts) {
401
- if (!embedCache.has(p.prompt))
402
- embedCache.set(p.prompt, await (0, embedder_js_1.embed)(p.prompt));
403
- }
404
- const neverCalledEmbed = async () => {
405
- throw new Error("relevance-eval: retrieve() called the embedFn — queryEmbedding was not honored. Eval aborted (results would be invalid).");
406
- };
407
- const results = [];
408
- for (const sp of prompts) {
409
- const queryEmbedding = embedCache.get(sp.prompt);
410
- for (const mode of ["off", "on"]) {
411
- const scopeOpts = mode === "on"
412
- ? sp.project !== null
413
- ? { project: sp.project }
414
- : { missionDomains: sp.missionDomains }
415
- : {};
416
- const retrieved = await (0, retrieval_js_1.retrieve)(db, neverCalledEmbed, sp.prompt, {
417
- limit: RESULT_K,
418
- queryEmbedding,
419
- noStrengthen: true,
420
- // #458: the pinned clock (undefined = live) — identical selections
421
- // across before/after runs regardless of when each run executes.
422
- ...(opts?.now ? { now: opts.now } : {}),
423
- ...scopeOpts,
424
- });
425
- const topIds = retrieved.map((r) => r.id);
426
- const contents = {};
427
- const rows = {};
428
- for (const r of retrieved) {
429
- contents[r.id] = r.content;
430
- rows[r.id] = r;
431
- }
432
- results.push({ prompt: sp, mode, topIds, contents, rows, fullVerdicts: [], lineVerdicts: [], redundancy: null });
433
- }
434
- }
435
- return results;
436
- }
437
- // ---------------------------------------------------------------------------
438
- // Rendering — production surface (imported, never reimplemented) + one
439
- // exploratory variant (title + first sentence) that has no production
440
- // equivalent, so it lives here.
441
- // ---------------------------------------------------------------------------
442
- const RECALL_BLOCK_HEADER = [
443
- "## Memory recall (auto)",
444
- // Copied verbatim from recall-index.ts's handleRecallIndex block header —
445
- // DISPLAY TEXT only, not rendering LOGIC (the logic under measurement,
446
- // formatIndexLine/memoryTitle, IS imported). Keep in sync if it changes.
447
- "Possibly relevant memories — dates matter, newer supersedes older. Fetch with `hicortex_get(id)` when an entry could change your action. Cite what you rely on by id + date, and mark it `FETCHED` if you read the full memory or `SNIPPET` if you're citing the one-line entry unread — don't pass a SNIPPET citation off as established fact.",
448
- ];
449
- /** Not a production surface — spec §4.2's exploratory 4th variant. Title
450
- * (same memoryTitle() production helper) + one sentence of body, so the
451
- * sweep can measure whether ONE extra sentence (not just a longer title)
452
- * closes the snippet gap. Date formatting is a local dd.mm.yyyy helper (not
453
- * exported from recall-index.ts) since this whole render path is synthetic. */
454
- function ddmmyyyy(iso) {
455
- const d = new Date(iso);
456
- if (isNaN(d.getTime()))
457
- return "";
458
- const dd = String(d.getDate()).padStart(2, "0");
459
- const mm = String(d.getMonth() + 1).padStart(2, "0");
460
- return `${dd}.${mm}.${d.getFullYear()}`;
461
- }
462
- function firstSentence(text) {
463
- const cleaned = text
464
- .replace(/^#+\s*/, "")
465
- .replace(/^Session Memory:\s*/i, "")
466
- .replace(/^Lesson:\s*/i, "")
467
- .trim();
468
- const m = cleaned.match(/^(.*?[.!?])(\s|$)/);
469
- return (m ? m[1] : cleaned).trim();
470
- }
471
- function renderTitleFirstSentenceLine(r) {
472
- const title = (0, recall_index_js_1.memoryTitle)(r.content, 100);
473
- const lines = r.content
474
- .split("\n")
475
- .map((l) => l.trim())
476
- .filter((l) => l.length > 0);
477
- const restAfterTitleLine = lines.slice(1).join(" ");
478
- const sentence = firstSentence(restAfterTitleLine.length > 0 ? restAfterTitleLine : r.content);
479
- const body = sentence && sentence.toLowerCase() !== title.toLowerCase() ? `${title} — ${sentence}` : title;
480
- const meta = [ddmmyyyy(r.created_at), r.domain ?? r.project ?? undefined, r.source_agent ?? undefined, r.memory_type]
481
- .filter(Boolean)
482
- .join(", ");
483
- return `- [${r.id}] ${body}${meta ? ` (${meta})` : ""}`;
484
- }
485
- function renderVariantLine(r, variant) {
486
- if (variant === "title1sent")
487
- return renderTitleFirstSentenceLine(r);
488
- return (0, recall_index_js_1.formatIndexLine)(r, parseInt(variant, 10));
489
- }
490
- /** The ACTUAL production block: relevance-gated (passesRelevanceGate,
491
- * imported), capped at maxItems, rendered via formatIndexLine (imported). */
492
- function renderProductionBlock(rows, minSimilarity, maxItems, maxLen = PROD_TITLE_CHARS) {
493
- const picked = rows.filter((r) => (0, recall_index_js_1.passesRelevanceGate)(r, minSimilarity)).slice(0, maxItems);
494
- const lines = {};
495
- for (const r of picked)
496
- lines[r.id] = (0, recall_index_js_1.formatIndexLine)(r, maxLen);
497
- const block = [...RECALL_BLOCK_HEADER, ...picked.map((r) => lines[r.id])].join("\n");
498
- return { block, shownIds: picked.map((r) => r.id), lines };
499
- }
500
- /** char/4 token estimate — no tokenizer available locally; stated in the
501
- * report. Good enough for a relative (variant-to-variant, K-to-K) comparison. */
502
- function estimateTokens(s) {
503
- return Math.ceil(s.length / 4);
504
- }
505
- // ---------------------------------------------------------------------------
506
- // Judge prompts
507
- // ---------------------------------------------------------------------------
508
- function buildFullContentJudgePrompt(userPrompt, topIds, contents) {
509
- const memBlock = [];
510
- for (let i = 0; i < topIds.length; i++) {
511
- const id = topIds[i];
512
- const label = `Q${i + 1}`;
513
- const full = contents[id] ?? "";
514
- const trimmed = full.length > MAX_MEM_CHARS_IN_JUDGE ? full.slice(0, MAX_MEM_CHARS_IN_JUDGE) + " […truncated]" : full;
515
- memBlock.push(`[${label}] (memory id ${id})\n${trimmed}`);
516
- }
517
- return ("You are a STRICT relevance judge for an AI coding agent's memory system. " +
518
- "A real user prompt and the memories the retrieval system surfaced are " +
519
- "below. Grade EACH memory independently against the SPECIFIC prompt.\n\n" +
520
- "Verdicts (pick exactly one per memory):\n" +
521
- "- relevant — genuinely useful for handling the user's actual need. The memory " +
522
- "carries information, context, or a lesson that would help answer or " +
523
- "contextualize the prompt. A tangential keyword overlap is NOT relevant.\n" +
524
- "- noise — irrelevant. Off-topic, wrong scope, or surfaced purely on a " +
525
- "keyword / lexical collision with no genuine connection to what the user is " +
526
- "asking.\n" +
527
- "- stale — once relevant but now outdated or superseded. The decision was " +
528
- "reversed, the value moved, the API was removed, the lesson no longer holds.\n\n" +
529
- "Grade relevance to the SPECIFIC prompt, not how interesting the memory is " +
530
- "in general. When in doubt between 'relevant' and 'noise', prefer 'noise' — " +
531
- "relevance inflation defeats the eval.\n\n" +
532
- `USER PROMPT:\n${userPrompt}\n\n` +
533
- `MEMORIES (Q1 = top-ranked):\n${memBlock.join("\n\n")}\n\n` +
534
- "Return ONLY a JSON array — one entry per memory, in this exact shape, no " +
535
- "prose, no markdown fences:\n" +
536
- '[{"id": "Q1", "verdict": "relevant|noise|stale", "reason": "<= 12 words"}, ...]');
537
- }
538
- /** Spec §4.1 — the line judge is BLIND to full content: a separate call that
539
- * sees ONLY the rendered production one-liner (or a sweep variant of it). */
540
- function buildLineJudgePrompt(userPrompt, topIds, lines) {
541
- const memBlock = [];
542
- for (let i = 0; i < topIds.length; i++) {
543
- const id = topIds[i];
544
- memBlock.push(`[Q${i + 1}] ${lines[id] ?? ""}`);
545
- }
546
- return ("You are a STRICT relevance judge for an AI coding agent's memory system. " +
547
- "The agent is shown ONLY the one-line index entry below per memory — " +
548
- "NEVER the full content. Judge whether the AGENT COULD TELL, FROM THE LINE " +
549
- "ALONE, that the memory is relevant to its actual need. Do not assume " +
550
- "information that isn't visible in the line.\n\n" +
551
- "Verdicts (pick exactly one per line):\n" +
552
- "- relevant — the line itself signals genuine relevance to the prompt.\n" +
553
- "- noise — the line signals no genuine connection, or is too vague/generic " +
554
- "to tell anything from.\n" +
555
- "- stale — the line itself signals outdated/superseded content.\n\n" +
556
- "When in doubt, prefer 'noise' — relevance inflation defeats this measurement.\n\n" +
557
- `USER PROMPT:\n${userPrompt}\n\n` +
558
- `INDEX LINES (Q1 = top-ranked):\n${memBlock.join("\n")}\n\n` +
559
- "Return ONLY a JSON array, same shape, no prose, no markdown fences:\n" +
560
- '[{"id": "Q1", "verdict": "relevant|noise|stale", "reason": "<= 12 words"}, ...]');
561
- }
562
- function buildRedundancyPrompt(userPrompt, shownLines) {
563
- const block = shownLines.map((l, i) => `[R${i + 1}] ${l}`).join("\n");
564
- return ("You are grading REDUNDANCY in a set of memory index lines an AI agent " +
565
- "would see TOGETHER for one prompt. Two lines are redundant if reading one " +
566
- "makes the other add NOTHING NEW (near-duplicates, restatements, or a " +
567
- "supersede-pair where only the newer one matters).\n\n" +
568
- `USER PROMPT:\n${userPrompt}\n\n` +
569
- `INDEX LINES SHOWN TOGETHER:\n${block}\n\n` +
570
- `Return ONLY JSON, no prose, no markdown fences: {"non_redundant_count": <integer 0-${shownLines.length}>, "reason": "<= 15 words"}\n` +
571
- "non_redundant_count = how many DISTINCT pieces of information are represented " +
572
- "(total lines minus duplicates counted beyond the first in each redundant group).");
573
- }
574
- // ---------------------------------------------------------------------------
575
- // Lenient JSON parse (v1, unchanged) — full/line/sweep share this parser.
576
- // ---------------------------------------------------------------------------
577
- const VALID_VERDICTS = new Set(["relevant", "noise", "stale"]);
578
- function normalizeVerdict(v) {
579
- if (typeof v !== "string")
580
- return null;
581
- const lower = v.toLowerCase().trim();
582
- if (lower.startsWith("rel"))
583
- return "relevant";
584
- if (lower.startsWith("noise") || lower === "irrelevant")
585
- return "noise";
586
- if (lower.startsWith("stale") || lower.startsWith("outdated") || lower.startsWith("superse"))
587
- return "stale";
588
- return VALID_VERDICTS.has(lower) ? lower : null;
589
- }
590
- function parseJudgeOutput(raw, expectedLabels) {
591
- const cleaned = raw.replace(/```(?:json)?/gi, "").trim();
592
- const start = cleaned.indexOf("[");
593
- const end = cleaned.lastIndexOf("]");
594
- if (start >= 0 && end > start) {
595
- try {
596
- const parsed = JSON.parse(cleaned.slice(start, end + 1));
597
- if (Array.isArray(parsed)) {
598
- const out = [];
599
- for (const item of parsed) {
600
- if (typeof item !== "object" || item === null)
601
- continue;
602
- const id = String(item.id ?? "").toUpperCase();
603
- if (!expectedLabels.has(id))
604
- continue;
605
- const verdict = normalizeVerdict(item.verdict);
606
- if (!verdict)
607
- continue;
608
- const reason = String(item.reason ?? "").trim();
609
- out.push({ id, verdict, reason });
610
- }
611
- if (out.length > 0)
612
- return out;
613
- }
614
- }
615
- catch {
616
- // fall through to heuristic
617
- }
618
- }
619
- const out = [];
620
- for (const line of cleaned.split("\n")) {
621
- const m = line.match(/\b(Q\d+)\b[\s\-:)]*([A-Za-z]+)/);
622
- if (!m)
623
- continue;
624
- const id = m[1].toUpperCase();
625
- if (!expectedLabels.has(id))
626
- continue;
627
- const verdict = normalizeVerdict(m[2]);
628
- if (!verdict)
629
- continue;
630
- const tail = line.slice((m.index ?? 0) + m[0].length).replace(/^[\s\-:;.,)]+/, "");
631
- const reason = tail.split(/[.\n]/)[0].trim().slice(0, 80);
632
- out.push({ id, verdict, reason });
633
- }
634
- return out.length > 0 ? out : null;
635
- }
636
- function parseRedundancyOutput(raw, shown) {
637
- const cleaned = raw.replace(/```(?:json)?/gi, "").trim();
638
- const start = cleaned.indexOf("{");
639
- const end = cleaned.lastIndexOf("}");
640
- if (start >= 0 && end > start) {
641
- try {
642
- const obj = JSON.parse(cleaned.slice(start, end + 1));
643
- const n = Number(obj.non_redundant_count);
644
- if (Number.isFinite(n)) {
645
- return {
646
- nonRedundantCount: Math.max(0, Math.min(shown, Math.round(n))),
647
- reason: String(obj.reason ?? "").trim(),
648
- };
649
- }
650
- }
651
- catch {
652
- // fall through
653
- }
654
- }
655
- const m = cleaned.match(/(\d+)/);
656
- if (m)
657
- return { nonRedundantCount: Math.max(0, Math.min(shown, parseInt(m[1], 10))), reason: "" };
658
- return null;
659
- }
660
- // ---------------------------------------------------------------------------
661
- // Rate limiting + resumability (spec §6b, MANDATORY)
662
- // ---------------------------------------------------------------------------
663
- class JudgeHttpError extends Error {
664
- status;
665
- retryAfterMs;
666
- constructor(status, message, retryAfterMs) {
667
- super(message);
668
- this.name = "JudgeHttpError";
669
- this.status = status;
670
- this.retryAfterMs = retryAfterMs;
671
- }
672
- }
673
- /** Thrown when the run must stop (budget or error-rate abort). Caught once at
674
- * the top of the judging pipeline in main() — whatever completed so far is
675
- * still persisted and reported, clearly marked as an aborted/partial run. */
676
- class RunAbortedError extends Error {
677
- }
678
- function sleep(ms) {
679
- return new Promise((resolve) => setTimeout(resolve, ms));
680
- }
681
- /** ONE raw HTTP attempt against z.ai's Anthropic-compatible endpoint. Mirrors
682
- * llm.ts#completeAnthropic's request shape exactly (model/messages/max_tokens,
683
- * x-api-key header) but with NO built-in retry — the backoff wrapper below
684
- * owns retries so the schedule matches spec §6b precisely. */
685
- async function callZaiOnce(apiKey, prompt, maxTokens, timeoutMs = 90_000) {
686
- const url = `${ZAI_BASE_URL.replace(/\/$/, "")}/v1/messages`;
687
- let resp;
688
- try {
689
- resp = await fetch(url, {
690
- method: "POST",
691
- headers: {
692
- "Content-Type": "application/json",
693
- "x-api-key": apiKey,
694
- "anthropic-version": "2023-06-01",
695
- },
696
- body: JSON.stringify({ model: ZAI_MODEL, messages: [{ role: "user", content: prompt }], max_tokens: maxTokens }),
697
- signal: AbortSignal.timeout(timeoutMs),
698
- });
699
- }
700
- catch (err) {
701
- const msg = err instanceof Error ? err.message : String(err);
702
- throw new JudgeHttpError(0, `network/timeout: ${msg}`); // status 0 = retryable (network/timeout)
703
- }
704
- if (resp.status === 429 || resp.status >= 500) {
705
- const retryAfterHeader = resp.headers.get("retry-after");
706
- const retryAfterMs = retryAfterHeader ? parseInt(retryAfterHeader, 10) * 1000 : undefined;
707
- const text = await resp.text().catch(() => "");
708
- throw new JudgeHttpError(resp.status, `HTTP ${resp.status}: ${text.slice(0, 200)}`, retryAfterMs);
709
- }
710
- if (!resp.ok) {
711
- const text = await resp.text().catch(() => "");
712
- throw new JudgeHttpError(resp.status, `HTTP ${resp.status} (non-retryable): ${text.slice(0, 200)}`);
713
- }
714
- const data = (await resp.json());
715
- const block = data.content?.find((c) => c.type === "text");
716
- return (block?.text ?? "").trim();
717
- }
718
- function jitter(ms) {
719
- return Math.round(ms * (0.8 + Math.random() * 0.4)); // ±20%
720
- }
721
- /** Retryable = 429, 5xx, or network/timeout (status 0). Any other non-ok
722
- * status (400/401/403/404) fails immediately — retrying a malformed request
723
- * or an auth failure would waste the call budget on a guaranteed repeat. */
724
- function isRetryable(err) {
725
- return err instanceof JudgeHttpError && (err.status === 429 || err.status >= 500 || err.status === 0);
726
- }
727
- async function callZaiWithBackoff(prompt, maxTokens, state) {
728
- let lastErr;
729
- for (let attempt = 0; attempt <= BACKOFF_SCHEDULE_MS.length; attempt++) {
730
- try {
731
- return await callZaiOnce(state.apiKey, prompt, maxTokens);
732
- }
733
- catch (err) {
734
- lastErr = err;
735
- if (!isRetryable(err) || attempt === BACKOFF_SCHEDULE_MS.length)
736
- throw err;
737
- const scheduled = BACKOFF_SCHEDULE_MS[attempt];
738
- const waitMs = err.retryAfterMs ?? jitter(scheduled); // Retry-After wins over the schedule.
739
- state.totalRetries++;
740
- console.log(`[relevance-eval] judge call failed (${err.message.slice(0, 100)}), ` +
741
- `retry ${attempt + 1}/${BACKOFF_SCHEDULE_MS.length} in ${Math.round(waitMs / 1000)}s...`);
742
- await sleep(waitMs);
743
- }
744
- }
745
- throw lastErr;
746
- }
747
- function loadResumeMap(sidecarPath) {
748
- const map = new Map();
749
- if (!(0, node_fs_1.existsSync)(sidecarPath))
750
- return map;
751
- const raw = (0, node_fs_1.readFileSync)(sidecarPath, "utf-8");
752
- for (const line of raw.split("\n")) {
753
- if (!line.trim())
754
- continue;
755
- try {
756
- const obj = JSON.parse(line);
757
- if (obj && typeof obj.key === "string")
758
- map.set(obj.key, obj.result);
759
- }
760
- catch {
761
- // corrupt checkpoint line — skip, don't crash the whole resume.
762
- }
763
- }
764
- return map;
765
- }
766
- /** One judge UNIT: resume-cache check → budget check → delay → call w/
767
- * backoff → parse → checkpoint-append → progress log → error-rate check. */
768
- async function runJudgeUnit(state, key, buildPrompt, maxTokens, parse, parseErrorLabel) {
769
- const cached = state.resumed.get(key);
770
- if (cached !== undefined) {
771
- console.log(`[relevance-eval] resumed: ${key}`);
772
- return cached;
773
- }
774
- if (state.aborted)
775
- throw new RunAbortedError(state.aborted.reason);
776
- if (state.totalCalls >= state.maxCalls) {
777
- state.aborted = { reason: `--max-calls budget (${state.maxCalls}) reached` };
778
- throw new RunAbortedError(state.aborted.reason);
779
- }
780
- await sleep(state.judgeDelayMs);
781
- state.totalCalls++;
782
- let result;
783
- try {
784
- const raw = await callZaiWithBackoff(buildPrompt(), maxTokens, state);
785
- const parsed = parse(raw);
786
- result = parsed ?? { judgeError: `${parseErrorLabel}; raw head: ${raw.slice(0, 200).replace(/\s+/g, " ")}` };
787
- }
788
- catch (err) {
789
- result = { judgeError: (err instanceof Error ? err.message : String(err)).slice(0, 300) };
790
- }
791
- state.completedUnits++;
792
- if (typeof result === "object" && result !== null && "judgeError" in result) {
793
- state.errorUnits++;
794
- }
795
- (0, node_fs_1.appendFileSync)(state.sidecarPath, JSON.stringify({ key, result }) + "\n", "utf-8");
796
- const elapsedS = Math.round((Date.now() - state.startTime) / 1000);
797
- console.log(`[relevance-eval] judged ${state.completedUnits} (calls ${state.totalCalls}/${state.maxCalls}, ` +
798
- `${elapsedS}s elapsed, ${state.totalRetries} retries, ${state.errorUnits} judge_error) — ${key}`);
799
- if (state.completedUnits >= ERROR_RATE_MIN_SAMPLE && state.errorUnits / state.completedUnits > ERROR_RATE_ABORT_THRESHOLD) {
800
- state.aborted = {
801
- reason: `judge-error rate ${((state.errorUnits / state.completedUnits) * 100).toFixed(1)}% ` +
802
- `exceeds the 10% abort threshold after ${state.completedUnits} units`,
803
- };
804
- throw new RunAbortedError(state.aborted.reason);
805
- }
806
- return result;
807
- }
808
- async function judgeVerdictUnit(state, key, promptText, expectedLabels) {
809
- if (expectedLabels.size === 0)
810
- return { judgeError: "retrieve() surfaced 0 memories — nothing to grade" };
811
- return runJudgeUnit(state, key, () => promptText, 2048, (raw) => parseJudgeOutput(raw, expectedLabels), "parse failed");
812
- }
813
- async function judgeRedundancyUnit(state, key, promptText, shown) {
814
- return runJudgeUnit(state, key, () => promptText, 512, (raw) => parseRedundancyOutput(raw, shown), "parse failed");
815
- }
816
- function computeMetricsAtK(results, k) {
817
- const judged = results.filter((r) => Array.isArray(r.fullVerdicts));
818
- if (judged.length === 0)
819
- return null;
820
- const inK = (id) => {
821
- const idx = parseInt(id.slice(1), 10);
822
- return Number.isInteger(idx) && idx >= 1 && idx <= k;
823
- };
824
- let relTotal = 0, noiseTotal = 0, staleTotal = 0, verdictTotal = 0;
825
- const perPrompt = [];
826
- for (const r of judged) {
827
- let promptRel = 0;
828
- for (const v of r.fullVerdicts) {
829
- if (!inK(v.id))
830
- continue;
831
- verdictTotal++;
832
- if (v.verdict === "relevant") {
833
- relTotal++;
834
- promptRel++;
835
- }
836
- else if (v.verdict === "noise")
837
- noiseTotal++;
838
- else
839
- staleTotal++;
840
- }
841
- perPrompt.push(promptRel / k);
842
- }
843
- return {
844
- precision: perPrompt.reduce((a, b) => a + b, 0) / perPrompt.length,
845
- noiseRate: verdictTotal > 0 ? noiseTotal / verdictTotal : 0,
846
- staleRate: verdictTotal > 0 ? staleTotal / verdictTotal : 0,
847
- n: judged.length,
848
- perPrompt,
849
- };
850
- }
851
- function computePromptHistogram(results, mode) {
852
- const buckets = new Array(RESULT_K + 1).fill(0);
853
- for (const r of results.filter((r) => r.mode === mode)) {
854
- if (!Array.isArray(r.fullVerdicts))
855
- continue;
856
- const rel = r.fullVerdicts.filter((v) => v.verdict === "relevant").length;
857
- if (rel >= 0 && rel <= RESULT_K)
858
- buckets[rel]++;
859
- }
860
- return buckets;
861
- }
862
- function findWorstBestPrompts(results, mode, worstRelMax, bestRelMin) {
863
- const named = [];
864
- for (const r of results.filter((r) => r.mode === mode)) {
865
- if (!Array.isArray(r.fullVerdicts))
866
- continue;
867
- let rel = 0, noise = 0, stale = 0;
868
- const perRank = [];
869
- for (const v of r.fullVerdicts) {
870
- const rank = parseInt(v.id.slice(1), 10);
871
- const memId = r.topIds[rank - 1] ?? "?";
872
- if (v.verdict === "relevant")
873
- rel++;
874
- else if (v.verdict === "noise")
875
- noise++;
876
- else
877
- stale++;
878
- perRank.push({ rank, memory_id: memId, verdict: v.verdict, reason: v.reason, content: r.contents[memId] ?? "" });
879
- }
880
- named.push({ prompt: r.prompt.prompt, agent: r.prompt.agent, relevant: rel, noise, stale, perRank });
881
- }
882
- return {
883
- worst: named.filter((p) => p.relevant <= worstRelMax),
884
- best: named.filter((p) => p.relevant >= bestRelMin),
885
- };
886
- }
887
- function rankChronicMemories(results, metaById) {
888
- const byId = new Map();
889
- for (const r of results) {
890
- if (!Array.isArray(r.fullVerdicts))
891
- continue;
892
- const seenNoiseForPrompt = new Set();
893
- const seenStaleForPrompt = new Set();
894
- const seenSurfaceForPrompt = new Set();
895
- for (const v of r.fullVerdicts) {
896
- const rank = parseInt(v.id.slice(1), 10);
897
- const memId = r.topIds[rank - 1];
898
- if (!memId)
899
- continue;
900
- let entry = byId.get(memId);
901
- if (!entry) {
902
- const meta = metaById.get(memId) ?? { content: r.contents[memId] ?? "", memory_type: null, project: null };
903
- entry = { memory_id: memId, noisePrompts: 0, stalePrompts: 0, totalSurfaces: 0, ...meta };
904
- byId.set(memId, entry);
905
- }
906
- if (!seenSurfaceForPrompt.has(memId)) {
907
- entry.totalSurfaces++;
908
- seenSurfaceForPrompt.add(memId);
909
- }
910
- if (v.verdict === "noise" && !seenNoiseForPrompt.has(memId)) {
911
- entry.noisePrompts++;
912
- seenNoiseForPrompt.add(memId);
913
- }
914
- else if (v.verdict === "stale" && !seenStaleForPrompt.has(memId)) {
915
- entry.stalePrompts++;
916
- seenStaleForPrompt.add(memId);
917
- }
918
- }
919
- }
920
- return Array.from(byId.values());
921
- }
922
- function lookupMemoryMeta(db, results) {
923
- const ids = new Set();
924
- for (const r of results)
925
- for (const id of r.topIds)
926
- ids.add(id);
927
- const out = new Map();
928
- if (ids.size === 0)
929
- return out;
930
- const placeholders = Array.from(ids, () => "?").join(",");
931
- try {
932
- const rows = db
933
- .prepare(`SELECT id, content, memory_type, project FROM memories WHERE id IN (${placeholders})`)
934
- .all(...ids);
935
- for (const row of rows) {
936
- out.set(row.id, { content: row.content ?? "", memory_type: row.memory_type, project: row.project });
937
- }
938
- }
939
- catch {
940
- // Schema drift — empty map; callers fall back to cached content.
941
- }
942
- for (const r of results) {
943
- for (const id of r.topIds) {
944
- if (!out.has(id))
945
- out.set(id, { content: r.contents[id] ?? "", memory_type: null, project: null });
946
- }
947
- }
948
- return out;
949
- }
950
- function computeRankMetrics(results) {
951
- const judged = results.filter((r) => Array.isArray(r.fullVerdicts));
952
- const rel = new Array(RESULT_K).fill(0);
953
- const noise = new Array(RESULT_K).fill(0);
954
- const stale = new Array(RESULT_K).fill(0);
955
- const n = new Array(RESULT_K).fill(0);
956
- for (const r of judged) {
957
- for (const v of r.fullVerdicts) {
958
- const idx = parseInt(v.id.slice(1), 10) - 1;
959
- if (!Number.isInteger(idx) || idx < 0 || idx >= RESULT_K)
960
- continue;
961
- n[idx]++;
962
- if (v.verdict === "relevant")
963
- rel[idx]++;
964
- else if (v.verdict === "noise")
965
- noise[idx]++;
966
- else
967
- stale[idx]++;
968
- }
969
- }
970
- const rows = [];
971
- for (let i = 0; i < RESULT_K; i++) {
972
- rows.push({
973
- rank: i + 1,
974
- n: n[i],
975
- relevantRate: n[i] > 0 ? rel[i] / n[i] : 0,
976
- noiseRate: n[i] > 0 ? noise[i] / n[i] : 0,
977
- staleRate: n[i] > 0 ? stale[i] / n[i] : 0,
978
- });
979
- }
980
- return rows;
981
- }
982
- function aggregateRankBand(results, fromRank, toRank) {
983
- const judged = results.filter((r) => Array.isArray(r.fullVerdicts));
984
- const want = new Set();
985
- for (let r = fromRank; r <= toRank; r++)
986
- want.add(`Q${r}`);
987
- let noise = 0, total = 0;
988
- for (const r of judged) {
989
- for (const v of r.fullVerdicts) {
990
- if (!want.has(v.id))
991
- continue;
992
- total++;
993
- if (v.verdict === "noise")
994
- noise++;
995
- }
996
- }
997
- return { noise, total, rate: total > 0 ? noise / total : 0 };
998
- }
999
- function computeSnippetGap(results) {
1000
- let actionable = 0, snippetFailure = 0, misleadingLine = 0, fullRelevantTotal = 0, n = 0;
1001
- for (const r of results) {
1002
- if (!Array.isArray(r.fullVerdicts) || !Array.isArray(r.lineVerdicts))
1003
- continue;
1004
- const lineById = new Map(r.lineVerdicts.map((v) => [v.id, v]));
1005
- for (const fv of r.fullVerdicts) {
1006
- const lv = lineById.get(fv.id);
1007
- if (!lv)
1008
- continue;
1009
- n++;
1010
- const fullRel = fv.verdict === "relevant";
1011
- const lineRel = lv.verdict === "relevant";
1012
- if (fullRel)
1013
- fullRelevantTotal++;
1014
- if (fullRel && lineRel)
1015
- actionable++;
1016
- else if (fullRel && !lineRel)
1017
- snippetFailure++;
1018
- else if (!fullRel && lineRel)
1019
- misleadingLine++;
1020
- }
1021
- }
1022
- return { actionable, snippetFailure, misleadingLine, fullRelevantTotal, n };
1023
- }
1024
- function findNamedSnippetFailures(results, limit) {
1025
- const out = [];
1026
- for (const r of results) {
1027
- if (!Array.isArray(r.fullVerdicts) || !Array.isArray(r.lineVerdicts))
1028
- continue;
1029
- const lineById = new Map(r.lineVerdicts.map((v) => [v.id, v]));
1030
- for (const fv of r.fullVerdicts) {
1031
- const lv = lineById.get(fv.id);
1032
- if (!lv)
1033
- continue;
1034
- if (fv.verdict === "relevant" && lv.verdict !== "relevant") {
1035
- const rank = parseInt(fv.id.slice(1), 10);
1036
- const memId = r.topIds[rank - 1] ?? "?";
1037
- const row = r.rows[memId];
1038
- out.push({
1039
- prompt: r.prompt.prompt,
1040
- agent: r.prompt.agent,
1041
- mode: r.mode,
1042
- memory_id: memId,
1043
- rendered_line: row ? (0, recall_index_js_1.formatIndexLine)(row, PROD_TITLE_CHARS) : "(unavailable)",
1044
- full_reason: fv.reason,
1045
- line_reason: lv.reason,
1046
- content: r.contents[memId] ?? "",
1047
- });
1048
- }
1049
- }
1050
- }
1051
- return out.slice(0, limit);
1052
- }
1053
- function mean(xs) {
1054
- return xs.length === 0 ? NaN : xs.reduce((a, b) => a + b, 0) / xs.length;
1055
- }
1056
- function sd(xs) {
1057
- if (xs.length < 2)
1058
- return 0;
1059
- const m = mean(xs);
1060
- return Math.sqrt(xs.reduce((a, b) => a + (b - m) ** 2, 0) / (xs.length - 1));
1061
- }
1062
- function ci95(xs) {
1063
- const m = mean(xs);
1064
- const s = sd(xs);
1065
- const se = xs.length > 0 ? s / Math.sqrt(xs.length) : NaN;
1066
- return { mean: m, sd: s, se, lo: m - 1.96 * se, hi: m + 1.96 * se, n: xs.length };
1067
- }
1068
- /** Similarity buckets for §5.9 — width-0.05 bands from 0.30 to 1.00, plus a
1069
- * catch-all "<0.30" and a "n/a" bucket for FTS/graph hits (no measured
1070
- * cosine — they bypass the floor entirely per passesRelevanceGate). */
1071
- const SIM_BUCKET_EDGES = [0.3, 0.35, 0.4, 0.45, 0.5, 0.55, 0.6, 0.65, 0.7, 0.75, 0.8, 0.85, 0.9, 0.95, 1.001];
1072
- function similarityBucketLabel(sim) {
1073
- if (sim === null || sim === undefined)
1074
- return "n/a (fts/graph, no cosine)";
1075
- if (sim < SIM_BUCKET_EDGES[0])
1076
- return "<0.30";
1077
- for (let i = 0; i < SIM_BUCKET_EDGES.length - 1; i++) {
1078
- if (sim >= SIM_BUCKET_EDGES[i] && sim < SIM_BUCKET_EDGES[i + 1]) {
1079
- const hi = Math.min(SIM_BUCKET_EDGES[i + 1], 1);
1080
- return `${SIM_BUCKET_EDGES[i].toFixed(2)}–${hi.toFixed(2)}`;
1081
- }
1082
- }
1083
- return ">=1.00";
1084
- }
1085
- function buildVerdictRows(prompts, results, sweepRows, runsTotal = 1) {
1086
- const byKey = new Map();
1087
- for (const r of results) {
1088
- const pIdx = prompts.findIndex((p) => p.prompt === r.prompt.prompt && p.agent === r.prompt.agent);
1089
- byKey.set(`${pIdx}:${r.mode}`, r);
1090
- }
1091
- const rows = [];
1092
- // Baseline (variant PROD_TITLE_CHARS, production render) — one row per (prompt, mode, rank).
1093
- for (let i = 0; i < prompts.length; i++) {
1094
- for (const mode of ["off", "on"]) {
1095
- const r = byKey.get(`${i}:${mode}`);
1096
- if (!r)
1097
- continue;
1098
- const fullOk = Array.isArray(r.fullVerdicts);
1099
- const lineOk = Array.isArray(r.lineVerdicts);
1100
- const fullById = fullOk ? new Map(r.fullVerdicts.map((v) => [v.id, v])) : null;
1101
- const lineById = lineOk ? new Map(r.lineVerdicts.map((v) => [v.id, v])) : null;
1102
- const ranks = r.topIds.length > 0 ? r.topIds.map((_, idx) => idx + 1) : [0];
1103
- for (const rank of ranks) {
1104
- const label = `Q${rank}`;
1105
- const memId = rank > 0 ? r.topIds[rank - 1] : "";
1106
- const row = memId ? r.rows[memId] : undefined;
1107
- const fv = fullById?.get(label);
1108
- const lv = lineById?.get(label);
1109
- rows.push({
1110
- prompt_idx: i,
1111
- prompt: r.prompt.prompt,
1112
- source: r.prompt.sourceTag,
1113
- agent: r.prompt.agent,
1114
- mode,
1115
- rank,
1116
- memory_id: memId,
1117
- memory_type: row?.memory_type ?? null,
1118
- similarity: row?.similarity ?? null,
1119
- retrieval_source: row?.source ?? null,
1120
- full_verdict: fv ? fv.verdict : fullOk ? null : "judge_error",
1121
- full_reason: fv ? fv.reason : fullOk ? "" : r.fullVerdicts.judgeError,
1122
- line_verdict: lv ? lv.verdict : lineOk ? null : "judge_error",
1123
- line_reason: lv ? lv.reason : lineOk ? "" : r.lineVerdicts.judgeError,
1124
- rendered_line: row ? (0, recall_index_js_1.formatIndexLine)(row, PROD_TITLE_CHARS) : "",
1125
- snippet_len_variant: String(PROD_TITLE_CHARS),
1126
- ...(runsTotal > 1 ? { run: 1 } : {}),
1127
- });
1128
- }
1129
- }
1130
- }
1131
- // Sweep rows (variants 100/120/200/title1sent — PROD_TITLE_CHARS (150)
1132
- // already covered above by the baseline pass; mode is always "off" per the
1133
- // resource-bounded design, spec §4.2).
1134
- for (const sw of sweepRows) {
1135
- if (sw.variant === String(PROD_TITLE_CHARS))
1136
- continue; // covered by the baseline pass above.
1137
- const r = byKey.get(`${sw.promptIdx}:off`);
1138
- if (!r)
1139
- continue;
1140
- const fullById = Array.isArray(r.fullVerdicts) ? new Map(r.fullVerdicts.map((v) => [v.id, v])) : null;
1141
- const lineOk = Array.isArray(sw.verdicts);
1142
- const lineById = lineOk ? new Map(sw.verdicts.map((v) => [v.id, v])) : null;
1143
- for (let rank = 1; rank <= r.topIds.length; rank++) {
1144
- const label = `Q${rank}`;
1145
- const memId = r.topIds[rank - 1];
1146
- const row = r.rows[memId];
1147
- const fv = fullById?.get(label);
1148
- const lv = lineById?.get(label);
1149
- rows.push({
1150
- prompt_idx: sw.promptIdx,
1151
- prompt: r.prompt.prompt,
1152
- source: r.prompt.sourceTag,
1153
- agent: r.prompt.agent,
1154
- mode: "off",
1155
- rank,
1156
- memory_id: memId,
1157
- memory_type: row?.memory_type ?? null,
1158
- similarity: row?.similarity ?? null,
1159
- retrieval_source: row?.source ?? null,
1160
- full_verdict: fv ? fv.verdict : null, // carried over from the baseline full judging; not re-run per variant.
1161
- full_reason: fv ? fv.reason : "",
1162
- line_verdict: lv ? lv.verdict : lineOk ? null : "judge_error",
1163
- line_reason: lv ? lv.reason : lineOk ? "" : sw.verdicts.judgeError,
1164
- rendered_line: sw.lines[memId] ?? "",
1165
- snippet_len_variant: sw.variant,
1166
- });
1167
- }
1168
- }
1169
- return rows;
1170
- }
1171
- /** Parsed and exported for tests (#458). Throws on an invalid --runs value
1172
- * (NaN or <1) — fail-explicit, never a silent default. */
1173
- function parseArgs(argv) {
1174
- const positionals = [];
1175
- let judgeDelayMs = DEFAULT_JUDGE_DELAY_MS;
1176
- let maxCalls = DEFAULT_MAX_CALLS;
1177
- let maxCallsExplicit = false;
1178
- let resume = false;
1179
- let verdictsJsonPath = VERDICTS_JSON_PATH;
1180
- let verdictsJsonlPath = VERDICTS_JSONL_PATH;
1181
- let nowIso;
1182
- let runs = 1;
1183
- for (const arg of argv) {
1184
- if (arg === "--resume") {
1185
- resume = true;
1186
- continue;
1187
- }
1188
- const m = arg.match(/^--([a-z-]+)=(.+)$/);
1189
- if (m) {
1190
- if (m[1] === "judge-delay-ms")
1191
- judgeDelayMs = parseInt(m[2], 10);
1192
- else if (m[1] === "max-calls") {
1193
- maxCalls = parseInt(m[2], 10);
1194
- maxCallsExplicit = true;
1195
- }
1196
- else if (m[1] === "verdicts-json")
1197
- verdictsJsonPath = m[2];
1198
- else if (m[1] === "verdicts-jsonl")
1199
- verdictsJsonlPath = m[2];
1200
- else if (m[1] === "now")
1201
- nowIso = m[2];
1202
- else if (m[1] === "runs") {
1203
- const parsed = Number(m[2]);
1204
- if (!Number.isInteger(parsed) || parsed < 1) {
1205
- throw new Error(`--runs must be an integer >= 1 (got "${m[2]}")`);
1206
- }
1207
- runs = parsed;
1208
- }
1209
- continue;
1210
- }
1211
- positionals.push(arg);
1212
- }
1213
- return {
1214
- positionals,
1215
- judgeDelayMs,
1216
- maxCalls,
1217
- maxCallsExplicit,
1218
- resume,
1219
- verdictsJsonPath,
1220
- verdictsJsonlPath,
1221
- nowIso,
1222
- runs,
1223
- };
1224
- }
1225
- /**
1226
- * #458 — resolve the effective --max-calls budget. The ONE shared budget
1227
- * spans ALL judge runs, so when --runs>1 and the caller did NOT pass
1228
- * --max-calls explicitly, the default scales ×N (each run re-judges every
1229
- * unit). An explicit budget always wins — the caller sized it for the whole
1230
- * invocation. Pure, exported for tests.
1231
- */
1232
- function scaleMaxCalls(runs, explicitMaxCalls, defaultMaxCalls) {
1233
- if (explicitMaxCalls !== undefined)
1234
- return explicitMaxCalls;
1235
- return runs > 1 ? defaultMaxCalls * runs : defaultMaxCalls;
1236
- }
1237
- // ---------------------------------------------------------------------------
1238
- // Judge variance (#458) — median, spread, and selection-identical flip
1239
- // counts over N runs of the SAME selections. Pure + synthetic-verdict
1240
- // testable; the orchestration (running the judge N times) lives in main.
1241
- // ---------------------------------------------------------------------------
1242
- /** Median of a numeric sample (even count → mean of the two middle values).
1243
- * Empty input → NaN, the mean() convention above. Does not mutate input. */
1244
- function median(xs) {
1245
- if (xs.length === 0)
1246
- return NaN;
1247
- const sorted = [...xs].sort((a, b) => a - b);
1248
- const mid = Math.floor(sorted.length / 2);
1249
- return sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
1250
- }
1251
- /**
1252
- * Compute the #458 judge-variance account over N runs of the same selections:
1253
- * per-metric median + spread (the noise floor), and pairwise verdict flip
1254
- * counts (rows = one Q-label judgment on one (prompt,mode) unit; units flip
1255
- * if any row does). `perRunVerdicts[r]` is run r+1's units in a stable
1256
- * order/identity; `metricSeries[s].values[r]` is run r+1's headline value.
1257
- */
1258
- function computeJudgeVariance(perRunVerdicts, metricSeries) {
1259
- const runs = perRunVerdicts.length;
1260
- const flips = [];
1261
- let maxRows = 0;
1262
- let maxUnits = 0;
1263
- for (let i = 0; i < runs; i++) {
1264
- for (let j = i + 1; j < runs; j++) {
1265
- const byKeyI = new Map(perRunVerdicts[i].map((u) => [u.key, u.verdicts]));
1266
- let rowsCompared = 0;
1267
- let rowsFlipped = 0;
1268
- let unitsCompared = 0;
1269
- let unitsFlipped = 0;
1270
- for (const unitJ of perRunVerdicts[j]) {
1271
- const vi = byKeyI.get(unitJ.key);
1272
- if (!vi || !Array.isArray(vi) || !Array.isArray(unitJ.verdicts))
1273
- continue; // judge_error/absent — not comparable
1274
- const vj = unitJ.verdicts;
1275
- unitsCompared++;
1276
- let unitFlipped = false;
1277
- const byLabelJ = new Map(vj.map((v) => [v.id, v.verdict]));
1278
- for (const rowI of vi) {
1279
- const verdictJ = byLabelJ.get(rowI.id);
1280
- if (verdictJ === undefined)
1281
- continue; // label omitted in run j — not comparable
1282
- rowsCompared++;
1283
- if (verdictJ !== rowI.verdict) {
1284
- rowsFlipped++;
1285
- unitFlipped = true;
1286
- }
1287
- }
1288
- if (unitFlipped)
1289
- unitsFlipped++;
1290
- }
1291
- flips.push({ runA: i + 1, runB: j + 1, rowsCompared, rowsFlipped, unitsCompared, unitsFlipped });
1292
- maxRows = Math.max(maxRows, rowsFlipped);
1293
- maxUnits = Math.max(maxUnits, unitsFlipped);
1294
- }
1295
- }
1296
- const metrics = metricSeries.map((s) => {
1297
- const usable = s.values.filter((v) => v !== null && Number.isFinite(v));
1298
- if (usable.length === 0)
1299
- return { ...s, median: null, spreadPts: null };
1300
- return {
1301
- ...s,
1302
- median: median(usable),
1303
- spreadPts: (Math.max(...usable) - Math.min(...usable)) * 100,
1304
- };
1305
- });
1306
- const judgeErrorUnitsPerRun = perRunVerdicts.map((units) => units.filter((u) => !Array.isArray(u.verdicts)).length);
1307
- return {
1308
- runs,
1309
- metrics,
1310
- flips,
1311
- maxRowsFlipped: maxRows,
1312
- maxUnitsFlipped: maxUnits,
1313
- meanRowsFlipped: flips.length > 0 ? mean(flips.map((f) => f.rowsFlipped)) : 0,
1314
- meanUnitsFlipped: flips.length > 0 ? mean(flips.map((f) => f.unitsFlipped)) : 0,
1315
- totalComparableRows: flips.length > 0 ? flips[0].rowsCompared : 0,
1316
- judgeErrorUnitsPerRun,
1317
- };
1318
- }
1319
- // ---------------------------------------------------------------------------
1320
- // Reporting
1321
- // ---------------------------------------------------------------------------
1322
- function pct(x) {
1323
- return Number.isFinite(x) ? `${(x * 100).toFixed(1)}%` : "n/a";
1324
- }
1325
- function pts(x) {
1326
- return Number.isFinite(x) ? `${(x * 100).toFixed(1)}pts` : "n/a";
1327
- }
1328
- function fmtNum(x, decimals = 2) {
1329
- return Number.isFinite(x) ? x.toFixed(decimals) : "n/a";
1330
- }
1331
- function head(s, n) {
1332
- const flat = s.replace(/\s+/g, " ").trim();
1333
- return flat.length > n ? flat.slice(0, n) + "…" : flat;
1334
- }
1335
- function renderReport(args) {
1336
- const { prompts, results, sweepRows, sweepSubsetIdx, verdictRows, judgeState } = args;
1337
- const off = results.filter((r) => r.mode === "off");
1338
- const on = results.filter((r) => r.mode === "on");
1339
- const L = [];
1340
- if (judgeState.aborted) {
1341
- L.push(`# ⚠ ABORTED RUN — ${judgeState.aborted.reason}\n\n` +
1342
- `_The report below reflects only what completed before the abort. ${judgeState.completedUnits} judge ` +
1343
- `units completed, ${judgeState.errorUnits} judge_error, ${judgeState.totalCalls} fresh calls this ` +
1344
- `invocation. Re-run with \`--resume\` after addressing the cause._\n`);
1345
- }
1346
- L.push("# Recall Relevance + Snippet Eval — v2\n");
1347
- L.push(`Snapshot: \`${args.snapshotPath}\` \n` +
1348
- `Prompts: \`${args.promptsPath}\` (${prompts.length} prompts; requested ${SOURCE_TAGS.length * N_PER_SOURCE} — ` +
1349
- `see §0 for any per-source shortfall) \n` +
1350
- `Generated: ${args.generatedAt} \n` +
1351
- `Clock: ${args.clock} (the instant the retrieve sweep scored against, #458)${args.variance ? ` \nJudge runs: ${args.variance.runs} (phases 1+2 re-judged over fixed selections — see §15b)` : ""}\n`);
1352
- L.push(`_Spec: \`specs/2026-08-02-relevance-eval.md\`. Real agent prompts replayed through \`retrieve()\` on a ` +
1353
- `READONLY snapshot; GLM-5.2 (z.ai) is the JUDGE only — retrieve() is LLM-free. Two independent, BLIND judge ` +
1354
- `calls per (prompt × mode): \`full_verdict\` (up to 2000 chars) and \`line_verdict\` (ONLY the rendered ` +
1355
- `production one-liner, imported from \`recall-index.ts#formatIndexLine\`/\`memoryTitle\` — never ` +
1356
- `reimplemented). Static DB (\`noStrengthen: true\`); real bge-small-en-v1.5 embedder; embed-once via ` +
1357
- `\`queryEmbedding\` with a \`neverCalledEmbed\` self-check. Retrieve K=${RESULT_K}; precision@4/@${PROD_MENU_K}/@${RESULT_K} ` +
1358
- `all derived from the SAME verdicts. Token counts are a char/4 ESTIMATE (no tokenizer available locally — stated ` +
1359
- `explicitly wherever used). Raw per-verdict rows saved to \`${args.verdictsJsonPath}\` for ad-hoc tuning ` +
1360
- `analysis without re-judging._\n`);
1361
- // =====================================================================
1362
- // §0 — corpus construction (source counts, sweep subset)
1363
- // =====================================================================
1364
- L.push("## 0. Corpus\n");
1365
- const bySource = new Map();
1366
- for (const p of prompts)
1367
- bySource.set(p.sourceTag, (bySource.get(p.sourceTag) ?? 0) + 1);
1368
- L.push("| source | prompts | of requested |");
1369
- L.push("|---|---|---|");
1370
- for (const tag of SOURCE_TAGS) {
1371
- L.push(`| ${tag} | ${bySource.get(tag) ?? 0} | ${N_PER_SOURCE} |`);
1372
- }
1373
- L.push("");
1374
- L.push(`Length-sweep fixed subset (spec §4.2): ${sweepSubsetIdx.length} prompts (${SWEEP_N_PER_SOURCE}/source), ` +
1375
- `scope OFF only (the true baseline).\n`);
1376
- // =====================================================================
1377
- // §1 — per-prompt precision histogram (v1, unchanged shape)
1378
- // =====================================================================
1379
- L.push(`## 1. Per-prompt precision histogram (relevant-count in Q1-Q${RESULT_K}, full_verdict)\n`);
1380
- const offHist = computePromptHistogram(results, "off");
1381
- const onHist = computePromptHistogram(results, "on");
1382
- L.push("| relevant-count | OFF prompt-count | ON prompt-count |");
1383
- L.push("|---|---|---|");
1384
- for (let i = 0; i <= RESULT_K; i++) {
1385
- const marker = i <= 1 ? " ← worst" : i >= RESULT_K - 1 ? " ← best" : "";
1386
- L.push(`| ${i}${marker} | ${offHist[i]} | ${onHist[i]} |`);
1387
- }
1388
- L.push("");
1389
- // =====================================================================
1390
- // §2 — worst/best named prompts
1391
- // =====================================================================
1392
- L.push(`## 2. Worst (0-1 relevant) and best (${RESULT_K - 1}-${RESULT_K} relevant) prompts — scope OFF\n`);
1393
- const offWb = findWorstBestPrompts(results, "off", 1, RESULT_K - 1);
1394
- L.push(`### Worst — ${offWb.worst.length} prompt(s)\n`);
1395
- if (offWb.worst.length === 0) {
1396
- L.push("_(none)_\n");
1397
- }
1398
- else {
1399
- for (const p of offWb.worst.slice(0, 6)) {
1400
- L.push(`- **[${p.agent}]** \`${head(p.prompt, 160)}\``);
1401
- L.push(` - tally: ${p.relevant} relevant / ${p.noise} noise / ${p.stale} stale`);
1402
- for (const v of p.perRank) {
1403
- L.push(` - Q${v.rank} \`${v.memory_id.slice(0, 8)}\` **${v.verdict}** — ${v.reason || "_(no reason)_"} — \`${head(v.content, 100)}\``);
1404
- }
1405
- }
1406
- L.push("");
1407
- }
1408
- L.push(`### Best — ${offWb.best.length} prompt(s)\n`);
1409
- if (offWb.best.length === 0) {
1410
- L.push("_(none)_\n");
1411
- }
1412
- else {
1413
- for (const p of offWb.best.slice(0, 4)) {
1414
- L.push(`- **[${p.agent}]** \`${head(p.prompt, 160)}\` — ${p.relevant} relevant / ${p.noise} noise / ${p.stale} stale`);
1415
- }
1416
- L.push("");
1417
- }
1418
- // =====================================================================
1419
- // §3/§4 — chronic-noise / chronic-stale
1420
- // =====================================================================
1421
- const metaById = lookupMemoryMeta(args.db, results);
1422
- const chronic = rankChronicMemories(results, metaById);
1423
- const chronicNoise = [...chronic].sort((a, b) => b.noisePrompts - a.noisePrompts).filter((c) => c.noisePrompts > 0);
1424
- const chronicStale = [...chronic].sort((a, b) => b.stalePrompts - a.stalePrompts).filter((c) => c.stalePrompts > 0);
1425
- L.push(`## 3. Chronic-noise memories (ranked pollution sources) — HEADLINE\n`);
1426
- if (chronicNoise.length === 0) {
1427
- L.push("_(none)_\n");
1428
- }
1429
- else {
1430
- L.push("| # | noise-prompts | stale-prompts | surfaces | type | project | id | content snippet |");
1431
- L.push("|---|---|---|---|---|---|---|---|");
1432
- for (let i = 0; i < Math.min(15, chronicNoise.length); i++) {
1433
- const c = chronicNoise[i];
1434
- L.push(`| ${i + 1} | ${c.noisePrompts} | ${c.stalePrompts} | ${c.totalSurfaces} | ${c.memory_type ?? "-"} | ` +
1435
- `${c.project ?? "-"} | \`${c.memory_id.slice(0, 8)}\` | \`${head(c.content, 100)}\` |`);
1436
- }
1437
- L.push("");
1438
- }
1439
- L.push(`## 4. Chronic-stale memories (supersession gaps)\n`);
1440
- if (chronicStale.length === 0) {
1441
- L.push("_(none)_\n");
1442
- }
1443
- else {
1444
- L.push("| # | stale-prompts | noise-prompts | surfaces | type | project | id | content snippet |");
1445
- L.push("|---|---|---|---|---|---|---|---|");
1446
- for (let i = 0; i < Math.min(15, chronicStale.length); i++) {
1447
- const c = chronicStale[i];
1448
- L.push(`| ${i + 1} | ${c.stalePrompts} | ${c.noisePrompts} | ${c.totalSurfaces} | ${c.memory_type ?? "-"} | ` +
1449
- `${c.project ?? "-"} | \`${c.memory_id.slice(0, 8)}\` | \`${head(c.content, 100)}\` |`);
1450
- }
1451
- L.push("");
1452
- }
1453
- // =====================================================================
1454
- // §5 — K-sweep + per-rank + top/bottom + bubble (v1, supporting)
1455
- // =====================================================================
1456
- L.push(`## 5. K-sweep (supporting — precision@4 / @${PROD_MENU_K} / @${RESULT_K}, full_verdict)\n`);
1457
- L.push("| mode | precision@4 | precision@6 | precision@8 | noise@4 | noise@6 | noise@8 | stale@6 |");
1458
- L.push("|---|---|---|---|---|---|---|---|");
1459
- const m4off = computeMetricsAtK(off, 4), m6off = computeMetricsAtK(off, 6), m8off = computeMetricsAtK(off, 8);
1460
- const m4on = computeMetricsAtK(on, 4), m6on = computeMetricsAtK(on, 6), m8on = computeMetricsAtK(on, 8);
1461
- if (m4off && m6off && m8off) {
1462
- L.push(`| OFF | ${pct(m4off.precision)} | ${pct(m6off.precision)} | ${pct(m8off.precision)} | ${pct(m4off.noiseRate)} | ${pct(m6off.noiseRate)} | ${pct(m8off.noiseRate)} | ${pct(m6off.staleRate)} |`);
1463
- }
1464
- if (m4on && m6on && m8on) {
1465
- L.push(`| ON | ${pct(m4on.precision)} | ${pct(m6on.precision)} | ${pct(m8on.precision)} | ${pct(m4on.noiseRate)} | ${pct(m6on.noiseRate)} | ${pct(m8on.noiseRate)} | ${pct(m6on.staleRate)} |`);
1466
- }
1467
- L.push("");
1468
- L.push(`## 6. Per-rank verdict distribution (Q1..Q${RESULT_K}, full_verdict)\n`);
1469
- for (const [label, rs] of [["OFF", off], ["ON", on]]) {
1470
- const rows = computeRankMetrics(rs);
1471
- L.push(`### Scope ${label}\n`);
1472
- L.push("| rank | band | N | relevant | noise | stale |");
1473
- L.push("|---|---|---|---|---|---|");
1474
- for (const r of rows) {
1475
- const band = r.rank <= PROD_MENU_K ? "menu" : "bubble";
1476
- L.push(`| Q${r.rank} | ${band} | ${r.n} | ${Math.round(r.relevantRate * r.n)} | ${Math.round(r.noiseRate * r.n)} | ${Math.round(r.staleRate * r.n)} |`);
1477
- }
1478
- L.push("");
1479
- }
1480
- const offTop = aggregateRankBand(off, 1, 3);
1481
- const offBot = aggregateRankBand(off, 4, PROD_MENU_K);
1482
- const onTop = aggregateRankBand(on, 1, 3);
1483
- const onBot = aggregateRankBand(on, 4, PROD_MENU_K);
1484
- L.push(`## 7. Top-3 vs bottom-3 noise split + bubble (supporting)\n`);
1485
- L.push("| mode | band | noise / N | noise-rate |");
1486
- L.push("|---|---|---|---|");
1487
- L.push(`| OFF | top-3 | ${offTop.noise}/${offTop.total} | **${pct(offTop.rate)}** |`);
1488
- L.push(`| OFF | bottom-3 | ${offBot.noise}/${offBot.total} | ${pct(offBot.rate)} |`);
1489
- L.push(`| ON | top-3 | ${onTop.noise}/${onTop.total} | **${pct(onTop.rate)}** |`);
1490
- L.push(`| ON | bottom-3 | ${onBot.noise}/${onBot.total} | ${pct(onBot.rate)} |`);
1491
- L.push("");
1492
- const rateFromRows = (rows) => {
1493
- const rel = rows.reduce((s, r) => s + Math.round(r.relevantRate * r.n), 0);
1494
- const total = rows.reduce((s, r) => s + r.n, 0);
1495
- return { rel, total, rate: total > 0 ? rel / total : 0 };
1496
- };
1497
- const offMenuRel = rateFromRows(computeRankMetrics(off).slice(0, PROD_MENU_K));
1498
- const onMenuRel = rateFromRows(computeRankMetrics(on).slice(0, PROD_MENU_K));
1499
- const offBubRel = rateFromRows(computeRankMetrics(off).slice(PROD_MENU_K));
1500
- const onBubRel = rateFromRows(computeRankMetrics(on).slice(PROD_MENU_K));
1501
- L.push(`Bubble (Q${PROD_MENU_K + 1}-Q${RESULT_K}) relevant-rate vs the menu.\n`);
1502
- L.push("| mode | band | relevant / N | relevant-rate |");
1503
- L.push("|---|---|---|---|");
1504
- L.push(`| OFF | menu | ${offMenuRel.rel}/${offMenuRel.total} | ${pct(offMenuRel.rate)} |`);
1505
- L.push(`| OFF | bubble | ${offBubRel.rel}/${offBubRel.total} | **${pct(offBubRel.rate)}** |`);
1506
- L.push(`| ON | menu | ${onMenuRel.rel}/${onMenuRel.total} | ${pct(onMenuRel.rate)} |`);
1507
- L.push(`| ON | bubble | ${onBubRel.rel}/${onBubRel.total} | **${pct(onBubRel.rate)}** |`);
1508
- L.push("");
1509
- // =====================================================================
1510
- // §8 — per-source breakdown (spec §5.7)
1511
- // =====================================================================
1512
- L.push("## 8. Per-source breakdown (scope OFF, precision@6)\n");
1513
- L.push("| source | precision@6 | noise@6 | stale@6 | n prompts |");
1514
- L.push("|---|---|---|---|---|");
1515
- for (const tag of SOURCE_TAGS) {
1516
- const rs = off.filter((r) => r.prompt.sourceTag === tag);
1517
- const m = computeMetricsAtK(rs, PROD_MENU_K);
1518
- L.push(`| ${tag} | ${m ? pct(m.precision) : "n/a"} | ${m ? pct(m.noiseRate) : "n/a"} | ${m ? pct(m.staleRate) : "n/a"} | ${m ? m.n : 0} |`);
1519
- }
1520
- L.push("");
1521
- // =====================================================================
1522
- // §9 — snippet layer: dual verdict + named failures (spec §4.1, §5.8) — HEADLINE
1523
- // =====================================================================
1524
- L.push("## 9. Snippet layer — dual verdict (full_verdict vs line_verdict) — HEADLINE\n");
1525
- const gapOff = computeSnippetGap(off);
1526
- const gapOn = computeSnippetGap(on);
1527
- const gapAll = computeSnippetGap(results);
1528
- const snippetFailureRate = (g) => (g.fullRelevantTotal > 0 ? g.snippetFailure / g.fullRelevantTotal : NaN);
1529
- L.push("| scope | full-relevant (denominator) | actionable | snippet_failure | snippet_failure_rate | misleading_line |");
1530
- L.push("|---|---|---|---|---|---|");
1531
- L.push(`| OFF | ${gapOff.fullRelevantTotal} | ${gapOff.actionable} | ${gapOff.snippetFailure} | **${pct(snippetFailureRate(gapOff))}** | ${gapOff.misleadingLine} |`);
1532
- L.push(`| ON | ${gapOn.fullRelevantTotal} | ${gapOn.actionable} | ${gapOn.snippetFailure} | **${pct(snippetFailureRate(gapOn))}** | ${gapOn.misleadingLine} |`);
1533
- L.push(`| BOTH | ${gapAll.fullRelevantTotal} | ${gapAll.actionable} | ${gapAll.snippetFailure} | **${pct(snippetFailureRate(gapAll))}** | ${gapAll.misleadingLine} |`);
1534
- L.push("");
1535
- L.push(`_snippet_failure_rate = snippet_failure / full-relevant — "of the memories retrieve() got RIGHT, what fraction ` +
1536
- `does the ~100-char production line fail to communicate as relevant?" misleading_line = the agent is BAITED ` +
1537
- `by a line that reads relevant for a memory that (on full inspection) is not._\n`);
1538
- const namedFailures = findNamedSnippetFailures(results, 15);
1539
- L.push(`### Named worst snippet failures (up to 15)\n`);
1540
- if (namedFailures.length === 0) {
1541
- L.push("_(none — every full-relevant memory's line also read as relevant)_\n");
1542
- }
1543
- else {
1544
- for (const f of namedFailures) {
1545
- L.push(`- **[${f.mode}/${f.agent}]** prompt: \`${head(f.prompt, 120)}\``);
1546
- L.push(` - rendered line: \`${f.rendered_line}\``);
1547
- L.push(` - full_verdict=relevant (${f.full_reason || "no reason"}) but line_verdict≠relevant (${f.line_reason || "no reason"})`);
1548
- L.push(` - full content: \`${head(f.content, 140)}\``);
1549
- }
1550
- L.push("");
1551
- }
1552
- // =====================================================================
1553
- // §10 — snippet-length sweep (spec §4.2, §5.8) — HEADLINE
1554
- // =====================================================================
1555
- L.push(`## 10. Snippet-length sweep (N=${sweepSubsetIdx.length}, scope OFF, Q1-Q${RESULT_K}) — HEADLINE\n`);
1556
- L.push(`_Same prompts, same memories — only the rendering changes per variant. CI is ~1.6× wider than the N=100 tables ` +
1557
- `(spec §8); only effects ≥ ~15pts are distinguishable here. ${PROD_TITLE_CHARS} is the PRODUCTION baseline ` +
1558
- `(shipped \`recallTitleChars\`) — deltas are reported against it, not against 100._\n`);
1559
- L.push(`| variant | line-relevant-rate | 95% CI | Δ vs ${PROD_TITLE_CHARS} (prod) | mean tokens/block (prod-6, char/4) |`);
1560
- L.push("|---|---|---|---|---|");
1561
- const sweepByVariant = new Map();
1562
- for (const v of SWEEP_VARIANTS)
1563
- sweepByVariant.set(v, sweepRows.filter((s) => s.variant === v));
1564
- const sweepVariantCi = new Map();
1565
- const sweepVariantTokens = new Map();
1566
- // Pass 1: compute CI + token estimate per variant (order-independent — the
1567
- // baseline lookup below must not depend on iterating variants in a
1568
- // particular order, since PROD_TITLE_CHARS (150) is no longer first in
1569
- // SWEEP_VARIANTS' display order).
1570
- for (const variant of SWEEP_VARIANTS) {
1571
- const rowsForVariant = sweepByVariant.get(variant) ?? [];
1572
- const perPrompt = [];
1573
- const tokenSamples = [];
1574
- for (const sw of rowsForVariant) {
1575
- if (!Array.isArray(sw.verdicts))
1576
- continue;
1577
- const relCount = sw.verdicts.filter((v) => v.verdict === "relevant").length;
1578
- const denom = sw.verdicts.length || 1;
1579
- perPrompt.push(relCount / denom);
1580
- const off6 = off.find((r) => prompts.indexOf(r.prompt) === sw.promptIdx);
1581
- if (off6) {
1582
- const rowsArr = off6.topIds.map((id) => off6.rows[id]).filter(Boolean);
1583
- const gated = rowsArr.filter((r) => (0, recall_index_js_1.passesRelevanceGate)(r, PROD_MIN_SIMILARITY)).slice(0, PROD_MENU_K);
1584
- const block = [...RECALL_BLOCK_HEADER, ...gated.map((r) => sw.lines[r.id] ?? "")].join("\n");
1585
- tokenSamples.push(estimateTokens(block));
1586
- }
1587
- }
1588
- sweepVariantCi.set(variant, ci95(perPrompt));
1589
- sweepVariantTokens.set(variant, tokenSamples.length ? mean(tokenSamples) : NaN);
1590
- }
1591
- // Pass 2: render rows in SWEEP_VARIANTS order, delta always vs the
1592
- // production baseline (PROD_TITLE_CHARS), looked up by key not by position.
1593
- const baselineCi = sweepVariantCi.get(String(PROD_TITLE_CHARS)) ?? null;
1594
- for (const variant of SWEEP_VARIANTS) {
1595
- const c = sweepVariantCi.get(variant);
1596
- const tokenSamples = sweepVariantTokens.get(variant);
1597
- const isBaseline = variant === String(PROD_TITLE_CHARS);
1598
- const delta = baselineCi && !isBaseline ? c.mean - baselineCi.mean : null;
1599
- const variantLabel = isBaseline ? `${variant} (prod)` : variant;
1600
- L.push(`| ${variantLabel} | ${pct(c.mean)} | [${pct(c.lo)}, ${pct(c.hi)}] | ${delta === null ? "—" : pts(delta)} | ${Number.isFinite(tokenSamples) ? tokenSamples.toFixed(0) : "n/a"} |`);
1601
- }
1602
- L.push("");
1603
- // =====================================================================
1604
- // §11 — similarity-floor + retrieval-source analysis (spec §5.9)
1605
- // =====================================================================
1606
- L.push(`## 11. Similarity-floor + retrieval-source analysis (full_verdict, variant=${PROD_TITLE_CHARS} baseline only)\n`);
1607
- const baseRows = verdictRows.filter((r) => r.snippet_len_variant === String(PROD_TITLE_CHARS) && r.full_verdict !== null);
1608
- const byBucket = new Map();
1609
- for (const r of baseRows) {
1610
- if (r.full_verdict === "judge_error")
1611
- continue;
1612
- const label = similarityBucketLabel(r.similarity);
1613
- const e = byBucket.get(label) ?? { relevant: 0, noise: 0, stale: 0, total: 0 };
1614
- e.total++;
1615
- if (r.full_verdict === "relevant")
1616
- e.relevant++;
1617
- else if (r.full_verdict === "noise")
1618
- e.noise++;
1619
- else if (r.full_verdict === "stale")
1620
- e.stale++;
1621
- byBucket.set(label, e);
1622
- }
1623
- L.push("### By similarity band\n");
1624
- L.push("| band | n | relevant | noise | stale | noise-rate |");
1625
- L.push("|---|---|---|---|---|---|");
1626
- const bucketOrder = ["n/a (fts/graph, no cosine)", "<0.30", ...SIM_BUCKET_EDGES.slice(0, -1).map((_, i) => `${SIM_BUCKET_EDGES[i].toFixed(2)}–${Math.min(SIM_BUCKET_EDGES[i + 1], 1).toFixed(2)}`)];
1627
- for (const label of bucketOrder) {
1628
- const e = byBucket.get(label);
1629
- if (!e)
1630
- continue;
1631
- L.push(`| ${label} | ${e.total} | ${e.relevant} | ${e.noise} | ${e.stale} | ${pct(e.total > 0 ? e.noise / e.total : 0)} |`);
1632
- }
1633
- L.push("");
1634
- L.push("### By retrieval source (channel)\n");
1635
- const bySrc = new Map();
1636
- for (const r of baseRows) {
1637
- if (r.full_verdict === "judge_error")
1638
- continue;
1639
- const label = r.retrieval_source ?? "(unknown)";
1640
- const e = bySrc.get(label) ?? { relevant: 0, noise: 0, stale: 0, total: 0 };
1641
- e.total++;
1642
- if (r.full_verdict === "relevant")
1643
- e.relevant++;
1644
- else if (r.full_verdict === "noise")
1645
- e.noise++;
1646
- else if (r.full_verdict === "stale")
1647
- e.stale++;
1648
- bySrc.set(label, e);
1649
- }
1650
- L.push("| retrieval_source | n | relevant | noise | stale | noise-rate |");
1651
- L.push("|---|---|---|---|---|---|");
1652
- for (const [label, e] of Array.from(bySrc.entries()).sort((a, b) => b[1].total - a[1].total)) {
1653
- L.push(`| ${label} | ${e.total} | ${e.relevant} | ${e.noise} | ${e.stale} | ${pct(e.total > 0 ? e.noise / e.total : 0)} |`);
1654
- }
1655
- L.push("");
1656
- // =====================================================================
1657
- // §12 — token cost per K (spec §5.10)
1658
- // =====================================================================
1659
- L.push(`## 12. Token cost per K (production render, maxLen=${PROD_TITLE_CHARS}, scope OFF, char/4 estimate)\n`);
1660
- L.push("| K | mean tokens/block | mean chars/block |");
1661
- L.push("|---|---|---|");
1662
- for (const k of SWEEP_KS) {
1663
- const tokenSamples = [];
1664
- for (const r of off) {
1665
- const rowsArr = r.topIds.map((id) => r.rows[id]).filter(Boolean);
1666
- const { block } = renderProductionBlock(rowsArr, PROD_MIN_SIMILARITY, k, PROD_TITLE_CHARS);
1667
- tokenSamples.push(estimateTokens(block));
1668
- }
1669
- L.push(`| ${k} | ${fmtNum(mean(tokenSamples), 0)} | ${fmtNum(mean(tokenSamples) * 4, 0)} |`);
1670
- }
1671
- L.push("\n_Per-variant token cost is in §10's sweep table (same char/4 estimate, prod-6-gated block)._\n");
1672
- // =====================================================================
1673
- // §13 — rendered production blocks (spec §5.11) — ~20 mixed best/worst
1674
- // =====================================================================
1675
- L.push(`## 13. Rendered ACTUAL production blocks (mixed best/worst, scope OFF, maxLen=${PROD_TITLE_CHARS})\n`);
1676
- L.push(`_Exactly what \`/recall-index\` would render: \`passesRelevanceGate\` (minSimilarity=${PROD_MIN_SIMILARITY}) then ` +
1677
- `capped at ${PROD_MENU_K}. NOTE: candidate pool here is Q1-Q${RESULT_K} (this eval's replay window); production ` +
1678
- `over-fetches \`maxItems×3\` candidates before gating, so a real production block could differ slightly on a ` +
1679
- `sparse query — a known, documented simplification, not a bug._\n`);
1680
- const worstBlocks = offWb.worst.slice(0, 10);
1681
- const bestBlocks = offWb.best.slice(0, 10);
1682
- let shown = 0;
1683
- for (const label of ["worst", "best"]) {
1684
- const set = label === "worst" ? worstBlocks : bestBlocks;
1685
- for (const p of set) {
1686
- if (shown >= 20)
1687
- break;
1688
- const r = off.find((x) => x.prompt.prompt === p.prompt && x.prompt.agent === p.agent);
1689
- if (!r)
1690
- continue;
1691
- const rowsArr = r.topIds.map((id) => r.rows[id]).filter(Boolean);
1692
- const { block } = renderProductionBlock(rowsArr, PROD_MIN_SIMILARITY, PROD_MENU_K, PROD_TITLE_CHARS);
1693
- L.push(`### [${label}] ${p.agent}: \`${head(p.prompt, 100)}\`\n`);
1694
- L.push("```");
1695
- L.push(block || "(null — no candidate passed the relevance gate)");
1696
- L.push("```\n");
1697
- shown++;
1698
- }
1699
- }
1700
- // =====================================================================
1701
- // §14 — redundancy (spec §5.12)
1702
- // =====================================================================
1703
- L.push("## 14. Redundancy over the production 6 (set-level judge call per prompt × mode)\n");
1704
- const redundancySamples = [];
1705
- for (const r of results) {
1706
- if (!r.redundancy)
1707
- continue;
1708
- if ("skipped" in r.redundancy)
1709
- redundancySamples.push({ mode: r.mode, shown: r.redundancy.shown, nonRedundant: r.redundancy.shown });
1710
- else if (!("judgeError" in r.redundancy))
1711
- redundancySamples.push({ mode: r.mode, shown: r.redundancy.shown, nonRedundant: r.redundancy.nonRedundantCount });
1712
- }
1713
- L.push("| mode | n prompts | mean shown | mean non-redundant | mean redundant slots |");
1714
- L.push("|---|---|---|---|---|");
1715
- for (const label of ["off", "on"]) {
1716
- const rs = redundancySamples.filter((s) => s.mode === label);
1717
- const meanShown = mean(rs.map((s) => s.shown));
1718
- const meanNr = mean(rs.map((s) => s.nonRedundant));
1719
- L.push(`| ${label.toUpperCase()} | ${rs.length} | ${fmtNum(meanShown)} | ${fmtNum(meanNr)} | ${fmtNum(meanShown - meanNr)} |`);
1720
- }
1721
- L.push("");
1722
- // =====================================================================
1723
- // §15 — uncertainty (spec §5.13, §8)
1724
- // =====================================================================
1725
- L.push("## 15. Uncertainty — SD / SE / 95% CI\n");
1726
- const p6off = computeMetricsAtK(off, PROD_MENU_K);
1727
- const p6on = computeMetricsAtK(on, PROD_MENU_K);
1728
- L.push("| metric | mode | mean | SD | SE | 95% CI |");
1729
- L.push("|---|---|---|---|---|---|");
1730
- if (p6off) {
1731
- const c = ci95(p6off.perPrompt);
1732
- L.push(`| precision@${PROD_MENU_K} | OFF | ${pct(c.mean)} | ${pct(c.sd)} | ${pct(c.se)} | [${pct(c.lo)}, ${pct(c.hi)}] |`);
1733
- }
1734
- if (p6on) {
1735
- const c = ci95(p6on.perPrompt);
1736
- L.push(`| precision@${PROD_MENU_K} | ON | ${pct(c.mean)} | ${pct(c.sd)} | ${pct(c.se)} | [${pct(c.lo)}, ${pct(c.hi)}] |`);
1737
- }
1738
- L.push("");
1739
- let distinguishable = "n/a";
1740
- let deltaStr = "n/a";
1741
- if (p6off && p6on) {
1742
- const cOff = ci95(p6off.perPrompt);
1743
- const cOn = ci95(p6on.perPrompt);
1744
- const delta = cOn.mean - cOff.mean;
1745
- const seDelta = Math.sqrt(cOff.se ** 2 + cOn.se ** 2);
1746
- const thresholdPts = 1.96 * seDelta;
1747
- distinguishable = Math.abs(delta) > thresholdPts ? "YES" : "NO — within noise";
1748
- deltaStr = `${pts(delta)} (SE_delta=${pts(seDelta)}, 95% threshold=±${pts(thresholdPts)})`;
1749
- L.push(`**Scope ON−OFF delta (precision@${PROD_MENU_K}):** ${deltaStr} — **${distinguishable}**\n`);
1750
- }
1751
- // =====================================================================
1752
- // §15b — judge variance (#458): the measured run-to-run noise floor.
1753
- // Rendered ONLY when --runs>1.
1754
- // =====================================================================
1755
- if (args.variance) {
1756
- const v = args.variance;
1757
- L.push(`## 15b. Judge variance — run-to-run noise floor (#458, ${v.runs} runs)\n`);
1758
- L.push(`_Judge phases 1+2 ran ${v.runs} times over the SAME selections (one retrieve sweep feeds every run — ` +
1759
- `the sweep precedes judging, so selections are fixed under a live clock too). ALL drift measured here is ` +
1760
- `judge-only instrument noise, the class PR C evidenced at 38/380 flipped verdicts. Read every body section ` +
1761
- `above as run 1; this section carries the median and the band._\n`);
1762
- L.push("| metric | " + v.metrics.map((_, i) => `run ${i + 1}`).join(" | ") + " | median | spread (max−min) |");
1763
- L.push("|---|" + v.metrics.map(() => "---").join("|") + "|---|---|");
1764
- for (const m of v.metrics) {
1765
- const cells = m.values.map((x) => (x === null || !Number.isFinite(x) ? "n/a" : pct(x))).join(" | ");
1766
- L.push(`| ${m.label} | ${cells} | **${m.median !== null && Number.isFinite(m.median) ? pct(m.median) : "n/a"}** | ${m.spreadPts !== null && Number.isFinite(m.spreadPts) ? `${m.spreadPts.toFixed(1)}pts` : "n/a"} |`);
1767
- }
1768
- L.push("");
1769
- L.push(`_Pairwise verdict flips on selection-identical units (full_verdict rows; judge_error units excluded from ` +
1770
- `the denominators — error units per run: ${v.judgeErrorUnitsPerRun.map((e, i) => `r${i + 1} ${e}`).join(", ")}). ` +
1771
- `A verdict ROW is one Q-label judgment on one (prompt, mode) unit; a UNIT flips if any of its rows differ._\n`);
1772
- L.push("| pair | rows flipped / compared | units flipped / compared |");
1773
- L.push("|---|---|---|");
1774
- for (const f of v.flips) {
1775
- L.push(`| r${f.runA}↔r${f.runB} | ${f.rowsFlipped} / ${f.rowsCompared} | ${f.unitsFlipped} / ${f.unitsCompared} |`);
1776
- }
1777
- L.push("");
1778
- L.push(`**Noise floor:** max pairwise flips ${v.maxRowsFlipped} rows / ${v.maxUnitsFlipped} units; mean ` +
1779
- `${v.meanRowsFlipped.toFixed(1)} rows / ${v.meanUnitsFlipped.toFixed(1)} units over ${v.flips.length} pair(s); ` +
1780
- `total comparable rows ${v.totalComparableRows}. A before/after headline delta smaller than the spread row ` +
1781
- `above is NOT resolvable by this instrument — report it as within judge noise._\n`);
1782
- }
1783
- // =====================================================================
1784
- // §16 — DECISION THRESHOLDS (spec §9) — filled with measured values
1785
- // =====================================================================
1786
- L.push("## 16. Decision thresholds (spec §9) — measured values + triggered actions\n");
1787
- const noiseRate6off = m6off ? m6off.noiseRate : NaN;
1788
- const noiseHighTriggered = Number.isFinite(noiseRate6off) && noiseRate6off > 0.3;
1789
- // Worst similarity band with meaningful sample (n>=5), excluding n/a.
1790
- let worstBand = null;
1791
- for (const [label, e] of byBucket.entries()) {
1792
- if (label.startsWith("n/a"))
1793
- continue;
1794
- if (e.total < 5)
1795
- continue;
1796
- const rate = e.noise / e.total;
1797
- if (!worstBand || rate > worstBand.rate)
1798
- worstBand = { label, rate, n: e.total };
1799
- }
1800
- const bandTriggered = worstBand !== null && worstBand.rate >= 0.6;
1801
- const ftsE = bySrc.get("fts");
1802
- const vecE = bySrc.get("vector");
1803
- const ftsNoiseRate = ftsE && ftsE.total > 0 ? ftsE.noise / ftsE.total : NaN;
1804
- const vecNoiseRate = vecE && vecE.total > 0 ? vecE.noise / vecE.total : NaN;
1805
- const ftsRatio = Number.isFinite(ftsNoiseRate) && Number.isFinite(vecNoiseRate) && vecNoiseRate > 0 ? ftsNoiseRate / vecNoiseRate : NaN;
1806
- const ftsTriggered = Number.isFinite(ftsRatio) && ftsRatio > 1.5;
1807
- const snippetFailOffRate = snippetFailureRate(gapOff);
1808
- const snippetFailTriggered = Number.isFinite(snippetFailOffRate) && snippetFailOffRate > 0.2;
1809
- // Baseline is now PROD_TITLE_CHARS (150), not 100 — 300 was dropped from the
1810
- // sweep (replaced by 120) in the 2026-08-02 post-rewrite run config.
1811
- const ciBase = sweepVariantCi.get(String(PROD_TITLE_CHARS));
1812
- const ciLong = sweepVariantCi.get("200");
1813
- const deltaBaseToLong = ciBase && ciLong ? ciLong.mean - ciBase.mean : NaN;
1814
- const lengthHelpsTriggered = Number.isFinite(deltaBaseToLong) && deltaBaseToLong >= 0.05;
1815
- // Owner question (2026-08-02): can maxLen be LOWERED 150→120 without losing
1816
- // signal (token savings)? "No meaningful loss" = delta not more negative
1817
- // than -5pts.
1818
- const ciShort = sweepVariantCi.get("120");
1819
- const deltaBaseToShort = ciBase && ciShort ? ciShort.mean - ciBase.mean : NaN;
1820
- const canLowerTo120 = Number.isFinite(deltaBaseToShort) && deltaBaseToShort >= -0.05;
1821
- const bubbleGapPts = offMenuRel.rate - offBubRel.rate; // menu minus bubble
1822
- const bubbleWithin5 = Number.isFinite(bubbleGapPts) && Math.abs(bubbleGapPts) <= 0.05;
1823
- const bubbleMuchLower = Number.isFinite(bubbleGapPts) && bubbleGapPts > 0.15;
1824
- const topLoaded = offTop.total > 0 && offBot.total > 0 && offTop.rate > offBot.rate;
1825
- const topChronicNoise = chronicNoise[0] ?? null;
1826
- const chronicTriggered = topChronicNoise !== null && topChronicNoise.noisePrompts > 10;
1827
- const offRedundancy = redundancySamples.filter((s) => s.mode === "off");
1828
- const meanRedundantSlotsOff = offRedundancy.length > 0 ? mean(offRedundancy.map((s) => s.shown - s.nonRedundant)) : NaN;
1829
- const redundancyTriggered = Number.isFinite(meanRedundantSlotsOff) && meanRedundantSlotsOff > 1.5;
1830
- const scopeDeltaPts = p6off && p6on ? (ci95(p6on.perPrompt).mean - ci95(p6off.perPrompt).mean) * 100 : NaN;
1831
- L.push("| finding | measured | threshold | triggered? | action |");
1832
- L.push("|---|---|---|---|---|");
1833
- L.push(`| Overall noise-rate (prod K=6, OFF) | ${pct(noiseRate6off)} | >30% | ${noiseHighTriggered ? "**YES**" : "no"} | Raise \`recallMinSimilarity\` (0.55 → floor indicated by the band table) |`);
1834
- L.push(`| Noise concentrated below a band | worst band ${worstBand ? `${worstBand.label} (${pct(worstBand.rate)}, n=${worstBand.n})` : "n/a"} | band ≥60% noise | ${bandTriggered ? "**YES**" : "no"} | Set \`recallMinSimilarity\` at that band's lower edge |`);
1835
- L.push(`| FTS-matched hits disproportionately noise | fts ${pct(ftsNoiseRate)} vs vector ${pct(vecNoiseRate)} (ratio ${Number.isFinite(ftsRatio) ? ftsRatio.toFixed(2) : "n/a"}×) | >1.5× | ${ftsTriggered ? "**YES**" : "no"} | Stop letting FTS bypass the floor (\`passesRelevanceGate\`) |`);
1836
- L.push(`| Snippet failures | ${pct(snippetFailOffRate)} (OFF) | >20% | ${snippetFailTriggered ? "**YES**" : "no"} | Raise \`maxLen\` to the sweep's best variant, or switch to title+first-sentence |`);
1837
- L.push(`| Longer snippet doesn't improve line-relevance (${PROD_TITLE_CHARS}→200) | Δ ${Number.isFinite(deltaBaseToLong) ? pts(deltaBaseToLong) : "n/a"} | <5pts | ${!lengthHelpsTriggered ? "**YES (no improvement)**" : "no (improves ≥5pts)"} | Keep \`maxLen=${PROD_TITLE_CHARS}\` — the info problem is distillation quality, not length |`);
1838
- L.push(`| Can \`maxLen\` be LOWERED ${PROD_TITLE_CHARS}→120 without losing signal? | Δ ${Number.isFinite(deltaBaseToShort) ? pts(deltaBaseToShort) : "n/a"} | ≥-5pts (no meaningful loss) | ${canLowerTo120 ? "**YES**" : "no"} | ${canLowerTo120 ? "120 is a viable token-saving downgrade — no measurable quality loss on this evidence" : `Keep \`maxLen=${PROD_TITLE_CHARS}\` — 120 measurably loses signal`} |`);
1839
- L.push(`| Bubble (Q7-8) relevant-rate within ~5pts of menu | menu ${pct(offMenuRel.rate)} vs bubble ${pct(offBubRel.rate)} (gap ${pts(bubbleGapPts)}) | within ±5pts | ${bubbleWithin5 ? "**YES**" : "no"} | Raise \`recallMaxItems\` 6→8 if token cost is acceptable |`);
1840
- L.push(`| Bubble much lower than menu | gap ${pts(bubbleGapPts)} | >15pts | ${bubbleMuchLower ? "**YES**" : "no"} | Keep 6, or lower to 4 if Q5-6 are also weak |`);
1841
- L.push(`| Noise top-loaded (Q1-3 worse than Q4-6) | top-3 ${pct(offTop.rate)} vs bottom-3 ${pct(offBot.rate)} | top > bottom | ${topLoaded ? "**YES**" : "no"} | Ranking bug — investigate \`computeScore\`, don't tune the cap |`);
1842
- L.push(`| Chronic-noise memory | top offender pollutes ${topChronicNoise ? topChronicNoise.noisePrompts : 0} prompts${topChronicNoise ? ` (\`${topChronicNoise.memory_id.slice(0, 8)}\`)` : ""} | >10 prompts | ${chronicTriggered ? "**YES**" : "no"} | Named target: dedup / demote / re-distill |`);
1843
- L.push(`| Redundancy in the production 6 | mean ${Number.isFinite(meanRedundantSlotsOff) ? meanRedundantSlotsOff.toFixed(2) : "n/a"} redundant slots (OFF) | >1.5 avg | ${redundancyTriggered ? "**YES**" : "no"} | Tighten \`dedupMergeThreshold\` or add a diversity penalty |`);
1844
- L.push(`| Scope ON−OFF delta | ${Number.isFinite(scopeDeltaPts) ? scopeDeltaPts.toFixed(1) + "pts" : "n/a"} (empirical 95% threshold ±${p6off && p6on ? (1.96 * Math.sqrt(ci95(p6off.perPrompt).se ** 2 + ci95(p6on.perPrompt).se ** 2) * 100).toFixed(1) : "n/a"}pts; spec's illustrative ±8.5pts) | >8.5pts (spec illustrative) / empirical 95% CI | ${distinguishable === "YES" ? "**YES — distinguishable**" : "no — within noise"} | ${distinguishable === "YES" ? "Declaring mission domains in production is justified" : "Report \"within noise\" — do NOT ship scoping on this evidence"} |`);
1845
- L.push("");
1846
- // =====================================================================
1847
- // §17 — what to tune next
1848
- // =====================================================================
1849
- L.push("## 17. What to tune next\n");
1850
- const actionable = [];
1851
- const withinNoise = [];
1852
- if (noiseHighTriggered)
1853
- actionable.push(`\`recallMinSimilarity\`: overall noise-rate ${pct(noiseRate6off)} > 30% at prod K=6 (OFF).`);
1854
- else
1855
- withinNoise.push(`Overall noise-rate at prod K=6 (${pct(noiseRate6off)}) is under the 30% action threshold.`);
1856
- if (bandTriggered && worstBand)
1857
- actionable.push(`\`recallMinSimilarity\` → ~${worstBand.label.split("–")[1] ?? worstBand.label}: similarity band ${worstBand.label} is ${pct(worstBand.rate)} noise (n=${worstBand.n}).`);
1858
- if (ftsTriggered)
1859
- actionable.push(`\`passesRelevanceGate\` (recall-index.ts): FTS-matched noise-rate (${pct(ftsNoiseRate)}) is ${ftsRatio.toFixed(2)}× vector's (${pct(vecNoiseRate)}) — stop letting FTS bypass the similarity floor.`);
1860
- else if (Number.isFinite(ftsRatio))
1861
- withinNoise.push(`FTS-vs-vector noise ratio (${ftsRatio.toFixed(2)}×) is under the 1.5× action threshold.`);
1862
- if (snippetFailTriggered)
1863
- actionable.push(`\`maxLen\` (recall-index.ts \`memoryTitle\`): snippet_failure_rate ${pct(snippetFailOffRate)} (OFF) > 20% — the ${PROD_MENU_K}-line menu is losing real information at the render layer, not the retrieval layer.`);
1864
- else
1865
- withinNoise.push(`snippet_failure_rate (${pct(snippetFailOffRate)}, OFF) is under the 20% action threshold.`);
1866
- if (!lengthHelpsTriggered)
1867
- withinNoise.push(`Length sweep ${PROD_TITLE_CHARS}→200: Δ ${Number.isFinite(deltaBaseToLong) ? pts(deltaBaseToLong) : "n/a"} — within noise at N=${sweepSubsetIdx.length} (±~10pts CI); do not raise \`maxLen\` further on this evidence alone. If snippet_failure IS high, the fix is likely distillation quality (writing tighter titles), not raw length.`);
1868
- else
1869
- actionable.push(`\`maxLen\`: ${PROD_TITLE_CHARS}→200 improved line-relevance by ${pts(deltaBaseToLong)} (≥5pts) at N=${sweepSubsetIdx.length} — worth a targeted follow-up at full N before shipping.`);
1870
- if (canLowerTo120)
1871
- actionable.push(`\`maxLen\`: ${PROD_TITLE_CHARS}→120 shows Δ ${pts(deltaBaseToShort)} (no meaningful loss, ≥-5pts) at N=${sweepSubsetIdx.length} — a token-saving downgrade the owner asked about is evidence-backed.`);
1872
- else
1873
- withinNoise.push(`\`maxLen\` ${PROD_TITLE_CHARS}→120: Δ ${Number.isFinite(deltaBaseToShort) ? pts(deltaBaseToShort) : "n/a"} — do NOT lower to 120 on this evidence, it measurably costs signal.`);
1874
- if (bubbleWithin5)
1875
- actionable.push(`\`recallMaxItems\` 6→8: bubble (Q7-8, ${pct(offBubRel.rate)}) is within 5pts of the menu (${pct(offMenuRel.rate)}) — check token cost (§12) before raising.`);
1876
- else if (bubbleMuchLower)
1877
- withinNoise.push(`Bubble relevant-rate (${pct(offBubRel.rate)}) is >15pts below the menu (${pct(offMenuRel.rate)}) — \`recallMaxItems=6\` is not leaving good memories on the table.`);
1878
- if (topLoaded)
1879
- actionable.push(`\`computeScore\` (retrieval.ts): noise is top-loaded (Q1-3 ${pct(offTop.rate)} > Q4-6 ${pct(offBot.rate)}) — this is a RANKING bug, not a cap-size question.`);
1880
- else
1881
- withinNoise.push(`Top-3 vs bottom-3 noise split (${pct(offTop.rate)} vs ${pct(offBot.rate)}) shows no top-loading.`);
1882
- if (chronicTriggered && topChronicNoise)
1883
- actionable.push(`Dedup/demote/re-distill \`${topChronicNoise.memory_id.slice(0, 8)}\` (${topChronicNoise.memory_type ?? "unknown type"}): pollutes ${topChronicNoise.noisePrompts} prompts.`);
1884
- else
1885
- withinNoise.push(`No single memory pollutes >10 prompts (top offender: ${topChronicNoise ? `${topChronicNoise.noisePrompts} prompts` : "none"}).`);
1886
- if (redundancyTriggered)
1887
- actionable.push(`\`dedupMergeThreshold\` / diversity penalty: mean ${meanRedundantSlotsOff.toFixed(2)} redundant slots in the production 6 (OFF) > 1.5.`);
1888
- else
1889
- withinNoise.push(`Redundancy in the production 6 (mean ${Number.isFinite(meanRedundantSlotsOff) ? meanRedundantSlotsOff.toFixed(2) : "n/a"} redundant slots, OFF) is under the 1.5 action threshold.`);
1890
- if (distinguishable === "YES")
1891
- actionable.push(`Scope (\`missionDomains\`/\`project\` affinity): ON−OFF delta ${deltaStr} is distinguishable at 95% — declaring mission domains in production is evidence-backed.`);
1892
- else
1893
- withinNoise.push(`Scope ON−OFF delta (${deltaStr}) is within noise at N=${prompts.length} — do NOT ship mission-domain scoping on this evidence; it would need N≈1800 to detect a ~2pt true effect (spec §8).`);
1894
- L.push("**Actionable now:**\n");
1895
- if (actionable.length === 0)
1896
- L.push("_(none crossed a decision threshold this run)_\n");
1897
- else
1898
- for (const a of actionable)
1899
- L.push(`- ${a}`);
1900
- L.push("\n**Within noise (not actionable on this evidence):**\n");
1901
- if (withinNoise.length === 0)
1902
- L.push("_(none)_\n");
1903
- else
1904
- for (const w of withinNoise)
1905
- L.push(`- ${w}`);
1906
- L.push("");
1907
- if (judgeState.errorUnits > 0) {
1908
- L.push(`_${judgeState.errorUnits}/${judgeState.completedUnits} judge units returned \`judge_error\` (excluded from every ` +
1909
- `denominator above) — see \`${judgeState.sidecarPath}\` for the raw error messages._\n`);
1910
- }
1911
- return L.join("\n");
1912
- }
1913
- // ---------------------------------------------------------------------------
1914
- // main
1915
- // ---------------------------------------------------------------------------
1916
- function usage() {
1917
- console.error("Usage: relevance-eval.js <snapshot.db> [prompts.json] [report.md] " +
1918
- "[--judge-delay-ms=2000] [--max-calls=900] [--resume] " +
1919
- "[--verdicts-json=path] [--verdicts-jsonl=path] " +
1920
- "[--now=<ISO>] [--runs=N]\n" +
1921
- " <snapshot.db> — REQUIRED. Readonly bedrock snapshot (opened via openSnapshot).\n" +
1922
- " [prompts.json] — omitted: build the v2 corpus fresh (20/source × 5 sources, spec §2),\n" +
1923
- " saved to data/prompts.json. A valid existing v2 set is reused verbatim.\n" +
1924
- " <path.json>: load a saved stable set verbatim (no resample).\n" +
1925
- " [report.md] — output report path (default data/relevance-eval-report.md).\n" +
1926
- " --verdicts-json / --verdicts-jsonl — override the raw-verdicts JSON / checkpoint-sidecar\n" +
1927
- " JSONL paths (default data/relevance-verdicts.json[l]). ALWAYS override\n" +
1928
- " these for a re-run against a new snapshot — otherwise a re-run silently\n" +
1929
- " clobbers a prior baseline's raw verdicts.\n" +
1930
- " --now=<ISO> — #458: pin the clock the retrieve sweep scores against (full ISO 8601\n" +
1931
- " instant). Before/after runs become wall-clock-independent. Default: the\n" +
1932
- " live clock. Invalid input is an error, never a silent live fallback.\n" +
1933
- " --runs=N — #458: run judge phases 1+2 N times over the SAME selections (one\n" +
1934
- " retrieve sweep) and report the run-to-run noise floor (median + spread\n" +
1935
- " + selection-identical flip counts, §15b). Default 1 (single-run\n" +
1936
- " behavior). The shared --max-calls budget spans ALL runs; without an\n" +
1937
- " explicit --max-calls the default scales ×N. Phases 3+4 stay single-shot.\n" +
1938
- "Env: ZAI_API_KEY (required — GLM-5.2 via z.ai is the judge)\n" +
1939
- "Preconditions: /tmp/hermes-{lenny,raider,nano}-state.db (readonly Hermes state DBs) when\n" +
1940
- "building the corpus fresh; ~/.claude/projects/.../infrastructure + .../aironic-marine dirs.");
1941
- process.exitCode = 1;
1942
- process.exit(1);
1943
- }
1944
- async function main() {
1945
- // Pin all retrieval knobs to shipped defaults — measure the SHIPPED ranker.
1946
- (0, retrieval_js_1.configureScoring)();
1947
- (0, retrieval_js_1.configureDecay)();
1948
- (0, retrieval_js_1.configureRecall)();
1949
- (0, retrieval_js_1.configureSessionIntent)();
1950
- let parsed;
1951
- try {
1952
- parsed = parseArgs(process.argv.slice(2));
1953
- }
1954
- catch (err) {
1955
- console.error(`[relevance-eval] ${err instanceof Error ? err.message : err}`);
1956
- usage();
1957
- }
1958
- const { positionals, judgeDelayMs, maxCalls: maxCallsArg, maxCallsExplicit, resume, verdictsJsonPath, verdictsJsonlPath, nowIso, runs, } = parsed;
1959
- const [snapshotArg, promptsSourceArg, reportArg] = positionals;
1960
- if (!snapshotArg)
1961
- usage();
1962
- // #458: pin the eval clock (invalid → usage-style error, never a silent
1963
- // live fallback — a run that believed it was pinned would silently drift).
1964
- let now;
1965
- try {
1966
- now = (0, eval_clock_js_1.parsePinnedNow)(nowIso);
1967
- }
1968
- catch (err) {
1969
- console.error(`[relevance-eval] ${err instanceof Error ? err.message : err}`);
1970
- usage();
1971
- }
1972
- console.log(`[relevance-eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}${now === null ? " (pass --now=<ISO> to pin before/after runs)" : ""}`);
1973
- // #458: the ONE shared call budget spans ALL judge runs. When --runs>1 and
1974
- // --max-calls was not passed explicitly, scale the default ×N — loudly.
1975
- const maxCalls = scaleMaxCalls(runs, maxCallsExplicit ? maxCallsArg : undefined, DEFAULT_MAX_CALLS);
1976
- if (runs > 1 && !maxCallsExplicit) {
1977
- console.log(`[relevance-eval] --runs=${runs} without an explicit --max-calls: scaling the default budget ` +
1978
- `${DEFAULT_MAX_CALLS} × ${runs} = ${maxCalls} (the budget spans ALL runs)`);
1979
- }
1980
- if (runs > 1) {
1981
- console.log(`[relevance-eval] judge variance protocol: phases 1+2 run ${runs}× over the fixed selections (see §15b)`);
1982
- }
1983
- const apiKey = process.env.ZAI_API_KEY;
1984
- if (!apiKey) {
1985
- console.error("relevance-eval: ZAI_API_KEY env var is required (GLM-5.2 via z.ai is the judge)");
1986
- process.exitCode = 1;
1987
- return;
1988
- }
1989
- // ---- prompt corpus (spec §2) ----
1990
- let prompts;
1991
- let promptsPath;
1992
- if (promptsSourceArg) {
1993
- promptsPath = promptsSourceArg;
1994
- prompts = loadPromptsJson(promptsSourceArg);
1995
- console.log(`[relevance-eval] loaded ${prompts.length} prompts from ${promptsPath}`);
1996
- }
1997
- else {
1998
- promptsPath = PROMPTS_SAVE_PATH;
1999
- if ((0, node_fs_1.existsSync)(promptsPath)) {
2000
- const existing = loadPromptsJson(promptsPath);
2001
- if (isValidV2PromptSet(existing)) {
2002
- prompts = existing;
2003
- console.log(`[relevance-eval] reusing existing v2 prompt set: ${promptsPath} (${prompts.length} prompts, stable across runs)`);
2004
- }
2005
- else {
2006
- console.log(`[relevance-eval] existing ${promptsPath} does not satisfy the v2 corpus contract (5 sources) — resampling`);
2007
- prompts = buildV2Corpus();
2008
- savePromptsJson(promptsPath, prompts);
2009
- }
2010
- }
2011
- else {
2012
- prompts = buildV2Corpus();
2013
- savePromptsJson(promptsPath, prompts);
2014
- }
2015
- console.log(`[relevance-eval] corpus: ${prompts.length} prompts across ${new Set(prompts.map((p) => p.sourceTag)).size}/${SOURCE_TAGS.length} sources; saved to ${promptsPath}`);
2016
- }
2017
- const sweepSubsetIdx = computeSweepSubsetIndices(prompts);
2018
- const db = (0, eval_db_js_1.openSnapshot)(snapshotArg);
2019
- try {
2020
- console.log(`[relevance-eval] snapshot opened readonly: ${snapshotArg} (${prompts.length} prompts × 2 modes = ${prompts.length * 2} retrieve() calls)`);
2021
- console.log("[relevance-eval] loading bge-small-en-v1.5 + replaying retrieve()...");
2022
- const t0 = Date.now();
2023
- const results = await runRetrieveSweep(db, prompts, { now: now ?? undefined });
2024
- console.log(`[relevance-eval] retrieve sweep done in ${Date.now() - t0}ms (${results.length} prompt×mode batches, LLM-free, clock ${(0, eval_clock_js_1.clockLabel)(now)})`);
2025
- // ---- judge run state (rate limiting + resumability, spec §6b) ----
2026
- (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(verdictsJsonlPath), { recursive: true });
2027
- if (!resume)
2028
- (0, node_fs_1.writeFileSync)(verdictsJsonlPath, "", "utf-8"); // fresh run: don't let a stale sidecar leak in.
2029
- const resumedMap = resume ? loadResumeMap(verdictsJsonlPath) : new Map();
2030
- let preResumedErrors = 0;
2031
- for (const v of resumedMap.values()) {
2032
- if (v && typeof v === "object" && "judgeError" in v)
2033
- preResumedErrors++;
2034
- }
2035
- const judgeState = {
2036
- apiKey,
2037
- judgeDelayMs,
2038
- maxCalls,
2039
- totalCalls: 0,
2040
- totalRetries: 0,
2041
- completedUnits: resumedMap.size,
2042
- errorUnits: preResumedErrors,
2043
- startTime: Date.now(),
2044
- resumed: resumedMap,
2045
- sidecarPath: verdictsJsonlPath,
2046
- aborted: null,
2047
- };
2048
- if (resume) {
2049
- console.log(`[relevance-eval] --resume: ${resumedMap.size} units already checkpointed in ${verdictsJsonlPath}`);
2050
- }
2051
- const sweepRows = [];
2052
- // #458: judge phases 1+2 run `runs` times over the SAME selections (the
2053
- // retrieve sweep above ran ONCE, before judging — so selections are fixed
2054
- // across runs under a live clock too). Per-run verdict arrays feed the
2055
- // §15b variance section; run 1's verdicts land in `results` so every body
2056
- // section renders exactly as a single-run report. Checkpoint keys carry a
2057
- // 1-based run prefix when runs>1 (`full|r1|…`) so sidecar entries never
2058
- // collide across runs; the legacy single-run keys are byte-unchanged when
2059
- // runs==1 (old single-run sidecars still resume — an N-run invocation
2060
- // starts a fresh sidecar by default).
2061
- const perRunFull = [];
2062
- const perRunLine = [];
2063
- const judgeRunKey = (phase, run, ...parts) => runs > 1 ? [phase, `r${run}`, ...parts].join("|") : [phase, ...parts].join("|");
2064
- try {
2065
- // ---- Phase 1: full_verdict (all prompts × both modes), N runs ----
2066
- for (let run = 1; run <= runs; run++) {
2067
- console.log(`[relevance-eval] phase 1/4: full-content judging (${results.length} units, run ${run}/${runs})...`);
2068
- const units = [];
2069
- for (const r of results) {
2070
- const pIdx = prompts.indexOf(r.prompt);
2071
- const key = judgeRunKey("full", run, pIdx, r.mode);
2072
- const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
2073
- const promptText = buildFullContentJudgePrompt(r.prompt.prompt, r.topIds, r.contents);
2074
- const verdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
2075
- units.push({ key: `${pIdx}|${r.mode}`, verdicts });
2076
- if (run === 1)
2077
- r.fullVerdicts = verdicts;
2078
- }
2079
- perRunFull.push(units);
2080
- }
2081
- // ---- Phase 2: line_verdict at maxLen=PROD_TITLE_CHARS (production render), N runs ----
2082
- for (let run = 1; run <= runs; run++) {
2083
- console.log(`[relevance-eval] phase 2/4: line-only judging at production maxLen=${PROD_TITLE_CHARS} (${results.length} units, run ${run}/${runs})...`);
2084
- const units = [];
2085
- for (const r of results) {
2086
- const pIdx = prompts.indexOf(r.prompt);
2087
- const key = judgeRunKey("line", run, pIdx, r.mode);
2088
- const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
2089
- const lines = {};
2090
- for (const id of r.topIds)
2091
- lines[id] = (0, recall_index_js_1.formatIndexLine)(r.rows[id], PROD_TITLE_CHARS);
2092
- const promptText = buildLineJudgePrompt(r.prompt.prompt, r.topIds, lines);
2093
- const verdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
2094
- units.push({ key: `${pIdx}|${r.mode}`, verdicts });
2095
- if (run === 1)
2096
- r.lineVerdicts = verdicts;
2097
- }
2098
- perRunLine.push(units);
2099
- }
2100
- // ---- Phase 3: snippet-length sweep (40-subset, scope OFF, 4 variants) ----
2101
- const offResultByPromptIdx = new Map();
2102
- for (const r of results) {
2103
- if (r.mode === "off")
2104
- offResultByPromptIdx.set(prompts.indexOf(r.prompt), r);
2105
- }
2106
- console.log(`[relevance-eval] phase 3/4: snippet-length sweep (${sweepSubsetIdx.length} prompts × ${SWEEP_VARIANTS.length} variants)...`);
2107
- for (const promptIdx of sweepSubsetIdx) {
2108
- const r = offResultByPromptIdx.get(promptIdx);
2109
- if (!r)
2110
- continue;
2111
- for (const variant of SWEEP_VARIANTS) {
2112
- const key = `sweep|${promptIdx}|off|${variant}`;
2113
- const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
2114
- const lines = {};
2115
- for (const id of r.topIds)
2116
- lines[id] = renderVariantLine(r.rows[id], variant);
2117
- const promptText = buildLineJudgePrompt(r.prompt.prompt, r.topIds, lines);
2118
- const verdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
2119
- sweepRows.push({ promptIdx, variant, verdicts, lines });
2120
- }
2121
- }
2122
- // ---- Phase 4: redundancy over the production 6 (per prompt × mode) ----
2123
- console.log(`[relevance-eval] phase 4/4: redundancy judging (${results.length} units, some skipped when <2 shown)...`);
2124
- for (const r of results) {
2125
- const rowsArr = r.topIds.map((id) => r.rows[id]).filter(Boolean);
2126
- const { shownIds, lines } = renderProductionBlock(rowsArr, PROD_MIN_SIMILARITY, PROD_MENU_K, PROD_TITLE_CHARS);
2127
- if (shownIds.length < 2) {
2128
- r.redundancy = { skipped: true, shown: shownIds.length };
2129
- continue;
2130
- }
2131
- const key = `redundancy|${prompts.indexOf(r.prompt)}|${r.mode}`;
2132
- const shownLines = shownIds.map((id) => lines[id]);
2133
- const promptText = buildRedundancyPrompt(r.prompt.prompt, shownLines);
2134
- const parsed = await judgeRedundancyUnit(judgeState, key, promptText, shownIds.length);
2135
- r.redundancy = "judgeError" in parsed ? parsed : { shown: shownIds.length, ...parsed };
2136
- }
2137
- }
2138
- catch (err) {
2139
- if (err instanceof RunAbortedError) {
2140
- console.error(`[relevance-eval] RUN ABORTED: ${err.message}`);
2141
- judgeState.aborted = judgeState.aborted ?? { reason: err.message };
2142
- }
2143
- else {
2144
- throw err;
2145
- }
2146
- }
2147
- // ---- #458: per-run headline metrics + the judge-variance account ----
2148
- // Abort-tolerant like everything downstream of the judge loop: a run
2149
- // stopped by the budget / error-rate gate leaves partial per-run arrays,
2150
- // and the report still renders (run-1 verdicts + a null variance).
2151
- let variance = null;
2152
- const varianceComplete = perRunFull.length === runs &&
2153
- perRunLine.length === runs &&
2154
- perRunFull.every((u) => u.length === results.length) &&
2155
- perRunLine.every((u) => u.length === results.length);
2156
- if (runs > 1 && !varianceComplete) {
2157
- console.log(`[relevance-eval] judge runs incomplete (${perRunFull.length}/${runs} full-verdict runs filled) — §15b variance skipped`);
2158
- }
2159
- if (runs > 1 && varianceComplete) {
2160
- const runResults = (run) => results.map((r, i) => ({
2161
- ...r,
2162
- fullVerdicts: perRunFull[run - 1][i].verdicts,
2163
- lineVerdicts: perRunLine[run - 1][i].verdicts,
2164
- }));
2165
- const series = [
2166
- { label: `precision@${PROD_MENU_K} OFF`, values: [] },
2167
- { label: `precision@${PROD_MENU_K} ON`, values: [] },
2168
- { label: "snippet_failure_rate OFF", values: [] },
2169
- ];
2170
- for (let run = 1; run <= runs; run++) {
2171
- const rs = runResults(run);
2172
- const offRun = rs.filter((r) => r.mode === "off");
2173
- const onRun = rs.filter((r) => r.mode === "on");
2174
- const mOff = computeMetricsAtK(offRun, PROD_MENU_K);
2175
- const mOn = computeMetricsAtK(onRun, PROD_MENU_K);
2176
- const gap = computeSnippetGap(offRun);
2177
- series[0].values.push(mOff ? mOff.precision : null);
2178
- series[1].values.push(mOn ? mOn.precision : null);
2179
- series[2].values.push(gap.fullRelevantTotal > 0 ? gap.snippetFailure / gap.fullRelevantTotal : null);
2180
- }
2181
- variance = computeJudgeVariance(perRunFull, series);
2182
- console.log(`[relevance-eval] judge variance over ${runs} runs (selections fixed): ` +
2183
- variance.metrics.map((m) => `${m.label} median ${m.median !== null ? pct(m.median) : "n/a"}, spread ${m.spreadPts !== null ? m.spreadPts.toFixed(1) + "pts" : "n/a"}`).join("; ") +
2184
- `; max pairwise flips ${variance.maxRowsFlipped}/${variance.totalComparableRows} rows, ${variance.maxUnitsFlipped} units`);
2185
- }
2186
- // ---- persist raw verdicts (source of truth, spec §6) ----
2187
- const verdictRows = buildVerdictRows(prompts, results, sweepRows, runs);
2188
- (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(verdictsJsonPath), { recursive: true });
2189
- (0, node_fs_1.writeFileSync)(verdictsJsonPath, JSON.stringify(verdictRows, null, 2), "utf-8");
2190
- console.log(`[relevance-eval] raw verdicts saved: ${verdictsJsonPath} (${verdictRows.length} rows), checkpoint sidecar: ${verdictsJsonlPath}`);
2191
- const report = renderReport({
2192
- snapshotPath: snapshotArg,
2193
- promptsPath,
2194
- verdictsJsonPath,
2195
- prompts,
2196
- results,
2197
- sweepRows,
2198
- sweepSubsetIdx,
2199
- verdictRows,
2200
- judgeState,
2201
- generatedAt: new Date().toISOString(),
2202
- db,
2203
- clock: (0, eval_clock_js_1.clockLabel)(now),
2204
- variance,
2205
- });
2206
- const reportPath = reportArg ?? DEFAULT_REPORT_PATH;
2207
- (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(reportPath), { recursive: true });
2208
- (0, node_fs_1.writeFileSync)(reportPath, report, "utf-8");
2209
- console.log("\n=== relevance-eval summary ===");
2210
- const off6 = computeMetricsAtK(results.filter((r) => r.mode === "off"), PROD_MENU_K);
2211
- const on6 = computeMetricsAtK(results.filter((r) => r.mode === "on"), PROD_MENU_K);
2212
- console.log(`precision@${PROD_MENU_K} (menu) OFF ${off6 ? pct(off6.precision) : "n/a"} → ON ${on6 ? pct(on6.precision) : "n/a"}`);
2213
- const gapOffSummary = computeSnippetGap(results.filter((r) => r.mode === "off"));
2214
- console.log(`snippet_failure_rate (OFF) ${gapOffSummary.fullRelevantTotal > 0 ? pct(gapOffSummary.snippetFailure / gapOffSummary.fullRelevantTotal) : "n/a"}`);
2215
- console.log(`judge units: ${judgeState.completedUnits} completed, ${judgeState.errorUnits} judge_error, ${judgeState.totalCalls} fresh calls, ${judgeState.totalRetries} retries`);
2216
- if (judgeState.aborted)
2217
- console.log(`ABORTED: ${judgeState.aborted.reason}`);
2218
- console.log(`\n[relevance-eval] full report: ${reportPath}`);
2219
- if (judgeState.aborted)
2220
- process.exitCode = 1;
2221
- }
2222
- finally {
2223
- try {
2224
- db.close();
2225
- }
2226
- catch {
2227
- /* already closed */
2228
- }
2229
- }
2230
- }
2231
- // The importance-eval.ts guard: run main() only when executed directly, so
2232
- // the exported pure helpers (parseArgs, scaleMaxCalls, median,
2233
- // computeJudgeVariance — #458) are importable from the vitest suite without
2234
- // firing the CLI.
2235
- if (process.argv[1] && process.argv[1].endsWith("relevance-eval.js")) {
2236
- main().catch((err) => {
2237
- console.error("[relevance-eval] FAILED:", err instanceof Error ? err.stack : String(err));
2238
- process.exitCode = 1;
2239
- });
2240
- }