@gamaze/hicortex 0.22.0 → 0.22.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/assets/dashboard.html +209 -76
  2. package/dist/calibration.d.ts +92 -12
  3. package/dist/calibration.js +102 -15
  4. package/dist/classify-domains.js +6 -1
  5. package/dist/consolidate.js +87 -19
  6. package/dist/dashboard.d.ts +11 -0
  7. package/dist/dashboard.js +9 -0
  8. package/dist/db.js +74 -0
  9. package/dist/eval/decay-eval.d.ts +4 -2
  10. package/dist/eval/decay-eval.js +4 -4
  11. package/dist/eval/eval-clock.d.ts +32 -0
  12. package/dist/eval/eval-clock.js +47 -0
  13. package/dist/eval/graph-eval.d.ts +15 -2
  14. package/dist/eval/graph-eval.js +51 -5
  15. package/dist/eval/planted-eval.d.ts +4 -0
  16. package/dist/eval/planted-eval.js +27 -2
  17. package/dist/eval/planted-harness.d.ts +7 -0
  18. package/dist/eval/planted-harness.js +2 -0
  19. package/dist/eval/ranking-battery.d.ts +49 -2
  20. package/dist/eval/ranking-battery.js +110 -2
  21. package/dist/eval/ranking-eval.d.ts +26 -6
  22. package/dist/eval/ranking-eval.js +197 -34
  23. package/dist/eval/ranking-fixtures.d.ts +41 -1
  24. package/dist/eval/ranking-fixtures.js +261 -2
  25. package/dist/eval/recall-sweep.d.ts +7 -2
  26. package/dist/eval/recall-sweep.js +42 -13
  27. package/dist/eval/relevance-eval.d.ts +115 -1
  28. package/dist/eval/relevance-eval.js +318 -32
  29. package/dist/eval/run-eval.d.ts +7 -4
  30. package/dist/eval/run-eval.js +36 -9
  31. package/dist/mcp-server.js +37 -9
  32. package/dist/nightly.js +14 -0
  33. package/dist/recall-index.d.ts +46 -3
  34. package/dist/recall-index.js +83 -26
  35. package/dist/recall-precision.d.ts +212 -0
  36. package/dist/recall-precision.js +381 -0
  37. package/dist/retrieval.d.ts +34 -15
  38. package/dist/retrieval.js +132 -59
  39. package/dist/types.d.ts +7 -0
  40. package/package.json +1 -1
  41. package/server.json +3 -3
@@ -60,17 +60,33 @@
60
60
  * Run:
61
61
  * npm run eval:relevance -- <snapshot.db> [prompts.json] [report.md] \
62
62
  * [--judge-delay-ms=2000] [--max-calls=900] [--resume] \
63
- * [--verdicts-json=path] [--verdicts-jsonl=path]
63
+ * [--verdicts-json=path] [--verdicts-jsonl=path] \
64
+ * [--now=<ISO>] [--runs=N]
65
+ *
66
+ * #458 clock pin + judge variance protocol:
67
+ * - `--now=<ISO>` pins the clock the retrieve sweep scores against (the
68
+ * retrieve() `now` seam) — before/after runs become wall-clock-independent
69
+ * (default: live clock, the pre-#458 behavior).
70
+ * - `--runs=N` (default 1) runs judge phases 1+2 N times over the SAME
71
+ * selections (one retrieve sweep) and reports the run-to-run noise floor
72
+ * (median + spread + selection-identical flip counts, §15b). Phases 3+4
73
+ * stay single-shot; the shared --max-calls budget spans all runs and the
74
+ * default scales ×N unless --max-calls was passed explicitly.
64
75
  */
65
76
  var __importDefault = (this && this.__importDefault) || function (mod) {
66
77
  return (mod && mod.__esModule) ? mod : { "default": mod };
67
78
  };
68
79
  Object.defineProperty(exports, "__esModule", { value: true });
80
+ exports.parseArgs = parseArgs;
81
+ exports.scaleMaxCalls = scaleMaxCalls;
82
+ exports.median = median;
83
+ exports.computeJudgeVariance = computeJudgeVariance;
69
84
  const node_fs_1 = require("node:fs");
70
85
  const node_path_1 = require("node:path");
71
86
  const node_os_1 = require("node:os");
72
87
  const better_sqlite3_1 = __importDefault(require("better-sqlite3"));
73
88
  const eval_db_js_1 = require("./eval-db.js");
89
+ const eval_clock_js_1 = require("./eval-clock.js");
74
90
  const embedder_js_1 = require("../embedder.js");
75
91
  const retrieval_js_1 = require("../retrieval.js");
76
92
  const recall_index_js_1 = require("../recall-index.js");
@@ -379,7 +395,7 @@ function computeSweepSubsetIndices(prompts) {
379
395
  // ---------------------------------------------------------------------------
380
396
  // Retrieve sweep (scope OFF vs ON) — LLM-free system-under-test
381
397
  // ---------------------------------------------------------------------------
382
- async function runRetrieveSweep(db, prompts) {
398
+ async function runRetrieveSweep(db, prompts, opts) {
383
399
  const embedCache = new Map();
384
400
  for (const p of prompts) {
385
401
  if (!embedCache.has(p.prompt))
@@ -401,6 +417,9 @@ async function runRetrieveSweep(db, prompts) {
401
417
  limit: RESULT_K,
402
418
  queryEmbedding,
403
419
  noStrengthen: true,
420
+ // #458: the pinned clock (undefined = live) — identical selections
421
+ // across before/after runs regardless of when each run executes.
422
+ ...(opts?.now ? { now: opts.now } : {}),
404
423
  ...scopeOpts,
405
424
  });
406
425
  const topIds = retrieved.map((r) => r.id);
@@ -1063,7 +1082,7 @@ function similarityBucketLabel(sim) {
1063
1082
  }
1064
1083
  return ">=1.00";
1065
1084
  }
1066
- function buildVerdictRows(prompts, results, sweepRows) {
1085
+ function buildVerdictRows(prompts, results, sweepRows, runsTotal = 1) {
1067
1086
  const byKey = new Map();
1068
1087
  for (const r of results) {
1069
1088
  const pIdx = prompts.findIndex((p) => p.prompt === r.prompt.prompt && p.agent === r.prompt.agent);
@@ -1104,6 +1123,7 @@ function buildVerdictRows(prompts, results, sweepRows) {
1104
1123
  line_reason: lv ? lv.reason : lineOk ? "" : r.lineVerdicts.judgeError,
1105
1124
  rendered_line: row ? (0, recall_index_js_1.formatIndexLine)(row, PROD_TITLE_CHARS) : "",
1106
1125
  snippet_len_variant: String(PROD_TITLE_CHARS),
1126
+ ...(runsTotal > 1 ? { run: 1 } : {}),
1107
1127
  });
1108
1128
  }
1109
1129
  }
@@ -1148,13 +1168,18 @@ function buildVerdictRows(prompts, results, sweepRows) {
1148
1168
  }
1149
1169
  return rows;
1150
1170
  }
1171
+ /** Parsed and exported for tests (#458). Throws on an invalid --runs value
1172
+ * (NaN or <1) — fail-explicit, never a silent default. */
1151
1173
  function parseArgs(argv) {
1152
1174
  const positionals = [];
1153
1175
  let judgeDelayMs = DEFAULT_JUDGE_DELAY_MS;
1154
1176
  let maxCalls = DEFAULT_MAX_CALLS;
1177
+ let maxCallsExplicit = false;
1155
1178
  let resume = false;
1156
1179
  let verdictsJsonPath = VERDICTS_JSON_PATH;
1157
1180
  let verdictsJsonlPath = VERDICTS_JSONL_PATH;
1181
+ let nowIso;
1182
+ let runs = 1;
1158
1183
  for (const arg of argv) {
1159
1184
  if (arg === "--resume") {
1160
1185
  resume = true;
@@ -1164,17 +1189,132 @@ function parseArgs(argv) {
1164
1189
  if (m) {
1165
1190
  if (m[1] === "judge-delay-ms")
1166
1191
  judgeDelayMs = parseInt(m[2], 10);
1167
- else if (m[1] === "max-calls")
1192
+ else if (m[1] === "max-calls") {
1168
1193
  maxCalls = parseInt(m[2], 10);
1194
+ maxCallsExplicit = true;
1195
+ }
1169
1196
  else if (m[1] === "verdicts-json")
1170
1197
  verdictsJsonPath = m[2];
1171
1198
  else if (m[1] === "verdicts-jsonl")
1172
1199
  verdictsJsonlPath = m[2];
1200
+ else if (m[1] === "now")
1201
+ nowIso = m[2];
1202
+ else if (m[1] === "runs") {
1203
+ const parsed = Number(m[2]);
1204
+ if (!Number.isInteger(parsed) || parsed < 1) {
1205
+ throw new Error(`--runs must be an integer >= 1 (got "${m[2]}")`);
1206
+ }
1207
+ runs = parsed;
1208
+ }
1173
1209
  continue;
1174
1210
  }
1175
1211
  positionals.push(arg);
1176
1212
  }
1177
- return { positionals, judgeDelayMs, maxCalls, resume, verdictsJsonPath, verdictsJsonlPath };
1213
+ return {
1214
+ positionals,
1215
+ judgeDelayMs,
1216
+ maxCalls,
1217
+ maxCallsExplicit,
1218
+ resume,
1219
+ verdictsJsonPath,
1220
+ verdictsJsonlPath,
1221
+ nowIso,
1222
+ runs,
1223
+ };
1224
+ }
1225
+ /**
1226
+ * #458 — resolve the effective --max-calls budget. The ONE shared budget
1227
+ * spans ALL judge runs, so when --runs>1 and the caller did NOT pass
1228
+ * --max-calls explicitly, the default scales ×N (each run re-judges every
1229
+ * unit). An explicit budget always wins — the caller sized it for the whole
1230
+ * invocation. Pure, exported for tests.
1231
+ */
1232
+ function scaleMaxCalls(runs, explicitMaxCalls, defaultMaxCalls) {
1233
+ if (explicitMaxCalls !== undefined)
1234
+ return explicitMaxCalls;
1235
+ return runs > 1 ? defaultMaxCalls * runs : defaultMaxCalls;
1236
+ }
1237
+ // ---------------------------------------------------------------------------
1238
+ // Judge variance (#458) — median, spread, and selection-identical flip
1239
+ // counts over N runs of the SAME selections. Pure + synthetic-verdict
1240
+ // testable; the orchestration (running the judge N times) lives in main.
1241
+ // ---------------------------------------------------------------------------
1242
+ /** Median of a numeric sample (even count → mean of the two middle values).
1243
+ * Empty input → NaN, the mean() convention above. Does not mutate input. */
1244
+ function median(xs) {
1245
+ if (xs.length === 0)
1246
+ return NaN;
1247
+ const sorted = [...xs].sort((a, b) => a - b);
1248
+ const mid = Math.floor(sorted.length / 2);
1249
+ return sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
1250
+ }
1251
+ /**
1252
+ * Compute the #458 judge-variance account over N runs of the same selections:
1253
+ * per-metric median + spread (the noise floor), and pairwise verdict flip
1254
+ * counts (rows = one Q-label judgment on one (prompt,mode) unit; units flip
1255
+ * if any row does). `perRunVerdicts[r]` is run r+1's units in a stable
1256
+ * order/identity; `metricSeries[s].values[r]` is run r+1's headline value.
1257
+ */
1258
+ function computeJudgeVariance(perRunVerdicts, metricSeries) {
1259
+ const runs = perRunVerdicts.length;
1260
+ const flips = [];
1261
+ let maxRows = 0;
1262
+ let maxUnits = 0;
1263
+ for (let i = 0; i < runs; i++) {
1264
+ for (let j = i + 1; j < runs; j++) {
1265
+ const byKeyI = new Map(perRunVerdicts[i].map((u) => [u.key, u.verdicts]));
1266
+ let rowsCompared = 0;
1267
+ let rowsFlipped = 0;
1268
+ let unitsCompared = 0;
1269
+ let unitsFlipped = 0;
1270
+ for (const unitJ of perRunVerdicts[j]) {
1271
+ const vi = byKeyI.get(unitJ.key);
1272
+ if (!vi || !Array.isArray(vi) || !Array.isArray(unitJ.verdicts))
1273
+ continue; // judge_error/absent — not comparable
1274
+ const vj = unitJ.verdicts;
1275
+ unitsCompared++;
1276
+ let unitFlipped = false;
1277
+ const byLabelJ = new Map(vj.map((v) => [v.id, v.verdict]));
1278
+ for (const rowI of vi) {
1279
+ const verdictJ = byLabelJ.get(rowI.id);
1280
+ if (verdictJ === undefined)
1281
+ continue; // label omitted in run j — not comparable
1282
+ rowsCompared++;
1283
+ if (verdictJ !== rowI.verdict) {
1284
+ rowsFlipped++;
1285
+ unitFlipped = true;
1286
+ }
1287
+ }
1288
+ if (unitFlipped)
1289
+ unitsFlipped++;
1290
+ }
1291
+ flips.push({ runA: i + 1, runB: j + 1, rowsCompared, rowsFlipped, unitsCompared, unitsFlipped });
1292
+ maxRows = Math.max(maxRows, rowsFlipped);
1293
+ maxUnits = Math.max(maxUnits, unitsFlipped);
1294
+ }
1295
+ }
1296
+ const metrics = metricSeries.map((s) => {
1297
+ const usable = s.values.filter((v) => v !== null && Number.isFinite(v));
1298
+ if (usable.length === 0)
1299
+ return { ...s, median: null, spreadPts: null };
1300
+ return {
1301
+ ...s,
1302
+ median: median(usable),
1303
+ spreadPts: (Math.max(...usable) - Math.min(...usable)) * 100,
1304
+ };
1305
+ });
1306
+ const judgeErrorUnitsPerRun = perRunVerdicts.map((units) => units.filter((u) => !Array.isArray(u.verdicts)).length);
1307
+ return {
1308
+ runs,
1309
+ metrics,
1310
+ flips,
1311
+ maxRowsFlipped: maxRows,
1312
+ maxUnitsFlipped: maxUnits,
1313
+ meanRowsFlipped: flips.length > 0 ? mean(flips.map((f) => f.rowsFlipped)) : 0,
1314
+ meanUnitsFlipped: flips.length > 0 ? mean(flips.map((f) => f.unitsFlipped)) : 0,
1315
+ totalComparableRows: flips.length > 0 ? flips[0].rowsCompared : 0,
1316
+ judgeErrorUnitsPerRun,
1317
+ };
1178
1318
  }
1179
1319
  // ---------------------------------------------------------------------------
1180
1320
  // Reporting
@@ -1207,7 +1347,8 @@ function renderReport(args) {
1207
1347
  L.push(`Snapshot: \`${args.snapshotPath}\` \n` +
1208
1348
  `Prompts: \`${args.promptsPath}\` (${prompts.length} prompts; requested ${SOURCE_TAGS.length * N_PER_SOURCE} — ` +
1209
1349
  `see §0 for any per-source shortfall) \n` +
1210
- `Generated: ${args.generatedAt}\n`);
1350
+ `Generated: ${args.generatedAt} \n` +
1351
+ `Clock: ${args.clock} (the instant the retrieve sweep scored against, #458)${args.variance ? ` \nJudge runs: ${args.variance.runs} (phases 1+2 re-judged over fixed selections — see §15b)` : ""}\n`);
1211
1352
  L.push(`_Spec: \`specs/2026-08-02-relevance-eval.md\`. Real agent prompts replayed through \`retrieve()\` on a ` +
1212
1353
  `READONLY snapshot; GLM-5.2 (z.ai) is the JUDGE only — retrieve() is LLM-free. Two independent, BLIND judge ` +
1213
1354
  `calls per (prompt × mode): \`full_verdict\` (up to 2000 chars) and \`line_verdict\` (ONLY the rendered ` +
@@ -1608,6 +1749,38 @@ function renderReport(args) {
1608
1749
  L.push(`**Scope ON−OFF delta (precision@${PROD_MENU_K}):** ${deltaStr} — **${distinguishable}**\n`);
1609
1750
  }
1610
1751
  // =====================================================================
1752
+ // §15b — judge variance (#458): the measured run-to-run noise floor.
1753
+ // Rendered ONLY when --runs>1.
1754
+ // =====================================================================
1755
+ if (args.variance) {
1756
+ const v = args.variance;
1757
+ L.push(`## 15b. Judge variance — run-to-run noise floor (#458, ${v.runs} runs)\n`);
1758
+ L.push(`_Judge phases 1+2 ran ${v.runs} times over the SAME selections (one retrieve sweep feeds every run — ` +
1759
+ `the sweep precedes judging, so selections are fixed under a live clock too). ALL drift measured here is ` +
1760
+ `judge-only instrument noise, the class PR C evidenced at 38/380 flipped verdicts. Read every body section ` +
1761
+ `above as run 1; this section carries the median and the band._\n`);
1762
+ L.push("| metric | " + v.metrics.map((_, i) => `run ${i + 1}`).join(" | ") + " | median | spread (max−min) |");
1763
+ L.push("|---|" + v.metrics.map(() => "---").join("|") + "|---|---|");
1764
+ for (const m of v.metrics) {
1765
+ const cells = m.values.map((x) => (x === null || !Number.isFinite(x) ? "n/a" : pct(x))).join(" | ");
1766
+ L.push(`| ${m.label} | ${cells} | **${m.median !== null && Number.isFinite(m.median) ? pct(m.median) : "n/a"}** | ${m.spreadPts !== null && Number.isFinite(m.spreadPts) ? `${m.spreadPts.toFixed(1)}pts` : "n/a"} |`);
1767
+ }
1768
+ L.push("");
1769
+ L.push(`_Pairwise verdict flips on selection-identical units (full_verdict rows; judge_error units excluded from ` +
1770
+ `the denominators — error units per run: ${v.judgeErrorUnitsPerRun.map((e, i) => `r${i + 1} ${e}`).join(", ")}). ` +
1771
+ `A verdict ROW is one Q-label judgment on one (prompt, mode) unit; a UNIT flips if any of its rows differ._\n`);
1772
+ L.push("| pair | rows flipped / compared | units flipped / compared |");
1773
+ L.push("|---|---|---|");
1774
+ for (const f of v.flips) {
1775
+ L.push(`| r${f.runA}↔r${f.runB} | ${f.rowsFlipped} / ${f.rowsCompared} | ${f.unitsFlipped} / ${f.unitsCompared} |`);
1776
+ }
1777
+ L.push("");
1778
+ L.push(`**Noise floor:** max pairwise flips ${v.maxRowsFlipped} rows / ${v.maxUnitsFlipped} units; mean ` +
1779
+ `${v.meanRowsFlipped.toFixed(1)} rows / ${v.meanUnitsFlipped.toFixed(1)} units over ${v.flips.length} pair(s); ` +
1780
+ `total comparable rows ${v.totalComparableRows}. A before/after headline delta smaller than the spread row ` +
1781
+ `above is NOT resolvable by this instrument — report it as within judge noise._\n`);
1782
+ }
1783
+ // =====================================================================
1611
1784
  // §16 — DECISION THRESHOLDS (spec §9) — filled with measured values
1612
1785
  // =====================================================================
1613
1786
  L.push("## 16. Decision thresholds (spec §9) — measured values + triggered actions\n");
@@ -1743,7 +1916,8 @@ function renderReport(args) {
1743
1916
  function usage() {
1744
1917
  console.error("Usage: relevance-eval.js <snapshot.db> [prompts.json] [report.md] " +
1745
1918
  "[--judge-delay-ms=2000] [--max-calls=900] [--resume] " +
1746
- "[--verdicts-json=path] [--verdicts-jsonl=path]\n" +
1919
+ "[--verdicts-json=path] [--verdicts-jsonl=path] " +
1920
+ "[--now=<ISO>] [--runs=N]\n" +
1747
1921
  " <snapshot.db> — REQUIRED. Readonly bedrock snapshot (opened via openSnapshot).\n" +
1748
1922
  " [prompts.json] — omitted: build the v2 corpus fresh (20/source × 5 sources, spec §2),\n" +
1749
1923
  " saved to data/prompts.json. A valid existing v2 set is reused verbatim.\n" +
@@ -1753,6 +1927,14 @@ function usage() {
1753
1927
  " JSONL paths (default data/relevance-verdicts.json[l]). ALWAYS override\n" +
1754
1928
  " these for a re-run against a new snapshot — otherwise a re-run silently\n" +
1755
1929
  " clobbers a prior baseline's raw verdicts.\n" +
1930
+ " --now=<ISO> — #458: pin the clock the retrieve sweep scores against (full ISO 8601\n" +
1931
+ " instant). Before/after runs become wall-clock-independent. Default: the\n" +
1932
+ " live clock. Invalid input is an error, never a silent live fallback.\n" +
1933
+ " --runs=N — #458: run judge phases 1+2 N times over the SAME selections (one\n" +
1934
+ " retrieve sweep) and report the run-to-run noise floor (median + spread\n" +
1935
+ " + selection-identical flip counts, §15b). Default 1 (single-run\n" +
1936
+ " behavior). The shared --max-calls budget spans ALL runs; without an\n" +
1937
+ " explicit --max-calls the default scales ×N. Phases 3+4 stay single-shot.\n" +
1756
1938
  "Env: ZAI_API_KEY (required — GLM-5.2 via z.ai is the judge)\n" +
1757
1939
  "Preconditions: /tmp/hermes-{lenny,raider,nano}-state.db (readonly Hermes state DBs) when\n" +
1758
1940
  "building the corpus fresh; ~/.claude/projects/.../infrastructure + .../aironic-marine dirs.");
@@ -1765,10 +1947,39 @@ async function main() {
1765
1947
  (0, retrieval_js_1.configureDecay)();
1766
1948
  (0, retrieval_js_1.configureRecall)();
1767
1949
  (0, retrieval_js_1.configureSessionIntent)();
1768
- const { positionals, judgeDelayMs, maxCalls, resume, verdictsJsonPath, verdictsJsonlPath } = parseArgs(process.argv.slice(2));
1950
+ let parsed;
1951
+ try {
1952
+ parsed = parseArgs(process.argv.slice(2));
1953
+ }
1954
+ catch (err) {
1955
+ console.error(`[relevance-eval] ${err instanceof Error ? err.message : err}`);
1956
+ usage();
1957
+ }
1958
+ const { positionals, judgeDelayMs, maxCalls: maxCallsArg, maxCallsExplicit, resume, verdictsJsonPath, verdictsJsonlPath, nowIso, runs, } = parsed;
1769
1959
  const [snapshotArg, promptsSourceArg, reportArg] = positionals;
1770
1960
  if (!snapshotArg)
1771
1961
  usage();
1962
+ // #458: pin the eval clock (invalid → usage-style error, never a silent
1963
+ // live fallback — a run that believed it was pinned would silently drift).
1964
+ let now;
1965
+ try {
1966
+ now = (0, eval_clock_js_1.parsePinnedNow)(nowIso);
1967
+ }
1968
+ catch (err) {
1969
+ console.error(`[relevance-eval] ${err instanceof Error ? err.message : err}`);
1970
+ usage();
1971
+ }
1972
+ console.log(`[relevance-eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}${now === null ? " (pass --now=<ISO> to pin before/after runs)" : ""}`);
1973
+ // #458: the ONE shared call budget spans ALL judge runs. When --runs>1 and
1974
+ // --max-calls was not passed explicitly, scale the default ×N — loudly.
1975
+ const maxCalls = scaleMaxCalls(runs, maxCallsExplicit ? maxCallsArg : undefined, DEFAULT_MAX_CALLS);
1976
+ if (runs > 1 && !maxCallsExplicit) {
1977
+ console.log(`[relevance-eval] --runs=${runs} without an explicit --max-calls: scaling the default budget ` +
1978
+ `${DEFAULT_MAX_CALLS} × ${runs} = ${maxCalls} (the budget spans ALL runs)`);
1979
+ }
1980
+ if (runs > 1) {
1981
+ console.log(`[relevance-eval] judge variance protocol: phases 1+2 run ${runs}× over the fixed selections (see §15b)`);
1982
+ }
1772
1983
  const apiKey = process.env.ZAI_API_KEY;
1773
1984
  if (!apiKey) {
1774
1985
  console.error("relevance-eval: ZAI_API_KEY env var is required (GLM-5.2 via z.ai is the judge)");
@@ -1809,8 +2020,8 @@ async function main() {
1809
2020
  console.log(`[relevance-eval] snapshot opened readonly: ${snapshotArg} (${prompts.length} prompts × 2 modes = ${prompts.length * 2} retrieve() calls)`);
1810
2021
  console.log("[relevance-eval] loading bge-small-en-v1.5 + replaying retrieve()...");
1811
2022
  const t0 = Date.now();
1812
- const results = await runRetrieveSweep(db, prompts);
1813
- console.log(`[relevance-eval] retrieve sweep done in ${Date.now() - t0}ms (${results.length} prompt×mode batches, LLM-free)`);
2023
+ const results = await runRetrieveSweep(db, prompts, { now: now ?? undefined });
2024
+ console.log(`[relevance-eval] retrieve sweep done in ${Date.now() - t0}ms (${results.length} prompt×mode batches, LLM-free, clock ${(0, eval_clock_js_1.clockLabel)(now)})`);
1814
2025
  // ---- judge run state (rate limiting + resumability, spec §6b) ----
1815
2026
  (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(verdictsJsonlPath), { recursive: true });
1816
2027
  if (!resume)
@@ -1838,25 +2049,53 @@ async function main() {
1838
2049
  console.log(`[relevance-eval] --resume: ${resumedMap.size} units already checkpointed in ${verdictsJsonlPath}`);
1839
2050
  }
1840
2051
  const sweepRows = [];
2052
+ // #458: judge phases 1+2 run `runs` times over the SAME selections (the
2053
+ // retrieve sweep above ran ONCE, before judging — so selections are fixed
2054
+ // across runs under a live clock too). Per-run verdict arrays feed the
2055
+ // §15b variance section; run 1's verdicts land in `results` so every body
2056
+ // section renders exactly as a single-run report. Checkpoint keys carry a
2057
+ // 1-based run prefix when runs>1 (`full|r1|…`) so sidecar entries never
2058
+ // collide across runs; the legacy single-run keys are byte-unchanged when
2059
+ // runs==1 (old single-run sidecars still resume — an N-run invocation
2060
+ // starts a fresh sidecar by default).
2061
+ const perRunFull = [];
2062
+ const perRunLine = [];
2063
+ const judgeRunKey = (phase, run, ...parts) => runs > 1 ? [phase, `r${run}`, ...parts].join("|") : [phase, ...parts].join("|");
1841
2064
  try {
1842
- // ---- Phase 1: full_verdict (all prompts × both modes) ----
1843
- console.log(`[relevance-eval] phase 1/4: full-content judging (${results.length} units)...`);
1844
- for (const r of results) {
1845
- const key = `full|${prompts.indexOf(r.prompt)}|${r.mode}`;
1846
- const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
1847
- const promptText = buildFullContentJudgePrompt(r.prompt.prompt, r.topIds, r.contents);
1848
- r.fullVerdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
2065
+ // ---- Phase 1: full_verdict (all prompts × both modes), N runs ----
2066
+ for (let run = 1; run <= runs; run++) {
2067
+ console.log(`[relevance-eval] phase 1/4: full-content judging (${results.length} units, run ${run}/${runs})...`);
2068
+ const units = [];
2069
+ for (const r of results) {
2070
+ const pIdx = prompts.indexOf(r.prompt);
2071
+ const key = judgeRunKey("full", run, pIdx, r.mode);
2072
+ const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
2073
+ const promptText = buildFullContentJudgePrompt(r.prompt.prompt, r.topIds, r.contents);
2074
+ const verdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
2075
+ units.push({ key: `${pIdx}|${r.mode}`, verdicts });
2076
+ if (run === 1)
2077
+ r.fullVerdicts = verdicts;
2078
+ }
2079
+ perRunFull.push(units);
1849
2080
  }
1850
- // ---- Phase 2: line_verdict at maxLen=PROD_TITLE_CHARS (production render) ----
1851
- console.log(`[relevance-eval] phase 2/4: line-only judging at production maxLen=${PROD_TITLE_CHARS} (${results.length} units)...`);
1852
- for (const r of results) {
1853
- const key = `line|${prompts.indexOf(r.prompt)}|${r.mode}`;
1854
- const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
1855
- const lines = {};
1856
- for (const id of r.topIds)
1857
- lines[id] = (0, recall_index_js_1.formatIndexLine)(r.rows[id], PROD_TITLE_CHARS);
1858
- const promptText = buildLineJudgePrompt(r.prompt.prompt, r.topIds, lines);
1859
- r.lineVerdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
2081
+ // ---- Phase 2: line_verdict at maxLen=PROD_TITLE_CHARS (production render), N runs ----
2082
+ for (let run = 1; run <= runs; run++) {
2083
+ console.log(`[relevance-eval] phase 2/4: line-only judging at production maxLen=${PROD_TITLE_CHARS} (${results.length} units, run ${run}/${runs})...`);
2084
+ const units = [];
2085
+ for (const r of results) {
2086
+ const pIdx = prompts.indexOf(r.prompt);
2087
+ const key = judgeRunKey("line", run, pIdx, r.mode);
2088
+ const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
2089
+ const lines = {};
2090
+ for (const id of r.topIds)
2091
+ lines[id] = (0, recall_index_js_1.formatIndexLine)(r.rows[id], PROD_TITLE_CHARS);
2092
+ const promptText = buildLineJudgePrompt(r.prompt.prompt, r.topIds, lines);
2093
+ const verdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
2094
+ units.push({ key: `${pIdx}|${r.mode}`, verdicts });
2095
+ if (run === 1)
2096
+ r.lineVerdicts = verdicts;
2097
+ }
2098
+ perRunLine.push(units);
1860
2099
  }
1861
2100
  // ---- Phase 3: snippet-length sweep (40-subset, scope OFF, 4 variants) ----
1862
2101
  const offResultByPromptIdx = new Map();
@@ -1905,8 +2144,47 @@ async function main() {
1905
2144
  throw err;
1906
2145
  }
1907
2146
  }
2147
+ // ---- #458: per-run headline metrics + the judge-variance account ----
2148
+ // Abort-tolerant like everything downstream of the judge loop: a run
2149
+ // stopped by the budget / error-rate gate leaves partial per-run arrays,
2150
+ // and the report still renders (run-1 verdicts + a null variance).
2151
+ let variance = null;
2152
+ const varianceComplete = perRunFull.length === runs &&
2153
+ perRunLine.length === runs &&
2154
+ perRunFull.every((u) => u.length === results.length) &&
2155
+ perRunLine.every((u) => u.length === results.length);
2156
+ if (runs > 1 && !varianceComplete) {
2157
+ console.log(`[relevance-eval] judge runs incomplete (${perRunFull.length}/${runs} full-verdict runs filled) — §15b variance skipped`);
2158
+ }
2159
+ if (runs > 1 && varianceComplete) {
2160
+ const runResults = (run) => results.map((r, i) => ({
2161
+ ...r,
2162
+ fullVerdicts: perRunFull[run - 1][i].verdicts,
2163
+ lineVerdicts: perRunLine[run - 1][i].verdicts,
2164
+ }));
2165
+ const series = [
2166
+ { label: `precision@${PROD_MENU_K} OFF`, values: [] },
2167
+ { label: `precision@${PROD_MENU_K} ON`, values: [] },
2168
+ { label: "snippet_failure_rate OFF", values: [] },
2169
+ ];
2170
+ for (let run = 1; run <= runs; run++) {
2171
+ const rs = runResults(run);
2172
+ const offRun = rs.filter((r) => r.mode === "off");
2173
+ const onRun = rs.filter((r) => r.mode === "on");
2174
+ const mOff = computeMetricsAtK(offRun, PROD_MENU_K);
2175
+ const mOn = computeMetricsAtK(onRun, PROD_MENU_K);
2176
+ const gap = computeSnippetGap(offRun);
2177
+ series[0].values.push(mOff ? mOff.precision : null);
2178
+ series[1].values.push(mOn ? mOn.precision : null);
2179
+ series[2].values.push(gap.fullRelevantTotal > 0 ? gap.snippetFailure / gap.fullRelevantTotal : null);
2180
+ }
2181
+ variance = computeJudgeVariance(perRunFull, series);
2182
+ console.log(`[relevance-eval] judge variance over ${runs} runs (selections fixed): ` +
2183
+ variance.metrics.map((m) => `${m.label} median ${m.median !== null ? pct(m.median) : "n/a"}, spread ${m.spreadPts !== null ? m.spreadPts.toFixed(1) + "pts" : "n/a"}`).join("; ") +
2184
+ `; max pairwise flips ${variance.maxRowsFlipped}/${variance.totalComparableRows} rows, ${variance.maxUnitsFlipped} units`);
2185
+ }
1908
2186
  // ---- persist raw verdicts (source of truth, spec §6) ----
1909
- const verdictRows = buildVerdictRows(prompts, results, sweepRows);
2187
+ const verdictRows = buildVerdictRows(prompts, results, sweepRows, runs);
1910
2188
  (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(verdictsJsonPath), { recursive: true });
1911
2189
  (0, node_fs_1.writeFileSync)(verdictsJsonPath, JSON.stringify(verdictRows, null, 2), "utf-8");
1912
2190
  console.log(`[relevance-eval] raw verdicts saved: ${verdictsJsonPath} (${verdictRows.length} rows), checkpoint sidecar: ${verdictsJsonlPath}`);
@@ -1922,6 +2200,8 @@ async function main() {
1922
2200
  judgeState,
1923
2201
  generatedAt: new Date().toISOString(),
1924
2202
  db,
2203
+ clock: (0, eval_clock_js_1.clockLabel)(now),
2204
+ variance,
1925
2205
  });
1926
2206
  const reportPath = reportArg ?? DEFAULT_REPORT_PATH;
1927
2207
  (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(reportPath), { recursive: true });
@@ -1948,7 +2228,13 @@ async function main() {
1948
2228
  }
1949
2229
  }
1950
2230
  }
1951
- main().catch((err) => {
1952
- console.error("[relevance-eval] FAILED:", err instanceof Error ? err.stack : String(err));
1953
- process.exitCode = 1;
1954
- });
2231
+ // The importance-eval.ts guard: run main() only when executed directly, so
2232
+ // the exported pure helpers (parseArgs, scaleMaxCalls, median,
2233
+ // computeJudgeVariance — #458) are importable from the vitest suite without
2234
+ // firing the CLI.
2235
+ if (process.argv[1] && process.argv[1].endsWith("relevance-eval.js")) {
2236
+ main().catch((err) => {
2237
+ console.error("[relevance-eval] FAILED:", err instanceof Error ? err.stack : String(err));
2238
+ process.exitCode = 1;
2239
+ });
2240
+ }
@@ -8,10 +8,13 @@
8
8
  * never `initDb()`), never writes to the DB. Not wired into `cli.ts` — this
9
9
  * is an internal measurement tool, run via the `eval` npm script:
10
10
  *
11
- * npm run eval -- <snapshot.db> [report.md]
11
+ * npm run eval -- <snapshot.db> [report.md] [--now=<ISO>]
12
12
  *
13
- * `state.json` is expected alongside the snapshot DB (same directory) for
14
- * the `domainCursor`/`relinkCursor` watermarks. Report path defaults to
15
- * `eval-report.md` next to the DB.
13
+ * `--now=<ISO>` (#458) pins the clock the age/decay-derived stats (structural
14
+ * strength, adoption age buckets) measure against, so before/after audits are
15
+ * wall-clock-independent. Default: the live clock; the report header records
16
+ * which clock produced it. `state.json` is expected alongside the snapshot DB
17
+ * (same directory) for the `domainCursor`/`relinkCursor` watermarks. Report
18
+ * path defaults to `eval-report.md` next to the DB.
16
19
  */
17
20
  export {};
@@ -9,16 +9,20 @@
9
9
  * never `initDb()`), never writes to the DB. Not wired into `cli.ts` — this
10
10
  * is an internal measurement tool, run via the `eval` npm script:
11
11
  *
12
- * npm run eval -- <snapshot.db> [report.md]
12
+ * npm run eval -- <snapshot.db> [report.md] [--now=<ISO>]
13
13
  *
14
- * `state.json` is expected alongside the snapshot DB (same directory) for
15
- * the `domainCursor`/`relinkCursor` watermarks. Report path defaults to
16
- * `eval-report.md` next to the DB.
14
+ * `--now=<ISO>` (#458) pins the clock the age/decay-derived stats (structural
15
+ * strength, adoption age buckets) measure against, so before/after audits are
16
+ * wall-clock-independent. Default: the live clock; the report header records
17
+ * which clock produced it. `state.json` is expected alongside the snapshot DB
18
+ * (same directory) for the `domainCursor`/`relinkCursor` watermarks. Report
19
+ * path defaults to `eval-report.md` next to the DB.
17
20
  */
18
21
  Object.defineProperty(exports, "__esModule", { value: true });
19
22
  const node_path_1 = require("node:path");
20
23
  const node_fs_1 = require("node:fs");
21
24
  const eval_db_js_1 = require("./eval-db.js");
25
+ const eval_clock_js_1 = require("./eval-clock.js");
22
26
  const dups_js_1 = require("./dups.js");
23
27
  const decay_eval_js_1 = require("./decay-eval.js");
24
28
  const graph_eval_js_1 = require("./graph-eval.js");
@@ -216,7 +220,7 @@ function renderPhaseB(dups, prune, structural, backlog, graph, adoption) {
216
220
  function renderReport(input) {
217
221
  const parts = [];
218
222
  parts.push("# Hicortex Mechanical Audit Baseline (#191)\n");
219
- parts.push(`Snapshot: \`${input.dbPath}\` \nGenerated: ${input.generatedAt} \nRead-only, zero LLM, zero human grading — mechanical measurement only.\n`);
223
+ parts.push(`Snapshot: \`${input.dbPath}\` \nGenerated: ${input.generatedAt} \nClock: ${input.clock} (#458 — the instant the age/decay-derived stats measured against) \nRead-only, zero LLM, zero human grading — mechanical measurement only.\n`);
220
224
  parts.push(renderDups(input.dups));
221
225
  parts.push(renderDecay(input.domainBacklog, input.pruneDryRun, input.structural, input.adoption));
222
226
  parts.push(renderGraph(input.graph));
@@ -228,12 +232,34 @@ function renderReport(input) {
228
232
  // CLI entry
229
233
  // ---------------------------------------------------------------------------
230
234
  function main() {
231
- const [, , dbPathArg, reportPathArg] = process.argv;
235
+ // #458: --now=<ISO> pins the age/decay-derived stats' clock; positionals
236
+ // (snapshot, report path) keep working unchanged.
237
+ const positionals = [];
238
+ let nowIso;
239
+ for (const arg of process.argv.slice(2)) {
240
+ const m = arg.match(/^--now=(.+)$/);
241
+ if (m)
242
+ nowIso = m[1];
243
+ else
244
+ positionals.push(arg);
245
+ }
246
+ const [dbPathArg, reportPathArg] = positionals;
232
247
  if (!dbPathArg) {
233
- console.error("Usage: run-eval.js <snapshot.db> [report.md]");
248
+ console.error("Usage: run-eval.js <snapshot.db> [report.md] [--now=<ISO>]");
249
+ process.exitCode = 1;
250
+ return;
251
+ }
252
+ let now;
253
+ try {
254
+ now = (0, eval_clock_js_1.parsePinnedNow)(nowIso);
255
+ }
256
+ catch (err) {
257
+ console.error(`[eval] ${err instanceof Error ? err.message : err}`);
234
258
  process.exitCode = 1;
235
259
  return;
236
260
  }
261
+ console.log(`[eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}`);
262
+ const nowArg = now ?? new Date(); // the live clock, made explicit for threading
237
263
  const dbPath = dbPathArg;
238
264
  const stateDir = (0, node_path_1.dirname)(dbPath);
239
265
  const reportPath = reportPathArg ?? (0, node_path_1.join)(stateDir, "eval-report.md");
@@ -246,9 +272,9 @@ function main() {
246
272
  console.log("[eval] D4 prune dry-run...");
247
273
  const pruneDryRun = (0, decay_eval_js_1.runPruneDryRun)(db, retrieval_js_1.DEFAULT_DECAY_HALF_LIFE_DAYS);
248
274
  console.log("[eval] D4 structural strength stats...");
249
- const structural = (0, decay_eval_js_1.runStructuralStrengthStats)(db);
275
+ const structural = (0, decay_eval_js_1.runStructuralStrengthStats)(db, nowArg);
250
276
  console.log("[eval] D4 adoption stats...");
251
- const adoption = (0, decay_eval_js_1.runAdoptionStats)(db);
277
+ const adoption = (0, decay_eval_js_1.runAdoptionStats)(db, nowArg);
252
278
  console.log("[eval] D6 link-graph audit...");
253
279
  const graph = (0, graph_eval_js_1.runGraphAudit)(db, stateDir);
254
280
  console.log("[eval] reflection census...");
@@ -256,6 +282,7 @@ function main() {
256
282
  const report = renderReport({
257
283
  dbPath,
258
284
  generatedAt: new Date().toISOString(),
285
+ clock: (0, eval_clock_js_1.clockLabel)(now),
259
286
  dups,
260
287
  domainBacklog,
261
288
  pruneDryRun,