@gamaze/hicortex 0.22.0 → 0.22.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +209 -76
- package/dist/calibration.d.ts +92 -12
- package/dist/calibration.js +102 -15
- package/dist/classify-domains.js +6 -1
- package/dist/consolidate.js +87 -19
- package/dist/dashboard.d.ts +11 -0
- package/dist/dashboard.js +9 -0
- package/dist/db.js +74 -0
- package/dist/eval/decay-eval.d.ts +4 -2
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/eval-clock.d.ts +32 -0
- package/dist/eval/eval-clock.js +47 -0
- package/dist/eval/graph-eval.d.ts +15 -2
- package/dist/eval/graph-eval.js +51 -5
- package/dist/eval/planted-eval.d.ts +4 -0
- package/dist/eval/planted-eval.js +27 -2
- package/dist/eval/planted-harness.d.ts +7 -0
- package/dist/eval/planted-harness.js +2 -0
- package/dist/eval/ranking-battery.d.ts +49 -2
- package/dist/eval/ranking-battery.js +110 -2
- package/dist/eval/ranking-eval.d.ts +26 -6
- package/dist/eval/ranking-eval.js +197 -34
- package/dist/eval/ranking-fixtures.d.ts +41 -1
- package/dist/eval/ranking-fixtures.js +261 -2
- package/dist/eval/recall-sweep.d.ts +7 -2
- package/dist/eval/recall-sweep.js +42 -13
- package/dist/eval/relevance-eval.d.ts +115 -1
- package/dist/eval/relevance-eval.js +318 -32
- package/dist/eval/run-eval.d.ts +7 -4
- package/dist/eval/run-eval.js +36 -9
- package/dist/mcp-server.js +37 -9
- package/dist/nightly.js +14 -0
- package/dist/recall-index.d.ts +46 -3
- package/dist/recall-index.js +83 -26
- package/dist/recall-precision.d.ts +212 -0
- package/dist/recall-precision.js +381 -0
- package/dist/retrieval.d.ts +34 -15
- package/dist/retrieval.js +132 -59
- package/dist/types.d.ts +7 -0
- package/package.json +1 -1
- package/server.json +3 -3
|
@@ -60,17 +60,33 @@
|
|
|
60
60
|
* Run:
|
|
61
61
|
* npm run eval:relevance -- <snapshot.db> [prompts.json] [report.md] \
|
|
62
62
|
* [--judge-delay-ms=2000] [--max-calls=900] [--resume] \
|
|
63
|
-
* [--verdicts-json=path] [--verdicts-jsonl=path]
|
|
63
|
+
* [--verdicts-json=path] [--verdicts-jsonl=path] \
|
|
64
|
+
* [--now=<ISO>] [--runs=N]
|
|
65
|
+
*
|
|
66
|
+
* #458 clock pin + judge variance protocol:
|
|
67
|
+
* - `--now=<ISO>` pins the clock the retrieve sweep scores against (the
|
|
68
|
+
* retrieve() `now` seam) — before/after runs become wall-clock-independent
|
|
69
|
+
* (default: live clock, the pre-#458 behavior).
|
|
70
|
+
* - `--runs=N` (default 1) runs judge phases 1+2 N times over the SAME
|
|
71
|
+
* selections (one retrieve sweep) and reports the run-to-run noise floor
|
|
72
|
+
* (median + spread + selection-identical flip counts, §15b). Phases 3+4
|
|
73
|
+
* stay single-shot; the shared --max-calls budget spans all runs and the
|
|
74
|
+
* default scales ×N unless --max-calls was passed explicitly.
|
|
64
75
|
*/
|
|
65
76
|
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
66
77
|
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
67
78
|
};
|
|
68
79
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
80
|
+
exports.parseArgs = parseArgs;
|
|
81
|
+
exports.scaleMaxCalls = scaleMaxCalls;
|
|
82
|
+
exports.median = median;
|
|
83
|
+
exports.computeJudgeVariance = computeJudgeVariance;
|
|
69
84
|
const node_fs_1 = require("node:fs");
|
|
70
85
|
const node_path_1 = require("node:path");
|
|
71
86
|
const node_os_1 = require("node:os");
|
|
72
87
|
const better_sqlite3_1 = __importDefault(require("better-sqlite3"));
|
|
73
88
|
const eval_db_js_1 = require("./eval-db.js");
|
|
89
|
+
const eval_clock_js_1 = require("./eval-clock.js");
|
|
74
90
|
const embedder_js_1 = require("../embedder.js");
|
|
75
91
|
const retrieval_js_1 = require("../retrieval.js");
|
|
76
92
|
const recall_index_js_1 = require("../recall-index.js");
|
|
@@ -379,7 +395,7 @@ function computeSweepSubsetIndices(prompts) {
|
|
|
379
395
|
// ---------------------------------------------------------------------------
|
|
380
396
|
// Retrieve sweep (scope OFF vs ON) — LLM-free system-under-test
|
|
381
397
|
// ---------------------------------------------------------------------------
|
|
382
|
-
async function runRetrieveSweep(db, prompts) {
|
|
398
|
+
async function runRetrieveSweep(db, prompts, opts) {
|
|
383
399
|
const embedCache = new Map();
|
|
384
400
|
for (const p of prompts) {
|
|
385
401
|
if (!embedCache.has(p.prompt))
|
|
@@ -401,6 +417,9 @@ async function runRetrieveSweep(db, prompts) {
|
|
|
401
417
|
limit: RESULT_K,
|
|
402
418
|
queryEmbedding,
|
|
403
419
|
noStrengthen: true,
|
|
420
|
+
// #458: the pinned clock (undefined = live) — identical selections
|
|
421
|
+
// across before/after runs regardless of when each run executes.
|
|
422
|
+
...(opts?.now ? { now: opts.now } : {}),
|
|
404
423
|
...scopeOpts,
|
|
405
424
|
});
|
|
406
425
|
const topIds = retrieved.map((r) => r.id);
|
|
@@ -1063,7 +1082,7 @@ function similarityBucketLabel(sim) {
|
|
|
1063
1082
|
}
|
|
1064
1083
|
return ">=1.00";
|
|
1065
1084
|
}
|
|
1066
|
-
function buildVerdictRows(prompts, results, sweepRows) {
|
|
1085
|
+
function buildVerdictRows(prompts, results, sweepRows, runsTotal = 1) {
|
|
1067
1086
|
const byKey = new Map();
|
|
1068
1087
|
for (const r of results) {
|
|
1069
1088
|
const pIdx = prompts.findIndex((p) => p.prompt === r.prompt.prompt && p.agent === r.prompt.agent);
|
|
@@ -1104,6 +1123,7 @@ function buildVerdictRows(prompts, results, sweepRows) {
|
|
|
1104
1123
|
line_reason: lv ? lv.reason : lineOk ? "" : r.lineVerdicts.judgeError,
|
|
1105
1124
|
rendered_line: row ? (0, recall_index_js_1.formatIndexLine)(row, PROD_TITLE_CHARS) : "",
|
|
1106
1125
|
snippet_len_variant: String(PROD_TITLE_CHARS),
|
|
1126
|
+
...(runsTotal > 1 ? { run: 1 } : {}),
|
|
1107
1127
|
});
|
|
1108
1128
|
}
|
|
1109
1129
|
}
|
|
@@ -1148,13 +1168,18 @@ function buildVerdictRows(prompts, results, sweepRows) {
|
|
|
1148
1168
|
}
|
|
1149
1169
|
return rows;
|
|
1150
1170
|
}
|
|
1171
|
+
/** Parsed and exported for tests (#458). Throws on an invalid --runs value
|
|
1172
|
+
* (NaN or <1) — fail-explicit, never a silent default. */
|
|
1151
1173
|
function parseArgs(argv) {
|
|
1152
1174
|
const positionals = [];
|
|
1153
1175
|
let judgeDelayMs = DEFAULT_JUDGE_DELAY_MS;
|
|
1154
1176
|
let maxCalls = DEFAULT_MAX_CALLS;
|
|
1177
|
+
let maxCallsExplicit = false;
|
|
1155
1178
|
let resume = false;
|
|
1156
1179
|
let verdictsJsonPath = VERDICTS_JSON_PATH;
|
|
1157
1180
|
let verdictsJsonlPath = VERDICTS_JSONL_PATH;
|
|
1181
|
+
let nowIso;
|
|
1182
|
+
let runs = 1;
|
|
1158
1183
|
for (const arg of argv) {
|
|
1159
1184
|
if (arg === "--resume") {
|
|
1160
1185
|
resume = true;
|
|
@@ -1164,17 +1189,132 @@ function parseArgs(argv) {
|
|
|
1164
1189
|
if (m) {
|
|
1165
1190
|
if (m[1] === "judge-delay-ms")
|
|
1166
1191
|
judgeDelayMs = parseInt(m[2], 10);
|
|
1167
|
-
else if (m[1] === "max-calls")
|
|
1192
|
+
else if (m[1] === "max-calls") {
|
|
1168
1193
|
maxCalls = parseInt(m[2], 10);
|
|
1194
|
+
maxCallsExplicit = true;
|
|
1195
|
+
}
|
|
1169
1196
|
else if (m[1] === "verdicts-json")
|
|
1170
1197
|
verdictsJsonPath = m[2];
|
|
1171
1198
|
else if (m[1] === "verdicts-jsonl")
|
|
1172
1199
|
verdictsJsonlPath = m[2];
|
|
1200
|
+
else if (m[1] === "now")
|
|
1201
|
+
nowIso = m[2];
|
|
1202
|
+
else if (m[1] === "runs") {
|
|
1203
|
+
const parsed = Number(m[2]);
|
|
1204
|
+
if (!Number.isInteger(parsed) || parsed < 1) {
|
|
1205
|
+
throw new Error(`--runs must be an integer >= 1 (got "${m[2]}")`);
|
|
1206
|
+
}
|
|
1207
|
+
runs = parsed;
|
|
1208
|
+
}
|
|
1173
1209
|
continue;
|
|
1174
1210
|
}
|
|
1175
1211
|
positionals.push(arg);
|
|
1176
1212
|
}
|
|
1177
|
-
return {
|
|
1213
|
+
return {
|
|
1214
|
+
positionals,
|
|
1215
|
+
judgeDelayMs,
|
|
1216
|
+
maxCalls,
|
|
1217
|
+
maxCallsExplicit,
|
|
1218
|
+
resume,
|
|
1219
|
+
verdictsJsonPath,
|
|
1220
|
+
verdictsJsonlPath,
|
|
1221
|
+
nowIso,
|
|
1222
|
+
runs,
|
|
1223
|
+
};
|
|
1224
|
+
}
|
|
1225
|
+
/**
|
|
1226
|
+
* #458 — resolve the effective --max-calls budget. The ONE shared budget
|
|
1227
|
+
* spans ALL judge runs, so when --runs>1 and the caller did NOT pass
|
|
1228
|
+
* --max-calls explicitly, the default scales ×N (each run re-judges every
|
|
1229
|
+
* unit). An explicit budget always wins — the caller sized it for the whole
|
|
1230
|
+
* invocation. Pure, exported for tests.
|
|
1231
|
+
*/
|
|
1232
|
+
function scaleMaxCalls(runs, explicitMaxCalls, defaultMaxCalls) {
|
|
1233
|
+
if (explicitMaxCalls !== undefined)
|
|
1234
|
+
return explicitMaxCalls;
|
|
1235
|
+
return runs > 1 ? defaultMaxCalls * runs : defaultMaxCalls;
|
|
1236
|
+
}
|
|
1237
|
+
// ---------------------------------------------------------------------------
|
|
1238
|
+
// Judge variance (#458) — median, spread, and selection-identical flip
|
|
1239
|
+
// counts over N runs of the SAME selections. Pure + synthetic-verdict
|
|
1240
|
+
// testable; the orchestration (running the judge N times) lives in main.
|
|
1241
|
+
// ---------------------------------------------------------------------------
|
|
1242
|
+
/** Median of a numeric sample (even count → mean of the two middle values).
|
|
1243
|
+
* Empty input → NaN, the mean() convention above. Does not mutate input. */
|
|
1244
|
+
function median(xs) {
|
|
1245
|
+
if (xs.length === 0)
|
|
1246
|
+
return NaN;
|
|
1247
|
+
const sorted = [...xs].sort((a, b) => a - b);
|
|
1248
|
+
const mid = Math.floor(sorted.length / 2);
|
|
1249
|
+
return sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
|
|
1250
|
+
}
|
|
1251
|
+
/**
|
|
1252
|
+
* Compute the #458 judge-variance account over N runs of the same selections:
|
|
1253
|
+
* per-metric median + spread (the noise floor), and pairwise verdict flip
|
|
1254
|
+
* counts (rows = one Q-label judgment on one (prompt,mode) unit; units flip
|
|
1255
|
+
* if any row does). `perRunVerdicts[r]` is run r+1's units in a stable
|
|
1256
|
+
* order/identity; `metricSeries[s].values[r]` is run r+1's headline value.
|
|
1257
|
+
*/
|
|
1258
|
+
function computeJudgeVariance(perRunVerdicts, metricSeries) {
|
|
1259
|
+
const runs = perRunVerdicts.length;
|
|
1260
|
+
const flips = [];
|
|
1261
|
+
let maxRows = 0;
|
|
1262
|
+
let maxUnits = 0;
|
|
1263
|
+
for (let i = 0; i < runs; i++) {
|
|
1264
|
+
for (let j = i + 1; j < runs; j++) {
|
|
1265
|
+
const byKeyI = new Map(perRunVerdicts[i].map((u) => [u.key, u.verdicts]));
|
|
1266
|
+
let rowsCompared = 0;
|
|
1267
|
+
let rowsFlipped = 0;
|
|
1268
|
+
let unitsCompared = 0;
|
|
1269
|
+
let unitsFlipped = 0;
|
|
1270
|
+
for (const unitJ of perRunVerdicts[j]) {
|
|
1271
|
+
const vi = byKeyI.get(unitJ.key);
|
|
1272
|
+
if (!vi || !Array.isArray(vi) || !Array.isArray(unitJ.verdicts))
|
|
1273
|
+
continue; // judge_error/absent — not comparable
|
|
1274
|
+
const vj = unitJ.verdicts;
|
|
1275
|
+
unitsCompared++;
|
|
1276
|
+
let unitFlipped = false;
|
|
1277
|
+
const byLabelJ = new Map(vj.map((v) => [v.id, v.verdict]));
|
|
1278
|
+
for (const rowI of vi) {
|
|
1279
|
+
const verdictJ = byLabelJ.get(rowI.id);
|
|
1280
|
+
if (verdictJ === undefined)
|
|
1281
|
+
continue; // label omitted in run j — not comparable
|
|
1282
|
+
rowsCompared++;
|
|
1283
|
+
if (verdictJ !== rowI.verdict) {
|
|
1284
|
+
rowsFlipped++;
|
|
1285
|
+
unitFlipped = true;
|
|
1286
|
+
}
|
|
1287
|
+
}
|
|
1288
|
+
if (unitFlipped)
|
|
1289
|
+
unitsFlipped++;
|
|
1290
|
+
}
|
|
1291
|
+
flips.push({ runA: i + 1, runB: j + 1, rowsCompared, rowsFlipped, unitsCompared, unitsFlipped });
|
|
1292
|
+
maxRows = Math.max(maxRows, rowsFlipped);
|
|
1293
|
+
maxUnits = Math.max(maxUnits, unitsFlipped);
|
|
1294
|
+
}
|
|
1295
|
+
}
|
|
1296
|
+
const metrics = metricSeries.map((s) => {
|
|
1297
|
+
const usable = s.values.filter((v) => v !== null && Number.isFinite(v));
|
|
1298
|
+
if (usable.length === 0)
|
|
1299
|
+
return { ...s, median: null, spreadPts: null };
|
|
1300
|
+
return {
|
|
1301
|
+
...s,
|
|
1302
|
+
median: median(usable),
|
|
1303
|
+
spreadPts: (Math.max(...usable) - Math.min(...usable)) * 100,
|
|
1304
|
+
};
|
|
1305
|
+
});
|
|
1306
|
+
const judgeErrorUnitsPerRun = perRunVerdicts.map((units) => units.filter((u) => !Array.isArray(u.verdicts)).length);
|
|
1307
|
+
return {
|
|
1308
|
+
runs,
|
|
1309
|
+
metrics,
|
|
1310
|
+
flips,
|
|
1311
|
+
maxRowsFlipped: maxRows,
|
|
1312
|
+
maxUnitsFlipped: maxUnits,
|
|
1313
|
+
meanRowsFlipped: flips.length > 0 ? mean(flips.map((f) => f.rowsFlipped)) : 0,
|
|
1314
|
+
meanUnitsFlipped: flips.length > 0 ? mean(flips.map((f) => f.unitsFlipped)) : 0,
|
|
1315
|
+
totalComparableRows: flips.length > 0 ? flips[0].rowsCompared : 0,
|
|
1316
|
+
judgeErrorUnitsPerRun,
|
|
1317
|
+
};
|
|
1178
1318
|
}
|
|
1179
1319
|
// ---------------------------------------------------------------------------
|
|
1180
1320
|
// Reporting
|
|
@@ -1207,7 +1347,8 @@ function renderReport(args) {
|
|
|
1207
1347
|
L.push(`Snapshot: \`${args.snapshotPath}\` \n` +
|
|
1208
1348
|
`Prompts: \`${args.promptsPath}\` (${prompts.length} prompts; requested ${SOURCE_TAGS.length * N_PER_SOURCE} — ` +
|
|
1209
1349
|
`see §0 for any per-source shortfall) \n` +
|
|
1210
|
-
`Generated: ${args.generatedAt}\n`
|
|
1350
|
+
`Generated: ${args.generatedAt} \n` +
|
|
1351
|
+
`Clock: ${args.clock} (the instant the retrieve sweep scored against, #458)${args.variance ? ` \nJudge runs: ${args.variance.runs} (phases 1+2 re-judged over fixed selections — see §15b)` : ""}\n`);
|
|
1211
1352
|
L.push(`_Spec: \`specs/2026-08-02-relevance-eval.md\`. Real agent prompts replayed through \`retrieve()\` on a ` +
|
|
1212
1353
|
`READONLY snapshot; GLM-5.2 (z.ai) is the JUDGE only — retrieve() is LLM-free. Two independent, BLIND judge ` +
|
|
1213
1354
|
`calls per (prompt × mode): \`full_verdict\` (up to 2000 chars) and \`line_verdict\` (ONLY the rendered ` +
|
|
@@ -1608,6 +1749,38 @@ function renderReport(args) {
|
|
|
1608
1749
|
L.push(`**Scope ON−OFF delta (precision@${PROD_MENU_K}):** ${deltaStr} — **${distinguishable}**\n`);
|
|
1609
1750
|
}
|
|
1610
1751
|
// =====================================================================
|
|
1752
|
+
// §15b — judge variance (#458): the measured run-to-run noise floor.
|
|
1753
|
+
// Rendered ONLY when --runs>1.
|
|
1754
|
+
// =====================================================================
|
|
1755
|
+
if (args.variance) {
|
|
1756
|
+
const v = args.variance;
|
|
1757
|
+
L.push(`## 15b. Judge variance — run-to-run noise floor (#458, ${v.runs} runs)\n`);
|
|
1758
|
+
L.push(`_Judge phases 1+2 ran ${v.runs} times over the SAME selections (one retrieve sweep feeds every run — ` +
|
|
1759
|
+
`the sweep precedes judging, so selections are fixed under a live clock too). ALL drift measured here is ` +
|
|
1760
|
+
`judge-only instrument noise, the class PR C evidenced at 38/380 flipped verdicts. Read every body section ` +
|
|
1761
|
+
`above as run 1; this section carries the median and the band._\n`);
|
|
1762
|
+
L.push("| metric | " + v.metrics.map((_, i) => `run ${i + 1}`).join(" | ") + " | median | spread (max−min) |");
|
|
1763
|
+
L.push("|---|" + v.metrics.map(() => "---").join("|") + "|---|---|");
|
|
1764
|
+
for (const m of v.metrics) {
|
|
1765
|
+
const cells = m.values.map((x) => (x === null || !Number.isFinite(x) ? "n/a" : pct(x))).join(" | ");
|
|
1766
|
+
L.push(`| ${m.label} | ${cells} | **${m.median !== null && Number.isFinite(m.median) ? pct(m.median) : "n/a"}** | ${m.spreadPts !== null && Number.isFinite(m.spreadPts) ? `${m.spreadPts.toFixed(1)}pts` : "n/a"} |`);
|
|
1767
|
+
}
|
|
1768
|
+
L.push("");
|
|
1769
|
+
L.push(`_Pairwise verdict flips on selection-identical units (full_verdict rows; judge_error units excluded from ` +
|
|
1770
|
+
`the denominators — error units per run: ${v.judgeErrorUnitsPerRun.map((e, i) => `r${i + 1} ${e}`).join(", ")}). ` +
|
|
1771
|
+
`A verdict ROW is one Q-label judgment on one (prompt, mode) unit; a UNIT flips if any of its rows differ._\n`);
|
|
1772
|
+
L.push("| pair | rows flipped / compared | units flipped / compared |");
|
|
1773
|
+
L.push("|---|---|---|");
|
|
1774
|
+
for (const f of v.flips) {
|
|
1775
|
+
L.push(`| r${f.runA}↔r${f.runB} | ${f.rowsFlipped} / ${f.rowsCompared} | ${f.unitsFlipped} / ${f.unitsCompared} |`);
|
|
1776
|
+
}
|
|
1777
|
+
L.push("");
|
|
1778
|
+
L.push(`**Noise floor:** max pairwise flips ${v.maxRowsFlipped} rows / ${v.maxUnitsFlipped} units; mean ` +
|
|
1779
|
+
`${v.meanRowsFlipped.toFixed(1)} rows / ${v.meanUnitsFlipped.toFixed(1)} units over ${v.flips.length} pair(s); ` +
|
|
1780
|
+
`total comparable rows ${v.totalComparableRows}. A before/after headline delta smaller than the spread row ` +
|
|
1781
|
+
`above is NOT resolvable by this instrument — report it as within judge noise._\n`);
|
|
1782
|
+
}
|
|
1783
|
+
// =====================================================================
|
|
1611
1784
|
// §16 — DECISION THRESHOLDS (spec §9) — filled with measured values
|
|
1612
1785
|
// =====================================================================
|
|
1613
1786
|
L.push("## 16. Decision thresholds (spec §9) — measured values + triggered actions\n");
|
|
@@ -1743,7 +1916,8 @@ function renderReport(args) {
|
|
|
1743
1916
|
function usage() {
|
|
1744
1917
|
console.error("Usage: relevance-eval.js <snapshot.db> [prompts.json] [report.md] " +
|
|
1745
1918
|
"[--judge-delay-ms=2000] [--max-calls=900] [--resume] " +
|
|
1746
|
-
"[--verdicts-json=path] [--verdicts-jsonl=path]
|
|
1919
|
+
"[--verdicts-json=path] [--verdicts-jsonl=path] " +
|
|
1920
|
+
"[--now=<ISO>] [--runs=N]\n" +
|
|
1747
1921
|
" <snapshot.db> — REQUIRED. Readonly bedrock snapshot (opened via openSnapshot).\n" +
|
|
1748
1922
|
" [prompts.json] — omitted: build the v2 corpus fresh (20/source × 5 sources, spec §2),\n" +
|
|
1749
1923
|
" saved to data/prompts.json. A valid existing v2 set is reused verbatim.\n" +
|
|
@@ -1753,6 +1927,14 @@ function usage() {
|
|
|
1753
1927
|
" JSONL paths (default data/relevance-verdicts.json[l]). ALWAYS override\n" +
|
|
1754
1928
|
" these for a re-run against a new snapshot — otherwise a re-run silently\n" +
|
|
1755
1929
|
" clobbers a prior baseline's raw verdicts.\n" +
|
|
1930
|
+
" --now=<ISO> — #458: pin the clock the retrieve sweep scores against (full ISO 8601\n" +
|
|
1931
|
+
" instant). Before/after runs become wall-clock-independent. Default: the\n" +
|
|
1932
|
+
" live clock. Invalid input is an error, never a silent live fallback.\n" +
|
|
1933
|
+
" --runs=N — #458: run judge phases 1+2 N times over the SAME selections (one\n" +
|
|
1934
|
+
" retrieve sweep) and report the run-to-run noise floor (median + spread\n" +
|
|
1935
|
+
" + selection-identical flip counts, §15b). Default 1 (single-run\n" +
|
|
1936
|
+
" behavior). The shared --max-calls budget spans ALL runs; without an\n" +
|
|
1937
|
+
" explicit --max-calls the default scales ×N. Phases 3+4 stay single-shot.\n" +
|
|
1756
1938
|
"Env: ZAI_API_KEY (required — GLM-5.2 via z.ai is the judge)\n" +
|
|
1757
1939
|
"Preconditions: /tmp/hermes-{lenny,raider,nano}-state.db (readonly Hermes state DBs) when\n" +
|
|
1758
1940
|
"building the corpus fresh; ~/.claude/projects/.../infrastructure + .../aironic-marine dirs.");
|
|
@@ -1765,10 +1947,39 @@ async function main() {
|
|
|
1765
1947
|
(0, retrieval_js_1.configureDecay)();
|
|
1766
1948
|
(0, retrieval_js_1.configureRecall)();
|
|
1767
1949
|
(0, retrieval_js_1.configureSessionIntent)();
|
|
1768
|
-
|
|
1950
|
+
let parsed;
|
|
1951
|
+
try {
|
|
1952
|
+
parsed = parseArgs(process.argv.slice(2));
|
|
1953
|
+
}
|
|
1954
|
+
catch (err) {
|
|
1955
|
+
console.error(`[relevance-eval] ${err instanceof Error ? err.message : err}`);
|
|
1956
|
+
usage();
|
|
1957
|
+
}
|
|
1958
|
+
const { positionals, judgeDelayMs, maxCalls: maxCallsArg, maxCallsExplicit, resume, verdictsJsonPath, verdictsJsonlPath, nowIso, runs, } = parsed;
|
|
1769
1959
|
const [snapshotArg, promptsSourceArg, reportArg] = positionals;
|
|
1770
1960
|
if (!snapshotArg)
|
|
1771
1961
|
usage();
|
|
1962
|
+
// #458: pin the eval clock (invalid → usage-style error, never a silent
|
|
1963
|
+
// live fallback — a run that believed it was pinned would silently drift).
|
|
1964
|
+
let now;
|
|
1965
|
+
try {
|
|
1966
|
+
now = (0, eval_clock_js_1.parsePinnedNow)(nowIso);
|
|
1967
|
+
}
|
|
1968
|
+
catch (err) {
|
|
1969
|
+
console.error(`[relevance-eval] ${err instanceof Error ? err.message : err}`);
|
|
1970
|
+
usage();
|
|
1971
|
+
}
|
|
1972
|
+
console.log(`[relevance-eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}${now === null ? " (pass --now=<ISO> to pin before/after runs)" : ""}`);
|
|
1973
|
+
// #458: the ONE shared call budget spans ALL judge runs. When --runs>1 and
|
|
1974
|
+
// --max-calls was not passed explicitly, scale the default ×N — loudly.
|
|
1975
|
+
const maxCalls = scaleMaxCalls(runs, maxCallsExplicit ? maxCallsArg : undefined, DEFAULT_MAX_CALLS);
|
|
1976
|
+
if (runs > 1 && !maxCallsExplicit) {
|
|
1977
|
+
console.log(`[relevance-eval] --runs=${runs} without an explicit --max-calls: scaling the default budget ` +
|
|
1978
|
+
`${DEFAULT_MAX_CALLS} × ${runs} = ${maxCalls} (the budget spans ALL runs)`);
|
|
1979
|
+
}
|
|
1980
|
+
if (runs > 1) {
|
|
1981
|
+
console.log(`[relevance-eval] judge variance protocol: phases 1+2 run ${runs}× over the fixed selections (see §15b)`);
|
|
1982
|
+
}
|
|
1772
1983
|
const apiKey = process.env.ZAI_API_KEY;
|
|
1773
1984
|
if (!apiKey) {
|
|
1774
1985
|
console.error("relevance-eval: ZAI_API_KEY env var is required (GLM-5.2 via z.ai is the judge)");
|
|
@@ -1809,8 +2020,8 @@ async function main() {
|
|
|
1809
2020
|
console.log(`[relevance-eval] snapshot opened readonly: ${snapshotArg} (${prompts.length} prompts × 2 modes = ${prompts.length * 2} retrieve() calls)`);
|
|
1810
2021
|
console.log("[relevance-eval] loading bge-small-en-v1.5 + replaying retrieve()...");
|
|
1811
2022
|
const t0 = Date.now();
|
|
1812
|
-
const results = await runRetrieveSweep(db, prompts);
|
|
1813
|
-
console.log(`[relevance-eval] retrieve sweep done in ${Date.now() - t0}ms (${results.length} prompt×mode batches, LLM-free)`);
|
|
2023
|
+
const results = await runRetrieveSweep(db, prompts, { now: now ?? undefined });
|
|
2024
|
+
console.log(`[relevance-eval] retrieve sweep done in ${Date.now() - t0}ms (${results.length} prompt×mode batches, LLM-free, clock ${(0, eval_clock_js_1.clockLabel)(now)})`);
|
|
1814
2025
|
// ---- judge run state (rate limiting + resumability, spec §6b) ----
|
|
1815
2026
|
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(verdictsJsonlPath), { recursive: true });
|
|
1816
2027
|
if (!resume)
|
|
@@ -1838,25 +2049,53 @@ async function main() {
|
|
|
1838
2049
|
console.log(`[relevance-eval] --resume: ${resumedMap.size} units already checkpointed in ${verdictsJsonlPath}`);
|
|
1839
2050
|
}
|
|
1840
2051
|
const sweepRows = [];
|
|
2052
|
+
// #458: judge phases 1+2 run `runs` times over the SAME selections (the
|
|
2053
|
+
// retrieve sweep above ran ONCE, before judging — so selections are fixed
|
|
2054
|
+
// across runs under a live clock too). Per-run verdict arrays feed the
|
|
2055
|
+
// §15b variance section; run 1's verdicts land in `results` so every body
|
|
2056
|
+
// section renders exactly as a single-run report. Checkpoint keys carry a
|
|
2057
|
+
// 1-based run prefix when runs>1 (`full|r1|…`) so sidecar entries never
|
|
2058
|
+
// collide across runs; the legacy single-run keys are byte-unchanged when
|
|
2059
|
+
// runs==1 (old single-run sidecars still resume — an N-run invocation
|
|
2060
|
+
// starts a fresh sidecar by default).
|
|
2061
|
+
const perRunFull = [];
|
|
2062
|
+
const perRunLine = [];
|
|
2063
|
+
const judgeRunKey = (phase, run, ...parts) => runs > 1 ? [phase, `r${run}`, ...parts].join("|") : [phase, ...parts].join("|");
|
|
1841
2064
|
try {
|
|
1842
|
-
// ---- Phase 1: full_verdict (all prompts × both modes) ----
|
|
1843
|
-
|
|
1844
|
-
|
|
1845
|
-
const
|
|
1846
|
-
const
|
|
1847
|
-
|
|
1848
|
-
|
|
2065
|
+
// ---- Phase 1: full_verdict (all prompts × both modes), N runs ----
|
|
2066
|
+
for (let run = 1; run <= runs; run++) {
|
|
2067
|
+
console.log(`[relevance-eval] phase 1/4: full-content judging (${results.length} units, run ${run}/${runs})...`);
|
|
2068
|
+
const units = [];
|
|
2069
|
+
for (const r of results) {
|
|
2070
|
+
const pIdx = prompts.indexOf(r.prompt);
|
|
2071
|
+
const key = judgeRunKey("full", run, pIdx, r.mode);
|
|
2072
|
+
const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
|
|
2073
|
+
const promptText = buildFullContentJudgePrompt(r.prompt.prompt, r.topIds, r.contents);
|
|
2074
|
+
const verdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
|
|
2075
|
+
units.push({ key: `${pIdx}|${r.mode}`, verdicts });
|
|
2076
|
+
if (run === 1)
|
|
2077
|
+
r.fullVerdicts = verdicts;
|
|
2078
|
+
}
|
|
2079
|
+
perRunFull.push(units);
|
|
1849
2080
|
}
|
|
1850
|
-
// ---- Phase 2: line_verdict at maxLen=PROD_TITLE_CHARS (production render) ----
|
|
1851
|
-
|
|
1852
|
-
|
|
1853
|
-
const
|
|
1854
|
-
const
|
|
1855
|
-
|
|
1856
|
-
|
|
1857
|
-
|
|
1858
|
-
|
|
1859
|
-
|
|
2081
|
+
// ---- Phase 2: line_verdict at maxLen=PROD_TITLE_CHARS (production render), N runs ----
|
|
2082
|
+
for (let run = 1; run <= runs; run++) {
|
|
2083
|
+
console.log(`[relevance-eval] phase 2/4: line-only judging at production maxLen=${PROD_TITLE_CHARS} (${results.length} units, run ${run}/${runs})...`);
|
|
2084
|
+
const units = [];
|
|
2085
|
+
for (const r of results) {
|
|
2086
|
+
const pIdx = prompts.indexOf(r.prompt);
|
|
2087
|
+
const key = judgeRunKey("line", run, pIdx, r.mode);
|
|
2088
|
+
const expectedLabels = new Set(r.topIds.map((_, i) => `Q${i + 1}`));
|
|
2089
|
+
const lines = {};
|
|
2090
|
+
for (const id of r.topIds)
|
|
2091
|
+
lines[id] = (0, recall_index_js_1.formatIndexLine)(r.rows[id], PROD_TITLE_CHARS);
|
|
2092
|
+
const promptText = buildLineJudgePrompt(r.prompt.prompt, r.topIds, lines);
|
|
2093
|
+
const verdicts = await judgeVerdictUnit(judgeState, key, promptText, expectedLabels);
|
|
2094
|
+
units.push({ key: `${pIdx}|${r.mode}`, verdicts });
|
|
2095
|
+
if (run === 1)
|
|
2096
|
+
r.lineVerdicts = verdicts;
|
|
2097
|
+
}
|
|
2098
|
+
perRunLine.push(units);
|
|
1860
2099
|
}
|
|
1861
2100
|
// ---- Phase 3: snippet-length sweep (40-subset, scope OFF, 4 variants) ----
|
|
1862
2101
|
const offResultByPromptIdx = new Map();
|
|
@@ -1905,8 +2144,47 @@ async function main() {
|
|
|
1905
2144
|
throw err;
|
|
1906
2145
|
}
|
|
1907
2146
|
}
|
|
2147
|
+
// ---- #458: per-run headline metrics + the judge-variance account ----
|
|
2148
|
+
// Abort-tolerant like everything downstream of the judge loop: a run
|
|
2149
|
+
// stopped by the budget / error-rate gate leaves partial per-run arrays,
|
|
2150
|
+
// and the report still renders (run-1 verdicts + a null variance).
|
|
2151
|
+
let variance = null;
|
|
2152
|
+
const varianceComplete = perRunFull.length === runs &&
|
|
2153
|
+
perRunLine.length === runs &&
|
|
2154
|
+
perRunFull.every((u) => u.length === results.length) &&
|
|
2155
|
+
perRunLine.every((u) => u.length === results.length);
|
|
2156
|
+
if (runs > 1 && !varianceComplete) {
|
|
2157
|
+
console.log(`[relevance-eval] judge runs incomplete (${perRunFull.length}/${runs} full-verdict runs filled) — §15b variance skipped`);
|
|
2158
|
+
}
|
|
2159
|
+
if (runs > 1 && varianceComplete) {
|
|
2160
|
+
const runResults = (run) => results.map((r, i) => ({
|
|
2161
|
+
...r,
|
|
2162
|
+
fullVerdicts: perRunFull[run - 1][i].verdicts,
|
|
2163
|
+
lineVerdicts: perRunLine[run - 1][i].verdicts,
|
|
2164
|
+
}));
|
|
2165
|
+
const series = [
|
|
2166
|
+
{ label: `precision@${PROD_MENU_K} OFF`, values: [] },
|
|
2167
|
+
{ label: `precision@${PROD_MENU_K} ON`, values: [] },
|
|
2168
|
+
{ label: "snippet_failure_rate OFF", values: [] },
|
|
2169
|
+
];
|
|
2170
|
+
for (let run = 1; run <= runs; run++) {
|
|
2171
|
+
const rs = runResults(run);
|
|
2172
|
+
const offRun = rs.filter((r) => r.mode === "off");
|
|
2173
|
+
const onRun = rs.filter((r) => r.mode === "on");
|
|
2174
|
+
const mOff = computeMetricsAtK(offRun, PROD_MENU_K);
|
|
2175
|
+
const mOn = computeMetricsAtK(onRun, PROD_MENU_K);
|
|
2176
|
+
const gap = computeSnippetGap(offRun);
|
|
2177
|
+
series[0].values.push(mOff ? mOff.precision : null);
|
|
2178
|
+
series[1].values.push(mOn ? mOn.precision : null);
|
|
2179
|
+
series[2].values.push(gap.fullRelevantTotal > 0 ? gap.snippetFailure / gap.fullRelevantTotal : null);
|
|
2180
|
+
}
|
|
2181
|
+
variance = computeJudgeVariance(perRunFull, series);
|
|
2182
|
+
console.log(`[relevance-eval] judge variance over ${runs} runs (selections fixed): ` +
|
|
2183
|
+
variance.metrics.map((m) => `${m.label} median ${m.median !== null ? pct(m.median) : "n/a"}, spread ${m.spreadPts !== null ? m.spreadPts.toFixed(1) + "pts" : "n/a"}`).join("; ") +
|
|
2184
|
+
`; max pairwise flips ${variance.maxRowsFlipped}/${variance.totalComparableRows} rows, ${variance.maxUnitsFlipped} units`);
|
|
2185
|
+
}
|
|
1908
2186
|
// ---- persist raw verdicts (source of truth, spec §6) ----
|
|
1909
|
-
const verdictRows = buildVerdictRows(prompts, results, sweepRows);
|
|
2187
|
+
const verdictRows = buildVerdictRows(prompts, results, sweepRows, runs);
|
|
1910
2188
|
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(verdictsJsonPath), { recursive: true });
|
|
1911
2189
|
(0, node_fs_1.writeFileSync)(verdictsJsonPath, JSON.stringify(verdictRows, null, 2), "utf-8");
|
|
1912
2190
|
console.log(`[relevance-eval] raw verdicts saved: ${verdictsJsonPath} (${verdictRows.length} rows), checkpoint sidecar: ${verdictsJsonlPath}`);
|
|
@@ -1922,6 +2200,8 @@ async function main() {
|
|
|
1922
2200
|
judgeState,
|
|
1923
2201
|
generatedAt: new Date().toISOString(),
|
|
1924
2202
|
db,
|
|
2203
|
+
clock: (0, eval_clock_js_1.clockLabel)(now),
|
|
2204
|
+
variance,
|
|
1925
2205
|
});
|
|
1926
2206
|
const reportPath = reportArg ?? DEFAULT_REPORT_PATH;
|
|
1927
2207
|
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(reportPath), { recursive: true });
|
|
@@ -1948,7 +2228,13 @@ async function main() {
|
|
|
1948
2228
|
}
|
|
1949
2229
|
}
|
|
1950
2230
|
}
|
|
1951
|
-
main()
|
|
1952
|
-
|
|
1953
|
-
|
|
1954
|
-
|
|
2231
|
+
// The importance-eval.ts guard: run main() only when executed directly, so
|
|
2232
|
+
// the exported pure helpers (parseArgs, scaleMaxCalls, median,
|
|
2233
|
+
// computeJudgeVariance — #458) are importable from the vitest suite without
|
|
2234
|
+
// firing the CLI.
|
|
2235
|
+
if (process.argv[1] && process.argv[1].endsWith("relevance-eval.js")) {
|
|
2236
|
+
main().catch((err) => {
|
|
2237
|
+
console.error("[relevance-eval] FAILED:", err instanceof Error ? err.stack : String(err));
|
|
2238
|
+
process.exitCode = 1;
|
|
2239
|
+
});
|
|
2240
|
+
}
|
package/dist/eval/run-eval.d.ts
CHANGED
|
@@ -8,10 +8,13 @@
|
|
|
8
8
|
* never `initDb()`), never writes to the DB. Not wired into `cli.ts` — this
|
|
9
9
|
* is an internal measurement tool, run via the `eval` npm script:
|
|
10
10
|
*
|
|
11
|
-
* npm run eval -- <snapshot.db> [report.md]
|
|
11
|
+
* npm run eval -- <snapshot.db> [report.md] [--now=<ISO>]
|
|
12
12
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
13
|
+
* `--now=<ISO>` (#458) pins the clock the age/decay-derived stats (structural
|
|
14
|
+
* strength, adoption age buckets) measure against, so before/after audits are
|
|
15
|
+
* wall-clock-independent. Default: the live clock; the report header records
|
|
16
|
+
* which clock produced it. `state.json` is expected alongside the snapshot DB
|
|
17
|
+
* (same directory) for the `domainCursor`/`relinkCursor` watermarks. Report
|
|
18
|
+
* path defaults to `eval-report.md` next to the DB.
|
|
16
19
|
*/
|
|
17
20
|
export {};
|
package/dist/eval/run-eval.js
CHANGED
|
@@ -9,16 +9,20 @@
|
|
|
9
9
|
* never `initDb()`), never writes to the DB. Not wired into `cli.ts` — this
|
|
10
10
|
* is an internal measurement tool, run via the `eval` npm script:
|
|
11
11
|
*
|
|
12
|
-
* npm run eval -- <snapshot.db> [report.md]
|
|
12
|
+
* npm run eval -- <snapshot.db> [report.md] [--now=<ISO>]
|
|
13
13
|
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
14
|
+
* `--now=<ISO>` (#458) pins the clock the age/decay-derived stats (structural
|
|
15
|
+
* strength, adoption age buckets) measure against, so before/after audits are
|
|
16
|
+
* wall-clock-independent. Default: the live clock; the report header records
|
|
17
|
+
* which clock produced it. `state.json` is expected alongside the snapshot DB
|
|
18
|
+
* (same directory) for the `domainCursor`/`relinkCursor` watermarks. Report
|
|
19
|
+
* path defaults to `eval-report.md` next to the DB.
|
|
17
20
|
*/
|
|
18
21
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
19
22
|
const node_path_1 = require("node:path");
|
|
20
23
|
const node_fs_1 = require("node:fs");
|
|
21
24
|
const eval_db_js_1 = require("./eval-db.js");
|
|
25
|
+
const eval_clock_js_1 = require("./eval-clock.js");
|
|
22
26
|
const dups_js_1 = require("./dups.js");
|
|
23
27
|
const decay_eval_js_1 = require("./decay-eval.js");
|
|
24
28
|
const graph_eval_js_1 = require("./graph-eval.js");
|
|
@@ -216,7 +220,7 @@ function renderPhaseB(dups, prune, structural, backlog, graph, adoption) {
|
|
|
216
220
|
function renderReport(input) {
|
|
217
221
|
const parts = [];
|
|
218
222
|
parts.push("# Hicortex Mechanical Audit Baseline (#191)\n");
|
|
219
|
-
parts.push(`Snapshot: \`${input.dbPath}\` \nGenerated: ${input.generatedAt} \nRead-only, zero LLM, zero human grading — mechanical measurement only.\n`);
|
|
223
|
+
parts.push(`Snapshot: \`${input.dbPath}\` \nGenerated: ${input.generatedAt} \nClock: ${input.clock} (#458 — the instant the age/decay-derived stats measured against) \nRead-only, zero LLM, zero human grading — mechanical measurement only.\n`);
|
|
220
224
|
parts.push(renderDups(input.dups));
|
|
221
225
|
parts.push(renderDecay(input.domainBacklog, input.pruneDryRun, input.structural, input.adoption));
|
|
222
226
|
parts.push(renderGraph(input.graph));
|
|
@@ -228,12 +232,34 @@ function renderReport(input) {
|
|
|
228
232
|
// CLI entry
|
|
229
233
|
// ---------------------------------------------------------------------------
|
|
230
234
|
function main() {
|
|
231
|
-
|
|
235
|
+
// #458: --now=<ISO> pins the age/decay-derived stats' clock; positionals
|
|
236
|
+
// (snapshot, report path) keep working unchanged.
|
|
237
|
+
const positionals = [];
|
|
238
|
+
let nowIso;
|
|
239
|
+
for (const arg of process.argv.slice(2)) {
|
|
240
|
+
const m = arg.match(/^--now=(.+)$/);
|
|
241
|
+
if (m)
|
|
242
|
+
nowIso = m[1];
|
|
243
|
+
else
|
|
244
|
+
positionals.push(arg);
|
|
245
|
+
}
|
|
246
|
+
const [dbPathArg, reportPathArg] = positionals;
|
|
232
247
|
if (!dbPathArg) {
|
|
233
|
-
console.error("Usage: run-eval.js <snapshot.db> [report.md]");
|
|
248
|
+
console.error("Usage: run-eval.js <snapshot.db> [report.md] [--now=<ISO>]");
|
|
249
|
+
process.exitCode = 1;
|
|
250
|
+
return;
|
|
251
|
+
}
|
|
252
|
+
let now;
|
|
253
|
+
try {
|
|
254
|
+
now = (0, eval_clock_js_1.parsePinnedNow)(nowIso);
|
|
255
|
+
}
|
|
256
|
+
catch (err) {
|
|
257
|
+
console.error(`[eval] ${err instanceof Error ? err.message : err}`);
|
|
234
258
|
process.exitCode = 1;
|
|
235
259
|
return;
|
|
236
260
|
}
|
|
261
|
+
console.log(`[eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}`);
|
|
262
|
+
const nowArg = now ?? new Date(); // the live clock, made explicit for threading
|
|
237
263
|
const dbPath = dbPathArg;
|
|
238
264
|
const stateDir = (0, node_path_1.dirname)(dbPath);
|
|
239
265
|
const reportPath = reportPathArg ?? (0, node_path_1.join)(stateDir, "eval-report.md");
|
|
@@ -246,9 +272,9 @@ function main() {
|
|
|
246
272
|
console.log("[eval] D4 prune dry-run...");
|
|
247
273
|
const pruneDryRun = (0, decay_eval_js_1.runPruneDryRun)(db, retrieval_js_1.DEFAULT_DECAY_HALF_LIFE_DAYS);
|
|
248
274
|
console.log("[eval] D4 structural strength stats...");
|
|
249
|
-
const structural = (0, decay_eval_js_1.runStructuralStrengthStats)(db);
|
|
275
|
+
const structural = (0, decay_eval_js_1.runStructuralStrengthStats)(db, nowArg);
|
|
250
276
|
console.log("[eval] D4 adoption stats...");
|
|
251
|
-
const adoption = (0, decay_eval_js_1.runAdoptionStats)(db);
|
|
277
|
+
const adoption = (0, decay_eval_js_1.runAdoptionStats)(db, nowArg);
|
|
252
278
|
console.log("[eval] D6 link-graph audit...");
|
|
253
279
|
const graph = (0, graph_eval_js_1.runGraphAudit)(db, stateDir);
|
|
254
280
|
console.log("[eval] reflection census...");
|
|
@@ -256,6 +282,7 @@ function main() {
|
|
|
256
282
|
const report = renderReport({
|
|
257
283
|
dbPath,
|
|
258
284
|
generatedAt: new Date().toISOString(),
|
|
285
|
+
clock: (0, eval_clock_js_1.clockLabel)(now),
|
|
259
286
|
dups,
|
|
260
287
|
domainBacklog,
|
|
261
288
|
pruneDryRun,
|