@gamaze/hicortex 0.23.1 → 0.23.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/assets/dashboard.html +25 -13
  2. package/dist/dashboard.d.ts +18 -7
  3. package/dist/dashboard.js +42 -10
  4. package/dist/dedup.js +2 -2
  5. package/dist/index.js +17 -2
  6. package/dist/learnings-identity.js +20 -1
  7. package/dist/mcp-server.js +6 -2
  8. package/dist/nightly.js +1 -1
  9. package/hermes-plugin/hicortex/provider.py +23 -0
  10. package/opencode-plugin/hicortex/index.ts +24 -1
  11. package/package.json +2 -1
  12. package/pi-extension/hicortex/index.ts +24 -1
  13. package/server.json +2 -2
  14. package/dist/eval/decay-eval.d.ts +0 -111
  15. package/dist/eval/decay-eval.js +0 -214
  16. package/dist/eval/dups.d.ts +0 -100
  17. package/dist/eval/dups.js +0 -174
  18. package/dist/eval/eval-clock.d.ts +0 -32
  19. package/dist/eval/eval-clock.js +0 -47
  20. package/dist/eval/eval-db.d.ts +0 -25
  21. package/dist/eval/eval-db.js +0 -67
  22. package/dist/eval/graph-eval.d.ts +0 -89
  23. package/dist/eval/graph-eval.js +0 -246
  24. package/dist/eval/importance-eval.d.ts +0 -85
  25. package/dist/eval/importance-eval.js +0 -286
  26. package/dist/eval/planted-eval.d.ts +0 -30
  27. package/dist/eval/planted-eval.js +0 -122
  28. package/dist/eval/planted-fixtures.d.ts +0 -107
  29. package/dist/eval/planted-fixtures.js +0 -283
  30. package/dist/eval/planted-harness.d.ts +0 -183
  31. package/dist/eval/planted-harness.js +0 -651
  32. package/dist/eval/ranking-battery.d.ts +0 -125
  33. package/dist/eval/ranking-battery.js +0 -289
  34. package/dist/eval/ranking-eval.d.ts +0 -61
  35. package/dist/eval/ranking-eval.js +0 -554
  36. package/dist/eval/ranking-fixtures.d.ts +0 -117
  37. package/dist/eval/ranking-fixtures.js +0 -485
  38. package/dist/eval/recall-sweep.d.ts +0 -87
  39. package/dist/eval/recall-sweep.js +0 -1030
  40. package/dist/eval/reflection-census.d.ts +0 -19
  41. package/dist/eval/reflection-census.js +0 -25
  42. package/dist/eval/relevance-eval.d.ts +0 -178
  43. package/dist/eval/relevance-eval.js +0 -2240
  44. package/dist/eval/run-eval.d.ts +0 -20
  45. package/dist/eval/run-eval.js +0 -299
@@ -1,554 +0,0 @@
1
- #!/usr/bin/env node
2
- "use strict";
3
- /**
4
- * Search-trust ranking gate (#425) — the AC1/AC2/AC3 before/after photo.
5
- * Extended by #449 (items 1+3, PR E): the case-2 photo gains per-row
6
- * results (scores + measured connections) and its gate becomes the
7
- * unlinked-row identity comparator; a new planted case 3 (linked_fallback)
8
- * pins the connections-reshape expectations.
9
- *
10
- * npm run eval:ranking [--snapshot <db>] [--photo <out.json>]
11
- * [--compare <baseline.json>] [--scoring '<json>']
12
- *
13
- * --snapshot <db> enables case 4: the deterministic real-query battery
14
- * against a READONLY snapshot (openSnapshot — never
15
- * initDb; the snapshot is never touched). Also reports
16
- * where the real "Sirnäs" memory ranks today (the
17
- * measured red photo pre-fix, AC10 pre-condition).
18
- * --photo <out.json> where to write the machine photo (defaults to
19
- * data/ranking-eval-photo.json). ALWAYS written, even
20
- * on a red run — the photo IS the measurement.
21
- * --compare <json> gate against a previously recorded photo: case 2's
22
- * UNLINKED rows must keep their scores (within the
23
- * wall-clock tolerance) and mutual order — linked-row
24
- * order drifts BY DESIGN under #449 — and the battery
25
- * (when both sides have one) must hold top-1
26
- * stability >= 90% with no query losing a both-channel
27
- * exact match from its top-3.
28
- * --scoring '<json>' a JSON object of ScoringWeights overrides passed to
29
- * configureScoring (the eval/test seam, #408) — the
30
- * sweep knob. Production runs omit it.
31
- * --now <ISO> #458: pin the eval clock — planted lastAccessed offsets
32
- * AND every retrieve() score against the SAME instant,
33
- * so before/after photos are wall-clock-independent.
34
- * Default: the live clock. The photo records the clock.
35
- *
36
- * Cases (fixtures: ranking-fixtures.ts, real bge-small-en-v1.5 embedder):
37
- * 1. sirnas_exact_match — planted corpus in a THROWAWAY writable DB;
38
- * declared expectation: the proper-noun exact-match row returns #1.
39
- * Exit code reflects the verdict (a pre-fix run is EXPECTED to exit
40
- * non-zero — that red photo is the before picture).
41
- * 2. no_match_control — same corpus, a "mamma"-class query with zero FTS
42
- * hits; records the full returned list (ids + per-row results) for the
43
- * unlinked-identity comparison.
44
- * 3. linked_fallback (#449) — planted corpus in its OWN throwaway DB: a
45
- * 16-link vs 20-link identical-content hub pair (saturation tie,
46
- * declared byte-equal post-change) + an unlinked higher-similarity row
47
- * vs a 4-link lower-similarity row (direction pin). Measured under the
48
- * eval seam rrfCompositeWeight=1 (finalScore = composite exactly —
49
- * identical-content hubs differ by an RRF epsilon at the default
50
- * blend). RED under the pre-#449 linear term: the recorded
51
- * before-picture.
52
- * 4. real-query battery (needs --snapshot) — ~30 queries derived
53
- * deterministically from the snapshot corpus (top/mid/rare-frequency
54
- * distinctive terms + proper nouns) + the fixed "Sirnäs" query; top-8
55
- * ids per query recorded to the photo.
56
- *
57
- * Honesty invariants: similarities are MEASURED (real embedder, embed-once,
58
- * queryEmbedding reuse), noStrengthen on every retrieve() (the probe never
59
- * perturbs the store), readonly snapshot access, and the planted corpora
60
- * live in mkdtemp dirs removed on exit. Never touches ~/.hicortex.
61
- */
62
- var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
63
- if (k2 === undefined) k2 = k;
64
- var desc = Object.getOwnPropertyDescriptor(m, k);
65
- if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
66
- desc = { enumerable: true, get: function() { return m[k]; } };
67
- }
68
- Object.defineProperty(o, k2, desc);
69
- }) : (function(o, m, k, k2) {
70
- if (k2 === undefined) k2 = k;
71
- o[k2] = m[k];
72
- }));
73
- var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
74
- Object.defineProperty(o, "default", { enumerable: true, value: v });
75
- }) : function(o, v) {
76
- o["default"] = v;
77
- });
78
- var __importStar = (this && this.__importStar) || (function () {
79
- var ownKeys = function(o) {
80
- ownKeys = Object.getOwnPropertyNames || function (o) {
81
- var ar = [];
82
- for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
83
- return ar;
84
- };
85
- return ownKeys(o);
86
- };
87
- return function (mod) {
88
- if (mod && mod.__esModule) return mod;
89
- var result = {};
90
- if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
91
- __setModuleDefault(result, mod);
92
- return result;
93
- };
94
- })();
95
- Object.defineProperty(exports, "__esModule", { value: true });
96
- const node_fs_1 = require("node:fs");
97
- const node_os_1 = require("node:os");
98
- const node_path_1 = require("node:path");
99
- const embedder_js_1 = require("../embedder.js");
100
- const db_js_1 = require("../db.js");
101
- const storage = __importStar(require("../storage.js"));
102
- const retrieval_js_1 = require("../retrieval.js");
103
- const eval_db_js_1 = require("./eval-db.js");
104
- const eval_clock_js_1 = require("./eval-clock.js");
105
- const ranking_fixtures_js_1 = require("./ranking-fixtures.js");
106
- const ranking_battery_js_1 = require("./ranking-battery.js");
107
- const DEFAULT_PHOTO_PATH = (0, node_path_1.join)(process.cwd(), "data", "ranking-eval-photo.json");
108
- /** Battery gate thresholds (sweep protocol, issue #425 commit-3 gates). */
109
- const BATTERY_TOP1_STABILITY_MIN = 0.9;
110
- /** Plant one fixture corpus into a writable DB with its seeded strength metadata.
111
- * `now` (#458): the instant the planted lastAccessed offsets are measured
112
- * from — the pinned clock when --now was passed, live otherwise. */
113
- async function plantRankingFixture(db, fixture, now) {
114
- const idByKey = new Map();
115
- const nowMs = (now ?? new Date()).getTime();
116
- for (const row of fixture.rows) {
117
- const embedding = await (0, embedder_js_1.embed)(row.content);
118
- const id = storage.insertMemory(db, row.content, embedding, {
119
- sourceAgent: ranking_fixtures_js_1.RANKING_SOURCE_AGENT,
120
- project: ranking_fixtures_js_1.RANKING_PROJECT,
121
- memoryType: row.memoryType,
122
- createdAt: row.createdAt,
123
- });
124
- const lastAccessed = new Date(nowMs - row.lastAccessedDaysAgo * 86_400_000).toISOString();
125
- storage.updateMemory(db, id, {
126
- base_strength: row.baseStrength,
127
- access_count: row.accessCount,
128
- last_accessed: lastAccessed,
129
- });
130
- idByKey.set(row.key, id);
131
- }
132
- for (const row of fixture.rows) {
133
- for (const to of row.linksTo) {
134
- const sourceId = idByKey.get(row.key);
135
- const targetId = idByKey.get(to);
136
- if (sourceId && targetId)
137
- storage.addLink(db, sourceId, targetId, "relates_to", 0.7);
138
- }
139
- }
140
- return idByKey;
141
- }
142
- /**
143
- * Run one planted case: full-corpus retrieve (limit = rows + headroom, so
144
- * the cold-exposure swap is inert — scored.length <= limit — and the
145
- * returned order IS the score order), results mapped back to fixture keys.
146
- */
147
- async function measurePlantedCase(db, fixture, idByKey, now) {
148
- const keyById = new Map([...idByKey.entries()].map(([k, v]) => [v, k]));
149
- const queryEmb = await (0, embedder_js_1.embed)(fixture.query);
150
- const ftsRows = storage.searchFts(db, fixture.query, 100, undefined);
151
- const results = await (0, retrieval_js_1.retrieve)(db, embedder_js_1.embed, fixture.query, {
152
- limit: fixture.rows.length + 8,
153
- noStrengthen: true,
154
- queryEmbedding: queryEmb,
155
- // #458: the pinned clock (undefined = live) — score terms are
156
- // wall-clock-independent across before/after photos.
157
- ...(now ? { now } : {}),
158
- });
159
- const measured = results.map((r, i) => ({
160
- key: keyById.get(r.id) ?? r.id,
161
- rank: i + 1,
162
- score: Math.round(r.score * 1e6) / 1e6,
163
- similarity: r.similarity ?? null,
164
- source: r.source ?? null,
165
- connections: r.connections,
166
- effectiveStrength: Math.round(r.effective_strength * 1e6) / 1e6,
167
- baseStrength: fixture.rows.find((x) => x.key === keyById.get(r.id))?.baseStrength ?? -1,
168
- accessCount: r.access_count,
169
- }));
170
- const pass = fixture.expectedTop1
171
- ? measured.length > 0 && measured[0].key === fixture.expectedTop1
172
- : null;
173
- const gap = measured.length >= 2
174
- ? Math.round((measured[0].score - measured[1].score) * 1e6) / 1e6
175
- : null;
176
- return {
177
- pass,
178
- query: fixture.query,
179
- ftsHitRows: ftsRows.length,
180
- results: measured,
181
- top1Top2ScoreGap: gap,
182
- };
183
- }
184
- /** Collapse a linked_fallback measurement into the #449 verdicts. */
185
- function case3Verdict(fixture, measured) {
186
- const exp = fixture.linkedExpectations;
187
- const byKey = new Map(measured.map((r) => [r.key, r]));
188
- const hubA = byKey.get(exp.saturationPair.hubA);
189
- const hubB = byKey.get(exp.saturationPair.hubB);
190
- const unlinked = byKey.get(exp.directionPair.unlinked);
191
- const linked = byKey.get(exp.directionPair.linked);
192
- const byteEqual = Object.is(hubA.score, hubB.score);
193
- const unlinkedAbove = unlinked.rank < linked.rank;
194
- const measuredGap = unlinked.similarity !== null && linked.similarity !== null
195
- ? Math.round((unlinked.similarity - linked.similarity) * 1e6) / 1e6
196
- : null;
197
- return {
198
- pass: byteEqual && unlinkedAbove,
199
- query: fixture.query,
200
- seam: { rrfCompositeWeight: 1 },
201
- saturation: {
202
- hubA: { key: hubA.key, connections: hubA.connections, score: hubA.score },
203
- hubB: { key: hubB.key, connections: hubB.connections, score: hubB.score },
204
- byteEqual,
205
- },
206
- direction: {
207
- unlinked: {
208
- key: unlinked.key,
209
- rank: unlinked.rank,
210
- similarity: unlinked.similarity,
211
- },
212
- linked: {
213
- key: linked.key,
214
- rank: linked.rank,
215
- similarity: linked.similarity,
216
- },
217
- measuredGap,
218
- unlinkedAbove,
219
- },
220
- results: measured,
221
- };
222
- }
223
- async function runBattery(db, now) {
224
- const rows = db
225
- .prepare("SELECT id, content FROM memories WHERE COALESCE(status, '') != 'absorbed'")
226
- .all();
227
- const queries = (0, ranking_battery_js_1.deriveBatteryQueries)(rows);
228
- const out = [];
229
- let sirnasRank = null;
230
- for (const { q, band } of queries) {
231
- const queryEmb = await (0, embedder_js_1.embed)(q);
232
- const results = await (0, retrieval_js_1.retrieve)(db, embedder_js_1.embed, q, {
233
- limit: 8,
234
- noStrengthen: true,
235
- queryEmbedding: queryEmb,
236
- ...(now ? { now } : {}), // #458: the pinned clock (undefined = live)
237
- });
238
- const top8 = results.map((r) => r.id);
239
- // Per-result provenance for the sweep analysis: the channel that
240
- // produced the row + (for term queries) whether the row's content
241
- // carries the query term — lets the gate distinguish "the D3 property
242
- // fired" (a both-channel exact match displaced a single-channel top-1)
243
- // from a real regression.
244
- const lower = q.toLowerCase();
245
- const meta = results.map((r) => ({
246
- id: r.id,
247
- source: r.source ?? null,
248
- term: q === ranking_battery_js_1.SIRNAS_QUERY
249
- ? null
250
- : (rows.find((x) => x.id === r.id)?.content ?? "").toLowerCase().includes(lower),
251
- }));
252
- out.push({ q, band, top8, meta });
253
- if (q === ranking_battery_js_1.SIRNAS_QUERY) {
254
- const idx = top8.indexOf(ranking_fixtures_js_1.REAL_SIRNAS_MEMORY_ID);
255
- sirnasRank = idx >= 0 ? idx + 1 : null;
256
- }
257
- }
258
- return { queries: out, sirnasRank };
259
- }
260
- // ---------------------------------------------------------------------------
261
- // Report rendering
262
- // ---------------------------------------------------------------------------
263
- function renderReport(args) {
264
- const L = [];
265
- const { case1, case2, case3, battery } = args;
266
- L.push("# Search-trust ranking gate (#425 + #449)\n");
267
- L.push(`Generated: ${new Date().toISOString()} \nClock: ${args.clock} (#458) \nScoring: ${JSON.stringify(args.scoring)}\n`);
268
- L.push("## Case 1 — sirnas_exact_match (AC1)\n");
269
- L.push(`Query "${case1.query}" — declared top-1: ${ranking_fixtures_js_1.SIRNAS_FIXTURE.expectedTop1} — ` +
270
- `**${case1.pass ? "PASS" : "FAIL"}**${case1.top1Top2ScoreGap !== null ? ` (top1-top2 score gap ${case1.top1Top2ScoreGap.toFixed(4)})` : ""}\n`);
271
- L.push("| rank | key | similarity | source | score | effStr | base | access |");
272
- L.push("|---|---|---|---|---|---|---|---|");
273
- for (const r of case1.results.slice(0, 6)) {
274
- L.push(`| ${r.rank} | ${r.key} | ${r.similarity?.toFixed(4) ?? "-"} | ${r.source ?? "-"} | ` +
275
- `${r.score.toFixed(4)} | ${r.effectiveStrength.toFixed(4)} | ${r.baseStrength} | ${r.accessCount} |`);
276
- }
277
- L.push("");
278
- L.push("## Case 2 — no_match_control (AC2)\n");
279
- L.push(`Query "${case2.query}" — FTS hits: ${case2.ftsHitRows} (must be 0 — no both-channel ` +
280
- `candidates can exist)` +
281
- (args.case2Identity === null
282
- ? ""
283
- : ` — vs baseline (unlinked-identity gate): **${args.case2Identity.pass ? "PASS" : "FAILED"}**` +
284
- (args.case2Identity.failures.length > 0
285
- ? ` — ${args.case2Identity.failures.slice(0, 3).join("; ")}` +
286
- (args.case2Identity.failures.length > 3 ? "; …" : "")
287
- : "")) +
288
- "\n");
289
- L.push("Returned order: " + case2.ids.map((k) => k).join(" → ") + "\n");
290
- L.push("| rank | key | similarity | source | conn | score |");
291
- L.push("|---|---|---|---|---|---|");
292
- for (const r of case2.results.slice(0, 6)) {
293
- L.push(`| ${r.rank} | ${r.key} | ${r.similarity?.toFixed(4) ?? "-"} | ${r.source ?? "-"} | ` +
294
- `${r.connections} | ${r.score.toFixed(4)} |`);
295
- }
296
- L.push("");
297
- L.push("## Case 3 — linked_fallback (#449 items 1+3)\n");
298
- L.push(`Query "${case3.query}" — measured under the eval seam rrfCompositeWeight=1 ` +
299
- `(finalScore = composite) — **${case3.pass ? "PASS" : "FAIL"}**\n`);
300
- L.push(`Saturation pair (declared post-#449: byte-equal scores — both hubs saturate at K=16; ` +
301
- `red under the linear term is the before-picture): ` +
302
- `${case3.saturation.hubA.key} (k=${case3.saturation.hubA.connections}) score ` +
303
- `${case3.saturation.hubA.score.toFixed(6)} vs ${case3.saturation.hubB.key} ` +
304
- `(k=${case3.saturation.hubB.connections}) score ${case3.saturation.hubB.score.toFixed(6)} — ` +
305
- `byte-equal: **${case3.saturation.byteEqual ? "YES" : "NO"}**\n`);
306
- L.push(`Direction pair (declared: unlinked ranks ABOVE the 4-link row — similarity dominates ` +
307
- `the saturated connections credit): ${case3.direction.unlinked.key} sim ` +
308
- `${case3.direction.unlinked.similarity?.toFixed(4) ?? "-"} rank #${case3.direction.unlinked.rank} vs ` +
309
- `${case3.direction.linked.key} sim ${case3.direction.linked.similarity?.toFixed(4) ?? "-"} rank ` +
310
- `#${case3.direction.linked.rank} — measured gap ` +
311
- `${case3.direction.measuredGap?.toFixed(4) ?? "-"} — unlinked above: ` +
312
- `**${case3.direction.unlinkedAbove ? "YES" : "NO"}**\n`);
313
- L.push("| rank | key | similarity | source | conn | score |");
314
- L.push("|---|---|---|---|---|---|");
315
- for (const r of case3.results.slice(0, 6)) {
316
- L.push(`| ${r.rank} | ${r.key} | ${r.similarity?.toFixed(4) ?? "-"} | ${r.source ?? "-"} | ` +
317
- `${r.connections} | ${r.score.toFixed(4)} |`);
318
- }
319
- L.push("");
320
- if (battery) {
321
- L.push("## Case 4 — real-query battery (AC3 no-regression photo)\n");
322
- L.push(`Queries: ${battery.queries.length}; Sirnäs rank for "${ranking_battery_js_1.SIRNAS_QUERY}": ` +
323
- (battery.sirnasRank === null ? "not in top-8" : `#${battery.sirnasRank}`) + "\n");
324
- if (args.comparison) {
325
- L.push(`vs baseline: top-1 stability ${(args.comparison.top1Stability * 100).toFixed(1)}% ` +
326
- `(gate >= ${(BATTERY_TOP1_STABILITY_MIN * 100).toFixed(0)}%; ` +
327
- `excluding intended D3 promotions ` +
328
- `${(args.comparison.top1StabilityExPromotions * 100).toFixed(1)}%) — ` +
329
- `${args.comparison.changedTop1.length} changed top-1` +
330
- (args.comparison.changedTop1.length > 0
331
- ? ` [${args.comparison.changedTop1.join(", ")}]`
332
- : "") +
333
- `; queries that LOST their both-channel exact match from top-3: ` +
334
- `${args.comparison.lostBothChannelExact.length}` +
335
- (args.comparison.lostBothChannelExact.length > 0
336
- ? ` [${args.comparison.lostBothChannelExact.join(", ")}]`
337
- : "") +
338
- `; per-id exact-match displacements from top-3 (reported, not gated): ` +
339
- `${args.comparison.lostExactMatches.length}` +
340
- "\n");
341
- }
342
- L.push("| band | query | top-1 id |");
343
- L.push("|---|---|---|");
344
- for (const b of battery.queries) {
345
- L.push(`| ${b.band} | ${b.q} | ${b.top8[0]?.slice(0, 8) ?? "-"} |`);
346
- }
347
- L.push("");
348
- }
349
- return L.join("\n");
350
- }
351
- // ---------------------------------------------------------------------------
352
- // main
353
- // ---------------------------------------------------------------------------
354
- function usage() {
355
- console.error("Usage: npm run eval:ranking -- [--snapshot <db>] [--photo <out.json>] [--compare <baseline.json>] [--scoring '<json>'] [--now <ISO>]");
356
- process.exit(1);
357
- }
358
- async function main() {
359
- const argv = process.argv.slice(2);
360
- const flag = (name) => {
361
- const i = argv.indexOf(name);
362
- return i !== -1 && argv[i + 1] ? argv[i + 1] : undefined;
363
- };
364
- const snapshotPath = flag("--snapshot");
365
- const photoPath = flag("--photo") ?? DEFAULT_PHOTO_PATH;
366
- const comparePath = flag("--compare");
367
- const scoringRaw = flag("--scoring");
368
- if (argv.includes("--help") || argv.includes("-h"))
369
- usage();
370
- // #458: pin the eval clock (planted lastAccessed offsets + every retrieve
371
- // score against the same instant). Invalid input fails explicit — never a
372
- // silent live fallback that would reintroduce wall-clock drift.
373
- let now;
374
- try {
375
- now = (0, eval_clock_js_1.parsePinnedNow)(flag("--now"));
376
- }
377
- catch (err) {
378
- console.error(`[ranking-eval] ${err instanceof Error ? err.message : err}`);
379
- process.exit(1);
380
- }
381
- console.log(`[ranking-eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}`);
382
- const nowArg = now ?? undefined;
383
- if (scoringRaw) {
384
- try {
385
- const parsed = JSON.parse(scoringRaw);
386
- (0, retrieval_js_1.configureScoring)(parsed);
387
- }
388
- catch (err) {
389
- console.error(`[ranking-eval] --scoring is not valid JSON: ${err instanceof Error ? err.message : err}`);
390
- process.exit(1);
391
- }
392
- }
393
- for (const f of [ranking_fixtures_js_1.SIRNAS_FIXTURE, ranking_fixtures_js_1.NO_MATCH_FIXTURE, ranking_fixtures_js_1.LINKED_FALLBACK_FIXTURE]) {
394
- const problems = (0, ranking_fixtures_js_1.checkFixtureInvariants)(f);
395
- if (problems.length > 0) {
396
- console.error(`[ranking-eval] fixture ${f.key} invariant violations: ${problems.join("; ")}`);
397
- process.exit(1);
398
- }
399
- }
400
- console.log("[ranking-eval] loading real bge-small-en-v1.5 embedder...");
401
- const t0 = Date.now();
402
- // Cases 1 + 2 share one planted corpus DB (same rows; two queries).
403
- const workDir = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-ranking-eval-"));
404
- let case1;
405
- let case2;
406
- try {
407
- const db = (0, db_js_1.initDb)((0, node_path_1.join)(workDir, "planted.db"));
408
- try {
409
- console.log("[ranking-eval] planting corpus (real embedder)...");
410
- const idByKey = await plantRankingFixture(db, ranking_fixtures_js_1.SIRNAS_FIXTURE, nowArg);
411
- case1 = await measurePlantedCase(db, ranking_fixtures_js_1.SIRNAS_FIXTURE, idByKey, nowArg);
412
- const case2Measured = await measurePlantedCase(db, ranking_fixtures_js_1.NO_MATCH_FIXTURE, idByKey, nowArg);
413
- // The returned list keyed by fixture key (readable photos; ids would
414
- // churn across replanting).
415
- case2 = { ...case2Measured, ids: case2Measured.results.map((r) => r.key) };
416
- }
417
- finally {
418
- db.close();
419
- }
420
- }
421
- finally {
422
- (0, node_fs_1.rmSync)(workDir, { recursive: true, force: true });
423
- }
424
- // Case 3 (#449): planted in its OWN throwaway DB, measured under the eval
425
- // seam — rrfCompositeWeight 1 makes finalScore = composite exactly (the
426
- // savedWeights save/restore pattern from tests/ranking-flip.test.ts; the
427
- // --scoring CLI overrides, if any, are what gets saved and restored).
428
- let case3;
429
- {
430
- const savedWeights = (0, retrieval_js_1.getScoringWeights)();
431
- (0, retrieval_js_1.configureScoring)({ ...savedWeights, rrfCompositeWeight: 1 });
432
- const workDir3 = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-ranking-eval-linked-"));
433
- try {
434
- const db = (0, db_js_1.initDb)((0, node_path_1.join)(workDir3, "linked.db"));
435
- try {
436
- console.log("[ranking-eval] planting linked_fallback corpus (real embedder)...");
437
- const idByKey3 = await plantRankingFixture(db, ranking_fixtures_js_1.LINKED_FALLBACK_FIXTURE, nowArg);
438
- const measured3 = await measurePlantedCase(db, ranking_fixtures_js_1.LINKED_FALLBACK_FIXTURE, idByKey3, nowArg);
439
- case3 = case3Verdict(ranking_fixtures_js_1.LINKED_FALLBACK_FIXTURE, measured3.results);
440
- }
441
- finally {
442
- db.close();
443
- }
444
- }
445
- finally {
446
- (0, node_fs_1.rmSync)(workDir3, { recursive: true, force: true });
447
- (0, retrieval_js_1.configureScoring)(savedWeights); // restore — battery + photo run at the caller's weights
448
- }
449
- }
450
- let battery = null;
451
- let comparison = null;
452
- let case2Identity = null;
453
- // Content lookup for the exact-match gate (the SAME corpus both photos
454
- // were recorded against — the snapshot, when comparing).
455
- const contentById = new Map();
456
- if (snapshotPath) {
457
- console.log(`[ranking-eval] battery against readonly snapshot: ${snapshotPath}`);
458
- const sdb = (0, eval_db_js_1.openSnapshot)(snapshotPath);
459
- try {
460
- battery = await runBattery(sdb, nowArg);
461
- for (const r of sdb
462
- .prepare("SELECT id, content FROM memories WHERE COALESCE(status, '') != 'absorbed'")
463
- .all()) {
464
- contentById.set(r.id, r.content);
465
- }
466
- }
467
- finally {
468
- sdb.close();
469
- }
470
- }
471
- if (comparePath) {
472
- let baseline;
473
- try {
474
- baseline = JSON.parse((0, node_fs_1.readFileSync)(comparePath, "utf-8"));
475
- }
476
- catch (err) {
477
- console.error(`[ranking-eval] cannot read baseline photo ${comparePath}: ${err instanceof Error ? err.message : err}`);
478
- process.exit(1);
479
- }
480
- if (!baseline.case2) {
481
- console.error("[ranking-eval] baseline photo lacks case2 — record one with --photo first");
482
- process.exit(1);
483
- }
484
- // #449: the case-2 gate is the unlinked-row identity comparator (linked
485
- // rows drift BY DESIGN); it reads the per-row results the baseline must
486
- // already carry and fails with a re-record instruction when it doesn't.
487
- case2Identity = (0, ranking_battery_js_1.compareCase2Unlinked)(baseline.case2, { ftsHitRows: case2.ftsHitRows, ids: case2.ids, results: case2.results });
488
- // The battery comparison runs only when BOTH sides have one (the
489
- // baseline was recorded with --snapshot AND this run passes it).
490
- if (baseline.battery && battery) {
491
- comparison = (0, ranking_battery_js_1.batteryComparison)(baseline.battery, battery, contentById);
492
- }
493
- else if (baseline.battery && !battery) {
494
- console.error("[ranking-eval] baseline photo has a battery but this run recorded none — pass --snapshot to compare it");
495
- process.exit(1);
496
- }
497
- }
498
- const photo = {
499
- generatedAt: new Date().toISOString(),
500
- /** #458: self-describing clock — which instant the decay/recency terms
501
- * scored against ({mode:"pinned", now} or {mode:"live"}). */
502
- clock: now === null
503
- ? { mode: "live" }
504
- : { mode: "pinned", now: now.toISOString() },
505
- scoring: (0, retrieval_js_1.getScoringWeights)(),
506
- case1: {
507
- pass: case1.pass,
508
- expectedTop1: ranking_fixtures_js_1.SIRNAS_FIXTURE.expectedTop1,
509
- top1Top2ScoreGap: case1.top1Top2ScoreGap,
510
- results: case1.results,
511
- },
512
- case2: { ftsHitRows: case2.ftsHitRows, ids: case2.ids, results: case2.results },
513
- case3,
514
- battery,
515
- comparison,
516
- case2Identity,
517
- };
518
- (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(photoPath), { recursive: true });
519
- (0, node_fs_1.writeFileSync)(photoPath, JSON.stringify(photo, null, 2), "utf-8");
520
- console.log(`[ranking-eval] photo written: ${photoPath}`);
521
- const report = renderReport({
522
- case1,
523
- case2,
524
- case3,
525
- battery,
526
- comparison,
527
- case2Identity,
528
- scoring: { ...photo.scoring },
529
- clock: (0, eval_clock_js_1.clockLabel)(now),
530
- });
531
- console.log(report);
532
- const reportPath = photoPath.endsWith(".json")
533
- ? photoPath.slice(0, -5) + ".md"
534
- : photoPath + ".md";
535
- (0, node_fs_1.writeFileSync)(reportPath, report, "utf-8");
536
- const batteryOk = comparison === null
537
- ? true
538
- : comparison.top1StabilityExPromotions >= BATTERY_TOP1_STABILITY_MIN &&
539
- comparison.lostBothChannelExact.length === 0;
540
- const case2Ok = case2Identity === null ? true : case2Identity.pass;
541
- const case3Ok = case3.pass;
542
- const ok = case1.pass === true && case2Ok && case3Ok && batteryOk;
543
- console.log(`[ranking-eval] ${ok ? "GATE PASS" : "GATE FAIL"} (case1 ${case1.pass ? "pass" : "FAIL"}, ` +
544
- `case2 ${case2Ok ? "ok" : "DRIFT"}, case3 ${case3Ok ? "pass" : "FAIL"}, ` +
545
- `battery ${batteryOk ? "ok" : "DRIFT"} ` +
546
- `[top-1 ${((comparison?.top1Stability ?? 1) * 100).toFixed(1)}% raw / ` +
547
- `${((comparison?.top1StabilityExPromotions ?? 1) * 100).toFixed(1)}% ex-promotions, ` +
548
- `lost both-channel exact ${comparison?.lostBothChannelExact.length ?? 0}]) in ${Date.now() - t0}ms`);
549
- process.exitCode = ok ? 0 : 1;
550
- }
551
- main().catch((err) => {
552
- console.error("[ranking-eval] FAILED:", err instanceof Error ? err.stack : String(err));
553
- process.exitCode = 1;
554
- });
@@ -1,117 +0,0 @@
1
- /**
2
- * Planted ranking fixtures for the search-trust calibration gate (#425).
3
- *
4
- * Two fixture cases, both shaped like the field failure that opened the
5
- * issue (the owner's live console test, 2026-09-13): a user searches a
6
- * DISTINCTIVE PROPER NOUN they know exists; the one memory carrying that
7
- * token has the best similarity of all candidates AND the literal FTS hit,
8
- * yet ranks behind high-strength but less-relevant memories that win on
9
- * effective strength + graph connections.
10
- *
11
- * (a) sirnas_exact_match — the AC1 fixture: a low-strength, unlinked,
12
- * never-accessed memory carrying an invented proper noun against
13
- * base-≥0.9 rivals with links among themselves, access history, and
14
- * older creation dates. The rivals share the query's DOMAIN
15
- * vocabulary (harbours, docks, boats) but never the proper noun.
16
- * Declared expectation: the proper-noun row ranks #1.
17
- * (b) no_match_control — the AC2 fixture: a "mamma"-class query with no
18
- * relevant match and NO FTS hits anywhere in the corpus. Whatever
19
- * nearest-neighbor junk vector search returns must be byte-stable
20
- * under the ranking work (no both-channel candidates exist, so the
21
- * boost can never fire — the case pins that the no-regression
22
- * guarantee is structural, not lucky).
23
- * (c) linked_fallback — the #449 (items 1+3) connections fixture: a
24
- * 16-link vs 20-link identical-content hub pair (the saturation tie)
25
- * plus an unlinked higher-similarity row vs a 4-link lower-similarity
26
- * row (the direction pin). Planted in its OWN throwaway DB — the 40
27
- * link-neighbor rows would pollute the case-1/2 corpus.
28
- *
29
- * Honesty rules (the planted-pairs #393 A discipline):
30
- * - The texts are synthetic and generic (this package publishes to npm —
31
- * no real infrastructure, people, or fleet detail), but SHAPED like the
32
- * field failure: the proper noun is invented (it must appear nowhere
33
- * else — verified by an invariant test), the rivals carry the metadata
34
- * the production pipeline itself writes (base strength, access history,
35
- * link counts).
36
- * - Similarities are MEASURED at run time with the real bge-small-en-v1.5
37
- * embedder (ranking-eval.ts) — never asserted into existence. The
38
- * declared expectation is about ORDER under the shipped ranking, and
39
- * the report prints the measured similarity/score of every top row.
40
- * - Strength metadata is written through the production helpers
41
- * (insertMemory + updateMemory + addLink), not hand-rolled SQL, so the
42
- * fixture exercises the same rows retrieval() will see.
43
- */
44
- export interface RankingRow {
45
- /** Stable identifier used by expectations, links, and the report. */
46
- key: string;
47
- content: string;
48
- /** ISO timestamp. */
49
- createdAt: string;
50
- /** knowledge | experience | decision. */
51
- memoryType: string;
52
- /** Birth base_strength — the value stageImportance would have written. */
53
- baseStrength: number;
54
- /** Access history (the decay clock + the #448 promotion baseline). */
55
- accessCount: number;
56
- /** Days before the run's `now` of the last access (0 = touched today). */
57
- lastAccessedDaysAgo: number;
58
- /** Keys this row links to (links are created after all inserts). */
59
- linksTo: string[];
60
- }
61
- export interface RankingFixture {
62
- /** Case identifier (report + photo keys). */
63
- key: string;
64
- description: string;
65
- rows: RankingRow[];
66
- query: string;
67
- /**
68
- * The token that makes the query distinctive — invariant-tested to appear
69
- * in exactly one row (the target) and in no rival/filler content.
70
- */
71
- distinctiveToken: string;
72
- /** Declared expectation: this row's key must be returned at position 1. */
73
- expectedTop1: string;
74
- /**
75
- * Declared undirected link degree per row key (#449): own linksTo entries
76
- * PLUS the times other rows reference it (each linksTo entry is one link
77
- * row, mirroring storage.getLinks(id, "both").length). Optional — case 1
78
- * and case 3 declare it; a corpus that silently drifts from its declared
79
- * link-cluster shape would fake a pass/fail.
80
- */
81
- expectedUndirectedDegree?: Record<string, number>;
82
- /**
83
- * #449 case-3 (linked_fallback) expectations — absent for cases 1-2. The
84
- * verdict logic (byte-equality + rank order) lives in ranking-eval.ts.
85
- */
86
- linkedExpectations?: {
87
- /**
88
- * The saturation pair: two identical-content hubs at the declared
89
- * degrees — post-#449 (log-saturating term, K = 16) their scores must
90
- * be byte-EQUAL (both saturate); under the pre-#449 linear term they
91
- * differ by ~0.03 composite (the recorded before-picture).
92
- */
93
- saturationPair: {
94
- hubA: string;
95
- hubB: string;
96
- };
97
- /**
98
- * The direction pair: unlinked (higher measured similarity, carries the
99
- * distinctive token) vs linked (exactly 4 links, ~0.20 lower measured
100
- * similarity, equal metadata otherwise). Declared (Option A, owner
101
- * decision 2026-09-17): the unlinked row ranks ABOVE the linked one.
102
- */
103
- directionPair: {
104
- unlinked: string;
105
- linked: string;
106
- };
107
- };
108
- }
109
- /** All planted rows share project + source_agent (metadata never refuses). */
110
- export declare const RANKING_PROJECT = "ranking-eval";
111
- export declare const RANKING_SOURCE_AGENT = "ranking-eval";
112
- export declare const SIRNAS_FIXTURE: RankingFixture;
113
- export declare const NO_MATCH_FIXTURE: RankingFixture;
114
- /** The battery runs against a real snapshot copy (ranking-eval.ts case 4). */
115
- export declare const REAL_SIRNAS_MEMORY_ID = "12980c9d-0eba-4963-95eb-0548bc2aa93e";
116
- export declare const LINKED_FALLBACK_FIXTURE: RankingFixture;
117
- export declare function checkFixtureInvariants(f: RankingFixture): string[];