@gamaze/hicortex 0.20.7 → 0.20.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +18 -41
  2. package/assets/dashboard.html +3989 -836
  3. package/dist/calibration.d.ts +293 -0
  4. package/dist/calibration.js +379 -0
  5. package/dist/capture-health.d.ts +87 -0
  6. package/dist/capture-health.js +106 -0
  7. package/dist/capture-pause.d.ts +86 -0
  8. package/dist/capture-pause.js +127 -0
  9. package/dist/capture.d.ts +24 -3
  10. package/dist/capture.js +11 -1
  11. package/dist/classify-domains.d.ts +6 -0
  12. package/dist/classify-domains.js +7 -1
  13. package/dist/cli.js +38 -3
  14. package/dist/config-read.d.ts +1 -1
  15. package/dist/config-read.js +96 -9
  16. package/dist/consolidate.d.ts +114 -68
  17. package/dist/consolidate.js +302 -182
  18. package/dist/dashboard.d.ts +326 -6
  19. package/dist/dashboard.js +592 -7
  20. package/dist/db.js +105 -0
  21. package/dist/dedup.d.ts +34 -26
  22. package/dist/dedup.js +91 -57
  23. package/dist/distiller.js +1 -1
  24. package/dist/domain-classify.d.ts +7 -6
  25. package/dist/domain-classify.js +12 -10
  26. package/dist/eval/decay-eval.d.ts +3 -3
  27. package/dist/eval/decay-eval.js +4 -4
  28. package/dist/eval/importance-eval.d.ts +85 -0
  29. package/dist/eval/importance-eval.js +286 -0
  30. package/dist/eval/planted-eval.d.ts +26 -0
  31. package/dist/eval/planted-eval.js +97 -0
  32. package/dist/eval/planted-fixtures.d.ts +107 -0
  33. package/dist/eval/planted-fixtures.js +283 -0
  34. package/dist/eval/planted-harness.d.ts +176 -0
  35. package/dist/eval/planted-harness.js +649 -0
  36. package/dist/eval/ranking-battery.d.ts +78 -0
  37. package/dist/eval/ranking-battery.js +181 -0
  38. package/dist/eval/ranking-eval.d.ts +41 -0
  39. package/dist/eval/ranking-eval.js +391 -0
  40. package/dist/eval/ranking-fixtures.d.ts +77 -0
  41. package/dist/eval/ranking-fixtures.js +226 -0
  42. package/dist/identity-store.d.ts +21 -0
  43. package/dist/identity-store.js +49 -0
  44. package/dist/index.js +4 -3
  45. package/dist/init.d.ts +23 -3
  46. package/dist/init.js +84 -9
  47. package/dist/llm.d.ts +43 -58
  48. package/dist/llm.js +87 -101
  49. package/dist/mcp-server.d.ts +12 -0
  50. package/dist/mcp-server.js +213 -32
  51. package/dist/nightly.d.ts +9 -1
  52. package/dist/nightly.js +164 -110
  53. package/dist/nofit.d.ts +4 -11
  54. package/dist/nofit.js +6 -23
  55. package/dist/prompts.d.ts +10 -0
  56. package/dist/prompts.js +28 -5
  57. package/dist/recall-index.d.ts +30 -28
  58. package/dist/recall-index.js +21 -18
  59. package/dist/recall-registry.d.ts +2 -1
  60. package/dist/recall-registry.js +35 -1
  61. package/dist/reconsolidation.d.ts +168 -87
  62. package/dist/reconsolidation.js +818 -377
  63. package/dist/relink.js +3 -4
  64. package/dist/rescore-importance.d.ts +80 -0
  65. package/dist/rescore-importance.js +236 -0
  66. package/dist/retrieval.d.ts +80 -35
  67. package/dist/retrieval.js +322 -105
  68. package/dist/run-deadline.d.ts +62 -0
  69. package/dist/run-deadline.js +73 -0
  70. package/dist/schema-prototypes.d.ts +3 -3
  71. package/dist/schema-prototypes.js +3 -3
  72. package/dist/stages.d.ts +37 -0
  73. package/dist/stages.js +51 -0
  74. package/dist/state.d.ts +34 -9
  75. package/dist/storage.d.ts +50 -18
  76. package/dist/storage.js +125 -30
  77. package/dist/telemetry.d.ts +8 -7
  78. package/dist/token-budget.js +3 -4
  79. package/dist/type-classify.js +4 -4
  80. package/dist/types.d.ts +143 -155
  81. package/domains.example.json +4 -5
  82. package/hermes-plugin/hicortex/README.md +2 -2
  83. package/openclaw.plugin.json +1 -1
  84. package/package.json +4 -1
  85. package/pi-extension/hicortex/README.md +1 -1
  86. package/server.json +3 -3
@@ -0,0 +1,391 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * Search-trust ranking gate (#425) — the AC1/AC2/AC3 before/after photo.
5
+ *
6
+ * npm run eval:ranking [--snapshot <db>] [--photo <out.json>]
7
+ * [--compare <baseline.json>] [--scoring '<json>']
8
+ *
9
+ * --snapshot <db> enables case 3: the deterministic real-query battery
10
+ * against a READONLY snapshot (openSnapshot — never
11
+ * initDb; the snapshot is never touched). Also reports
12
+ * where the real "Sirnäs" memory ranks today (the
13
+ * measured red photo pre-fix, AC10 pre-condition).
14
+ * --photo <out.json> where to write the machine photo (defaults to
15
+ * data/ranking-eval-photo.json). ALWAYS written, even
16
+ * on a red run — the photo IS the measurement.
17
+ * --compare <json> gate against a previously recorded photo: case 2's
18
+ * returned list must be byte-identical, and the battery
19
+ * must hold top-1 stability >= 90% with no query losing
20
+ * a both-channel exact match from its top-3.
21
+ * --scoring '<json>' a JSON object of ScoringWeights overrides passed to
22
+ * configureScoring (the eval/test seam, #408) — the
23
+ * sweep knob. Production runs omit it.
24
+ *
25
+ * Cases (fixtures: ranking-fixtures.ts, real bge-small-en-v1.5 embedder):
26
+ * 1. sirnas_exact_match — planted corpus in a THROWAWAY writable DB;
27
+ * declared expectation: the proper-noun exact-match row returns #1.
28
+ * Exit code reflects the verdict (a pre-fix run is EXPECTED to exit
29
+ * non-zero — that red photo is the before picture).
30
+ * 2. no_match_control — same corpus, a "mamma"-class query with zero FTS
31
+ * hits; records the full returned list for byte-stability comparison.
32
+ * 3. real-query battery (needs --snapshot) — ~30 queries derived
33
+ * deterministically from the snapshot corpus (top/mid/rare-frequency
34
+ * distinctive terms + proper nouns) + the fixed "Sirnäs" query; top-8
35
+ * ids per query recorded to the photo.
36
+ *
37
+ * Honesty invariants: similarities are MEASURED (real embedder, embed-once,
38
+ * queryEmbedding reuse), noStrengthen on every retrieve() (the probe never
39
+ * perturbs the store), readonly snapshot access, and the planted corpora
40
+ * live in mkdtemp dirs removed on exit. Never touches ~/.hicortex.
41
+ */
42
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
43
+ if (k2 === undefined) k2 = k;
44
+ var desc = Object.getOwnPropertyDescriptor(m, k);
45
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
46
+ desc = { enumerable: true, get: function() { return m[k]; } };
47
+ }
48
+ Object.defineProperty(o, k2, desc);
49
+ }) : (function(o, m, k, k2) {
50
+ if (k2 === undefined) k2 = k;
51
+ o[k2] = m[k];
52
+ }));
53
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
54
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
55
+ }) : function(o, v) {
56
+ o["default"] = v;
57
+ });
58
+ var __importStar = (this && this.__importStar) || (function () {
59
+ var ownKeys = function(o) {
60
+ ownKeys = Object.getOwnPropertyNames || function (o) {
61
+ var ar = [];
62
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
63
+ return ar;
64
+ };
65
+ return ownKeys(o);
66
+ };
67
+ return function (mod) {
68
+ if (mod && mod.__esModule) return mod;
69
+ var result = {};
70
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
71
+ __setModuleDefault(result, mod);
72
+ return result;
73
+ };
74
+ })();
75
+ Object.defineProperty(exports, "__esModule", { value: true });
76
+ const node_fs_1 = require("node:fs");
77
+ const node_os_1 = require("node:os");
78
+ const node_path_1 = require("node:path");
79
+ const embedder_js_1 = require("../embedder.js");
80
+ const db_js_1 = require("../db.js");
81
+ const storage = __importStar(require("../storage.js"));
82
+ const retrieval_js_1 = require("../retrieval.js");
83
+ const eval_db_js_1 = require("./eval-db.js");
84
+ const ranking_fixtures_js_1 = require("./ranking-fixtures.js");
85
+ const ranking_battery_js_1 = require("./ranking-battery.js");
86
+ const DEFAULT_PHOTO_PATH = (0, node_path_1.join)(process.cwd(), "data", "ranking-eval-photo.json");
87
+ /** Battery gate thresholds (sweep protocol, issue #425 commit-3 gates). */
88
+ const BATTERY_TOP1_STABILITY_MIN = 0.9;
89
+ /** Plant one fixture corpus into a writable DB with its hardened metadata. */
90
+ async function plantRankingFixture(db, fixture) {
91
+ const idByKey = new Map();
92
+ const now = Date.now();
93
+ for (const row of fixture.rows) {
94
+ const embedding = await (0, embedder_js_1.embed)(row.content);
95
+ const id = storage.insertMemory(db, row.content, embedding, {
96
+ sourceAgent: ranking_fixtures_js_1.RANKING_SOURCE_AGENT,
97
+ project: ranking_fixtures_js_1.RANKING_PROJECT,
98
+ memoryType: row.memoryType,
99
+ createdAt: row.createdAt,
100
+ });
101
+ const lastAccessed = new Date(now - row.lastAccessedDaysAgo * 86_400_000).toISOString();
102
+ storage.updateMemory(db, id, {
103
+ base_strength: row.baseStrength,
104
+ access_count: row.accessCount,
105
+ last_accessed: lastAccessed,
106
+ });
107
+ idByKey.set(row.key, id);
108
+ }
109
+ for (const row of fixture.rows) {
110
+ for (const to of row.linksTo) {
111
+ const sourceId = idByKey.get(row.key);
112
+ const targetId = idByKey.get(to);
113
+ if (sourceId && targetId)
114
+ storage.addLink(db, sourceId, targetId, "relates_to", 0.7);
115
+ }
116
+ }
117
+ return idByKey;
118
+ }
119
+ /**
120
+ * Run one planted case: full-corpus retrieve (limit = rows + headroom, so
121
+ * the cold-exposure swap is inert — scored.length <= limit — and the
122
+ * returned order IS the score order), results mapped back to fixture keys.
123
+ */
124
+ async function measurePlantedCase(db, fixture, idByKey) {
125
+ const keyById = new Map([...idByKey.entries()].map(([k, v]) => [v, k]));
126
+ const queryEmb = await (0, embedder_js_1.embed)(fixture.query);
127
+ const ftsRows = storage.searchFts(db, fixture.query, 100, undefined);
128
+ const results = await (0, retrieval_js_1.retrieve)(db, embedder_js_1.embed, fixture.query, {
129
+ limit: fixture.rows.length + 8,
130
+ noStrengthen: true,
131
+ queryEmbedding: queryEmb,
132
+ });
133
+ const measured = results.map((r, i) => ({
134
+ key: keyById.get(r.id) ?? r.id,
135
+ rank: i + 1,
136
+ score: Math.round(r.score * 1e6) / 1e6,
137
+ similarity: r.similarity ?? null,
138
+ source: r.source ?? null,
139
+ effectiveStrength: Math.round(r.effective_strength * 1e6) / 1e6,
140
+ baseStrength: fixture.rows.find((x) => x.key === keyById.get(r.id))?.baseStrength ?? -1,
141
+ accessCount: r.access_count,
142
+ }));
143
+ const pass = fixture.expectedTop1
144
+ ? measured.length > 0 && measured[0].key === fixture.expectedTop1
145
+ : null;
146
+ const gap = measured.length >= 2
147
+ ? Math.round((measured[0].score - measured[1].score) * 1e6) / 1e6
148
+ : null;
149
+ return {
150
+ pass,
151
+ query: fixture.query,
152
+ ftsHitRows: ftsRows.length,
153
+ results: measured,
154
+ top1Top2ScoreGap: gap,
155
+ };
156
+ }
157
+ async function runBattery(db) {
158
+ const rows = db
159
+ .prepare("SELECT id, content FROM memories WHERE COALESCE(status, '') != 'absorbed'")
160
+ .all();
161
+ const queries = (0, ranking_battery_js_1.deriveBatteryQueries)(rows);
162
+ const out = [];
163
+ let sirnasRank = null;
164
+ for (const { q, band } of queries) {
165
+ const queryEmb = await (0, embedder_js_1.embed)(q);
166
+ const results = await (0, retrieval_js_1.retrieve)(db, embedder_js_1.embed, q, {
167
+ limit: 8,
168
+ noStrengthen: true,
169
+ queryEmbedding: queryEmb,
170
+ });
171
+ const top8 = results.map((r) => r.id);
172
+ // Per-result provenance for the sweep analysis: the channel that
173
+ // produced the row + (for term queries) whether the row's content
174
+ // carries the query term — lets the gate distinguish "the D3 property
175
+ // fired" (a both-channel exact match displaced a single-channel top-1)
176
+ // from a real regression.
177
+ const lower = q.toLowerCase();
178
+ const meta = results.map((r) => ({
179
+ id: r.id,
180
+ source: r.source ?? null,
181
+ term: q === ranking_battery_js_1.SIRNAS_QUERY
182
+ ? null
183
+ : (rows.find((x) => x.id === r.id)?.content ?? "").toLowerCase().includes(lower),
184
+ }));
185
+ out.push({ q, band, top8, meta });
186
+ if (q === ranking_battery_js_1.SIRNAS_QUERY) {
187
+ const idx = top8.indexOf(ranking_fixtures_js_1.REAL_SIRNAS_MEMORY_ID);
188
+ sirnasRank = idx >= 0 ? idx + 1 : null;
189
+ }
190
+ }
191
+ return { queries: out, sirnasRank };
192
+ }
193
+ // ---------------------------------------------------------------------------
194
+ // Report rendering
195
+ // ---------------------------------------------------------------------------
196
+ function renderReport(args) {
197
+ const L = [];
198
+ const { case1, case2, battery } = args;
199
+ L.push("# Search-trust ranking gate (#425)\n");
200
+ L.push(`Generated: ${new Date().toISOString()} \nScoring: ${JSON.stringify(args.scoring)}\n`);
201
+ L.push("## Case 1 — sirnas_exact_match (AC1)\n");
202
+ L.push(`Query "${case1.query}" — declared top-1: ${ranking_fixtures_js_1.SIRNAS_FIXTURE.expectedTop1} — ` +
203
+ `**${case1.pass ? "PASS" : "FAIL"}**${case1.top1Top2ScoreGap !== null ? ` (top1-top2 score gap ${case1.top1Top2ScoreGap.toFixed(4)})` : ""}\n`);
204
+ L.push("| rank | key | similarity | source | score | effStr | base | access |");
205
+ L.push("|---|---|---|---|---|---|---|---|");
206
+ for (const r of case1.results.slice(0, 6)) {
207
+ L.push(`| ${r.rank} | ${r.key} | ${r.similarity?.toFixed(4) ?? "-"} | ${r.source ?? "-"} | ` +
208
+ `${r.score.toFixed(4)} | ${r.effectiveStrength.toFixed(4)} | ${r.baseStrength} | ${r.accessCount} |`);
209
+ }
210
+ L.push("");
211
+ L.push("## Case 2 — no_match_control (AC2)\n");
212
+ L.push(`Query "${case2.query}" — FTS hits: ${case2.ftsHitRows} (must be 0 — no both-channel ` +
213
+ `candidates can exist)` +
214
+ (args.case2Stable === null ? "" : ` — vs baseline: **${args.case2Stable ? "identical" : "DRIFTED"}**`) +
215
+ "\n");
216
+ L.push("Returned order: " + case2.ids.map((k) => k).join(" → ") + "\n");
217
+ if (battery) {
218
+ L.push("## Case 3 — real-query battery (AC3 no-regression photo)\n");
219
+ L.push(`Queries: ${battery.queries.length}; Sirnäs rank for "${ranking_battery_js_1.SIRNAS_QUERY}": ` +
220
+ (battery.sirnasRank === null ? "not in top-8" : `#${battery.sirnasRank}`) + "\n");
221
+ if (args.comparison) {
222
+ L.push(`vs baseline: top-1 stability ${(args.comparison.top1Stability * 100).toFixed(1)}% ` +
223
+ `(gate >= ${(BATTERY_TOP1_STABILITY_MIN * 100).toFixed(0)}%; ` +
224
+ `excluding intended D3 promotions ` +
225
+ `${(args.comparison.top1StabilityExPromotions * 100).toFixed(1)}%) — ` +
226
+ `${args.comparison.changedTop1.length} changed top-1` +
227
+ (args.comparison.changedTop1.length > 0
228
+ ? ` [${args.comparison.changedTop1.join(", ")}]`
229
+ : "") +
230
+ `; queries that LOST their both-channel exact match from top-3: ` +
231
+ `${args.comparison.lostBothChannelExact.length}` +
232
+ (args.comparison.lostBothChannelExact.length > 0
233
+ ? ` [${args.comparison.lostBothChannelExact.join(", ")}]`
234
+ : "") +
235
+ `; per-id exact-match displacements from top-3 (reported, not gated): ` +
236
+ `${args.comparison.lostExactMatches.length}` +
237
+ "\n");
238
+ }
239
+ L.push("| band | query | top-1 id |");
240
+ L.push("|---|---|---|");
241
+ for (const b of battery.queries) {
242
+ L.push(`| ${b.band} | ${b.q} | ${b.top8[0]?.slice(0, 8) ?? "-"} |`);
243
+ }
244
+ L.push("");
245
+ }
246
+ return L.join("\n");
247
+ }
248
+ // ---------------------------------------------------------------------------
249
+ // main
250
+ // ---------------------------------------------------------------------------
251
+ function usage() {
252
+ console.error("Usage: npm run eval:ranking -- [--snapshot <db>] [--photo <out.json>] [--compare <baseline.json>] [--scoring '<json>']");
253
+ process.exit(1);
254
+ }
255
+ async function main() {
256
+ const argv = process.argv.slice(2);
257
+ const flag = (name) => {
258
+ const i = argv.indexOf(name);
259
+ return i !== -1 && argv[i + 1] ? argv[i + 1] : undefined;
260
+ };
261
+ const snapshotPath = flag("--snapshot");
262
+ const photoPath = flag("--photo") ?? DEFAULT_PHOTO_PATH;
263
+ const comparePath = flag("--compare");
264
+ const scoringRaw = flag("--scoring");
265
+ if (argv.includes("--help") || argv.includes("-h"))
266
+ usage();
267
+ if (scoringRaw) {
268
+ try {
269
+ const parsed = JSON.parse(scoringRaw);
270
+ (0, retrieval_js_1.configureScoring)(parsed);
271
+ }
272
+ catch (err) {
273
+ console.error(`[ranking-eval] --scoring is not valid JSON: ${err instanceof Error ? err.message : err}`);
274
+ process.exit(1);
275
+ }
276
+ }
277
+ for (const f of [ranking_fixtures_js_1.SIRNAS_FIXTURE, ranking_fixtures_js_1.NO_MATCH_FIXTURE]) {
278
+ const problems = (0, ranking_fixtures_js_1.checkFixtureInvariants)(f);
279
+ if (problems.length > 0) {
280
+ console.error(`[ranking-eval] fixture ${f.key} invariant violations: ${problems.join("; ")}`);
281
+ process.exit(1);
282
+ }
283
+ }
284
+ console.log("[ranking-eval] loading real bge-small-en-v1.5 embedder...");
285
+ const t0 = Date.now();
286
+ // Cases 1 + 2 share one planted corpus DB (same rows; two queries).
287
+ const workDir = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-ranking-eval-"));
288
+ let case1;
289
+ let case2;
290
+ try {
291
+ const db = (0, db_js_1.initDb)((0, node_path_1.join)(workDir, "planted.db"));
292
+ try {
293
+ console.log("[ranking-eval] planting corpus (real embedder)...");
294
+ const idByKey = await plantRankingFixture(db, ranking_fixtures_js_1.SIRNAS_FIXTURE);
295
+ case1 = await measurePlantedCase(db, ranking_fixtures_js_1.SIRNAS_FIXTURE, idByKey);
296
+ const case2Measured = await measurePlantedCase(db, ranking_fixtures_js_1.NO_MATCH_FIXTURE, idByKey);
297
+ // The returned list keyed by fixture key (readable photos; ids would
298
+ // churn across replanting).
299
+ case2 = { ...case2Measured, ids: case2Measured.results.map((r) => r.key) };
300
+ }
301
+ finally {
302
+ db.close();
303
+ }
304
+ }
305
+ finally {
306
+ (0, node_fs_1.rmSync)(workDir, { recursive: true, force: true });
307
+ }
308
+ let battery = null;
309
+ let comparison = null;
310
+ let case2Stable = null;
311
+ // Content lookup for the exact-match gate (the SAME corpus both photos
312
+ // were recorded against — the snapshot, when comparing).
313
+ const contentById = new Map();
314
+ if (snapshotPath) {
315
+ console.log(`[ranking-eval] battery against readonly snapshot: ${snapshotPath}`);
316
+ const sdb = (0, eval_db_js_1.openSnapshot)(snapshotPath);
317
+ try {
318
+ battery = await runBattery(sdb);
319
+ for (const r of sdb
320
+ .prepare("SELECT id, content FROM memories WHERE COALESCE(status, '') != 'absorbed'")
321
+ .all()) {
322
+ contentById.set(r.id, r.content);
323
+ }
324
+ }
325
+ finally {
326
+ sdb.close();
327
+ }
328
+ }
329
+ if (comparePath) {
330
+ let baseline;
331
+ try {
332
+ baseline = JSON.parse((0, node_fs_1.readFileSync)(comparePath, "utf-8"));
333
+ }
334
+ catch (err) {
335
+ console.error(`[ranking-eval] cannot read baseline photo ${comparePath}: ${err instanceof Error ? err.message : err}`);
336
+ process.exit(1);
337
+ }
338
+ if (!baseline.case2 || !baseline.battery || !battery) {
339
+ console.error("[ranking-eval] baseline photo lacks case2/battery — record one with --photo on the same snapshot first");
340
+ process.exit(1);
341
+ }
342
+ case2Stable = (0, ranking_battery_js_1.compareCase2)(baseline.case2.ids, case2.ids);
343
+ comparison = (0, ranking_battery_js_1.batteryComparison)(baseline.battery, battery, contentById);
344
+ }
345
+ const photo = {
346
+ generatedAt: new Date().toISOString(),
347
+ scoring: (0, retrieval_js_1.getScoringWeights)(),
348
+ case1: {
349
+ pass: case1.pass,
350
+ expectedTop1: ranking_fixtures_js_1.SIRNAS_FIXTURE.expectedTop1,
351
+ top1Top2ScoreGap: case1.top1Top2ScoreGap,
352
+ results: case1.results,
353
+ },
354
+ case2: { ftsHitRows: case2.ftsHitRows, ids: case2.ids },
355
+ battery,
356
+ comparison,
357
+ case2Stable,
358
+ };
359
+ (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(photoPath), { recursive: true });
360
+ (0, node_fs_1.writeFileSync)(photoPath, JSON.stringify(photo, null, 2), "utf-8");
361
+ console.log(`[ranking-eval] photo written: ${photoPath}`);
362
+ const report = renderReport({
363
+ case1,
364
+ case2,
365
+ battery,
366
+ comparison,
367
+ case2Stable,
368
+ scoring: { ...photo.scoring },
369
+ });
370
+ console.log(report);
371
+ const reportPath = photoPath.endsWith(".json")
372
+ ? photoPath.slice(0, -5) + ".md"
373
+ : photoPath + ".md";
374
+ (0, node_fs_1.writeFileSync)(reportPath, report, "utf-8");
375
+ const batteryOk = comparison === null
376
+ ? true
377
+ : comparison.top1StabilityExPromotions >= BATTERY_TOP1_STABILITY_MIN &&
378
+ comparison.lostBothChannelExact.length === 0;
379
+ const case2Ok = case2Stable === null ? true : case2Stable;
380
+ const ok = case1.pass === true && case2Ok && batteryOk;
381
+ console.log(`[ranking-eval] ${ok ? "GATE PASS" : "GATE FAIL"} (case1 ${case1.pass ? "pass" : "FAIL"}, ` +
382
+ `case2 ${case2Ok ? "ok" : "DRIFT"}, battery ${batteryOk ? "ok" : "DRIFT"} ` +
383
+ `[top-1 ${((comparison?.top1Stability ?? 1) * 100).toFixed(1)}% raw / ` +
384
+ `${((comparison?.top1StabilityExPromotions ?? 1) * 100).toFixed(1)}% ex-promotions, ` +
385
+ `lost both-channel exact ${comparison?.lostBothChannelExact.length ?? 0}]) in ${Date.now() - t0}ms`);
386
+ process.exitCode = ok ? 0 : 1;
387
+ }
388
+ main().catch((err) => {
389
+ console.error("[ranking-eval] FAILED:", err instanceof Error ? err.stack : String(err));
390
+ process.exitCode = 1;
391
+ });
@@ -0,0 +1,77 @@
1
+ /**
2
+ * Planted ranking fixtures for the search-trust calibration gate (#425).
3
+ *
4
+ * Two fixture cases, both shaped like the field failure that opened the
5
+ * issue (the owner's live console test, 2026-09-13): a user searches a
6
+ * DISTINCTIVE PROPER NOUN they know exists; the one memory carrying that
7
+ * token has the best similarity of all candidates AND the literal FTS hit,
8
+ * yet ranks behind hardened but less-relevant memories that win on
9
+ * effective strength + graph connections.
10
+ *
11
+ * (a) sirnas_exact_match — the AC1 fixture: a low-strength, unlinked,
12
+ * never-accessed memory carrying an invented proper noun against
13
+ * base-≥0.9 rivals with links among themselves, access history, and
14
+ * older creation dates. The rivals share the query's DOMAIN
15
+ * vocabulary (harbours, docks, boats) but never the proper noun.
16
+ * Declared expectation: the proper-noun row ranks #1.
17
+ * (b) no_match_control — the AC2 fixture: a "mamma"-class query with no
18
+ * relevant match and NO FTS hits anywhere in the corpus. Whatever
19
+ * nearest-neighbor junk vector search returns must be byte-stable
20
+ * under the ranking work (no both-channel candidates exist, so the
21
+ * boost can never fire — the case pins that the no-regression
22
+ * guarantee is structural, not lucky).
23
+ *
24
+ * Honesty rules (the planted-pairs #393 A discipline):
25
+ * - The texts are synthetic and generic (this package publishes to npm —
26
+ * no real infrastructure, people, or fleet detail), but SHAPED like the
27
+ * field failure: the proper noun is invented (it must appear nowhere
28
+ * else — verified by an invariant test), the rivals are hardened with
29
+ * metadata the production pipeline itself writes (base strength, access
30
+ * hardening, link counts).
31
+ * - Similarities are MEASURED at run time with the real bge-small-en-v1.5
32
+ * embedder (ranking-eval.ts) — never asserted into existence. The
33
+ * declared expectation is about ORDER under the shipped ranking, and
34
+ * the report prints the measured similarity/score of every top row.
35
+ * - Strength metadata is written through the production helpers
36
+ * (insertMemory + updateMemory + addLink), not hand-rolled SQL, so the
37
+ * fixture exercises the same rows retrieval() will see.
38
+ */
39
+ export interface RankingRow {
40
+ /** Stable identifier used by expectations, links, and the report. */
41
+ key: string;
42
+ content: string;
43
+ /** ISO timestamp. */
44
+ createdAt: string;
45
+ /** knowledge | experience | decision. */
46
+ memoryType: string;
47
+ /** Birth base_strength — the value stageImportance would have written. */
48
+ baseStrength: number;
49
+ /** Access history (hardening + the decay clock). */
50
+ accessCount: number;
51
+ /** Days before the run's `now` of the last access (0 = touched today). */
52
+ lastAccessedDaysAgo: number;
53
+ /** Keys this row links to (links are created after all inserts). */
54
+ linksTo: string[];
55
+ }
56
+ export interface RankingFixture {
57
+ /** Case identifier (report + photo keys). */
58
+ key: string;
59
+ description: string;
60
+ rows: RankingRow[];
61
+ query: string;
62
+ /**
63
+ * The token that makes the query distinctive — invariant-tested to appear
64
+ * in exactly one row (the target) and in no rival/filler content.
65
+ */
66
+ distinctiveToken: string;
67
+ /** Declared expectation: this row's key must be returned at position 1. */
68
+ expectedTop1: string;
69
+ }
70
+ /** All planted rows share project + source_agent (metadata never refuses). */
71
+ export declare const RANKING_PROJECT = "ranking-eval";
72
+ export declare const RANKING_SOURCE_AGENT = "ranking-eval";
73
+ export declare const SIRNAS_FIXTURE: RankingFixture;
74
+ export declare const NO_MATCH_FIXTURE: RankingFixture;
75
+ /** The battery runs against a real snapshot copy (ranking-eval.ts case 3). */
76
+ export declare const REAL_SIRNAS_MEMORY_ID = "12980c9d-0eba-4963-95eb-0548bc2aa93e";
77
+ export declare function checkFixtureInvariants(f: RankingFixture): string[];