@gamaze/hicortex 0.20.9 → 0.20.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +8 -0
  2. package/assets/dashboard.html +3989 -836
  3. package/dist/calibration.d.ts +119 -0
  4. package/dist/calibration.js +149 -1
  5. package/dist/capture-health.d.ts +87 -0
  6. package/dist/capture-health.js +106 -0
  7. package/dist/capture-pause.d.ts +86 -0
  8. package/dist/capture-pause.js +127 -0
  9. package/dist/capture.d.ts +9 -0
  10. package/dist/capture.js +2 -1
  11. package/dist/cli.js +36 -0
  12. package/dist/consolidate.d.ts +35 -0
  13. package/dist/consolidate.js +85 -9
  14. package/dist/dashboard.d.ts +322 -3
  15. package/dist/dashboard.js +592 -7
  16. package/dist/db.js +105 -0
  17. package/dist/eval/importance-eval.d.ts +85 -0
  18. package/dist/eval/importance-eval.js +286 -0
  19. package/dist/eval/planted-fixtures.d.ts +1 -1
  20. package/dist/eval/ranking-battery.d.ts +78 -0
  21. package/dist/eval/ranking-battery.js +181 -0
  22. package/dist/eval/ranking-eval.d.ts +41 -0
  23. package/dist/eval/ranking-eval.js +391 -0
  24. package/dist/eval/ranking-fixtures.d.ts +77 -0
  25. package/dist/eval/ranking-fixtures.js +226 -0
  26. package/dist/identity-store.d.ts +21 -0
  27. package/dist/identity-store.js +49 -0
  28. package/dist/init.d.ts +14 -0
  29. package/dist/init.js +32 -0
  30. package/dist/mcp-server.d.ts +12 -0
  31. package/dist/mcp-server.js +184 -3
  32. package/dist/nightly.d.ts +9 -1
  33. package/dist/nightly.js +59 -7
  34. package/dist/prompts.d.ts +10 -0
  35. package/dist/prompts.js +28 -5
  36. package/dist/reconsolidation.d.ts +59 -30
  37. package/dist/reconsolidation.js +526 -296
  38. package/dist/rescore-importance.d.ts +80 -0
  39. package/dist/rescore-importance.js +236 -0
  40. package/dist/retrieval.d.ts +12 -0
  41. package/dist/retrieval.js +30 -1
  42. package/dist/stages.d.ts +37 -0
  43. package/dist/stages.js +51 -0
  44. package/dist/state.d.ts +32 -6
  45. package/dist/storage.d.ts +34 -2
  46. package/dist/storage.js +63 -6
  47. package/dist/types.d.ts +48 -0
  48. package/package.json +3 -1
@@ -0,0 +1,181 @@
1
+ "use strict";
2
+ /**
3
+ * The real-query battery for the #425 ranking gate — the DETERMINISTIC,
4
+ * dependency-free half (kept out of ranking-eval.ts so vitest can import
5
+ * it without pulling the ONNX embedder; same split as planted-fixtures vs
6
+ * planted-eval).
7
+ *
8
+ * The battery is a stable set of ~30 queries derived from the SNAPSHOT
9
+ * CORPUS ITSELF: fixed df windows over content tokens, fixed tie-breaks —
10
+ * same snapshot, same queries, every run. Bands: 10 high-df (50-400 docs),
11
+ * 10 mid-df (10-50), 5 rare-df (3-10), 4 mid-sentence capitalized proper
12
+ * nouns (df 2-8), plus the fixed "Sirnäs" query (the owner's live case).
13
+ *
14
+ * Also carries the photo-comparison gates the sweep protocol defines:
15
+ * - case 2 byte-stability (no-match queries must not drift),
16
+ * - battery top-1 stability >= 90%,
17
+ * - no query losing a both-channel exact match from its top-3.
18
+ */
19
+ Object.defineProperty(exports, "__esModule", { value: true });
20
+ exports.STOPWORDS = exports.SIRNAS_QUERY = void 0;
21
+ exports.deriveBatteryQueries = deriveBatteryQueries;
22
+ exports.compareCase2 = compareCase2;
23
+ exports.batteryComparison = batteryComparison;
24
+ exports.SIRNAS_QUERY = "Sirnäs";
25
+ /** Compact English stopword block (battery derivation only, not product). */
26
+ exports.STOPWORDS = new Set(("the a an and or but if while with without for from into onto about after before during under over " +
27
+ "this that these those there here when where which who whom whose what why how all any both each " +
28
+ "few more most other some such no nor not only own same so than too very can will just should now " +
29
+ "was were been being are was is am be have has had having do does did doing would could might must " +
30
+ "also because through between against above below off again further then once its it their they " +
31
+ "them we you your our i me my he she his her him us of as at by in on to per via due using use " +
32
+ "used using make made get gets got set sets let run ran running work works working need needs " +
33
+ "added add adds change changed new next last first second one two three session sessions " +
34
+ "memory memories project projects user users system systems thing things stuff lot bit way ways " +
35
+ "time times day days week weeks month months year years")
36
+ .split(/\s+/));
37
+ /**
38
+ * Derive the battery from live corpus contents — DETERMINISTIC given the
39
+ * snapshot: fixed stopword list, fixed df windows, fixed tie-breaks
40
+ * (df desc, then token asc).
41
+ */
42
+ function deriveBatteryQueries(rows) {
43
+ const docs = rows.map((r) => r.content.toLowerCase());
44
+ const df = new Map();
45
+ for (const doc of docs) {
46
+ const seen = new Set();
47
+ for (const m of doc.matchAll(/[\p{L}\p{N}]{5,}/gu)) {
48
+ const t = m[0];
49
+ if (exports.STOPWORDS.has(t) || seen.has(t))
50
+ continue;
51
+ seen.add(t);
52
+ df.set(t, (df.get(t) ?? 0) + 1);
53
+ }
54
+ }
55
+ const inWindow = (lo, hi) => [...df.entries()]
56
+ .filter(([t, d]) => d >= lo && d < hi && !exports.STOPWORDS.has(t))
57
+ .sort((a, b) => (b[1] - a[1] !== 0 ? b[1] - a[1] : a[0].localeCompare(b[0])))
58
+ .map(([t, d]) => [t, d]);
59
+ const queries = [
60
+ { q: exports.SIRNAS_QUERY, band: "proper-noun-fixed" },
61
+ ];
62
+ for (const [t] of inWindow(50, 401).slice(0, 10))
63
+ queries.push({ q: t, band: "high-df" });
64
+ for (const [t] of inWindow(10, 50).slice(0, 10))
65
+ queries.push({ q: t, band: "mid-df" });
66
+ for (const [t] of inWindow(3, 10).slice(0, 5))
67
+ queries.push({ q: t, band: "rare-df" });
68
+ // Proper nouns: capitalized, appearing mid-sentence (not the first word
69
+ // after a sentence break — cheap deterministic heuristic), df 2-8.
70
+ // "Sirnäs" itself may or may not surface here; the fixed query above
71
+ // guarantees coverage.
72
+ const capDf = new Map();
73
+ for (const content of rows.map((r) => r.content)) {
74
+ const seen = new Set();
75
+ for (const m of content.matchAll(/[^\s.!?;:]\s+([\p{Lu}][\p{L}\p{N}\p{M}-]{3,})/gu)) {
76
+ const t = m[1];
77
+ if (!exports.STOPWORDS.has(t.toLowerCase()))
78
+ seen.add(t);
79
+ }
80
+ for (const t of seen)
81
+ capDf.set(t, (capDf.get(t) ?? 0) + 1);
82
+ }
83
+ const properNouns = [...capDf.entries()]
84
+ .filter(([t, d]) => d >= 2 && d <= 8)
85
+ .sort((a, b) => (b[1] - a[1] !== 0 ? b[1] - a[1] : a[0].localeCompare(b[0])))
86
+ .map(([t]) => t);
87
+ for (const t of properNouns.slice(0, 4))
88
+ queries.push({ q: t, band: "proper-noun" });
89
+ return queries;
90
+ }
91
+ /** Case 2 gate: the FULL returned list must be byte-identical (ids + order). */
92
+ function compareCase2(baseline, current) {
93
+ return (baseline.length === current.length &&
94
+ baseline.every((id, i) => id === current[i]));
95
+ }
96
+ /** A both-channel row carrying the query term — the genuine-match signature. */
97
+ function isBothChannelExact(m, id, term, contentById) {
98
+ if (m)
99
+ return m.source === "both" && m.term === true;
100
+ // Baseline photos recorded before `meta` existed: fall back to content
101
+ // containment; conservatively treat source as unknown (not "both").
102
+ const content = contentById.get(id);
103
+ return Boolean(content && content.toLowerCase().includes(term));
104
+ }
105
+ /**
106
+ * Battery gates vs a recorded baseline photo. `contentById` maps memory id →
107
+ * content for the corpus BOTH photos were recorded against.
108
+ *
109
+ * Gates: (a) any query whose baseline top-3 held a BOTH-CHANNEL exact match
110
+ * (vector+FTS agreeing on a term-carrying row) must still hold one — the
111
+ * genuine-match guarantee never regresses; (b) top-1 stability. Intended D3
112
+ * promotions (a both-channel exact match taking top-1 from a non-exact or
113
+ * single-channel row) are counted separately — they are the feature firing,
114
+ * and the sweep record shows both numbers.
115
+ */
116
+ function batteryComparison(baseline, current, contentById) {
117
+ const baseByQ = new Map(baseline.queries.map((b) => [b.q, b]));
118
+ const baseMeta = (b, rank) => b.meta?.[rank];
119
+ let stable = 0;
120
+ let stableExPromotions = 0;
121
+ let compared = 0;
122
+ const changedTop1 = [];
123
+ const lostBothChannelExact = [];
124
+ const lostExactMatches = [];
125
+ for (const cur of current.queries) {
126
+ const base = baseByQ.get(cur.q);
127
+ if (!base)
128
+ continue;
129
+ compared++;
130
+ if (cur.top8[0] === base.top8[0]) {
131
+ stable++;
132
+ stableExPromotions++;
133
+ continue;
134
+ }
135
+ changedTop1.push(cur.q);
136
+ const term = cur.q.toLowerCase();
137
+ const oldExact = baseMeta(base, 0) !== undefined
138
+ ? isBothChannelExact(baseMeta(base, 0), base.top8[0], term, contentById)
139
+ : null;
140
+ const newExact = cur.meta?.[0] !== undefined
141
+ ? isBothChannelExact(cur.meta[0], cur.top8[0], term, contentById)
142
+ : null;
143
+ // An intended promotion: the new top-1 is a both-channel exact match the
144
+ // old top-1 was not. (Null = provenance unknown — pre-meta baseline —
145
+ // counts as a plain change.)
146
+ if (newExact === true && oldExact === false)
147
+ stableExPromotions++;
148
+ }
149
+ for (const cur of current.queries) {
150
+ const base = baseByQ.get(cur.q);
151
+ if (!base || cur.q === exports.SIRNAS_QUERY)
152
+ continue;
153
+ const term = cur.q.toLowerCase();
154
+ const curTop3 = new Set(cur.top8.slice(0, 3));
155
+ let baseHadBothChannelExact = false;
156
+ for (let i = 0; i < Math.min(3, base.top8.length); i++) {
157
+ const id = base.top8[i];
158
+ const content = contentById.get(id);
159
+ const carries = Boolean(content && content.toLowerCase().includes(term));
160
+ if (carries && !curTop3.has(id))
161
+ lostExactMatches.push(`${cur.q}: ${id.slice(0, 8)}`);
162
+ const m = base.meta?.[i];
163
+ if (m ? m.source === "both" && m.term === true : false)
164
+ baseHadBothChannelExact = true;
165
+ }
166
+ if (baseHadBothChannelExact) {
167
+ const stillHas = cur.top8
168
+ .slice(0, 3)
169
+ .some((id, i) => cur.meta?.[i]?.source === "both" && cur.meta?.[i]?.term === true);
170
+ if (!stillHas)
171
+ lostBothChannelExact.push(cur.q);
172
+ }
173
+ }
174
+ return {
175
+ top1Stability: compared > 0 ? stable / compared : 1,
176
+ top1StabilityExPromotions: compared > 0 ? stableExPromotions / compared : 1,
177
+ changedTop1,
178
+ lostBothChannelExact,
179
+ lostExactMatches,
180
+ };
181
+ }
@@ -0,0 +1,41 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Search-trust ranking gate (#425) — the AC1/AC2/AC3 before/after photo.
4
+ *
5
+ * npm run eval:ranking [--snapshot <db>] [--photo <out.json>]
6
+ * [--compare <baseline.json>] [--scoring '<json>']
7
+ *
8
+ * --snapshot <db> enables case 3: the deterministic real-query battery
9
+ * against a READONLY snapshot (openSnapshot — never
10
+ * initDb; the snapshot is never touched). Also reports
11
+ * where the real "Sirnäs" memory ranks today (the
12
+ * measured red photo pre-fix, AC10 pre-condition).
13
+ * --photo <out.json> where to write the machine photo (defaults to
14
+ * data/ranking-eval-photo.json). ALWAYS written, even
15
+ * on a red run — the photo IS the measurement.
16
+ * --compare <json> gate against a previously recorded photo: case 2's
17
+ * returned list must be byte-identical, and the battery
18
+ * must hold top-1 stability >= 90% with no query losing
19
+ * a both-channel exact match from its top-3.
20
+ * --scoring '<json>' a JSON object of ScoringWeights overrides passed to
21
+ * configureScoring (the eval/test seam, #408) — the
22
+ * sweep knob. Production runs omit it.
23
+ *
24
+ * Cases (fixtures: ranking-fixtures.ts, real bge-small-en-v1.5 embedder):
25
+ * 1. sirnas_exact_match — planted corpus in a THROWAWAY writable DB;
26
+ * declared expectation: the proper-noun exact-match row returns #1.
27
+ * Exit code reflects the verdict (a pre-fix run is EXPECTED to exit
28
+ * non-zero — that red photo is the before picture).
29
+ * 2. no_match_control — same corpus, a "mamma"-class query with zero FTS
30
+ * hits; records the full returned list for byte-stability comparison.
31
+ * 3. real-query battery (needs --snapshot) — ~30 queries derived
32
+ * deterministically from the snapshot corpus (top/mid/rare-frequency
33
+ * distinctive terms + proper nouns) + the fixed "Sirnäs" query; top-8
34
+ * ids per query recorded to the photo.
35
+ *
36
+ * Honesty invariants: similarities are MEASURED (real embedder, embed-once,
37
+ * queryEmbedding reuse), noStrengthen on every retrieve() (the probe never
38
+ * perturbs the store), readonly snapshot access, and the planted corpora
39
+ * live in mkdtemp dirs removed on exit. Never touches ~/.hicortex.
40
+ */
41
+ export {};
@@ -0,0 +1,391 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * Search-trust ranking gate (#425) — the AC1/AC2/AC3 before/after photo.
5
+ *
6
+ * npm run eval:ranking [--snapshot <db>] [--photo <out.json>]
7
+ * [--compare <baseline.json>] [--scoring '<json>']
8
+ *
9
+ * --snapshot <db> enables case 3: the deterministic real-query battery
10
+ * against a READONLY snapshot (openSnapshot — never
11
+ * initDb; the snapshot is never touched). Also reports
12
+ * where the real "Sirnäs" memory ranks today (the
13
+ * measured red photo pre-fix, AC10 pre-condition).
14
+ * --photo <out.json> where to write the machine photo (defaults to
15
+ * data/ranking-eval-photo.json). ALWAYS written, even
16
+ * on a red run — the photo IS the measurement.
17
+ * --compare <json> gate against a previously recorded photo: case 2's
18
+ * returned list must be byte-identical, and the battery
19
+ * must hold top-1 stability >= 90% with no query losing
20
+ * a both-channel exact match from its top-3.
21
+ * --scoring '<json>' a JSON object of ScoringWeights overrides passed to
22
+ * configureScoring (the eval/test seam, #408) — the
23
+ * sweep knob. Production runs omit it.
24
+ *
25
+ * Cases (fixtures: ranking-fixtures.ts, real bge-small-en-v1.5 embedder):
26
+ * 1. sirnas_exact_match — planted corpus in a THROWAWAY writable DB;
27
+ * declared expectation: the proper-noun exact-match row returns #1.
28
+ * Exit code reflects the verdict (a pre-fix run is EXPECTED to exit
29
+ * non-zero — that red photo is the before picture).
30
+ * 2. no_match_control — same corpus, a "mamma"-class query with zero FTS
31
+ * hits; records the full returned list for byte-stability comparison.
32
+ * 3. real-query battery (needs --snapshot) — ~30 queries derived
33
+ * deterministically from the snapshot corpus (top/mid/rare-frequency
34
+ * distinctive terms + proper nouns) + the fixed "Sirnäs" query; top-8
35
+ * ids per query recorded to the photo.
36
+ *
37
+ * Honesty invariants: similarities are MEASURED (real embedder, embed-once,
38
+ * queryEmbedding reuse), noStrengthen on every retrieve() (the probe never
39
+ * perturbs the store), readonly snapshot access, and the planted corpora
40
+ * live in mkdtemp dirs removed on exit. Never touches ~/.hicortex.
41
+ */
42
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
43
+ if (k2 === undefined) k2 = k;
44
+ var desc = Object.getOwnPropertyDescriptor(m, k);
45
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
46
+ desc = { enumerable: true, get: function() { return m[k]; } };
47
+ }
48
+ Object.defineProperty(o, k2, desc);
49
+ }) : (function(o, m, k, k2) {
50
+ if (k2 === undefined) k2 = k;
51
+ o[k2] = m[k];
52
+ }));
53
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
54
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
55
+ }) : function(o, v) {
56
+ o["default"] = v;
57
+ });
58
+ var __importStar = (this && this.__importStar) || (function () {
59
+ var ownKeys = function(o) {
60
+ ownKeys = Object.getOwnPropertyNames || function (o) {
61
+ var ar = [];
62
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
63
+ return ar;
64
+ };
65
+ return ownKeys(o);
66
+ };
67
+ return function (mod) {
68
+ if (mod && mod.__esModule) return mod;
69
+ var result = {};
70
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
71
+ __setModuleDefault(result, mod);
72
+ return result;
73
+ };
74
+ })();
75
+ Object.defineProperty(exports, "__esModule", { value: true });
76
+ const node_fs_1 = require("node:fs");
77
+ const node_os_1 = require("node:os");
78
+ const node_path_1 = require("node:path");
79
+ const embedder_js_1 = require("../embedder.js");
80
+ const db_js_1 = require("../db.js");
81
+ const storage = __importStar(require("../storage.js"));
82
+ const retrieval_js_1 = require("../retrieval.js");
83
+ const eval_db_js_1 = require("./eval-db.js");
84
+ const ranking_fixtures_js_1 = require("./ranking-fixtures.js");
85
+ const ranking_battery_js_1 = require("./ranking-battery.js");
86
+ const DEFAULT_PHOTO_PATH = (0, node_path_1.join)(process.cwd(), "data", "ranking-eval-photo.json");
87
+ /** Battery gate thresholds (sweep protocol, issue #425 commit-3 gates). */
88
+ const BATTERY_TOP1_STABILITY_MIN = 0.9;
89
+ /** Plant one fixture corpus into a writable DB with its hardened metadata. */
90
+ async function plantRankingFixture(db, fixture) {
91
+ const idByKey = new Map();
92
+ const now = Date.now();
93
+ for (const row of fixture.rows) {
94
+ const embedding = await (0, embedder_js_1.embed)(row.content);
95
+ const id = storage.insertMemory(db, row.content, embedding, {
96
+ sourceAgent: ranking_fixtures_js_1.RANKING_SOURCE_AGENT,
97
+ project: ranking_fixtures_js_1.RANKING_PROJECT,
98
+ memoryType: row.memoryType,
99
+ createdAt: row.createdAt,
100
+ });
101
+ const lastAccessed = new Date(now - row.lastAccessedDaysAgo * 86_400_000).toISOString();
102
+ storage.updateMemory(db, id, {
103
+ base_strength: row.baseStrength,
104
+ access_count: row.accessCount,
105
+ last_accessed: lastAccessed,
106
+ });
107
+ idByKey.set(row.key, id);
108
+ }
109
+ for (const row of fixture.rows) {
110
+ for (const to of row.linksTo) {
111
+ const sourceId = idByKey.get(row.key);
112
+ const targetId = idByKey.get(to);
113
+ if (sourceId && targetId)
114
+ storage.addLink(db, sourceId, targetId, "relates_to", 0.7);
115
+ }
116
+ }
117
+ return idByKey;
118
+ }
119
+ /**
120
+ * Run one planted case: full-corpus retrieve (limit = rows + headroom, so
121
+ * the cold-exposure swap is inert — scored.length <= limit — and the
122
+ * returned order IS the score order), results mapped back to fixture keys.
123
+ */
124
+ async function measurePlantedCase(db, fixture, idByKey) {
125
+ const keyById = new Map([...idByKey.entries()].map(([k, v]) => [v, k]));
126
+ const queryEmb = await (0, embedder_js_1.embed)(fixture.query);
127
+ const ftsRows = storage.searchFts(db, fixture.query, 100, undefined);
128
+ const results = await (0, retrieval_js_1.retrieve)(db, embedder_js_1.embed, fixture.query, {
129
+ limit: fixture.rows.length + 8,
130
+ noStrengthen: true,
131
+ queryEmbedding: queryEmb,
132
+ });
133
+ const measured = results.map((r, i) => ({
134
+ key: keyById.get(r.id) ?? r.id,
135
+ rank: i + 1,
136
+ score: Math.round(r.score * 1e6) / 1e6,
137
+ similarity: r.similarity ?? null,
138
+ source: r.source ?? null,
139
+ effectiveStrength: Math.round(r.effective_strength * 1e6) / 1e6,
140
+ baseStrength: fixture.rows.find((x) => x.key === keyById.get(r.id))?.baseStrength ?? -1,
141
+ accessCount: r.access_count,
142
+ }));
143
+ const pass = fixture.expectedTop1
144
+ ? measured.length > 0 && measured[0].key === fixture.expectedTop1
145
+ : null;
146
+ const gap = measured.length >= 2
147
+ ? Math.round((measured[0].score - measured[1].score) * 1e6) / 1e6
148
+ : null;
149
+ return {
150
+ pass,
151
+ query: fixture.query,
152
+ ftsHitRows: ftsRows.length,
153
+ results: measured,
154
+ top1Top2ScoreGap: gap,
155
+ };
156
+ }
157
+ async function runBattery(db) {
158
+ const rows = db
159
+ .prepare("SELECT id, content FROM memories WHERE COALESCE(status, '') != 'absorbed'")
160
+ .all();
161
+ const queries = (0, ranking_battery_js_1.deriveBatteryQueries)(rows);
162
+ const out = [];
163
+ let sirnasRank = null;
164
+ for (const { q, band } of queries) {
165
+ const queryEmb = await (0, embedder_js_1.embed)(q);
166
+ const results = await (0, retrieval_js_1.retrieve)(db, embedder_js_1.embed, q, {
167
+ limit: 8,
168
+ noStrengthen: true,
169
+ queryEmbedding: queryEmb,
170
+ });
171
+ const top8 = results.map((r) => r.id);
172
+ // Per-result provenance for the sweep analysis: the channel that
173
+ // produced the row + (for term queries) whether the row's content
174
+ // carries the query term — lets the gate distinguish "the D3 property
175
+ // fired" (a both-channel exact match displaced a single-channel top-1)
176
+ // from a real regression.
177
+ const lower = q.toLowerCase();
178
+ const meta = results.map((r) => ({
179
+ id: r.id,
180
+ source: r.source ?? null,
181
+ term: q === ranking_battery_js_1.SIRNAS_QUERY
182
+ ? null
183
+ : (rows.find((x) => x.id === r.id)?.content ?? "").toLowerCase().includes(lower),
184
+ }));
185
+ out.push({ q, band, top8, meta });
186
+ if (q === ranking_battery_js_1.SIRNAS_QUERY) {
187
+ const idx = top8.indexOf(ranking_fixtures_js_1.REAL_SIRNAS_MEMORY_ID);
188
+ sirnasRank = idx >= 0 ? idx + 1 : null;
189
+ }
190
+ }
191
+ return { queries: out, sirnasRank };
192
+ }
193
+ // ---------------------------------------------------------------------------
194
+ // Report rendering
195
+ // ---------------------------------------------------------------------------
196
+ function renderReport(args) {
197
+ const L = [];
198
+ const { case1, case2, battery } = args;
199
+ L.push("# Search-trust ranking gate (#425)\n");
200
+ L.push(`Generated: ${new Date().toISOString()} \nScoring: ${JSON.stringify(args.scoring)}\n`);
201
+ L.push("## Case 1 — sirnas_exact_match (AC1)\n");
202
+ L.push(`Query "${case1.query}" — declared top-1: ${ranking_fixtures_js_1.SIRNAS_FIXTURE.expectedTop1} — ` +
203
+ `**${case1.pass ? "PASS" : "FAIL"}**${case1.top1Top2ScoreGap !== null ? ` (top1-top2 score gap ${case1.top1Top2ScoreGap.toFixed(4)})` : ""}\n`);
204
+ L.push("| rank | key | similarity | source | score | effStr | base | access |");
205
+ L.push("|---|---|---|---|---|---|---|---|");
206
+ for (const r of case1.results.slice(0, 6)) {
207
+ L.push(`| ${r.rank} | ${r.key} | ${r.similarity?.toFixed(4) ?? "-"} | ${r.source ?? "-"} | ` +
208
+ `${r.score.toFixed(4)} | ${r.effectiveStrength.toFixed(4)} | ${r.baseStrength} | ${r.accessCount} |`);
209
+ }
210
+ L.push("");
211
+ L.push("## Case 2 — no_match_control (AC2)\n");
212
+ L.push(`Query "${case2.query}" — FTS hits: ${case2.ftsHitRows} (must be 0 — no both-channel ` +
213
+ `candidates can exist)` +
214
+ (args.case2Stable === null ? "" : ` — vs baseline: **${args.case2Stable ? "identical" : "DRIFTED"}**`) +
215
+ "\n");
216
+ L.push("Returned order: " + case2.ids.map((k) => k).join(" → ") + "\n");
217
+ if (battery) {
218
+ L.push("## Case 3 — real-query battery (AC3 no-regression photo)\n");
219
+ L.push(`Queries: ${battery.queries.length}; Sirnäs rank for "${ranking_battery_js_1.SIRNAS_QUERY}": ` +
220
+ (battery.sirnasRank === null ? "not in top-8" : `#${battery.sirnasRank}`) + "\n");
221
+ if (args.comparison) {
222
+ L.push(`vs baseline: top-1 stability ${(args.comparison.top1Stability * 100).toFixed(1)}% ` +
223
+ `(gate >= ${(BATTERY_TOP1_STABILITY_MIN * 100).toFixed(0)}%; ` +
224
+ `excluding intended D3 promotions ` +
225
+ `${(args.comparison.top1StabilityExPromotions * 100).toFixed(1)}%) — ` +
226
+ `${args.comparison.changedTop1.length} changed top-1` +
227
+ (args.comparison.changedTop1.length > 0
228
+ ? ` [${args.comparison.changedTop1.join(", ")}]`
229
+ : "") +
230
+ `; queries that LOST their both-channel exact match from top-3: ` +
231
+ `${args.comparison.lostBothChannelExact.length}` +
232
+ (args.comparison.lostBothChannelExact.length > 0
233
+ ? ` [${args.comparison.lostBothChannelExact.join(", ")}]`
234
+ : "") +
235
+ `; per-id exact-match displacements from top-3 (reported, not gated): ` +
236
+ `${args.comparison.lostExactMatches.length}` +
237
+ "\n");
238
+ }
239
+ L.push("| band | query | top-1 id |");
240
+ L.push("|---|---|---|");
241
+ for (const b of battery.queries) {
242
+ L.push(`| ${b.band} | ${b.q} | ${b.top8[0]?.slice(0, 8) ?? "-"} |`);
243
+ }
244
+ L.push("");
245
+ }
246
+ return L.join("\n");
247
+ }
248
+ // ---------------------------------------------------------------------------
249
+ // main
250
+ // ---------------------------------------------------------------------------
251
+ function usage() {
252
+ console.error("Usage: npm run eval:ranking -- [--snapshot <db>] [--photo <out.json>] [--compare <baseline.json>] [--scoring '<json>']");
253
+ process.exit(1);
254
+ }
255
+ async function main() {
256
+ const argv = process.argv.slice(2);
257
+ const flag = (name) => {
258
+ const i = argv.indexOf(name);
259
+ return i !== -1 && argv[i + 1] ? argv[i + 1] : undefined;
260
+ };
261
+ const snapshotPath = flag("--snapshot");
262
+ const photoPath = flag("--photo") ?? DEFAULT_PHOTO_PATH;
263
+ const comparePath = flag("--compare");
264
+ const scoringRaw = flag("--scoring");
265
+ if (argv.includes("--help") || argv.includes("-h"))
266
+ usage();
267
+ if (scoringRaw) {
268
+ try {
269
+ const parsed = JSON.parse(scoringRaw);
270
+ (0, retrieval_js_1.configureScoring)(parsed);
271
+ }
272
+ catch (err) {
273
+ console.error(`[ranking-eval] --scoring is not valid JSON: ${err instanceof Error ? err.message : err}`);
274
+ process.exit(1);
275
+ }
276
+ }
277
+ for (const f of [ranking_fixtures_js_1.SIRNAS_FIXTURE, ranking_fixtures_js_1.NO_MATCH_FIXTURE]) {
278
+ const problems = (0, ranking_fixtures_js_1.checkFixtureInvariants)(f);
279
+ if (problems.length > 0) {
280
+ console.error(`[ranking-eval] fixture ${f.key} invariant violations: ${problems.join("; ")}`);
281
+ process.exit(1);
282
+ }
283
+ }
284
+ console.log("[ranking-eval] loading real bge-small-en-v1.5 embedder...");
285
+ const t0 = Date.now();
286
+ // Cases 1 + 2 share one planted corpus DB (same rows; two queries).
287
+ const workDir = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-ranking-eval-"));
288
+ let case1;
289
+ let case2;
290
+ try {
291
+ const db = (0, db_js_1.initDb)((0, node_path_1.join)(workDir, "planted.db"));
292
+ try {
293
+ console.log("[ranking-eval] planting corpus (real embedder)...");
294
+ const idByKey = await plantRankingFixture(db, ranking_fixtures_js_1.SIRNAS_FIXTURE);
295
+ case1 = await measurePlantedCase(db, ranking_fixtures_js_1.SIRNAS_FIXTURE, idByKey);
296
+ const case2Measured = await measurePlantedCase(db, ranking_fixtures_js_1.NO_MATCH_FIXTURE, idByKey);
297
+ // The returned list keyed by fixture key (readable photos; ids would
298
+ // churn across replanting).
299
+ case2 = { ...case2Measured, ids: case2Measured.results.map((r) => r.key) };
300
+ }
301
+ finally {
302
+ db.close();
303
+ }
304
+ }
305
+ finally {
306
+ (0, node_fs_1.rmSync)(workDir, { recursive: true, force: true });
307
+ }
308
+ let battery = null;
309
+ let comparison = null;
310
+ let case2Stable = null;
311
+ // Content lookup for the exact-match gate (the SAME corpus both photos
312
+ // were recorded against — the snapshot, when comparing).
313
+ const contentById = new Map();
314
+ if (snapshotPath) {
315
+ console.log(`[ranking-eval] battery against readonly snapshot: ${snapshotPath}`);
316
+ const sdb = (0, eval_db_js_1.openSnapshot)(snapshotPath);
317
+ try {
318
+ battery = await runBattery(sdb);
319
+ for (const r of sdb
320
+ .prepare("SELECT id, content FROM memories WHERE COALESCE(status, '') != 'absorbed'")
321
+ .all()) {
322
+ contentById.set(r.id, r.content);
323
+ }
324
+ }
325
+ finally {
326
+ sdb.close();
327
+ }
328
+ }
329
+ if (comparePath) {
330
+ let baseline;
331
+ try {
332
+ baseline = JSON.parse((0, node_fs_1.readFileSync)(comparePath, "utf-8"));
333
+ }
334
+ catch (err) {
335
+ console.error(`[ranking-eval] cannot read baseline photo ${comparePath}: ${err instanceof Error ? err.message : err}`);
336
+ process.exit(1);
337
+ }
338
+ if (!baseline.case2 || !baseline.battery || !battery) {
339
+ console.error("[ranking-eval] baseline photo lacks case2/battery — record one with --photo on the same snapshot first");
340
+ process.exit(1);
341
+ }
342
+ case2Stable = (0, ranking_battery_js_1.compareCase2)(baseline.case2.ids, case2.ids);
343
+ comparison = (0, ranking_battery_js_1.batteryComparison)(baseline.battery, battery, contentById);
344
+ }
345
+ const photo = {
346
+ generatedAt: new Date().toISOString(),
347
+ scoring: (0, retrieval_js_1.getScoringWeights)(),
348
+ case1: {
349
+ pass: case1.pass,
350
+ expectedTop1: ranking_fixtures_js_1.SIRNAS_FIXTURE.expectedTop1,
351
+ top1Top2ScoreGap: case1.top1Top2ScoreGap,
352
+ results: case1.results,
353
+ },
354
+ case2: { ftsHitRows: case2.ftsHitRows, ids: case2.ids },
355
+ battery,
356
+ comparison,
357
+ case2Stable,
358
+ };
359
+ (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(photoPath), { recursive: true });
360
+ (0, node_fs_1.writeFileSync)(photoPath, JSON.stringify(photo, null, 2), "utf-8");
361
+ console.log(`[ranking-eval] photo written: ${photoPath}`);
362
+ const report = renderReport({
363
+ case1,
364
+ case2,
365
+ battery,
366
+ comparison,
367
+ case2Stable,
368
+ scoring: { ...photo.scoring },
369
+ });
370
+ console.log(report);
371
+ const reportPath = photoPath.endsWith(".json")
372
+ ? photoPath.slice(0, -5) + ".md"
373
+ : photoPath + ".md";
374
+ (0, node_fs_1.writeFileSync)(reportPath, report, "utf-8");
375
+ const batteryOk = comparison === null
376
+ ? true
377
+ : comparison.top1StabilityExPromotions >= BATTERY_TOP1_STABILITY_MIN &&
378
+ comparison.lostBothChannelExact.length === 0;
379
+ const case2Ok = case2Stable === null ? true : case2Stable;
380
+ const ok = case1.pass === true && case2Ok && batteryOk;
381
+ console.log(`[ranking-eval] ${ok ? "GATE PASS" : "GATE FAIL"} (case1 ${case1.pass ? "pass" : "FAIL"}, ` +
382
+ `case2 ${case2Ok ? "ok" : "DRIFT"}, battery ${batteryOk ? "ok" : "DRIFT"} ` +
383
+ `[top-1 ${((comparison?.top1Stability ?? 1) * 100).toFixed(1)}% raw / ` +
384
+ `${((comparison?.top1StabilityExPromotions ?? 1) * 100).toFixed(1)}% ex-promotions, ` +
385
+ `lost both-channel exact ${comparison?.lostBothChannelExact.length ?? 0}]) in ${Date.now() - t0}ms`);
386
+ process.exitCode = ok ? 0 : 1;
387
+ }
388
+ main().catch((err) => {
389
+ console.error("[ranking-eval] FAILED:", err instanceof Error ? err.stack : String(err));
390
+ process.exitCode = 1;
391
+ });