@gamaze/hicortex 0.21.0 → 0.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +282 -27
- package/dist/calibration.d.ts +77 -12
- package/dist/calibration.js +82 -15
- package/dist/capture-health.d.ts +42 -5
- package/dist/capture-health.js +57 -7
- package/dist/capture-pause.d.ts +7 -4
- package/dist/capture-pause.js +7 -4
- package/dist/consolidate.d.ts +19 -0
- package/dist/consolidate.js +125 -51
- package/dist/dashboard.d.ts +41 -16
- package/dist/dashboard.js +110 -41
- package/dist/db.js +16 -0
- package/dist/dedup.js +4 -1
- package/dist/eval/decay-eval.d.ts +6 -5
- package/dist/eval/decay-eval.js +10 -48
- package/dist/eval/eval-clock.d.ts +32 -0
- package/dist/eval/eval-clock.js +47 -0
- package/dist/eval/graph-eval.d.ts +15 -2
- package/dist/eval/graph-eval.js +51 -5
- package/dist/eval/planted-eval.d.ts +4 -0
- package/dist/eval/planted-eval.js +27 -2
- package/dist/eval/planted-harness.d.ts +7 -0
- package/dist/eval/planted-harness.js +2 -0
- package/dist/eval/ranking-battery.d.ts +49 -2
- package/dist/eval/ranking-battery.js +110 -2
- package/dist/eval/ranking-eval.d.ts +26 -6
- package/dist/eval/ranking-eval.js +197 -34
- package/dist/eval/ranking-fixtures.d.ts +46 -6
- package/dist/eval/ranking-fixtures.js +268 -9
- package/dist/eval/recall-sweep.d.ts +7 -2
- package/dist/eval/recall-sweep.js +42 -13
- package/dist/eval/relevance-eval.d.ts +115 -1
- package/dist/eval/relevance-eval.js +318 -32
- package/dist/eval/run-eval.d.ts +7 -4
- package/dist/eval/run-eval.js +37 -10
- package/dist/graph.d.ts +1 -2
- package/dist/graph.js +3 -4
- package/dist/mcp-server.js +3 -2
- package/dist/nofit.d.ts +2 -1
- package/dist/nofit.js +2 -1
- package/dist/recall-index.d.ts +3 -2
- package/dist/recall-index.js +3 -2
- package/dist/retrieval.d.ts +43 -18
- package/dist/retrieval.js +148 -86
- package/dist/status.js +18 -0
- package/dist/storage.d.ts +2 -1
- package/dist/storage.js +6 -1
- package/dist/types.d.ts +25 -4
- package/package.json +1 -1
- package/server.json +4 -4
|
@@ -2,11 +2,15 @@
|
|
|
2
2
|
"use strict";
|
|
3
3
|
/**
|
|
4
4
|
* Search-trust ranking gate (#425) — the AC1/AC2/AC3 before/after photo.
|
|
5
|
+
* Extended by #449 (items 1+3, PR E): the case-2 photo gains per-row
|
|
6
|
+
* results (scores + measured connections) and its gate becomes the
|
|
7
|
+
* unlinked-row identity comparator; a new planted case 3 (linked_fallback)
|
|
8
|
+
* pins the connections-reshape expectations.
|
|
5
9
|
*
|
|
6
10
|
* npm run eval:ranking [--snapshot <db>] [--photo <out.json>]
|
|
7
11
|
* [--compare <baseline.json>] [--scoring '<json>']
|
|
8
12
|
*
|
|
9
|
-
* --snapshot <db> enables case
|
|
13
|
+
* --snapshot <db> enables case 4: the deterministic real-query battery
|
|
10
14
|
* against a READONLY snapshot (openSnapshot — never
|
|
11
15
|
* initDb; the snapshot is never touched). Also reports
|
|
12
16
|
* where the real "Sirnäs" memory ranks today (the
|
|
@@ -15,12 +19,19 @@
|
|
|
15
19
|
* data/ranking-eval-photo.json). ALWAYS written, even
|
|
16
20
|
* on a red run — the photo IS the measurement.
|
|
17
21
|
* --compare <json> gate against a previously recorded photo: case 2's
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
22
|
+
* UNLINKED rows must keep their scores (within the
|
|
23
|
+
* wall-clock tolerance) and mutual order — linked-row
|
|
24
|
+
* order drifts BY DESIGN under #449 — and the battery
|
|
25
|
+
* (when both sides have one) must hold top-1
|
|
26
|
+
* stability >= 90% with no query losing a both-channel
|
|
27
|
+
* exact match from its top-3.
|
|
21
28
|
* --scoring '<json>' a JSON object of ScoringWeights overrides passed to
|
|
22
29
|
* configureScoring (the eval/test seam, #408) — the
|
|
23
30
|
* sweep knob. Production runs omit it.
|
|
31
|
+
* --now <ISO> #458: pin the eval clock — planted lastAccessed offsets
|
|
32
|
+
* AND every retrieve() score against the SAME instant,
|
|
33
|
+
* so before/after photos are wall-clock-independent.
|
|
34
|
+
* Default: the live clock. The photo records the clock.
|
|
24
35
|
*
|
|
25
36
|
* Cases (fixtures: ranking-fixtures.ts, real bge-small-en-v1.5 embedder):
|
|
26
37
|
* 1. sirnas_exact_match — planted corpus in a THROWAWAY writable DB;
|
|
@@ -28,8 +39,17 @@
|
|
|
28
39
|
* Exit code reflects the verdict (a pre-fix run is EXPECTED to exit
|
|
29
40
|
* non-zero — that red photo is the before picture).
|
|
30
41
|
* 2. no_match_control — same corpus, a "mamma"-class query with zero FTS
|
|
31
|
-
* hits; records the full returned list
|
|
32
|
-
*
|
|
42
|
+
* hits; records the full returned list (ids + per-row results) for the
|
|
43
|
+
* unlinked-identity comparison.
|
|
44
|
+
* 3. linked_fallback (#449) — planted corpus in its OWN throwaway DB: a
|
|
45
|
+
* 16-link vs 20-link identical-content hub pair (saturation tie,
|
|
46
|
+
* declared byte-equal post-change) + an unlinked higher-similarity row
|
|
47
|
+
* vs a 4-link lower-similarity row (direction pin). Measured under the
|
|
48
|
+
* eval seam rrfCompositeWeight=1 (finalScore = composite exactly —
|
|
49
|
+
* identical-content hubs differ by an RRF epsilon at the default
|
|
50
|
+
* blend). RED under the pre-#449 linear term: the recorded
|
|
51
|
+
* before-picture.
|
|
52
|
+
* 4. real-query battery (needs --snapshot) — ~30 queries derived
|
|
33
53
|
* deterministically from the snapshot corpus (top/mid/rare-frequency
|
|
34
54
|
* distinctive terms + proper nouns) + the fixed "Sirnäs" query; top-8
|
|
35
55
|
* ids per query recorded to the photo.
|
|
@@ -81,15 +101,18 @@ const db_js_1 = require("../db.js");
|
|
|
81
101
|
const storage = __importStar(require("../storage.js"));
|
|
82
102
|
const retrieval_js_1 = require("../retrieval.js");
|
|
83
103
|
const eval_db_js_1 = require("./eval-db.js");
|
|
104
|
+
const eval_clock_js_1 = require("./eval-clock.js");
|
|
84
105
|
const ranking_fixtures_js_1 = require("./ranking-fixtures.js");
|
|
85
106
|
const ranking_battery_js_1 = require("./ranking-battery.js");
|
|
86
107
|
const DEFAULT_PHOTO_PATH = (0, node_path_1.join)(process.cwd(), "data", "ranking-eval-photo.json");
|
|
87
108
|
/** Battery gate thresholds (sweep protocol, issue #425 commit-3 gates). */
|
|
88
109
|
const BATTERY_TOP1_STABILITY_MIN = 0.9;
|
|
89
|
-
/** Plant one fixture corpus into a writable DB with its
|
|
90
|
-
|
|
110
|
+
/** Plant one fixture corpus into a writable DB with its seeded strength metadata.
|
|
111
|
+
* `now` (#458): the instant the planted lastAccessed offsets are measured
|
|
112
|
+
* from — the pinned clock when --now was passed, live otherwise. */
|
|
113
|
+
async function plantRankingFixture(db, fixture, now) {
|
|
91
114
|
const idByKey = new Map();
|
|
92
|
-
const
|
|
115
|
+
const nowMs = (now ?? new Date()).getTime();
|
|
93
116
|
for (const row of fixture.rows) {
|
|
94
117
|
const embedding = await (0, embedder_js_1.embed)(row.content);
|
|
95
118
|
const id = storage.insertMemory(db, row.content, embedding, {
|
|
@@ -98,7 +121,7 @@ async function plantRankingFixture(db, fixture) {
|
|
|
98
121
|
memoryType: row.memoryType,
|
|
99
122
|
createdAt: row.createdAt,
|
|
100
123
|
});
|
|
101
|
-
const lastAccessed = new Date(
|
|
124
|
+
const lastAccessed = new Date(nowMs - row.lastAccessedDaysAgo * 86_400_000).toISOString();
|
|
102
125
|
storage.updateMemory(db, id, {
|
|
103
126
|
base_strength: row.baseStrength,
|
|
104
127
|
access_count: row.accessCount,
|
|
@@ -121,7 +144,7 @@ async function plantRankingFixture(db, fixture) {
|
|
|
121
144
|
* the cold-exposure swap is inert — scored.length <= limit — and the
|
|
122
145
|
* returned order IS the score order), results mapped back to fixture keys.
|
|
123
146
|
*/
|
|
124
|
-
async function measurePlantedCase(db, fixture, idByKey) {
|
|
147
|
+
async function measurePlantedCase(db, fixture, idByKey, now) {
|
|
125
148
|
const keyById = new Map([...idByKey.entries()].map(([k, v]) => [v, k]));
|
|
126
149
|
const queryEmb = await (0, embedder_js_1.embed)(fixture.query);
|
|
127
150
|
const ftsRows = storage.searchFts(db, fixture.query, 100, undefined);
|
|
@@ -129,6 +152,9 @@ async function measurePlantedCase(db, fixture, idByKey) {
|
|
|
129
152
|
limit: fixture.rows.length + 8,
|
|
130
153
|
noStrengthen: true,
|
|
131
154
|
queryEmbedding: queryEmb,
|
|
155
|
+
// #458: the pinned clock (undefined = live) — score terms are
|
|
156
|
+
// wall-clock-independent across before/after photos.
|
|
157
|
+
...(now ? { now } : {}),
|
|
132
158
|
});
|
|
133
159
|
const measured = results.map((r, i) => ({
|
|
134
160
|
key: keyById.get(r.id) ?? r.id,
|
|
@@ -136,6 +162,7 @@ async function measurePlantedCase(db, fixture, idByKey) {
|
|
|
136
162
|
score: Math.round(r.score * 1e6) / 1e6,
|
|
137
163
|
similarity: r.similarity ?? null,
|
|
138
164
|
source: r.source ?? null,
|
|
165
|
+
connections: r.connections,
|
|
139
166
|
effectiveStrength: Math.round(r.effective_strength * 1e6) / 1e6,
|
|
140
167
|
baseStrength: fixture.rows.find((x) => x.key === keyById.get(r.id))?.baseStrength ?? -1,
|
|
141
168
|
accessCount: r.access_count,
|
|
@@ -154,7 +181,46 @@ async function measurePlantedCase(db, fixture, idByKey) {
|
|
|
154
181
|
top1Top2ScoreGap: gap,
|
|
155
182
|
};
|
|
156
183
|
}
|
|
157
|
-
|
|
184
|
+
/** Collapse a linked_fallback measurement into the #449 verdicts. */
|
|
185
|
+
function case3Verdict(fixture, measured) {
|
|
186
|
+
const exp = fixture.linkedExpectations;
|
|
187
|
+
const byKey = new Map(measured.map((r) => [r.key, r]));
|
|
188
|
+
const hubA = byKey.get(exp.saturationPair.hubA);
|
|
189
|
+
const hubB = byKey.get(exp.saturationPair.hubB);
|
|
190
|
+
const unlinked = byKey.get(exp.directionPair.unlinked);
|
|
191
|
+
const linked = byKey.get(exp.directionPair.linked);
|
|
192
|
+
const byteEqual = Object.is(hubA.score, hubB.score);
|
|
193
|
+
const unlinkedAbove = unlinked.rank < linked.rank;
|
|
194
|
+
const measuredGap = unlinked.similarity !== null && linked.similarity !== null
|
|
195
|
+
? Math.round((unlinked.similarity - linked.similarity) * 1e6) / 1e6
|
|
196
|
+
: null;
|
|
197
|
+
return {
|
|
198
|
+
pass: byteEqual && unlinkedAbove,
|
|
199
|
+
query: fixture.query,
|
|
200
|
+
seam: { rrfCompositeWeight: 1 },
|
|
201
|
+
saturation: {
|
|
202
|
+
hubA: { key: hubA.key, connections: hubA.connections, score: hubA.score },
|
|
203
|
+
hubB: { key: hubB.key, connections: hubB.connections, score: hubB.score },
|
|
204
|
+
byteEqual,
|
|
205
|
+
},
|
|
206
|
+
direction: {
|
|
207
|
+
unlinked: {
|
|
208
|
+
key: unlinked.key,
|
|
209
|
+
rank: unlinked.rank,
|
|
210
|
+
similarity: unlinked.similarity,
|
|
211
|
+
},
|
|
212
|
+
linked: {
|
|
213
|
+
key: linked.key,
|
|
214
|
+
rank: linked.rank,
|
|
215
|
+
similarity: linked.similarity,
|
|
216
|
+
},
|
|
217
|
+
measuredGap,
|
|
218
|
+
unlinkedAbove,
|
|
219
|
+
},
|
|
220
|
+
results: measured,
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
async function runBattery(db, now) {
|
|
158
224
|
const rows = db
|
|
159
225
|
.prepare("SELECT id, content FROM memories WHERE COALESCE(status, '') != 'absorbed'")
|
|
160
226
|
.all();
|
|
@@ -167,6 +233,7 @@ async function runBattery(db) {
|
|
|
167
233
|
limit: 8,
|
|
168
234
|
noStrengthen: true,
|
|
169
235
|
queryEmbedding: queryEmb,
|
|
236
|
+
...(now ? { now } : {}), // #458: the pinned clock (undefined = live)
|
|
170
237
|
});
|
|
171
238
|
const top8 = results.map((r) => r.id);
|
|
172
239
|
// Per-result provenance for the sweep analysis: the channel that
|
|
@@ -195,9 +262,9 @@ async function runBattery(db) {
|
|
|
195
262
|
// ---------------------------------------------------------------------------
|
|
196
263
|
function renderReport(args) {
|
|
197
264
|
const L = [];
|
|
198
|
-
const { case1, case2, battery } = args;
|
|
199
|
-
L.push("# Search-trust ranking gate (#425)\n");
|
|
200
|
-
L.push(`Generated: ${new Date().toISOString()} \nScoring: ${JSON.stringify(args.scoring)}\n`);
|
|
265
|
+
const { case1, case2, case3, battery } = args;
|
|
266
|
+
L.push("# Search-trust ranking gate (#425 + #449)\n");
|
|
267
|
+
L.push(`Generated: ${new Date().toISOString()} \nClock: ${args.clock} (#458) \nScoring: ${JSON.stringify(args.scoring)}\n`);
|
|
201
268
|
L.push("## Case 1 — sirnas_exact_match (AC1)\n");
|
|
202
269
|
L.push(`Query "${case1.query}" — declared top-1: ${ranking_fixtures_js_1.SIRNAS_FIXTURE.expectedTop1} — ` +
|
|
203
270
|
`**${case1.pass ? "PASS" : "FAIL"}**${case1.top1Top2ScoreGap !== null ? ` (top1-top2 score gap ${case1.top1Top2ScoreGap.toFixed(4)})` : ""}\n`);
|
|
@@ -211,11 +278,47 @@ function renderReport(args) {
|
|
|
211
278
|
L.push("## Case 2 — no_match_control (AC2)\n");
|
|
212
279
|
L.push(`Query "${case2.query}" — FTS hits: ${case2.ftsHitRows} (must be 0 — no both-channel ` +
|
|
213
280
|
`candidates can exist)` +
|
|
214
|
-
(args.
|
|
281
|
+
(args.case2Identity === null
|
|
282
|
+
? ""
|
|
283
|
+
: ` — vs baseline (unlinked-identity gate): **${args.case2Identity.pass ? "PASS" : "FAILED"}**` +
|
|
284
|
+
(args.case2Identity.failures.length > 0
|
|
285
|
+
? ` — ${args.case2Identity.failures.slice(0, 3).join("; ")}` +
|
|
286
|
+
(args.case2Identity.failures.length > 3 ? "; …" : "")
|
|
287
|
+
: "")) +
|
|
215
288
|
"\n");
|
|
216
289
|
L.push("Returned order: " + case2.ids.map((k) => k).join(" → ") + "\n");
|
|
290
|
+
L.push("| rank | key | similarity | source | conn | score |");
|
|
291
|
+
L.push("|---|---|---|---|---|---|");
|
|
292
|
+
for (const r of case2.results.slice(0, 6)) {
|
|
293
|
+
L.push(`| ${r.rank} | ${r.key} | ${r.similarity?.toFixed(4) ?? "-"} | ${r.source ?? "-"} | ` +
|
|
294
|
+
`${r.connections} | ${r.score.toFixed(4)} |`);
|
|
295
|
+
}
|
|
296
|
+
L.push("");
|
|
297
|
+
L.push("## Case 3 — linked_fallback (#449 items 1+3)\n");
|
|
298
|
+
L.push(`Query "${case3.query}" — measured under the eval seam rrfCompositeWeight=1 ` +
|
|
299
|
+
`(finalScore = composite) — **${case3.pass ? "PASS" : "FAIL"}**\n`);
|
|
300
|
+
L.push(`Saturation pair (declared post-#449: byte-equal scores — both hubs saturate at K=16; ` +
|
|
301
|
+
`red under the linear term is the before-picture): ` +
|
|
302
|
+
`${case3.saturation.hubA.key} (k=${case3.saturation.hubA.connections}) score ` +
|
|
303
|
+
`${case3.saturation.hubA.score.toFixed(6)} vs ${case3.saturation.hubB.key} ` +
|
|
304
|
+
`(k=${case3.saturation.hubB.connections}) score ${case3.saturation.hubB.score.toFixed(6)} — ` +
|
|
305
|
+
`byte-equal: **${case3.saturation.byteEqual ? "YES" : "NO"}**\n`);
|
|
306
|
+
L.push(`Direction pair (declared: unlinked ranks ABOVE the 4-link row — similarity dominates ` +
|
|
307
|
+
`the saturated connections credit): ${case3.direction.unlinked.key} sim ` +
|
|
308
|
+
`${case3.direction.unlinked.similarity?.toFixed(4) ?? "-"} rank #${case3.direction.unlinked.rank} vs ` +
|
|
309
|
+
`${case3.direction.linked.key} sim ${case3.direction.linked.similarity?.toFixed(4) ?? "-"} rank ` +
|
|
310
|
+
`#${case3.direction.linked.rank} — measured gap ` +
|
|
311
|
+
`${case3.direction.measuredGap?.toFixed(4) ?? "-"} — unlinked above: ` +
|
|
312
|
+
`**${case3.direction.unlinkedAbove ? "YES" : "NO"}**\n`);
|
|
313
|
+
L.push("| rank | key | similarity | source | conn | score |");
|
|
314
|
+
L.push("|---|---|---|---|---|---|");
|
|
315
|
+
for (const r of case3.results.slice(0, 6)) {
|
|
316
|
+
L.push(`| ${r.rank} | ${r.key} | ${r.similarity?.toFixed(4) ?? "-"} | ${r.source ?? "-"} | ` +
|
|
317
|
+
`${r.connections} | ${r.score.toFixed(4)} |`);
|
|
318
|
+
}
|
|
319
|
+
L.push("");
|
|
217
320
|
if (battery) {
|
|
218
|
-
L.push("## Case
|
|
321
|
+
L.push("## Case 4 — real-query battery (AC3 no-regression photo)\n");
|
|
219
322
|
L.push(`Queries: ${battery.queries.length}; Sirnäs rank for "${ranking_battery_js_1.SIRNAS_QUERY}": ` +
|
|
220
323
|
(battery.sirnasRank === null ? "not in top-8" : `#${battery.sirnasRank}`) + "\n");
|
|
221
324
|
if (args.comparison) {
|
|
@@ -249,7 +352,7 @@ function renderReport(args) {
|
|
|
249
352
|
// main
|
|
250
353
|
// ---------------------------------------------------------------------------
|
|
251
354
|
function usage() {
|
|
252
|
-
console.error("Usage: npm run eval:ranking -- [--snapshot <db>] [--photo <out.json>] [--compare <baseline.json>] [--scoring '<json>']");
|
|
355
|
+
console.error("Usage: npm run eval:ranking -- [--snapshot <db>] [--photo <out.json>] [--compare <baseline.json>] [--scoring '<json>'] [--now <ISO>]");
|
|
253
356
|
process.exit(1);
|
|
254
357
|
}
|
|
255
358
|
async function main() {
|
|
@@ -264,6 +367,19 @@ async function main() {
|
|
|
264
367
|
const scoringRaw = flag("--scoring");
|
|
265
368
|
if (argv.includes("--help") || argv.includes("-h"))
|
|
266
369
|
usage();
|
|
370
|
+
// #458: pin the eval clock (planted lastAccessed offsets + every retrieve
|
|
371
|
+
// score against the same instant). Invalid input fails explicit — never a
|
|
372
|
+
// silent live fallback that would reintroduce wall-clock drift.
|
|
373
|
+
let now;
|
|
374
|
+
try {
|
|
375
|
+
now = (0, eval_clock_js_1.parsePinnedNow)(flag("--now"));
|
|
376
|
+
}
|
|
377
|
+
catch (err) {
|
|
378
|
+
console.error(`[ranking-eval] ${err instanceof Error ? err.message : err}`);
|
|
379
|
+
process.exit(1);
|
|
380
|
+
}
|
|
381
|
+
console.log(`[ranking-eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}`);
|
|
382
|
+
const nowArg = now ?? undefined;
|
|
267
383
|
if (scoringRaw) {
|
|
268
384
|
try {
|
|
269
385
|
const parsed = JSON.parse(scoringRaw);
|
|
@@ -274,7 +390,7 @@ async function main() {
|
|
|
274
390
|
process.exit(1);
|
|
275
391
|
}
|
|
276
392
|
}
|
|
277
|
-
for (const f of [ranking_fixtures_js_1.SIRNAS_FIXTURE, ranking_fixtures_js_1.NO_MATCH_FIXTURE]) {
|
|
393
|
+
for (const f of [ranking_fixtures_js_1.SIRNAS_FIXTURE, ranking_fixtures_js_1.NO_MATCH_FIXTURE, ranking_fixtures_js_1.LINKED_FALLBACK_FIXTURE]) {
|
|
278
394
|
const problems = (0, ranking_fixtures_js_1.checkFixtureInvariants)(f);
|
|
279
395
|
if (problems.length > 0) {
|
|
280
396
|
console.error(`[ranking-eval] fixture ${f.key} invariant violations: ${problems.join("; ")}`);
|
|
@@ -291,9 +407,9 @@ async function main() {
|
|
|
291
407
|
const db = (0, db_js_1.initDb)((0, node_path_1.join)(workDir, "planted.db"));
|
|
292
408
|
try {
|
|
293
409
|
console.log("[ranking-eval] planting corpus (real embedder)...");
|
|
294
|
-
const idByKey = await plantRankingFixture(db, ranking_fixtures_js_1.SIRNAS_FIXTURE);
|
|
295
|
-
case1 = await measurePlantedCase(db, ranking_fixtures_js_1.SIRNAS_FIXTURE, idByKey);
|
|
296
|
-
const case2Measured = await measurePlantedCase(db, ranking_fixtures_js_1.NO_MATCH_FIXTURE, idByKey);
|
|
410
|
+
const idByKey = await plantRankingFixture(db, ranking_fixtures_js_1.SIRNAS_FIXTURE, nowArg);
|
|
411
|
+
case1 = await measurePlantedCase(db, ranking_fixtures_js_1.SIRNAS_FIXTURE, idByKey, nowArg);
|
|
412
|
+
const case2Measured = await measurePlantedCase(db, ranking_fixtures_js_1.NO_MATCH_FIXTURE, idByKey, nowArg);
|
|
297
413
|
// The returned list keyed by fixture key (readable photos; ids would
|
|
298
414
|
// churn across replanting).
|
|
299
415
|
case2 = { ...case2Measured, ids: case2Measured.results.map((r) => r.key) };
|
|
@@ -305,9 +421,35 @@ async function main() {
|
|
|
305
421
|
finally {
|
|
306
422
|
(0, node_fs_1.rmSync)(workDir, { recursive: true, force: true });
|
|
307
423
|
}
|
|
424
|
+
// Case 3 (#449): planted in its OWN throwaway DB, measured under the eval
|
|
425
|
+
// seam — rrfCompositeWeight 1 makes finalScore = composite exactly (the
|
|
426
|
+
// savedWeights save/restore pattern from tests/ranking-flip.test.ts; the
|
|
427
|
+
// --scoring CLI overrides, if any, are what gets saved and restored).
|
|
428
|
+
let case3;
|
|
429
|
+
{
|
|
430
|
+
const savedWeights = (0, retrieval_js_1.getScoringWeights)();
|
|
431
|
+
(0, retrieval_js_1.configureScoring)({ ...savedWeights, rrfCompositeWeight: 1 });
|
|
432
|
+
const workDir3 = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-ranking-eval-linked-"));
|
|
433
|
+
try {
|
|
434
|
+
const db = (0, db_js_1.initDb)((0, node_path_1.join)(workDir3, "linked.db"));
|
|
435
|
+
try {
|
|
436
|
+
console.log("[ranking-eval] planting linked_fallback corpus (real embedder)...");
|
|
437
|
+
const idByKey3 = await plantRankingFixture(db, ranking_fixtures_js_1.LINKED_FALLBACK_FIXTURE, nowArg);
|
|
438
|
+
const measured3 = await measurePlantedCase(db, ranking_fixtures_js_1.LINKED_FALLBACK_FIXTURE, idByKey3, nowArg);
|
|
439
|
+
case3 = case3Verdict(ranking_fixtures_js_1.LINKED_FALLBACK_FIXTURE, measured3.results);
|
|
440
|
+
}
|
|
441
|
+
finally {
|
|
442
|
+
db.close();
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
finally {
|
|
446
|
+
(0, node_fs_1.rmSync)(workDir3, { recursive: true, force: true });
|
|
447
|
+
(0, retrieval_js_1.configureScoring)(savedWeights); // restore — battery + photo run at the caller's weights
|
|
448
|
+
}
|
|
449
|
+
}
|
|
308
450
|
let battery = null;
|
|
309
451
|
let comparison = null;
|
|
310
|
-
let
|
|
452
|
+
let case2Identity = null;
|
|
311
453
|
// Content lookup for the exact-match gate (the SAME corpus both photos
|
|
312
454
|
// were recorded against — the snapshot, when comparing).
|
|
313
455
|
const contentById = new Map();
|
|
@@ -315,7 +457,7 @@ async function main() {
|
|
|
315
457
|
console.log(`[ranking-eval] battery against readonly snapshot: ${snapshotPath}`);
|
|
316
458
|
const sdb = (0, eval_db_js_1.openSnapshot)(snapshotPath);
|
|
317
459
|
try {
|
|
318
|
-
battery = await runBattery(sdb);
|
|
460
|
+
battery = await runBattery(sdb, nowArg);
|
|
319
461
|
for (const r of sdb
|
|
320
462
|
.prepare("SELECT id, content FROM memories WHERE COALESCE(status, '') != 'absorbed'")
|
|
321
463
|
.all()) {
|
|
@@ -335,15 +477,31 @@ async function main() {
|
|
|
335
477
|
console.error(`[ranking-eval] cannot read baseline photo ${comparePath}: ${err instanceof Error ? err.message : err}`);
|
|
336
478
|
process.exit(1);
|
|
337
479
|
}
|
|
338
|
-
if (!baseline.case2
|
|
339
|
-
console.error("[ranking-eval] baseline photo lacks case2
|
|
480
|
+
if (!baseline.case2) {
|
|
481
|
+
console.error("[ranking-eval] baseline photo lacks case2 — record one with --photo first");
|
|
482
|
+
process.exit(1);
|
|
483
|
+
}
|
|
484
|
+
// #449: the case-2 gate is the unlinked-row identity comparator (linked
|
|
485
|
+
// rows drift BY DESIGN); it reads the per-row results the baseline must
|
|
486
|
+
// already carry and fails with a re-record instruction when it doesn't.
|
|
487
|
+
case2Identity = (0, ranking_battery_js_1.compareCase2Unlinked)(baseline.case2, { ftsHitRows: case2.ftsHitRows, ids: case2.ids, results: case2.results });
|
|
488
|
+
// The battery comparison runs only when BOTH sides have one (the
|
|
489
|
+
// baseline was recorded with --snapshot AND this run passes it).
|
|
490
|
+
if (baseline.battery && battery) {
|
|
491
|
+
comparison = (0, ranking_battery_js_1.batteryComparison)(baseline.battery, battery, contentById);
|
|
492
|
+
}
|
|
493
|
+
else if (baseline.battery && !battery) {
|
|
494
|
+
console.error("[ranking-eval] baseline photo has a battery but this run recorded none — pass --snapshot to compare it");
|
|
340
495
|
process.exit(1);
|
|
341
496
|
}
|
|
342
|
-
case2Stable = (0, ranking_battery_js_1.compareCase2)(baseline.case2.ids, case2.ids);
|
|
343
|
-
comparison = (0, ranking_battery_js_1.batteryComparison)(baseline.battery, battery, contentById);
|
|
344
497
|
}
|
|
345
498
|
const photo = {
|
|
346
499
|
generatedAt: new Date().toISOString(),
|
|
500
|
+
/** #458: self-describing clock — which instant the decay/recency terms
|
|
501
|
+
* scored against ({mode:"pinned", now} or {mode:"live"}). */
|
|
502
|
+
clock: now === null
|
|
503
|
+
? { mode: "live" }
|
|
504
|
+
: { mode: "pinned", now: now.toISOString() },
|
|
347
505
|
scoring: (0, retrieval_js_1.getScoringWeights)(),
|
|
348
506
|
case1: {
|
|
349
507
|
pass: case1.pass,
|
|
@@ -351,10 +509,11 @@ async function main() {
|
|
|
351
509
|
top1Top2ScoreGap: case1.top1Top2ScoreGap,
|
|
352
510
|
results: case1.results,
|
|
353
511
|
},
|
|
354
|
-
case2: { ftsHitRows: case2.ftsHitRows, ids: case2.ids },
|
|
512
|
+
case2: { ftsHitRows: case2.ftsHitRows, ids: case2.ids, results: case2.results },
|
|
513
|
+
case3,
|
|
355
514
|
battery,
|
|
356
515
|
comparison,
|
|
357
|
-
|
|
516
|
+
case2Identity,
|
|
358
517
|
};
|
|
359
518
|
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(photoPath), { recursive: true });
|
|
360
519
|
(0, node_fs_1.writeFileSync)(photoPath, JSON.stringify(photo, null, 2), "utf-8");
|
|
@@ -362,10 +521,12 @@ async function main() {
|
|
|
362
521
|
const report = renderReport({
|
|
363
522
|
case1,
|
|
364
523
|
case2,
|
|
524
|
+
case3,
|
|
365
525
|
battery,
|
|
366
526
|
comparison,
|
|
367
|
-
|
|
527
|
+
case2Identity,
|
|
368
528
|
scoring: { ...photo.scoring },
|
|
529
|
+
clock: (0, eval_clock_js_1.clockLabel)(now),
|
|
369
530
|
});
|
|
370
531
|
console.log(report);
|
|
371
532
|
const reportPath = photoPath.endsWith(".json")
|
|
@@ -376,10 +537,12 @@ async function main() {
|
|
|
376
537
|
? true
|
|
377
538
|
: comparison.top1StabilityExPromotions >= BATTERY_TOP1_STABILITY_MIN &&
|
|
378
539
|
comparison.lostBothChannelExact.length === 0;
|
|
379
|
-
const case2Ok =
|
|
380
|
-
const
|
|
540
|
+
const case2Ok = case2Identity === null ? true : case2Identity.pass;
|
|
541
|
+
const case3Ok = case3.pass;
|
|
542
|
+
const ok = case1.pass === true && case2Ok && case3Ok && batteryOk;
|
|
381
543
|
console.log(`[ranking-eval] ${ok ? "GATE PASS" : "GATE FAIL"} (case1 ${case1.pass ? "pass" : "FAIL"}, ` +
|
|
382
|
-
`case2 ${case2Ok ? "ok" : "DRIFT"},
|
|
544
|
+
`case2 ${case2Ok ? "ok" : "DRIFT"}, case3 ${case3Ok ? "pass" : "FAIL"}, ` +
|
|
545
|
+
`battery ${batteryOk ? "ok" : "DRIFT"} ` +
|
|
383
546
|
`[top-1 ${((comparison?.top1Stability ?? 1) * 100).toFixed(1)}% raw / ` +
|
|
384
547
|
`${((comparison?.top1StabilityExPromotions ?? 1) * 100).toFixed(1)}% ex-promotions, ` +
|
|
385
548
|
`lost both-channel exact ${comparison?.lostBothChannelExact.length ?? 0}]) in ${Date.now() - t0}ms`);
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
* issue (the owner's live console test, 2026-09-13): a user searches a
|
|
6
6
|
* DISTINCTIVE PROPER NOUN they know exists; the one memory carrying that
|
|
7
7
|
* token has the best similarity of all candidates AND the literal FTS hit,
|
|
8
|
-
* yet ranks behind
|
|
8
|
+
* yet ranks behind high-strength but less-relevant memories that win on
|
|
9
9
|
* effective strength + graph connections.
|
|
10
10
|
*
|
|
11
11
|
* (a) sirnas_exact_match — the AC1 fixture: a low-strength, unlinked,
|
|
@@ -20,14 +20,19 @@
|
|
|
20
20
|
* under the ranking work (no both-channel candidates exist, so the
|
|
21
21
|
* boost can never fire — the case pins that the no-regression
|
|
22
22
|
* guarantee is structural, not lucky).
|
|
23
|
+
* (c) linked_fallback — the #449 (items 1+3) connections fixture: a
|
|
24
|
+
* 16-link vs 20-link identical-content hub pair (the saturation tie)
|
|
25
|
+
* plus an unlinked higher-similarity row vs a 4-link lower-similarity
|
|
26
|
+
* row (the direction pin). Planted in its OWN throwaway DB — the 40
|
|
27
|
+
* link-neighbor rows would pollute the case-1/2 corpus.
|
|
23
28
|
*
|
|
24
29
|
* Honesty rules (the planted-pairs #393 A discipline):
|
|
25
30
|
* - The texts are synthetic and generic (this package publishes to npm —
|
|
26
31
|
* no real infrastructure, people, or fleet detail), but SHAPED like the
|
|
27
32
|
* field failure: the proper noun is invented (it must appear nowhere
|
|
28
|
-
* else — verified by an invariant test), the rivals
|
|
29
|
-
*
|
|
30
|
-
*
|
|
33
|
+
* else — verified by an invariant test), the rivals carry the metadata
|
|
34
|
+
* the production pipeline itself writes (base strength, access history,
|
|
35
|
+
* link counts).
|
|
31
36
|
* - Similarities are MEASURED at run time with the real bge-small-en-v1.5
|
|
32
37
|
* embedder (ranking-eval.ts) — never asserted into existence. The
|
|
33
38
|
* declared expectation is about ORDER under the shipped ranking, and
|
|
@@ -46,7 +51,7 @@ export interface RankingRow {
|
|
|
46
51
|
memoryType: string;
|
|
47
52
|
/** Birth base_strength — the value stageImportance would have written. */
|
|
48
53
|
baseStrength: number;
|
|
49
|
-
/** Access history (
|
|
54
|
+
/** Access history (the decay clock + the #448 promotion baseline). */
|
|
50
55
|
accessCount: number;
|
|
51
56
|
/** Days before the run's `now` of the last access (0 = touched today). */
|
|
52
57
|
lastAccessedDaysAgo: number;
|
|
@@ -66,12 +71,47 @@ export interface RankingFixture {
|
|
|
66
71
|
distinctiveToken: string;
|
|
67
72
|
/** Declared expectation: this row's key must be returned at position 1. */
|
|
68
73
|
expectedTop1: string;
|
|
74
|
+
/**
|
|
75
|
+
* Declared undirected link degree per row key (#449): own linksTo entries
|
|
76
|
+
* PLUS the times other rows reference it (each linksTo entry is one link
|
|
77
|
+
* row, mirroring storage.getLinks(id, "both").length). Optional — case 1
|
|
78
|
+
* and case 3 declare it; a corpus that silently drifts from its declared
|
|
79
|
+
* link-cluster shape would fake a pass/fail.
|
|
80
|
+
*/
|
|
81
|
+
expectedUndirectedDegree?: Record<string, number>;
|
|
82
|
+
/**
|
|
83
|
+
* #449 case-3 (linked_fallback) expectations — absent for cases 1-2. The
|
|
84
|
+
* verdict logic (byte-equality + rank order) lives in ranking-eval.ts.
|
|
85
|
+
*/
|
|
86
|
+
linkedExpectations?: {
|
|
87
|
+
/**
|
|
88
|
+
* The saturation pair: two identical-content hubs at the declared
|
|
89
|
+
* degrees — post-#449 (log-saturating term, K = 16) their scores must
|
|
90
|
+
* be byte-EQUAL (both saturate); under the pre-#449 linear term they
|
|
91
|
+
* differ by ~0.03 composite (the recorded before-picture).
|
|
92
|
+
*/
|
|
93
|
+
saturationPair: {
|
|
94
|
+
hubA: string;
|
|
95
|
+
hubB: string;
|
|
96
|
+
};
|
|
97
|
+
/**
|
|
98
|
+
* The direction pair: unlinked (higher measured similarity, carries the
|
|
99
|
+
* distinctive token) vs linked (exactly 4 links, ~0.20 lower measured
|
|
100
|
+
* similarity, equal metadata otherwise). Declared (Option A, owner
|
|
101
|
+
* decision 2026-09-17): the unlinked row ranks ABOVE the linked one.
|
|
102
|
+
*/
|
|
103
|
+
directionPair: {
|
|
104
|
+
unlinked: string;
|
|
105
|
+
linked: string;
|
|
106
|
+
};
|
|
107
|
+
};
|
|
69
108
|
}
|
|
70
109
|
/** All planted rows share project + source_agent (metadata never refuses). */
|
|
71
110
|
export declare const RANKING_PROJECT = "ranking-eval";
|
|
72
111
|
export declare const RANKING_SOURCE_AGENT = "ranking-eval";
|
|
73
112
|
export declare const SIRNAS_FIXTURE: RankingFixture;
|
|
74
113
|
export declare const NO_MATCH_FIXTURE: RankingFixture;
|
|
75
|
-
/** The battery runs against a real snapshot copy (ranking-eval.ts case
|
|
114
|
+
/** The battery runs against a real snapshot copy (ranking-eval.ts case 4). */
|
|
76
115
|
export declare const REAL_SIRNAS_MEMORY_ID = "12980c9d-0eba-4963-95eb-0548bc2aa93e";
|
|
116
|
+
export declare const LINKED_FALLBACK_FIXTURE: RankingFixture;
|
|
77
117
|
export declare function checkFixtureInvariants(f: RankingFixture): string[];
|