@gamaze/hicortex 0.22.0 → 0.22.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +209 -76
- package/dist/calibration.d.ts +92 -12
- package/dist/calibration.js +102 -15
- package/dist/classify-domains.js +6 -1
- package/dist/consolidate.js +87 -19
- package/dist/dashboard.d.ts +11 -0
- package/dist/dashboard.js +9 -0
- package/dist/db.js +74 -0
- package/dist/eval/decay-eval.d.ts +4 -2
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/eval-clock.d.ts +32 -0
- package/dist/eval/eval-clock.js +47 -0
- package/dist/eval/graph-eval.d.ts +15 -2
- package/dist/eval/graph-eval.js +51 -5
- package/dist/eval/planted-eval.d.ts +4 -0
- package/dist/eval/planted-eval.js +27 -2
- package/dist/eval/planted-harness.d.ts +7 -0
- package/dist/eval/planted-harness.js +2 -0
- package/dist/eval/ranking-battery.d.ts +49 -2
- package/dist/eval/ranking-battery.js +110 -2
- package/dist/eval/ranking-eval.d.ts +26 -6
- package/dist/eval/ranking-eval.js +197 -34
- package/dist/eval/ranking-fixtures.d.ts +41 -1
- package/dist/eval/ranking-fixtures.js +261 -2
- package/dist/eval/recall-sweep.d.ts +7 -2
- package/dist/eval/recall-sweep.js +42 -13
- package/dist/eval/relevance-eval.d.ts +115 -1
- package/dist/eval/relevance-eval.js +318 -32
- package/dist/eval/run-eval.d.ts +7 -4
- package/dist/eval/run-eval.js +36 -9
- package/dist/mcp-server.js +37 -9
- package/dist/nightly.js +14 -0
- package/dist/recall-index.d.ts +46 -3
- package/dist/recall-index.js +83 -26
- package/dist/recall-precision.d.ts +212 -0
- package/dist/recall-precision.js +381 -0
- package/dist/retrieval.d.ts +34 -15
- package/dist/retrieval.js +132 -59
- package/dist/types.d.ts +7 -0
- package/package.json +1 -1
- package/server.json +3 -3
|
@@ -8,6 +8,13 @@
|
|
|
8
8
|
* corrected cosine formula, #145).
|
|
9
9
|
*/
|
|
10
10
|
import type Database from "better-sqlite3";
|
|
11
|
+
/**
|
|
12
|
+
* Deterministic sample of `rows` (#460): a seeded partial Fisher–Yates
|
|
13
|
+
* shuffles the LAST `size` slots from the back, and that shuffled tail is
|
|
14
|
+
* the sample. Rows must already be in a stable order (the caller sorts by
|
|
15
|
+
* PK) so the draw does not depend on SQLite's scan order.
|
|
16
|
+
*/
|
|
17
|
+
export declare function seededSample<T>(rows: T[], size: number, seed: number): T[];
|
|
11
18
|
export interface LinkRow {
|
|
12
19
|
source_id: string;
|
|
13
20
|
target_id: string;
|
|
@@ -50,8 +57,14 @@ export interface DriftReport {
|
|
|
50
57
|
maxAbsDrift: number;
|
|
51
58
|
skippedMissingEmbedding: number;
|
|
52
59
|
}
|
|
53
|
-
/**
|
|
54
|
-
|
|
60
|
+
/**
|
|
61
|
+
* Recompute cosine for a link sample and compare to stored `strength` (post
|
|
62
|
+
* migration-5 rescale, should track closely). The sample is deterministic
|
|
63
|
+
* (#460): all link rows in stable PK order, drawn by a seeded Fisher–Yates —
|
|
64
|
+
* two runs on the same build + snapshot produce an identical report, so
|
|
65
|
+
* before/after comparisons can require full-output identity.
|
|
66
|
+
*/
|
|
67
|
+
export declare function runDriftSample(db: Database.Database, embeddings: Map<string, Float32Array>, sampleSize?: number, seed?: number): DriftReport;
|
|
55
68
|
export interface PartitionStats {
|
|
56
69
|
linkCount: number;
|
|
57
70
|
byRelationship: Record<string, number>;
|
package/dist/eval/graph-eval.js
CHANGED
|
@@ -42,6 +42,7 @@ var __importStar = (this && this.__importStar) || (function () {
|
|
|
42
42
|
};
|
|
43
43
|
})();
|
|
44
44
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
45
|
+
exports.seededSample = seededSample;
|
|
45
46
|
exports.cosineBetween = cosineBetween;
|
|
46
47
|
exports.byRelationshipCounts = byRelationshipCounts;
|
|
47
48
|
exports.partitionByRelinkCursor = partitionByRelinkCursor;
|
|
@@ -55,8 +56,46 @@ const decay_eval_js_1 = require("./decay-eval.js");
|
|
|
55
56
|
/** Matches the middle D1 duplicate threshold — a link between near-dups is noise, not signal. */
|
|
56
57
|
const DUP_NOISE_THRESHOLD = 0.92;
|
|
57
58
|
const DRIFT_SAMPLE_SIZE = 500;
|
|
59
|
+
/**
|
|
60
|
+
* Seed for the drift sample's PRNG (#460). SQLite's `random()` cannot be
|
|
61
|
+
* seeded, so the draw happens client-side with a deterministic Fisher–Yates;
|
|
62
|
+
* this fixed seed makes the sample — and therefore the whole DriftReport —
|
|
63
|
+
* byte-identical across runs on the same build + snapshot, which lets
|
|
64
|
+
* before/after eval comparisons require full-output identity.
|
|
65
|
+
*/
|
|
66
|
+
const DRIFT_SAMPLE_SEED = 460;
|
|
58
67
|
const DEGREE_BUCKET_EDGES = [0, 1, 2, 3, 5, 10, 20, 50, 100];
|
|
59
68
|
const DRIFT_BUCKET_EDGES = [0, 0.01, 0.02, 0.05, 0.1, 0.2, 0.3, 0.5, 1.0];
|
|
69
|
+
/**
|
|
70
|
+
* mulberry32 — a 32-bit seeded PRNG (public-domain reference implementation).
|
|
71
|
+
* JS number ops are deterministic across platforms, so a fixed seed yields
|
|
72
|
+
* the same sequence everywhere.
|
|
73
|
+
*/
|
|
74
|
+
function mulberry32(seed) {
|
|
75
|
+
let a = seed >>> 0;
|
|
76
|
+
return () => {
|
|
77
|
+
a = (a + 0x6d2b79f5) | 0;
|
|
78
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
79
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
80
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
81
|
+
};
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* Deterministic sample of `rows` (#460): a seeded partial Fisher–Yates
|
|
85
|
+
* shuffles the LAST `size` slots from the back, and that shuffled tail is
|
|
86
|
+
* the sample. Rows must already be in a stable order (the caller sorts by
|
|
87
|
+
* PK) so the draw does not depend on SQLite's scan order.
|
|
88
|
+
*/
|
|
89
|
+
function seededSample(rows, size, seed) {
|
|
90
|
+
const rand = mulberry32(seed);
|
|
91
|
+
const arr = [...rows];
|
|
92
|
+
const n = Math.min(size, arr.length);
|
|
93
|
+
for (let i = arr.length - 1; i > arr.length - 1 - n; i--) {
|
|
94
|
+
const j = Math.floor(rand() * (i + 1));
|
|
95
|
+
[arr[i], arr[j]] = [arr[j], arr[i]];
|
|
96
|
+
}
|
|
97
|
+
return arr.slice(arr.length - n);
|
|
98
|
+
}
|
|
60
99
|
// ---------------------------------------------------------------------------
|
|
61
100
|
// Pure helpers (unit tested)
|
|
62
101
|
// ---------------------------------------------------------------------------
|
|
@@ -109,11 +148,18 @@ function runDegreeAudit(db, topN = 10) {
|
|
|
109
148
|
});
|
|
110
149
|
return { memoriesWithLinks: linkCounts.size, totalMemories, degreeHistogram, topHubs };
|
|
111
150
|
}
|
|
112
|
-
/**
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
151
|
+
/**
|
|
152
|
+
* Recompute cosine for a link sample and compare to stored `strength` (post
|
|
153
|
+
* migration-5 rescale, should track closely). The sample is deterministic
|
|
154
|
+
* (#460): all link rows in stable PK order, drawn by a seeded Fisher–Yates —
|
|
155
|
+
* two runs on the same build + snapshot produce an identical report, so
|
|
156
|
+
* before/after comparisons can require full-output identity.
|
|
157
|
+
*/
|
|
158
|
+
function runDriftSample(db, embeddings, sampleSize = DRIFT_SAMPLE_SIZE, seed = DRIFT_SAMPLE_SEED) {
|
|
159
|
+
const allLinks = db
|
|
160
|
+
.prepare("SELECT source_id, target_id, strength FROM memory_links ORDER BY source_id, target_id")
|
|
161
|
+
.all();
|
|
162
|
+
const sample = seededSample(allLinks, sampleSize, seed);
|
|
117
163
|
const drifts = [];
|
|
118
164
|
let skipped = 0;
|
|
119
165
|
for (const link of sample) {
|
|
@@ -20,6 +20,10 @@
|
|
|
20
20
|
* live-LLM real-corpus acceptance run is the release-soak step (refine Q3),
|
|
21
21
|
* not this one.
|
|
22
22
|
*
|
|
23
|
+
* `--now <ISO>` (#458) pins the clock the recall-surface retrieve() scores
|
|
24
|
+
* against — before/after gate runs become wall-clock-independent. Default:
|
|
25
|
+
* the live clock; the report records which clock produced it.
|
|
26
|
+
*
|
|
23
27
|
* Never touches ~/.hicortex: DB, state dir, and backups all live under a
|
|
24
28
|
* mkdtemp directory that is removed on exit.
|
|
25
29
|
*/
|
|
@@ -21,6 +21,10 @@
|
|
|
21
21
|
* live-LLM real-corpus acceptance run is the release-soak step (refine Q3),
|
|
22
22
|
* not this one.
|
|
23
23
|
*
|
|
24
|
+
* `--now <ISO>` (#458) pins the clock the recall-surface retrieve() scores
|
|
25
|
+
* against — before/after gate runs become wall-clock-independent. Default:
|
|
26
|
+
* the live clock; the report records which clock produced it.
|
|
27
|
+
*
|
|
24
28
|
* Never touches ~/.hicortex: DB, state dir, and backups all live under a
|
|
25
29
|
* mkdtemp directory that is removed on exit.
|
|
26
30
|
*/
|
|
@@ -32,12 +36,15 @@ const embedder_js_1 = require("../embedder.js");
|
|
|
32
36
|
const capture_js_1 = require("../capture.js");
|
|
33
37
|
const planted_fixtures_js_1 = require("./planted-fixtures.js");
|
|
34
38
|
const planted_harness_js_1 = require("./planted-harness.js");
|
|
39
|
+
const eval_clock_js_1 = require("./eval-clock.js");
|
|
35
40
|
const DEFAULT_REPORT_PATH = (0, node_path_1.join)(process.cwd(), "data", "planted-eval-report.md");
|
|
36
41
|
function usage() {
|
|
37
|
-
console.error("Usage: npm run eval:planted -- [report.md] [--snapshot <snapshot.db>]\n" +
|
|
42
|
+
console.error("Usage: npm run eval:planted -- [report.md] [--snapshot <snapshot.db>] [--now <ISO>]\n" +
|
|
38
43
|
" [report.md] output path (default data/planted-eval-report.md)\n" +
|
|
39
44
|
" --snapshot <db> plant into a COPY of a real snapshot (readonly source;\n" +
|
|
40
|
-
" the copy is writable, the snapshot is never touched)"
|
|
45
|
+
" the copy is writable, the snapshot is never touched)\n" +
|
|
46
|
+
" --now <ISO> #458: pin the eval clock (full ISO 8601 instant) the\n" +
|
|
47
|
+
" recall-surface retrieve() scores against. Default: live.");
|
|
41
48
|
process.exitCode = 1;
|
|
42
49
|
process.exit(1);
|
|
43
50
|
}
|
|
@@ -45,16 +52,32 @@ async function main() {
|
|
|
45
52
|
const argv = process.argv.slice(2);
|
|
46
53
|
let reportPath;
|
|
47
54
|
let snapshotPath;
|
|
55
|
+
let nowIso;
|
|
48
56
|
for (let i = 0; i < argv.length; i++) {
|
|
49
57
|
if (argv[i] === "--snapshot" && argv[i + 1]) {
|
|
50
58
|
snapshotPath = argv[++i];
|
|
51
59
|
continue;
|
|
52
60
|
}
|
|
61
|
+
if (argv[i] === "--now" && argv[i + 1]) {
|
|
62
|
+
nowIso = argv[++i];
|
|
63
|
+
continue;
|
|
64
|
+
}
|
|
53
65
|
if (!argv[i].startsWith("--"))
|
|
54
66
|
reportPath = argv[i];
|
|
55
67
|
}
|
|
56
68
|
if (argv.includes("--help") || argv.includes("-h"))
|
|
57
69
|
usage();
|
|
70
|
+
// #458: fail-explicit — an invalid pin is an error, never a live fallback.
|
|
71
|
+
let now;
|
|
72
|
+
try {
|
|
73
|
+
now = (0, eval_clock_js_1.parsePinnedNow)(nowIso);
|
|
74
|
+
}
|
|
75
|
+
catch (err) {
|
|
76
|
+
console.error(`[planted-eval] ${err instanceof Error ? err.message : err}`);
|
|
77
|
+
process.exitCode = 1;
|
|
78
|
+
return;
|
|
79
|
+
}
|
|
80
|
+
console.log(`[planted-eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}`);
|
|
58
81
|
const workDir = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-planted-eval-"));
|
|
59
82
|
try {
|
|
60
83
|
const dbPath = (0, node_path_1.join)(workDir, "planted.db");
|
|
@@ -70,6 +93,7 @@ async function main() {
|
|
|
70
93
|
stateDir,
|
|
71
94
|
embedFn: embedder_js_1.embed,
|
|
72
95
|
acquireLock: capture_js_1.acquireCaptureLock, // real lock, but on the temp stateDir only
|
|
96
|
+
now: now ?? undefined, // #458: pinned clock (undefined = live)
|
|
73
97
|
});
|
|
74
98
|
console.log(`[planted-eval] gate completed in ${Date.now() - t0}ms`);
|
|
75
99
|
const report = (0, planted_harness_js_1.renderPlantedReport)({
|
|
@@ -78,6 +102,7 @@ async function main() {
|
|
|
78
102
|
: "real-text planted corpus (synthetic texts, real embeddings)",
|
|
79
103
|
generatedAt: new Date().toISOString(),
|
|
80
104
|
embedderLabel: "real bge-small-en-v1.5 (Xenova, fp32) — cosines measured, never asserted",
|
|
105
|
+
clock: (0, eval_clock_js_1.clockLabel)(now),
|
|
81
106
|
result,
|
|
82
107
|
});
|
|
83
108
|
const outPath = reportPath ?? DEFAULT_REPORT_PATH;
|
|
@@ -162,6 +162,10 @@ export interface PlantedGateOptions {
|
|
|
162
162
|
acquireLock?: typeof acquireCaptureLock;
|
|
163
163
|
/** Zone ceiling. Default: the production default (0.92). */
|
|
164
164
|
threshold?: number;
|
|
165
|
+
/** #458: the pinned clock for the recall-surface retrieve() (undefined =
|
|
166
|
+
* live). Fixtures keep their absolute createdAt strings — only the
|
|
167
|
+
* scoring instant is pinned. */
|
|
168
|
+
now?: Date;
|
|
165
169
|
}
|
|
166
170
|
/**
|
|
167
171
|
* Build (or plant into) the DB, run the production stage, measure the gate.
|
|
@@ -172,5 +176,8 @@ export declare function renderPlantedReport(args: {
|
|
|
172
176
|
corpusLabel: string;
|
|
173
177
|
generatedAt: string;
|
|
174
178
|
embedderLabel: string;
|
|
179
|
+
/** #458: clock-mode label ("pinned <ISO>" | "live") — the instant the
|
|
180
|
+
* recall-surface retrieve() scored against. */
|
|
181
|
+
clock: string;
|
|
175
182
|
result: PlantedGateResult;
|
|
176
183
|
}): string;
|
|
@@ -521,6 +521,7 @@ async function runPlantedGate(corpus, opts) {
|
|
|
521
521
|
const recall = await (0, retrieval_js_1.retrieve)(db, opts.embedFn, queryContent, {
|
|
522
522
|
limit: 3,
|
|
523
523
|
noStrengthen: true,
|
|
524
|
+
...(opts.now ? { now: opts.now } : {}), // #458: pinned clock (undefined = live)
|
|
524
525
|
});
|
|
525
526
|
// Every surfaced row must be its own walk terminal — no superseded
|
|
526
527
|
// ancestor competes as a truth — and the linked chain's terminal
|
|
@@ -593,6 +594,7 @@ function renderPlantedReport(args) {
|
|
|
593
594
|
const L = [];
|
|
594
595
|
L.push("# Planted-Pairs Resolution Gate (#393 increment A)\n");
|
|
595
596
|
L.push(`Corpus: ${args.corpusLabel} \nGenerated: ${args.generatedAt} \nEmbedder: ${args.embedderLabel} \n` +
|
|
597
|
+
`Clock: ${args.clock} (#458) \n` +
|
|
596
598
|
`Judge: ground-truth stub (perfect-judge semantics — machinery isolation, per the refine Q3 ruling). \n` +
|
|
597
599
|
`Floors: correctionMinSimilarity 0.75 (judged-band floor) · dedupAutoMergeThreshold 0.92 (zone ceiling).\n`);
|
|
598
600
|
L.push("## Gate — per planted pair\n");
|
|
@@ -11,7 +11,8 @@
|
|
|
11
11
|
* nouns (df 2-8), plus the fixed "Sirnäs" query (the owner's live case).
|
|
12
12
|
*
|
|
13
13
|
* Also carries the photo-comparison gates the sweep protocol defines:
|
|
14
|
-
* - case 2
|
|
14
|
+
* - case 2 unlinked-row identity (#449: unlinked rows' scores + mutual
|
|
15
|
+
* order must not drift; linked-row order drifts by design),
|
|
15
16
|
* - battery top-1 stability >= 90%,
|
|
16
17
|
* - no query losing a both-channel exact match from its top-3.
|
|
17
18
|
*/
|
|
@@ -47,8 +48,54 @@ export declare function deriveBatteryQueries(rows: Array<{
|
|
|
47
48
|
id: string;
|
|
48
49
|
content: string;
|
|
49
50
|
}>): BatteryQuery[];
|
|
50
|
-
/**
|
|
51
|
+
/**
|
|
52
|
+
* Case 2 gate (pre-#449): the FULL returned list must be byte-identical
|
|
53
|
+
* (ids + order). Superseded as the --compare gate by `compareCase2Unlinked`
|
|
54
|
+
* in #449 PR E — linked rows' connections credit RISES by design under the
|
|
55
|
+
* reshape, so linked-row order drifts; kept for reference/re-recording old
|
|
56
|
+
* photos.
|
|
57
|
+
*/
|
|
51
58
|
export declare function compareCase2(baseline: string[], current: string[]): boolean;
|
|
59
|
+
/** One recorded case-2 result row (the #449 per-row photo extension). */
|
|
60
|
+
export interface Case2IdentityRow {
|
|
61
|
+
key: string;
|
|
62
|
+
score: number;
|
|
63
|
+
connections: number;
|
|
64
|
+
}
|
|
65
|
+
/** The photo sections the identity gate reads (baseline or current). */
|
|
66
|
+
export interface Case2IdentityPhoto {
|
|
67
|
+
ftsHitRows?: number;
|
|
68
|
+
ids?: string[];
|
|
69
|
+
results?: Case2IdentityRow[];
|
|
70
|
+
}
|
|
71
|
+
export interface Case2IdentityResult {
|
|
72
|
+
pass: boolean;
|
|
73
|
+
failures: string[];
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* #449 (spec AC-7): the unlinked-row identity gate REPLACING the ids+order
|
|
77
|
+
* byte-gate for this PR — linked-row order drifts BY DESIGN (their
|
|
78
|
+
* connections credit rises), while everything UNLINKED must be untouched:
|
|
79
|
+
* the log term contributes exactly +0 at k = 0, so an unlinked row's score
|
|
80
|
+
* is bit-identical pre/post change. On the extended photo (per-row results
|
|
81
|
+
* with scores + measured connections) it checks that
|
|
82
|
+
* (i) FTS hits stay 0 on both sides (no both-channel candidate can exist
|
|
83
|
+
* — the case's structural inertness),
|
|
84
|
+
* (ii) every connections===0 row's recorded score is identical to the
|
|
85
|
+
* baseline's within CASE2_SCORE_TOLERANCE (matched by key — the
|
|
86
|
+
* log term contributes exactly +0 at k = 0; the tolerance absorbs
|
|
87
|
+
* the wall-clock drift of the time terms between recordings),
|
|
88
|
+
* (iii) the subsequence of unlinked keys in the returned list is
|
|
89
|
+
* identical (bit-identical scores + stable sort ⇒ their mutual
|
|
90
|
+
* order cannot flip), and
|
|
91
|
+
* (iv) every position change involves at least one LINKED row.
|
|
92
|
+
*
|
|
93
|
+
* A baseline photo without per-row results (pre-#449 harness) fails with a
|
|
94
|
+
* re-record instruction — the gate refuses to compare what it cannot see.
|
|
95
|
+
*/
|
|
96
|
+
export declare function compareCase2Unlinked(baseline: Case2IdentityPhoto, current: Case2IdentityPhoto, opts?: {
|
|
97
|
+
ftsHitMax?: number;
|
|
98
|
+
}): Case2IdentityResult;
|
|
52
99
|
export interface BatteryComparison {
|
|
53
100
|
top1Stability: number;
|
|
54
101
|
/** Stability with intended D3 promotions excluded: old top-1 was NOT a
|
|
@@ -12,7 +12,8 @@
|
|
|
12
12
|
* nouns (df 2-8), plus the fixed "Sirnäs" query (the owner's live case).
|
|
13
13
|
*
|
|
14
14
|
* Also carries the photo-comparison gates the sweep protocol defines:
|
|
15
|
-
* - case 2
|
|
15
|
+
* - case 2 unlinked-row identity (#449: unlinked rows' scores + mutual
|
|
16
|
+
* order must not drift; linked-row order drifts by design),
|
|
16
17
|
* - battery top-1 stability >= 90%,
|
|
17
18
|
* - no query losing a both-channel exact match from its top-3.
|
|
18
19
|
*/
|
|
@@ -20,6 +21,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
20
21
|
exports.STOPWORDS = exports.SIRNAS_QUERY = void 0;
|
|
21
22
|
exports.deriveBatteryQueries = deriveBatteryQueries;
|
|
22
23
|
exports.compareCase2 = compareCase2;
|
|
24
|
+
exports.compareCase2Unlinked = compareCase2Unlinked;
|
|
23
25
|
exports.batteryComparison = batteryComparison;
|
|
24
26
|
exports.SIRNAS_QUERY = "Sirnäs";
|
|
25
27
|
/** Compact English stopword block (battery derivation only, not product). */
|
|
@@ -88,11 +90,117 @@ function deriveBatteryQueries(rows) {
|
|
|
88
90
|
queries.push({ q: t, band: "proper-noun" });
|
|
89
91
|
return queries;
|
|
90
92
|
}
|
|
91
|
-
/**
|
|
93
|
+
/**
|
|
94
|
+
* Case 2 gate (pre-#449): the FULL returned list must be byte-identical
|
|
95
|
+
* (ids + order). Superseded as the --compare gate by `compareCase2Unlinked`
|
|
96
|
+
* in #449 PR E — linked rows' connections credit RISES by design under the
|
|
97
|
+
* reshape, so linked-row order drifts; kept for reference/re-recording old
|
|
98
|
+
* photos.
|
|
99
|
+
*/
|
|
92
100
|
function compareCase2(baseline, current) {
|
|
93
101
|
return (baseline.length === current.length &&
|
|
94
102
|
baseline.every((id, i) => id === current[i]));
|
|
95
103
|
}
|
|
104
|
+
/**
|
|
105
|
+
* Score-comparison tolerance for gate (ii) — the WALL-CLOCK drift of the
|
|
106
|
+
* recorded scores, not slack. retrieve() scores with the live clock
|
|
107
|
+
* (computeScore's `now`), and the time-curve + decay terms move every
|
|
108
|
+
* row's score ≈1.5e-5/hour of wall time between photo recordings (measured
|
|
109
|
+
* 2026-09-17: two SAME-BUILD runs minutes apart differ by ~1e-6; the
|
|
110
|
+
* embedder itself is bit-deterministic across processes). The tolerance
|
|
111
|
+
* covers ~7 hours of drift between recordings while staying 290x below the
|
|
112
|
+
* smallest real movement this gate exists to catch — the k=1 linked-row
|
|
113
|
+
* credit change (0.0367 − 0.0075 = 0.0292). TRUE bit-identity of the k=0
|
|
114
|
+
* term is pinned at the unit level instead (tests/ranking-flip.test.ts,
|
|
115
|
+
* fixed NOW, synthetic distances).
|
|
116
|
+
*/
|
|
117
|
+
const CASE2_SCORE_TOLERANCE = 1e-4;
|
|
118
|
+
/**
|
|
119
|
+
* #449 (spec AC-7): the unlinked-row identity gate REPLACING the ids+order
|
|
120
|
+
* byte-gate for this PR — linked-row order drifts BY DESIGN (their
|
|
121
|
+
* connections credit rises), while everything UNLINKED must be untouched:
|
|
122
|
+
* the log term contributes exactly +0 at k = 0, so an unlinked row's score
|
|
123
|
+
* is bit-identical pre/post change. On the extended photo (per-row results
|
|
124
|
+
* with scores + measured connections) it checks that
|
|
125
|
+
* (i) FTS hits stay 0 on both sides (no both-channel candidate can exist
|
|
126
|
+
* — the case's structural inertness),
|
|
127
|
+
* (ii) every connections===0 row's recorded score is identical to the
|
|
128
|
+
* baseline's within CASE2_SCORE_TOLERANCE (matched by key — the
|
|
129
|
+
* log term contributes exactly +0 at k = 0; the tolerance absorbs
|
|
130
|
+
* the wall-clock drift of the time terms between recordings),
|
|
131
|
+
* (iii) the subsequence of unlinked keys in the returned list is
|
|
132
|
+
* identical (bit-identical scores + stable sort ⇒ their mutual
|
|
133
|
+
* order cannot flip), and
|
|
134
|
+
* (iv) every position change involves at least one LINKED row.
|
|
135
|
+
*
|
|
136
|
+
* A baseline photo without per-row results (pre-#449 harness) fails with a
|
|
137
|
+
* re-record instruction — the gate refuses to compare what it cannot see.
|
|
138
|
+
*/
|
|
139
|
+
function compareCase2Unlinked(baseline, current, opts) {
|
|
140
|
+
const failures = [];
|
|
141
|
+
const ftsMax = opts?.ftsHitMax ?? 0;
|
|
142
|
+
if ((baseline.ftsHitRows ?? ftsMax) > ftsMax) {
|
|
143
|
+
failures.push(`baseline FTS hits ${baseline.ftsHitRows} (expected <= ${ftsMax})`);
|
|
144
|
+
}
|
|
145
|
+
if ((current.ftsHitRows ?? ftsMax) > ftsMax) {
|
|
146
|
+
failures.push(`current FTS hits ${current.ftsHitRows} (expected <= ${ftsMax})`);
|
|
147
|
+
}
|
|
148
|
+
if (!baseline.results || baseline.results.length === 0 || !current.results) {
|
|
149
|
+
failures.push("baseline photo lacks case-2 per-row results — re-record it with the #449 harness " +
|
|
150
|
+
"(the identity gate reads recorded scores + connections)");
|
|
151
|
+
return { pass: false, failures };
|
|
152
|
+
}
|
|
153
|
+
// (ii) every unlinked row's score identical to the baseline's within the
|
|
154
|
+
// wall-clock tolerance (both directions: no unlinked row may appear,
|
|
155
|
+
// disappear, or rescore beyond clock drift).
|
|
156
|
+
const baseByKey = new Map(baseline.results.map((r) => [r.key, r]));
|
|
157
|
+
for (const r of current.results) {
|
|
158
|
+
if (r.connections !== 0)
|
|
159
|
+
continue;
|
|
160
|
+
const b = baseByKey.get(r.key);
|
|
161
|
+
if (!b) {
|
|
162
|
+
failures.push(`unlinked row ${r.key} missing from baseline results`);
|
|
163
|
+
continue;
|
|
164
|
+
}
|
|
165
|
+
if (Math.abs(b.score - r.score) > CASE2_SCORE_TOLERANCE) {
|
|
166
|
+
failures.push(`unlinked row ${r.key} score ${r.score} vs baseline ${b.score} (|Δ| ${Math.abs(b.score - r.score).toExponential(2)} > ${CASE2_SCORE_TOLERANCE})`);
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
const curKeys = new Set(current.results.map((r) => r.key));
|
|
170
|
+
for (const b of baseline.results) {
|
|
171
|
+
if (b.connections === 0 && !curKeys.has(b.key)) {
|
|
172
|
+
failures.push(`baseline unlinked row ${b.key} missing from current results`);
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
// (iii) the unlinked-key subsequence (order included) is identical.
|
|
176
|
+
const unlinkedKeys = (rows) => rows.filter((r) => r.connections === 0).map((r) => r.key);
|
|
177
|
+
const baseSub = unlinkedKeys(baseline.results);
|
|
178
|
+
const curSub = unlinkedKeys(current.results);
|
|
179
|
+
if (baseSub.join("") !== curSub.join("")) {
|
|
180
|
+
failures.push(`unlinked-key subsequence changed: [${baseSub.join(", ")}] -> [${curSub.join(", ")}]`);
|
|
181
|
+
}
|
|
182
|
+
// (iv) every position change involves at least one linked row.
|
|
183
|
+
const connByKey = (photo) => new Map(photo.map((r) => [r.key, r.connections]));
|
|
184
|
+
const baseConn = connByKey(baseline.results);
|
|
185
|
+
const curConn = connByKey(current.results);
|
|
186
|
+
if (baseline.results.length !== current.results.length) {
|
|
187
|
+
failures.push(`returned-list length changed (${baseline.results.length} -> ${current.results.length})`);
|
|
188
|
+
}
|
|
189
|
+
else {
|
|
190
|
+
for (let i = 0; i < current.results.length; i++) {
|
|
191
|
+
const oldKey = baseline.results[i].key;
|
|
192
|
+
const newKey = current.results[i].key;
|
|
193
|
+
if (oldKey === newKey)
|
|
194
|
+
continue;
|
|
195
|
+
const oldLinked = (baseConn.get(oldKey) ?? 0) > 0;
|
|
196
|
+
const newLinked = (curConn.get(newKey) ?? 0) > 0;
|
|
197
|
+
if (!oldLinked && !newLinked) {
|
|
198
|
+
failures.push(`position ${i + 1} changed ${oldKey} -> ${newKey} with BOTH rows unlinked`);
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
return { pass: failures.length === 0, failures };
|
|
203
|
+
}
|
|
96
204
|
/** A both-channel row carrying the query term — the genuine-match signature. */
|
|
97
205
|
function isBothChannelExact(m, id, term, contentById) {
|
|
98
206
|
if (m)
|
|
@@ -1,11 +1,15 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
/**
|
|
3
3
|
* Search-trust ranking gate (#425) — the AC1/AC2/AC3 before/after photo.
|
|
4
|
+
* Extended by #449 (items 1+3, PR E): the case-2 photo gains per-row
|
|
5
|
+
* results (scores + measured connections) and its gate becomes the
|
|
6
|
+
* unlinked-row identity comparator; a new planted case 3 (linked_fallback)
|
|
7
|
+
* pins the connections-reshape expectations.
|
|
4
8
|
*
|
|
5
9
|
* npm run eval:ranking [--snapshot <db>] [--photo <out.json>]
|
|
6
10
|
* [--compare <baseline.json>] [--scoring '<json>']
|
|
7
11
|
*
|
|
8
|
-
* --snapshot <db> enables case
|
|
12
|
+
* --snapshot <db> enables case 4: the deterministic real-query battery
|
|
9
13
|
* against a READONLY snapshot (openSnapshot — never
|
|
10
14
|
* initDb; the snapshot is never touched). Also reports
|
|
11
15
|
* where the real "Sirnäs" memory ranks today (the
|
|
@@ -14,12 +18,19 @@
|
|
|
14
18
|
* data/ranking-eval-photo.json). ALWAYS written, even
|
|
15
19
|
* on a red run — the photo IS the measurement.
|
|
16
20
|
* --compare <json> gate against a previously recorded photo: case 2's
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
21
|
+
* UNLINKED rows must keep their scores (within the
|
|
22
|
+
* wall-clock tolerance) and mutual order — linked-row
|
|
23
|
+
* order drifts BY DESIGN under #449 — and the battery
|
|
24
|
+
* (when both sides have one) must hold top-1
|
|
25
|
+
* stability >= 90% with no query losing a both-channel
|
|
26
|
+
* exact match from its top-3.
|
|
20
27
|
* --scoring '<json>' a JSON object of ScoringWeights overrides passed to
|
|
21
28
|
* configureScoring (the eval/test seam, #408) — the
|
|
22
29
|
* sweep knob. Production runs omit it.
|
|
30
|
+
* --now <ISO> #458: pin the eval clock — planted lastAccessed offsets
|
|
31
|
+
* AND every retrieve() score against the SAME instant,
|
|
32
|
+
* so before/after photos are wall-clock-independent.
|
|
33
|
+
* Default: the live clock. The photo records the clock.
|
|
23
34
|
*
|
|
24
35
|
* Cases (fixtures: ranking-fixtures.ts, real bge-small-en-v1.5 embedder):
|
|
25
36
|
* 1. sirnas_exact_match — planted corpus in a THROWAWAY writable DB;
|
|
@@ -27,8 +38,17 @@
|
|
|
27
38
|
* Exit code reflects the verdict (a pre-fix run is EXPECTED to exit
|
|
28
39
|
* non-zero — that red photo is the before picture).
|
|
29
40
|
* 2. no_match_control — same corpus, a "mamma"-class query with zero FTS
|
|
30
|
-
* hits; records the full returned list
|
|
31
|
-
*
|
|
41
|
+
* hits; records the full returned list (ids + per-row results) for the
|
|
42
|
+
* unlinked-identity comparison.
|
|
43
|
+
* 3. linked_fallback (#449) — planted corpus in its OWN throwaway DB: a
|
|
44
|
+
* 16-link vs 20-link identical-content hub pair (saturation tie,
|
|
45
|
+
* declared byte-equal post-change) + an unlinked higher-similarity row
|
|
46
|
+
* vs a 4-link lower-similarity row (direction pin). Measured under the
|
|
47
|
+
* eval seam rrfCompositeWeight=1 (finalScore = composite exactly —
|
|
48
|
+
* identical-content hubs differ by an RRF epsilon at the default
|
|
49
|
+
* blend). RED under the pre-#449 linear term: the recorded
|
|
50
|
+
* before-picture.
|
|
51
|
+
* 4. real-query battery (needs --snapshot) — ~30 queries derived
|
|
32
52
|
* deterministically from the snapshot corpus (top/mid/rare-frequency
|
|
33
53
|
* distinctive terms + proper nouns) + the fixed "Sirnäs" query; top-8
|
|
34
54
|
* ids per query recorded to the photo.
|