@gamaze/hicortex 0.23.1 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/assets/dashboard.html +64 -13
  2. package/dist/calibration.d.ts +31 -0
  3. package/dist/calibration.js +38 -1
  4. package/dist/capture.d.ts +7 -0
  5. package/dist/capture.js +10 -1
  6. package/dist/dashboard.d.ts +32 -7
  7. package/dist/dashboard.js +50 -10
  8. package/dist/db.js +45 -0
  9. package/dist/dedup.js +2 -2
  10. package/dist/distill-queue.d.ts +203 -0
  11. package/dist/distill-queue.js +440 -0
  12. package/dist/health.d.ts +13 -1
  13. package/dist/health.js +6 -1
  14. package/dist/hosted-boot.d.ts +1 -1
  15. package/dist/index.js +17 -2
  16. package/dist/learnings-identity.js +20 -1
  17. package/dist/localhost-bypass.js +1 -1
  18. package/dist/mcp-server.js +92 -7
  19. package/dist/nightly.js +88 -1
  20. package/dist/nofit.d.ts +1 -1
  21. package/dist/nofit.js +1 -1
  22. package/dist/schema-prototypes.d.ts +1 -1
  23. package/dist/schema-prototypes.js +1 -1
  24. package/dist/status.d.ts +11 -0
  25. package/dist/status.js +25 -0
  26. package/dist/types.d.ts +20 -0
  27. package/hermes-plugin/hicortex/README.md +13 -6
  28. package/hermes-plugin/hicortex/__init__.py +7 -0
  29. package/hermes-plugin/hicortex/client.py +64 -2
  30. package/hermes-plugin/hicortex/plugin.yaml +1 -1
  31. package/hermes-plugin/hicortex/provider.py +24 -1
  32. package/opencode-plugin/hicortex/index.ts +24 -1
  33. package/package.json +2 -1
  34. package/pi-extension/hicortex/index.ts +24 -1
  35. package/server.json +2 -2
  36. package/dist/eval/decay-eval.d.ts +0 -111
  37. package/dist/eval/decay-eval.js +0 -214
  38. package/dist/eval/dups.d.ts +0 -100
  39. package/dist/eval/dups.js +0 -174
  40. package/dist/eval/eval-clock.d.ts +0 -32
  41. package/dist/eval/eval-clock.js +0 -47
  42. package/dist/eval/eval-db.d.ts +0 -25
  43. package/dist/eval/eval-db.js +0 -67
  44. package/dist/eval/graph-eval.d.ts +0 -89
  45. package/dist/eval/graph-eval.js +0 -246
  46. package/dist/eval/importance-eval.d.ts +0 -85
  47. package/dist/eval/importance-eval.js +0 -286
  48. package/dist/eval/planted-eval.d.ts +0 -30
  49. package/dist/eval/planted-eval.js +0 -122
  50. package/dist/eval/planted-fixtures.d.ts +0 -107
  51. package/dist/eval/planted-fixtures.js +0 -283
  52. package/dist/eval/planted-harness.d.ts +0 -183
  53. package/dist/eval/planted-harness.js +0 -651
  54. package/dist/eval/ranking-battery.d.ts +0 -125
  55. package/dist/eval/ranking-battery.js +0 -289
  56. package/dist/eval/ranking-eval.d.ts +0 -61
  57. package/dist/eval/ranking-eval.js +0 -554
  58. package/dist/eval/ranking-fixtures.d.ts +0 -117
  59. package/dist/eval/ranking-fixtures.js +0 -485
  60. package/dist/eval/recall-sweep.d.ts +0 -87
  61. package/dist/eval/recall-sweep.js +0 -1030
  62. package/dist/eval/reflection-census.d.ts +0 -19
  63. package/dist/eval/reflection-census.js +0 -25
  64. package/dist/eval/relevance-eval.d.ts +0 -178
  65. package/dist/eval/relevance-eval.js +0 -2240
  66. package/dist/eval/run-eval.d.ts +0 -20
  67. package/dist/eval/run-eval.js +0 -299
@@ -1,125 +0,0 @@
1
- /**
2
- * The real-query battery for the #425 ranking gate — the DETERMINISTIC,
3
- * dependency-free half (kept out of ranking-eval.ts so vitest can import
4
- * it without pulling the ONNX embedder; same split as planted-fixtures vs
5
- * planted-eval).
6
- *
7
- * The battery is a stable set of ~30 queries derived from the SNAPSHOT
8
- * CORPUS ITSELF: fixed df windows over content tokens, fixed tie-breaks —
9
- * same snapshot, same queries, every run. Bands: 10 high-df (50-400 docs),
10
- * 10 mid-df (10-50), 5 rare-df (3-10), 4 mid-sentence capitalized proper
11
- * nouns (df 2-8), plus the fixed "Sirnäs" query (the owner's live case).
12
- *
13
- * Also carries the photo-comparison gates the sweep protocol defines:
14
- * - case 2 unlinked-row identity (#449: unlinked rows' scores + mutual
15
- * order must not drift; linked-row order drifts by design),
16
- * - battery top-1 stability >= 90%,
17
- * - no query losing a both-channel exact match from its top-3.
18
- */
19
- export declare const SIRNAS_QUERY = "Sirn\u00E4s";
20
- /** Compact English stopword block (battery derivation only, not product). */
21
- export declare const STOPWORDS: Set<string>;
22
- export interface BatteryQuery {
23
- q: string;
24
- band: "proper-noun-fixed" | "high-df" | "mid-df" | "rare-df" | "proper-noun";
25
- }
26
- export interface BatteryResult {
27
- q: string;
28
- band: string;
29
- top8: string[];
30
- /** Per-result provenance (ranking-eval fills it; the gates read it). */
31
- meta?: Array<{
32
- id: string;
33
- source: string | null;
34
- term: boolean | null;
35
- }>;
36
- }
37
- export interface BatteryPhoto {
38
- queries: BatteryResult[];
39
- /** Rank (1-based) of the real Sirnäs memory for the fixed query; null = not in top-8. */
40
- sirnasRank: number | null;
41
- }
42
- /**
43
- * Derive the battery from live corpus contents — DETERMINISTIC given the
44
- * snapshot: fixed stopword list, fixed df windows, fixed tie-breaks
45
- * (df desc, then token asc).
46
- */
47
- export declare function deriveBatteryQueries(rows: Array<{
48
- id: string;
49
- content: string;
50
- }>): BatteryQuery[];
51
- /**
52
- * Case 2 gate (pre-#449): the FULL returned list must be byte-identical
53
- * (ids + order). Superseded as the --compare gate by `compareCase2Unlinked`
54
- * in #449 PR E — linked rows' connections credit RISES by design under the
55
- * reshape, so linked-row order drifts; kept for reference/re-recording old
56
- * photos.
57
- */
58
- export declare function compareCase2(baseline: string[], current: string[]): boolean;
59
- /** One recorded case-2 result row (the #449 per-row photo extension). */
60
- export interface Case2IdentityRow {
61
- key: string;
62
- score: number;
63
- connections: number;
64
- }
65
- /** The photo sections the identity gate reads (baseline or current). */
66
- export interface Case2IdentityPhoto {
67
- ftsHitRows?: number;
68
- ids?: string[];
69
- results?: Case2IdentityRow[];
70
- }
71
- export interface Case2IdentityResult {
72
- pass: boolean;
73
- failures: string[];
74
- }
75
- /**
76
- * #449 (spec AC-7): the unlinked-row identity gate REPLACING the ids+order
77
- * byte-gate for this PR — linked-row order drifts BY DESIGN (their
78
- * connections credit rises), while everything UNLINKED must be untouched:
79
- * the log term contributes exactly +0 at k = 0, so an unlinked row's score
80
- * is bit-identical pre/post change. On the extended photo (per-row results
81
- * with scores + measured connections) it checks that
82
- * (i) FTS hits stay 0 on both sides (no both-channel candidate can exist
83
- * — the case's structural inertness),
84
- * (ii) every connections===0 row's recorded score is identical to the
85
- * baseline's within CASE2_SCORE_TOLERANCE (matched by key — the
86
- * log term contributes exactly +0 at k = 0; the tolerance absorbs
87
- * the wall-clock drift of the time terms between recordings),
88
- * (iii) the subsequence of unlinked keys in the returned list is
89
- * identical (bit-identical scores + stable sort ⇒ their mutual
90
- * order cannot flip), and
91
- * (iv) every position change involves at least one LINKED row.
92
- *
93
- * A baseline photo without per-row results (pre-#449 harness) fails with a
94
- * re-record instruction — the gate refuses to compare what it cannot see.
95
- */
96
- export declare function compareCase2Unlinked(baseline: Case2IdentityPhoto, current: Case2IdentityPhoto, opts?: {
97
- ftsHitMax?: number;
98
- }): Case2IdentityResult;
99
- export interface BatteryComparison {
100
- top1Stability: number;
101
- /** Stability with intended D3 promotions excluded: old top-1 was NOT a
102
- * both-channel exact match and the new top-1 IS one (the boost doing its
103
- * job on real queries — reported, not gated). */
104
- top1StabilityExPromotions: number;
105
- changedTop1: string[];
106
- /** Queries whose top-3 LOST every both-channel exact match (the gate:
107
- * must be empty). */
108
- lostBothChannelExact: string[];
109
- /** Per-id exact-match (term-carrying) rows displaced from top-3 — reported
110
- * for the sweep record, not gated (multiple genuine matches make per-id
111
- * membership arbitrary). */
112
- lostExactMatches: string[];
113
- }
114
- /**
115
- * Battery gates vs a recorded baseline photo. `contentById` maps memory id →
116
- * content for the corpus BOTH photos were recorded against.
117
- *
118
- * Gates: (a) any query whose baseline top-3 held a BOTH-CHANNEL exact match
119
- * (vector+FTS agreeing on a term-carrying row) must still hold one — the
120
- * genuine-match guarantee never regresses; (b) top-1 stability. Intended D3
121
- * promotions (a both-channel exact match taking top-1 from a non-exact or
122
- * single-channel row) are counted separately — they are the feature firing,
123
- * and the sweep record shows both numbers.
124
- */
125
- export declare function batteryComparison(baseline: BatteryPhoto, current: BatteryPhoto, contentById: Map<string, string>): BatteryComparison;
@@ -1,289 +0,0 @@
1
- "use strict";
2
- /**
3
- * The real-query battery for the #425 ranking gate — the DETERMINISTIC,
4
- * dependency-free half (kept out of ranking-eval.ts so vitest can import
5
- * it without pulling the ONNX embedder; same split as planted-fixtures vs
6
- * planted-eval).
7
- *
8
- * The battery is a stable set of ~30 queries derived from the SNAPSHOT
9
- * CORPUS ITSELF: fixed df windows over content tokens, fixed tie-breaks —
10
- * same snapshot, same queries, every run. Bands: 10 high-df (50-400 docs),
11
- * 10 mid-df (10-50), 5 rare-df (3-10), 4 mid-sentence capitalized proper
12
- * nouns (df 2-8), plus the fixed "Sirnäs" query (the owner's live case).
13
- *
14
- * Also carries the photo-comparison gates the sweep protocol defines:
15
- * - case 2 unlinked-row identity (#449: unlinked rows' scores + mutual
16
- * order must not drift; linked-row order drifts by design),
17
- * - battery top-1 stability >= 90%,
18
- * - no query losing a both-channel exact match from its top-3.
19
- */
20
- Object.defineProperty(exports, "__esModule", { value: true });
21
- exports.STOPWORDS = exports.SIRNAS_QUERY = void 0;
22
- exports.deriveBatteryQueries = deriveBatteryQueries;
23
- exports.compareCase2 = compareCase2;
24
- exports.compareCase2Unlinked = compareCase2Unlinked;
25
- exports.batteryComparison = batteryComparison;
26
- exports.SIRNAS_QUERY = "Sirnäs";
27
- /** Compact English stopword block (battery derivation only, not product). */
28
- exports.STOPWORDS = new Set(("the a an and or but if while with without for from into onto about after before during under over " +
29
- "this that these those there here when where which who whom whose what why how all any both each " +
30
- "few more most other some such no nor not only own same so than too very can will just should now " +
31
- "was were been being are was is am be have has had having do does did doing would could might must " +
32
- "also because through between against above below off again further then once its it their they " +
33
- "them we you your our i me my he she his her him us of as at by in on to per via due using use " +
34
- "used using make made get gets got set sets let run ran running work works working need needs " +
35
- "added add adds change changed new next last first second one two three session sessions " +
36
- "memory memories project projects user users system systems thing things stuff lot bit way ways " +
37
- "time times day days week weeks month months year years")
38
- .split(/\s+/));
39
- /**
40
- * Derive the battery from live corpus contents — DETERMINISTIC given the
41
- * snapshot: fixed stopword list, fixed df windows, fixed tie-breaks
42
- * (df desc, then token asc).
43
- */
44
- function deriveBatteryQueries(rows) {
45
- const docs = rows.map((r) => r.content.toLowerCase());
46
- const df = new Map();
47
- for (const doc of docs) {
48
- const seen = new Set();
49
- for (const m of doc.matchAll(/[\p{L}\p{N}]{5,}/gu)) {
50
- const t = m[0];
51
- if (exports.STOPWORDS.has(t) || seen.has(t))
52
- continue;
53
- seen.add(t);
54
- df.set(t, (df.get(t) ?? 0) + 1);
55
- }
56
- }
57
- const inWindow = (lo, hi) => [...df.entries()]
58
- .filter(([t, d]) => d >= lo && d < hi && !exports.STOPWORDS.has(t))
59
- .sort((a, b) => (b[1] - a[1] !== 0 ? b[1] - a[1] : a[0].localeCompare(b[0])))
60
- .map(([t, d]) => [t, d]);
61
- const queries = [
62
- { q: exports.SIRNAS_QUERY, band: "proper-noun-fixed" },
63
- ];
64
- for (const [t] of inWindow(50, 401).slice(0, 10))
65
- queries.push({ q: t, band: "high-df" });
66
- for (const [t] of inWindow(10, 50).slice(0, 10))
67
- queries.push({ q: t, band: "mid-df" });
68
- for (const [t] of inWindow(3, 10).slice(0, 5))
69
- queries.push({ q: t, band: "rare-df" });
70
- // Proper nouns: capitalized, appearing mid-sentence (not the first word
71
- // after a sentence break — cheap deterministic heuristic), df 2-8.
72
- // "Sirnäs" itself may or may not surface here; the fixed query above
73
- // guarantees coverage.
74
- const capDf = new Map();
75
- for (const content of rows.map((r) => r.content)) {
76
- const seen = new Set();
77
- for (const m of content.matchAll(/[^\s.!?;:]\s+([\p{Lu}][\p{L}\p{N}\p{M}-]{3,})/gu)) {
78
- const t = m[1];
79
- if (!exports.STOPWORDS.has(t.toLowerCase()))
80
- seen.add(t);
81
- }
82
- for (const t of seen)
83
- capDf.set(t, (capDf.get(t) ?? 0) + 1);
84
- }
85
- const properNouns = [...capDf.entries()]
86
- .filter(([t, d]) => d >= 2 && d <= 8)
87
- .sort((a, b) => (b[1] - a[1] !== 0 ? b[1] - a[1] : a[0].localeCompare(b[0])))
88
- .map(([t]) => t);
89
- for (const t of properNouns.slice(0, 4))
90
- queries.push({ q: t, band: "proper-noun" });
91
- return queries;
92
- }
93
- /**
94
- * Case 2 gate (pre-#449): the FULL returned list must be byte-identical
95
- * (ids + order). Superseded as the --compare gate by `compareCase2Unlinked`
96
- * in #449 PR E — linked rows' connections credit RISES by design under the
97
- * reshape, so linked-row order drifts; kept for reference/re-recording old
98
- * photos.
99
- */
100
- function compareCase2(baseline, current) {
101
- return (baseline.length === current.length &&
102
- baseline.every((id, i) => id === current[i]));
103
- }
104
- /**
105
- * Score-comparison tolerance for gate (ii) — the WALL-CLOCK drift of the
106
- * recorded scores, not slack. retrieve() scores with the live clock
107
- * (computeScore's `now`), and the time-curve + decay terms move every
108
- * row's score ≈1.5e-5/hour of wall time between photo recordings (measured
109
- * 2026-09-17: two SAME-BUILD runs minutes apart differ by ~1e-6; the
110
- * embedder itself is bit-deterministic across processes). The tolerance
111
- * covers ~7 hours of drift between recordings while staying 290x below the
112
- * smallest real movement this gate exists to catch — the k=1 linked-row
113
- * credit change (0.0367 − 0.0075 = 0.0292). TRUE bit-identity of the k=0
114
- * term is pinned at the unit level instead (tests/ranking-flip.test.ts,
115
- * fixed NOW, synthetic distances).
116
- */
117
- const CASE2_SCORE_TOLERANCE = 1e-4;
118
- /**
119
- * #449 (spec AC-7): the unlinked-row identity gate REPLACING the ids+order
120
- * byte-gate for this PR — linked-row order drifts BY DESIGN (their
121
- * connections credit rises), while everything UNLINKED must be untouched:
122
- * the log term contributes exactly +0 at k = 0, so an unlinked row's score
123
- * is bit-identical pre/post change. On the extended photo (per-row results
124
- * with scores + measured connections) it checks that
125
- * (i) FTS hits stay 0 on both sides (no both-channel candidate can exist
126
- * — the case's structural inertness),
127
- * (ii) every connections===0 row's recorded score is identical to the
128
- * baseline's within CASE2_SCORE_TOLERANCE (matched by key — the
129
- * log term contributes exactly +0 at k = 0; the tolerance absorbs
130
- * the wall-clock drift of the time terms between recordings),
131
- * (iii) the subsequence of unlinked keys in the returned list is
132
- * identical (bit-identical scores + stable sort ⇒ their mutual
133
- * order cannot flip), and
134
- * (iv) every position change involves at least one LINKED row.
135
- *
136
- * A baseline photo without per-row results (pre-#449 harness) fails with a
137
- * re-record instruction — the gate refuses to compare what it cannot see.
138
- */
139
- function compareCase2Unlinked(baseline, current, opts) {
140
- const failures = [];
141
- const ftsMax = opts?.ftsHitMax ?? 0;
142
- if ((baseline.ftsHitRows ?? ftsMax) > ftsMax) {
143
- failures.push(`baseline FTS hits ${baseline.ftsHitRows} (expected <= ${ftsMax})`);
144
- }
145
- if ((current.ftsHitRows ?? ftsMax) > ftsMax) {
146
- failures.push(`current FTS hits ${current.ftsHitRows} (expected <= ${ftsMax})`);
147
- }
148
- if (!baseline.results || baseline.results.length === 0 || !current.results) {
149
- failures.push("baseline photo lacks case-2 per-row results — re-record it with the #449 harness " +
150
- "(the identity gate reads recorded scores + connections)");
151
- return { pass: false, failures };
152
- }
153
- // (ii) every unlinked row's score identical to the baseline's within the
154
- // wall-clock tolerance (both directions: no unlinked row may appear,
155
- // disappear, or rescore beyond clock drift).
156
- const baseByKey = new Map(baseline.results.map((r) => [r.key, r]));
157
- for (const r of current.results) {
158
- if (r.connections !== 0)
159
- continue;
160
- const b = baseByKey.get(r.key);
161
- if (!b) {
162
- failures.push(`unlinked row ${r.key} missing from baseline results`);
163
- continue;
164
- }
165
- if (Math.abs(b.score - r.score) > CASE2_SCORE_TOLERANCE) {
166
- failures.push(`unlinked row ${r.key} score ${r.score} vs baseline ${b.score} (|Δ| ${Math.abs(b.score - r.score).toExponential(2)} > ${CASE2_SCORE_TOLERANCE})`);
167
- }
168
- }
169
- const curKeys = new Set(current.results.map((r) => r.key));
170
- for (const b of baseline.results) {
171
- if (b.connections === 0 && !curKeys.has(b.key)) {
172
- failures.push(`baseline unlinked row ${b.key} missing from current results`);
173
- }
174
- }
175
- // (iii) the unlinked-key subsequence (order included) is identical.
176
- const unlinkedKeys = (rows) => rows.filter((r) => r.connections === 0).map((r) => r.key);
177
- const baseSub = unlinkedKeys(baseline.results);
178
- const curSub = unlinkedKeys(current.results);
179
- if (baseSub.join("") !== curSub.join("")) {
180
- failures.push(`unlinked-key subsequence changed: [${baseSub.join(", ")}] -> [${curSub.join(", ")}]`);
181
- }
182
- // (iv) every position change involves at least one linked row.
183
- const connByKey = (photo) => new Map(photo.map((r) => [r.key, r.connections]));
184
- const baseConn = connByKey(baseline.results);
185
- const curConn = connByKey(current.results);
186
- if (baseline.results.length !== current.results.length) {
187
- failures.push(`returned-list length changed (${baseline.results.length} -> ${current.results.length})`);
188
- }
189
- else {
190
- for (let i = 0; i < current.results.length; i++) {
191
- const oldKey = baseline.results[i].key;
192
- const newKey = current.results[i].key;
193
- if (oldKey === newKey)
194
- continue;
195
- const oldLinked = (baseConn.get(oldKey) ?? 0) > 0;
196
- const newLinked = (curConn.get(newKey) ?? 0) > 0;
197
- if (!oldLinked && !newLinked) {
198
- failures.push(`position ${i + 1} changed ${oldKey} -> ${newKey} with BOTH rows unlinked`);
199
- }
200
- }
201
- }
202
- return { pass: failures.length === 0, failures };
203
- }
204
- /** A both-channel row carrying the query term — the genuine-match signature. */
205
- function isBothChannelExact(m, id, term, contentById) {
206
- if (m)
207
- return m.source === "both" && m.term === true;
208
- // Baseline photos recorded before `meta` existed: fall back to content
209
- // containment; conservatively treat source as unknown (not "both").
210
- const content = contentById.get(id);
211
- return Boolean(content && content.toLowerCase().includes(term));
212
- }
213
- /**
214
- * Battery gates vs a recorded baseline photo. `contentById` maps memory id →
215
- * content for the corpus BOTH photos were recorded against.
216
- *
217
- * Gates: (a) any query whose baseline top-3 held a BOTH-CHANNEL exact match
218
- * (vector+FTS agreeing on a term-carrying row) must still hold one — the
219
- * genuine-match guarantee never regresses; (b) top-1 stability. Intended D3
220
- * promotions (a both-channel exact match taking top-1 from a non-exact or
221
- * single-channel row) are counted separately — they are the feature firing,
222
- * and the sweep record shows both numbers.
223
- */
224
- function batteryComparison(baseline, current, contentById) {
225
- const baseByQ = new Map(baseline.queries.map((b) => [b.q, b]));
226
- const baseMeta = (b, rank) => b.meta?.[rank];
227
- let stable = 0;
228
- let stableExPromotions = 0;
229
- let compared = 0;
230
- const changedTop1 = [];
231
- const lostBothChannelExact = [];
232
- const lostExactMatches = [];
233
- for (const cur of current.queries) {
234
- const base = baseByQ.get(cur.q);
235
- if (!base)
236
- continue;
237
- compared++;
238
- if (cur.top8[0] === base.top8[0]) {
239
- stable++;
240
- stableExPromotions++;
241
- continue;
242
- }
243
- changedTop1.push(cur.q);
244
- const term = cur.q.toLowerCase();
245
- const oldExact = baseMeta(base, 0) !== undefined
246
- ? isBothChannelExact(baseMeta(base, 0), base.top8[0], term, contentById)
247
- : null;
248
- const newExact = cur.meta?.[0] !== undefined
249
- ? isBothChannelExact(cur.meta[0], cur.top8[0], term, contentById)
250
- : null;
251
- // An intended promotion: the new top-1 is a both-channel exact match the
252
- // old top-1 was not. (Null = provenance unknown — pre-meta baseline —
253
- // counts as a plain change.)
254
- if (newExact === true && oldExact === false)
255
- stableExPromotions++;
256
- }
257
- for (const cur of current.queries) {
258
- const base = baseByQ.get(cur.q);
259
- if (!base || cur.q === exports.SIRNAS_QUERY)
260
- continue;
261
- const term = cur.q.toLowerCase();
262
- const curTop3 = new Set(cur.top8.slice(0, 3));
263
- let baseHadBothChannelExact = false;
264
- for (let i = 0; i < Math.min(3, base.top8.length); i++) {
265
- const id = base.top8[i];
266
- const content = contentById.get(id);
267
- const carries = Boolean(content && content.toLowerCase().includes(term));
268
- if (carries && !curTop3.has(id))
269
- lostExactMatches.push(`${cur.q}: ${id.slice(0, 8)}`);
270
- const m = base.meta?.[i];
271
- if (m ? m.source === "both" && m.term === true : false)
272
- baseHadBothChannelExact = true;
273
- }
274
- if (baseHadBothChannelExact) {
275
- const stillHas = cur.top8
276
- .slice(0, 3)
277
- .some((id, i) => cur.meta?.[i]?.source === "both" && cur.meta?.[i]?.term === true);
278
- if (!stillHas)
279
- lostBothChannelExact.push(cur.q);
280
- }
281
- }
282
- return {
283
- top1Stability: compared > 0 ? stable / compared : 1,
284
- top1StabilityExPromotions: compared > 0 ? stableExPromotions / compared : 1,
285
- changedTop1,
286
- lostBothChannelExact,
287
- lostExactMatches,
288
- };
289
- }
@@ -1,61 +0,0 @@
1
- #!/usr/bin/env node
2
- /**
3
- * Search-trust ranking gate (#425) — the AC1/AC2/AC3 before/after photo.
4
- * Extended by #449 (items 1+3, PR E): the case-2 photo gains per-row
5
- * results (scores + measured connections) and its gate becomes the
6
- * unlinked-row identity comparator; a new planted case 3 (linked_fallback)
7
- * pins the connections-reshape expectations.
8
- *
9
- * npm run eval:ranking [--snapshot <db>] [--photo <out.json>]
10
- * [--compare <baseline.json>] [--scoring '<json>']
11
- *
12
- * --snapshot <db> enables case 4: the deterministic real-query battery
13
- * against a READONLY snapshot (openSnapshot — never
14
- * initDb; the snapshot is never touched). Also reports
15
- * where the real "Sirnäs" memory ranks today (the
16
- * measured red photo pre-fix, AC10 pre-condition).
17
- * --photo <out.json> where to write the machine photo (defaults to
18
- * data/ranking-eval-photo.json). ALWAYS written, even
19
- * on a red run — the photo IS the measurement.
20
- * --compare <json> gate against a previously recorded photo: case 2's
21
- * UNLINKED rows must keep their scores (within the
22
- * wall-clock tolerance) and mutual order — linked-row
23
- * order drifts BY DESIGN under #449 — and the battery
24
- * (when both sides have one) must hold top-1
25
- * stability >= 90% with no query losing a both-channel
26
- * exact match from its top-3.
27
- * --scoring '<json>' a JSON object of ScoringWeights overrides passed to
28
- * configureScoring (the eval/test seam, #408) — the
29
- * sweep knob. Production runs omit it.
30
- * --now <ISO> #458: pin the eval clock — planted lastAccessed offsets
31
- * AND every retrieve() score against the SAME instant,
32
- * so before/after photos are wall-clock-independent.
33
- * Default: the live clock. The photo records the clock.
34
- *
35
- * Cases (fixtures: ranking-fixtures.ts, real bge-small-en-v1.5 embedder):
36
- * 1. sirnas_exact_match — planted corpus in a THROWAWAY writable DB;
37
- * declared expectation: the proper-noun exact-match row returns #1.
38
- * Exit code reflects the verdict (a pre-fix run is EXPECTED to exit
39
- * non-zero — that red photo is the before picture).
40
- * 2. no_match_control — same corpus, a "mamma"-class query with zero FTS
41
- * hits; records the full returned list (ids + per-row results) for the
42
- * unlinked-identity comparison.
43
- * 3. linked_fallback (#449) — planted corpus in its OWN throwaway DB: a
44
- * 16-link vs 20-link identical-content hub pair (saturation tie,
45
- * declared byte-equal post-change) + an unlinked higher-similarity row
46
- * vs a 4-link lower-similarity row (direction pin). Measured under the
47
- * eval seam rrfCompositeWeight=1 (finalScore = composite exactly —
48
- * identical-content hubs differ by an RRF epsilon at the default
49
- * blend). RED under the pre-#449 linear term: the recorded
50
- * before-picture.
51
- * 4. real-query battery (needs --snapshot) — ~30 queries derived
52
- * deterministically from the snapshot corpus (top/mid/rare-frequency
53
- * distinctive terms + proper nouns) + the fixed "Sirnäs" query; top-8
54
- * ids per query recorded to the photo.
55
- *
56
- * Honesty invariants: similarities are MEASURED (real embedder, embed-once,
57
- * queryEmbedding reuse), noStrengthen on every retrieve() (the probe never
58
- * perturbs the store), readonly snapshot access, and the planted corpora
59
- * live in mkdtemp dirs removed on exit. Never touches ~/.hicortex.
60
- */
61
- export {};