@gamaze/hicortex 0.21.0 → 0.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +282 -27
- package/dist/calibration.d.ts +77 -12
- package/dist/calibration.js +82 -15
- package/dist/capture-health.d.ts +42 -5
- package/dist/capture-health.js +57 -7
- package/dist/capture-pause.d.ts +7 -4
- package/dist/capture-pause.js +7 -4
- package/dist/consolidate.d.ts +19 -0
- package/dist/consolidate.js +125 -51
- package/dist/dashboard.d.ts +41 -16
- package/dist/dashboard.js +110 -41
- package/dist/db.js +16 -0
- package/dist/dedup.js +4 -1
- package/dist/eval/decay-eval.d.ts +6 -5
- package/dist/eval/decay-eval.js +10 -48
- package/dist/eval/eval-clock.d.ts +32 -0
- package/dist/eval/eval-clock.js +47 -0
- package/dist/eval/graph-eval.d.ts +15 -2
- package/dist/eval/graph-eval.js +51 -5
- package/dist/eval/planted-eval.d.ts +4 -0
- package/dist/eval/planted-eval.js +27 -2
- package/dist/eval/planted-harness.d.ts +7 -0
- package/dist/eval/planted-harness.js +2 -0
- package/dist/eval/ranking-battery.d.ts +49 -2
- package/dist/eval/ranking-battery.js +110 -2
- package/dist/eval/ranking-eval.d.ts +26 -6
- package/dist/eval/ranking-eval.js +197 -34
- package/dist/eval/ranking-fixtures.d.ts +46 -6
- package/dist/eval/ranking-fixtures.js +268 -9
- package/dist/eval/recall-sweep.d.ts +7 -2
- package/dist/eval/recall-sweep.js +42 -13
- package/dist/eval/relevance-eval.d.ts +115 -1
- package/dist/eval/relevance-eval.js +318 -32
- package/dist/eval/run-eval.d.ts +7 -4
- package/dist/eval/run-eval.js +37 -10
- package/dist/graph.d.ts +1 -2
- package/dist/graph.js +3 -4
- package/dist/mcp-server.js +3 -2
- package/dist/nofit.d.ts +2 -1
- package/dist/nofit.js +2 -1
- package/dist/recall-index.d.ts +3 -2
- package/dist/recall-index.js +3 -2
- package/dist/retrieval.d.ts +43 -18
- package/dist/retrieval.js +148 -86
- package/dist/status.js +18 -0
- package/dist/storage.d.ts +2 -1
- package/dist/storage.js +6 -1
- package/dist/types.d.ts +25 -4
- package/package.json +1 -1
- package/server.json +4 -4
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
* issue (the owner's live console test, 2026-09-13): a user searches a
|
|
7
7
|
* DISTINCTIVE PROPER NOUN they know exists; the one memory carrying that
|
|
8
8
|
* token has the best similarity of all candidates AND the literal FTS hit,
|
|
9
|
-
* yet ranks behind
|
|
9
|
+
* yet ranks behind high-strength but less-relevant memories that win on
|
|
10
10
|
* effective strength + graph connections.
|
|
11
11
|
*
|
|
12
12
|
* (a) sirnas_exact_match — the AC1 fixture: a low-strength, unlinked,
|
|
@@ -21,14 +21,19 @@
|
|
|
21
21
|
* under the ranking work (no both-channel candidates exist, so the
|
|
22
22
|
* boost can never fire — the case pins that the no-regression
|
|
23
23
|
* guarantee is structural, not lucky).
|
|
24
|
+
* (c) linked_fallback — the #449 (items 1+3) connections fixture: a
|
|
25
|
+
* 16-link vs 20-link identical-content hub pair (the saturation tie)
|
|
26
|
+
* plus an unlinked higher-similarity row vs a 4-link lower-similarity
|
|
27
|
+
* row (the direction pin). Planted in its OWN throwaway DB — the 40
|
|
28
|
+
* link-neighbor rows would pollute the case-1/2 corpus.
|
|
24
29
|
*
|
|
25
30
|
* Honesty rules (the planted-pairs #393 A discipline):
|
|
26
31
|
* - The texts are synthetic and generic (this package publishes to npm —
|
|
27
32
|
* no real infrastructure, people, or fleet detail), but SHAPED like the
|
|
28
33
|
* field failure: the proper noun is invented (it must appear nowhere
|
|
29
|
-
* else — verified by an invariant test), the rivals
|
|
30
|
-
*
|
|
31
|
-
*
|
|
34
|
+
* else — verified by an invariant test), the rivals carry the metadata
|
|
35
|
+
* the production pipeline itself writes (base strength, access history,
|
|
36
|
+
* link counts).
|
|
32
37
|
* - Similarities are MEASURED at run time with the real bge-small-en-v1.5
|
|
33
38
|
* embedder (ranking-eval.ts) — never asserted into existence. The
|
|
34
39
|
* declared expectation is about ORDER under the shipped ranking, and
|
|
@@ -38,7 +43,7 @@
|
|
|
38
43
|
* fixture exercises the same rows retrieval() will see.
|
|
39
44
|
*/
|
|
40
45
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
41
|
-
exports.REAL_SIRNAS_MEMORY_ID = exports.NO_MATCH_FIXTURE = exports.SIRNAS_FIXTURE = exports.RANKING_SOURCE_AGENT = exports.RANKING_PROJECT = void 0;
|
|
46
|
+
exports.LINKED_FALLBACK_FIXTURE = exports.REAL_SIRNAS_MEMORY_ID = exports.NO_MATCH_FIXTURE = exports.SIRNAS_FIXTURE = exports.RANKING_SOURCE_AGENT = exports.RANKING_PROJECT = void 0;
|
|
42
47
|
exports.checkFixtureInvariants = checkFixtureInvariants;
|
|
43
48
|
/** All planted rows share project + source_agent (metadata never refuses). */
|
|
44
49
|
exports.RANKING_PROJECT = "ranking-eval";
|
|
@@ -50,13 +55,13 @@ exports.RANKING_SOURCE_AGENT = "ranking-eval";
|
|
|
50
55
|
// The target mirrors the field case's SHAPE with invented names: a home
|
|
51
56
|
// harbour memory carrying one distinctive proper noun, born weak (base 0.35,
|
|
52
57
|
// the "routine-but-curated" band), never accessed, unlinked, a few months
|
|
53
|
-
// old. The rivals are what the decay model produces for early-
|
|
58
|
+
// old. The rivals are what the decay model produces for early high-strength
|
|
54
59
|
// technical rows: base 0.90-0.95, mutual links, access history, created
|
|
55
60
|
// BEFORE the target. They share the harbour/boating domain vocabulary — so
|
|
56
61
|
// a bare proper-noun query lands in their neighborhood — but none carries
|
|
57
62
|
// the proper noun.
|
|
58
63
|
const SIRNAS_ROWS = [
|
|
59
|
-
// --- rivals:
|
|
64
|
+
// --- rivals: high-strength, connected, older, domain-adjacent ---
|
|
60
65
|
{
|
|
61
66
|
key: "rival_fees",
|
|
62
67
|
content: "Harbour fees and guest dock rules: pay at the machine by the fuel quay, electricity on the guest dock is metered, " +
|
|
@@ -158,11 +163,25 @@ const SIRNAS_ROWS = [
|
|
|
158
163
|
exports.SIRNAS_FIXTURE = {
|
|
159
164
|
key: "sirnas_exact_match",
|
|
160
165
|
description: "AC1: a distinctive proper-noun query must rank the exact-match (both-channel) memory first, " +
|
|
161
|
-
"ahead of
|
|
166
|
+
"ahead of high-strength domain-adjacent rivals that win on strength + connections.",
|
|
162
167
|
rows: SIRNAS_ROWS,
|
|
163
168
|
query: "Vindarsvik",
|
|
164
169
|
distinctiveToken: "vindarsvik",
|
|
165
170
|
expectedTop1: "target_harbour",
|
|
171
|
+
// Undirected degrees (#449 invariant extension): the rivals' mutual-link
|
|
172
|
+
// cluster — fees/channel 4 each, maintenance 3, passage 1 (each linksTo
|
|
173
|
+
// entry is one link row; both directions of a mutual pair count).
|
|
174
|
+
expectedUndirectedDegree: {
|
|
175
|
+
rival_fees: 4,
|
|
176
|
+
rival_channel: 4,
|
|
177
|
+
rival_maintenance: 3,
|
|
178
|
+
rival_passage: 1,
|
|
179
|
+
target_harbour: 0,
|
|
180
|
+
fill_recipe: 0,
|
|
181
|
+
fill_books: 0,
|
|
182
|
+
fill_lang: 0,
|
|
183
|
+
fill_garden: 0,
|
|
184
|
+
},
|
|
166
185
|
};
|
|
167
186
|
// ---------------------------------------------------------------------------
|
|
168
187
|
// Case 2 — no_match_control (AC2)
|
|
@@ -186,13 +205,213 @@ exports.NO_MATCH_FIXTURE = {
|
|
|
186
205
|
// records the list; the gate compares runs).
|
|
187
206
|
expectedTop1: "",
|
|
188
207
|
};
|
|
189
|
-
/** The battery runs against a real snapshot copy (ranking-eval.ts case
|
|
208
|
+
/** The battery runs against a real snapshot copy (ranking-eval.ts case 4). */
|
|
190
209
|
exports.REAL_SIRNAS_MEMORY_ID = "12980c9d-0eba-4963-95eb-0548bc2aa93e";
|
|
191
210
|
// ---------------------------------------------------------------------------
|
|
211
|
+
// Case 3 — linked_fallback (#449 items 1+3, PR E)
|
|
212
|
+
// ---------------------------------------------------------------------------
|
|
213
|
+
//
|
|
214
|
+
// The connections-term fixture for the log-saturation reshape. Two probes in
|
|
215
|
+
// ONE corpus (planted in its own throwaway DB by ranking-eval.ts — the 40
|
|
216
|
+
// link-neighbor rows stay out of the case-1/2 corpus):
|
|
217
|
+
//
|
|
218
|
+
// saturation pair — two hubs with IDENTICAL content strings and identical
|
|
219
|
+
// metadata (same baseStrength, accessCount, lastAccessedDaysAgo,
|
|
220
|
+
// createdAt, memoryType); hub A carries exactly 16 undirected links, hub B
|
|
221
|
+
// exactly 20, through disjoint off-domain neighbor rows. Declared
|
|
222
|
+
// expectation POST-change (Option A, K = 16): the two hubs' scores are
|
|
223
|
+
// byte-EQUAL — the term min(1, log1p(k)/log1p(16)) saturates at 1 for both.
|
|
224
|
+
// Under the pre-#449 linear term (connectionCount / maxConnections = k/20)
|
|
225
|
+
// they differ by 0.2 × 0.15 ≈ 0.03 composite — that red is the recorded
|
|
226
|
+
// before-picture; the harness does NOT make it pass.
|
|
227
|
+
//
|
|
228
|
+
// direction pair — an unlinked row carrying the query's distinctive proper
|
|
229
|
+
// noun (the highest measured similarity) vs a linked row (exactly 4 links)
|
|
230
|
+
// whose MEASURED similarity is engineered ~0.20 LOWER, with equal metadata
|
|
231
|
+
// otherwise (equal strength/dates). Declared expectation (Option A, owner
|
|
232
|
+
// decision 2026-09-17): the unlinked row ranks ABOVE the linked one — a
|
|
233
|
+
// 0.20 gap is worth 0.10 composite at the 0.50 similarity weight, more
|
|
234
|
+
// than the k=4 connections credit (0.0852 post-change). The gap must
|
|
235
|
+
// exceed 0.171 for the pin to be sound at k = 4; the wording is tuned
|
|
236
|
+
// until the real-embedder measurement (recorded in the photo/report)
|
|
237
|
+
// lands at ~0.20.
|
|
238
|
+
//
|
|
239
|
+
// Neighbors are DISTINCT filler rows, disjoint between the two hubs and from
|
|
240
|
+
// the direction pair's cluster, all off-domain from the query — they exist
|
|
241
|
+
// to carry the hub/dir-linked degrees and nothing else.
|
|
242
|
+
const HUB_CONTENT = "Community noticeboard archive: the autumn fair needs tombola volunteers, choir practice moves to " +
|
|
243
|
+
"Thursday evenings, and the book-swap shelf now carries the walking-map folder.";
|
|
244
|
+
/** Off-domain one-line filler rows (deterministic keys, uniform metadata). */
|
|
245
|
+
function fillerRows(prefix, contents) {
|
|
246
|
+
return contents.map((content, i) => ({
|
|
247
|
+
key: `${prefix}_n${i}`,
|
|
248
|
+
content,
|
|
249
|
+
createdAt: "2026-04-15T09:00:00.000Z",
|
|
250
|
+
memoryType: "experience",
|
|
251
|
+
baseStrength: 0.5,
|
|
252
|
+
accessCount: 0,
|
|
253
|
+
lastAccessedDaysAgo: 150,
|
|
254
|
+
linksTo: [],
|
|
255
|
+
}));
|
|
256
|
+
}
|
|
257
|
+
const HUB_A_NEIGHBOR_CONTENTS = [
|
|
258
|
+
"Slow-cooker beans: soak overnight, then four hours on low with a bay leaf and a piece of smoked pork.",
|
|
259
|
+
"The second detective novel in the series drags until the lighthouse keeper's ledger turns up.",
|
|
260
|
+
"German grammar drill: subordinate clauses shove the verb to the end, so read the whole sentence before translating.",
|
|
261
|
+
"Raised-bed rotation: legumes follow brassicas, and lime the bed the cabbage family just emptied.",
|
|
262
|
+
"October weather pattern: the first night frosts usually arrive right after the harvest moon.",
|
|
263
|
+
"Bike maintenance: true the wheel by tightening spokes a quarter turn at a time, listening for the ping.",
|
|
264
|
+
"Chess opening note: against the locked center, flank with pawns only after the pieces are developed.",
|
|
265
|
+
"The documentary about the alpine tunnels spent too long on the geologists' cafeteria.",
|
|
266
|
+
"Coffee grinding: a burr grind for the pour-over; the blade grinder is fine enough only for the French press.",
|
|
267
|
+
"Hiking the ridge trail: start before the thermals build, and carry an extra liter once the pass is in sight.",
|
|
268
|
+
"Piano practice: separate hands at half tempo until the fingering is automatic, then join them slowly.",
|
|
269
|
+
"Sewing patch pockets: interface the fabric, and topstitch the same distance from every edge.",
|
|
270
|
+
"Bird notes: the chiffchaff repeats its own name all day; the willow warbler slides down the scale.",
|
|
271
|
+
"Sourdough rhythm: feed the starter after work, shape in the morning, bake when the poke test springs back halfway.",
|
|
272
|
+
"Replacing the washing-machine hose: shut the valve, expect a cup of water, and check the jubilee clip.",
|
|
273
|
+
"Audiobook note: the narrator's accent makes every ship's name sound like a Hebridean island.",
|
|
274
|
+
];
|
|
275
|
+
const HUB_B_NEIGHBOR_CONTENTS = [
|
|
276
|
+
"Winter tyres: swap before the first ice, and store the summer set flat in the dark basement.",
|
|
277
|
+
"The boardgame group settled on tiles-and-traders for the long evenings; the card game stays in the drawer.",
|
|
278
|
+
"Compost thermometer: the pile is cooking properly when it reads hotter than a warm bath for a week.",
|
|
279
|
+
"Frost-proof terracotta: empty the pots, stack them dry, and the succulents overwinter on the south sill.",
|
|
280
|
+
"The photography club theme this month is doors; shoot in morning shade and bracket the exposures.",
|
|
281
|
+
"Robot vacuum note: keep the charging dock's floor clear, or it nudges the rug and gives up mid-run.",
|
|
282
|
+
"Knife sharpening: the ceramic rod for touch-ups, the whetstone when the tomato skin resists.",
|
|
283
|
+
"The lecture on Roman concrete held the room until the slide about seawater pozzolana.",
|
|
284
|
+
"Necklace repair: jump rings open sideways, never apart, or the circle never closes again.",
|
|
285
|
+
"Ferry timetable change: the winter crossing departs an hour earlier and stops running on Sunday evenings.",
|
|
286
|
+
"Ski waxing: glide wax for the thaw, grip wax for the freeze, and never both in the same zone.",
|
|
287
|
+
"The archive boxes are labelled by decade now; the negatives live in the fire cupboard.",
|
|
288
|
+
"Beekeeping note: heft the hive — if one side lifts easily, the winter stores are short.",
|
|
289
|
+
"Jigsaw strategy: edges first, then the sky, and give the lamp a clean bulb for the blue hours.",
|
|
290
|
+
"The museum's clock gallery is closed for restoration until the pendulum clock returns.",
|
|
291
|
+
"Preserving basil: pesto freezes flat in bags; the leaves kept in water go black within a week.",
|
|
292
|
+
"Draft excluder: the brush strip on the letterbox cut the hallway draught more than the door seal did.",
|
|
293
|
+
"The organ recital opened with a toccata nobody knew and closed with everyone humming it.",
|
|
294
|
+
"Train-set wiring: one feeder per rail section, or the far loop slows to a crawl.",
|
|
295
|
+
"The pottery class moved on to wheel work; my first cylinder collapsed into a bowl with charm.",
|
|
296
|
+
];
|
|
297
|
+
const DIR_NEIGHBOR_CONTENTS = [
|
|
298
|
+
"Vaccine appointment bookkeeping: the clinic sends both a text and a letter, and only the letter carries the time.",
|
|
299
|
+
"Reading glasses: the cheap pair from the station shop outlasted the optician's by two years.",
|
|
300
|
+
"The new debit card needs the reader for anything over the limit; the shop keeps the old machine as backup.",
|
|
301
|
+
"Laundering the down jacket: tennis balls in the dryer, low heat, and patience.",
|
|
302
|
+
];
|
|
303
|
+
const hubNeighbors = (prefix, contents) => contents.map((_, i) => `${prefix}_n${i}`);
|
|
304
|
+
const LINKED_FALLBACK_ROWS = [
|
|
305
|
+
// --- saturation pair: identical content + metadata, degrees 16 vs 20 ---
|
|
306
|
+
{
|
|
307
|
+
key: "hub_sat_16",
|
|
308
|
+
content: HUB_CONTENT,
|
|
309
|
+
createdAt: "2026-05-10T09:00:00.000Z",
|
|
310
|
+
memoryType: "knowledge",
|
|
311
|
+
baseStrength: 0.6,
|
|
312
|
+
accessCount: 0,
|
|
313
|
+
lastAccessedDaysAgo: 120,
|
|
314
|
+
linksTo: hubNeighbors("huba", HUB_A_NEIGHBOR_CONTENTS),
|
|
315
|
+
},
|
|
316
|
+
{
|
|
317
|
+
key: "hub_sat_20",
|
|
318
|
+
content: HUB_CONTENT,
|
|
319
|
+
createdAt: "2026-05-10T09:00:00.000Z",
|
|
320
|
+
memoryType: "knowledge",
|
|
321
|
+
baseStrength: 0.6,
|
|
322
|
+
accessCount: 0,
|
|
323
|
+
lastAccessedDaysAgo: 120,
|
|
324
|
+
linksTo: hubNeighbors("hubb", HUB_B_NEIGHBOR_CONTENTS),
|
|
325
|
+
},
|
|
326
|
+
// --- direction pair: equal metadata; similarity gap engineered ~0.20 ---
|
|
327
|
+
// MEASURED (bge-small-en-v1.5, 2026-09-17): unlinked 0.8233 vs linked
|
|
328
|
+
// 0.6022 — gap 0.2212. The linked row deliberately carries the plural
|
|
329
|
+
// "guests" (an FTS-DISTINCT token from the query's "guest" under the
|
|
330
|
+
// unicode61 tokenizer) so it stays single-channel: the pin isolates
|
|
331
|
+
// similarity vs connections credit, not the both-channel boost.
|
|
332
|
+
{
|
|
333
|
+
key: "dir_unlinked",
|
|
334
|
+
content: "Home anchorage: Anchorholme, east of the skerries. The guest mooring lies by the old timber jetty — " +
|
|
335
|
+
"arrive at slack water and mind the ferry wake inside the bay.",
|
|
336
|
+
createdAt: "2026-07-01T09:00:00.000Z",
|
|
337
|
+
memoryType: "knowledge",
|
|
338
|
+
baseStrength: 0.6,
|
|
339
|
+
accessCount: 0,
|
|
340
|
+
lastAccessedDaysAgo: 90,
|
|
341
|
+
linksTo: [],
|
|
342
|
+
},
|
|
343
|
+
{
|
|
344
|
+
key: "dir_linked",
|
|
345
|
+
content: "Anchoring for the night behind the skerries: lay out chain at least four times the depth in the crowded " +
|
|
346
|
+
"bay, take one of the buoys kept for guests if the harbour has room, and check the swing radius around " +
|
|
347
|
+
"the mooring field before dark.",
|
|
348
|
+
createdAt: "2026-07-01T09:00:00.000Z",
|
|
349
|
+
memoryType: "knowledge",
|
|
350
|
+
baseStrength: 0.6,
|
|
351
|
+
accessCount: 0,
|
|
352
|
+
lastAccessedDaysAgo: 90,
|
|
353
|
+
linksTo: ["dirn_n0", "dirn_n1", "dirn_n2", "dirn_n3"],
|
|
354
|
+
},
|
|
355
|
+
// --- link carriers: 40 disjoint off-domain filler rows (degrees all 1) ---
|
|
356
|
+
...fillerRows("huba", HUB_A_NEIGHBOR_CONTENTS),
|
|
357
|
+
...fillerRows("hubb", HUB_B_NEIGHBOR_CONTENTS),
|
|
358
|
+
...fillerRows("dirn", DIR_NEIGHBOR_CONTENTS),
|
|
359
|
+
];
|
|
360
|
+
/** Declared degrees: hubs 16/20, direction-linked 4, every neighbor 1. */
|
|
361
|
+
function linkedFallbackDegrees() {
|
|
362
|
+
const degrees = {
|
|
363
|
+
hub_sat_16: 16,
|
|
364
|
+
hub_sat_20: 20,
|
|
365
|
+
dir_linked: 4,
|
|
366
|
+
dir_unlinked: 0,
|
|
367
|
+
};
|
|
368
|
+
for (const prefix of ["huba", "hubb", "dirn"]) {
|
|
369
|
+
const count = prefix === "huba"
|
|
370
|
+
? HUB_A_NEIGHBOR_CONTENTS.length
|
|
371
|
+
: prefix === "hubb"
|
|
372
|
+
? HUB_B_NEIGHBOR_CONTENTS.length
|
|
373
|
+
: DIR_NEIGHBOR_CONTENTS.length;
|
|
374
|
+
for (let i = 0; i < count; i++)
|
|
375
|
+
degrees[`${prefix}_n${i}`] = 1;
|
|
376
|
+
}
|
|
377
|
+
return degrees;
|
|
378
|
+
}
|
|
379
|
+
exports.LINKED_FALLBACK_FIXTURE = {
|
|
380
|
+
key: "linked_fallback",
|
|
381
|
+
description: "#449 (items 1+3): the connections reshape — a 16-link and a 20-link identical-content hub " +
|
|
382
|
+
"must score byte-equal post-change (both saturate at K=16), and an unlinked row ~0.20 more " +
|
|
383
|
+
"similar must outrank a 4-link rival (similarity dominates the saturated connections credit).",
|
|
384
|
+
rows: LINKED_FALLBACK_ROWS,
|
|
385
|
+
query: "Anchorholme guest anchorage",
|
|
386
|
+
distinctiveToken: "anchorholme",
|
|
387
|
+
expectedTop1: "dir_unlinked",
|
|
388
|
+
expectedUndirectedDegree: linkedFallbackDegrees(),
|
|
389
|
+
linkedExpectations: {
|
|
390
|
+
saturationPair: { hubA: "hub_sat_16", hubB: "hub_sat_20" },
|
|
391
|
+
directionPair: { unlinked: "dir_unlinked", linked: "dir_linked" },
|
|
392
|
+
},
|
|
393
|
+
};
|
|
394
|
+
// ---------------------------------------------------------------------------
|
|
192
395
|
// Fixture invariants (checked by ranking-eval at run time AND pinned in
|
|
193
396
|
// tests/ranking-eval-tooling.test.ts — a corpus that silently drifts from
|
|
194
397
|
// its declared shape would fake a pass/fail)
|
|
195
398
|
// ---------------------------------------------------------------------------
|
|
399
|
+
/**
|
|
400
|
+
* Undirected degree per row key from the declared linksTo arrays (#449):
|
|
401
|
+
* a row's own linksTo entries PLUS the times other rows reference it — each
|
|
402
|
+
* linksTo entry is one link row, so a mutual pair counts twice, mirroring
|
|
403
|
+
* storage.getLinks(id, "both").length.
|
|
404
|
+
*/
|
|
405
|
+
function undirectedDegrees(rows) {
|
|
406
|
+
const degree = new Map(rows.map((r) => [r.key, 0]));
|
|
407
|
+
for (const r of rows) {
|
|
408
|
+
for (const to of r.linksTo) {
|
|
409
|
+
degree.set(r.key, (degree.get(r.key) ?? 0) + 1);
|
|
410
|
+
degree.set(to, (degree.get(to) ?? 0) + 1);
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
return degree;
|
|
414
|
+
}
|
|
196
415
|
function checkFixtureInvariants(f) {
|
|
197
416
|
const problems = [];
|
|
198
417
|
const keys = new Set(f.rows.map((r) => r.key));
|
|
@@ -209,6 +428,46 @@ function checkFixtureInvariants(f) {
|
|
|
209
428
|
if (f.expectedTop1 && !keys.has(f.expectedTop1)) {
|
|
210
429
|
problems.push(`expectedTop1 ${f.expectedTop1} is not a row key`);
|
|
211
430
|
}
|
|
431
|
+
if (f.linkedExpectations) {
|
|
432
|
+
const declared = [
|
|
433
|
+
["hubA", f.linkedExpectations.saturationPair.hubA],
|
|
434
|
+
["hubB", f.linkedExpectations.saturationPair.hubB],
|
|
435
|
+
["unlinked", f.linkedExpectations.directionPair.unlinked],
|
|
436
|
+
["linked", f.linkedExpectations.directionPair.linked],
|
|
437
|
+
];
|
|
438
|
+
for (const [role, key] of declared) {
|
|
439
|
+
if (!keys.has(key)) {
|
|
440
|
+
problems.push(`linkedExpectations ${role} ${key} is not a row key`);
|
|
441
|
+
}
|
|
442
|
+
}
|
|
443
|
+
const hubs = f.rows.filter((r) => r.key === f.linkedExpectations.saturationPair.hubA ||
|
|
444
|
+
r.key === f.linkedExpectations.saturationPair.hubB);
|
|
445
|
+
if (hubs.length === 2) {
|
|
446
|
+
const [a, b] = hubs;
|
|
447
|
+
const metaEqual = a.content === b.content &&
|
|
448
|
+
a.createdAt === b.createdAt &&
|
|
449
|
+
a.memoryType === b.memoryType &&
|
|
450
|
+
a.baseStrength === b.baseStrength &&
|
|
451
|
+
a.accessCount === b.accessCount &&
|
|
452
|
+
a.lastAccessedDaysAgo === b.lastAccessedDaysAgo;
|
|
453
|
+
if (!metaEqual) {
|
|
454
|
+
problems.push("saturation-pair hubs differ beyond their links (content/metadata must be identical)");
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
if (f.expectedUndirectedDegree) {
|
|
459
|
+
const degree = undirectedDegrees(f.rows);
|
|
460
|
+
for (const [key, expected] of Object.entries(f.expectedUndirectedDegree)) {
|
|
461
|
+
if (!keys.has(key)) {
|
|
462
|
+
problems.push(`expectedUndirectedDegree names unknown key ${key}`);
|
|
463
|
+
continue;
|
|
464
|
+
}
|
|
465
|
+
const actual = degree.get(key) ?? 0;
|
|
466
|
+
if (actual !== expected) {
|
|
467
|
+
problems.push(`${key} undirected degree is ${actual}, declared ${expected}`);
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
}
|
|
212
471
|
if (f.distinctiveToken) {
|
|
213
472
|
const token = f.distinctiveToken.toLowerCase();
|
|
214
473
|
const carriers = f.rows.filter((r) => r.content.toLowerCase().includes(token));
|
|
@@ -68,7 +68,7 @@
|
|
|
68
68
|
*
|
|
69
69
|
* Each hardware query runs TWICE on the same static DB:
|
|
70
70
|
* scope OFF — no `project` sent (byte-identical to pre-#203).
|
|
71
|
-
* scope ON — `project: "hardware"` (computeScore adds +
|
|
71
|
+
* scope ON — `project: "hardware"` (computeScore adds +scopeAffinity 0.15
|
|
72
72
|
* to hardware memories; marine gets 0; no hard filter).
|
|
73
73
|
*
|
|
74
74
|
* Metrics: contamination@5 (marine in top-5 / 5 — LOWER is better),
|
|
@@ -77,6 +77,11 @@
|
|
|
77
77
|
* `project` is the ONLY variable. Same invariants (noStrengthen, real
|
|
78
78
|
* embedder, uniform metadata, neverCalledEmbed self-check).
|
|
79
79
|
*
|
|
80
|
-
* Run: npm run eval:recall-sweep (== node dist/eval/recall-sweep.js)
|
|
80
|
+
* Run: npm run eval:recall-sweep [--now <ISO>] (== node dist/eval/recall-sweep.js)
|
|
81
|
+
*
|
|
82
|
+
* `--now <ISO>` (#458) pins the eval clock: the planted corpora's created_at
|
|
83
|
+
* AND every retrieve() call score against the SAME instant, so before/after
|
|
84
|
+
* runs are wall-clock-independent. Default: the live clock (the pre-#458
|
|
85
|
+
* behavior); the report header records which clock produced it.
|
|
81
86
|
*/
|
|
82
87
|
export {};
|
|
@@ -69,7 +69,7 @@
|
|
|
69
69
|
*
|
|
70
70
|
* Each hardware query runs TWICE on the same static DB:
|
|
71
71
|
* scope OFF — no `project` sent (byte-identical to pre-#203).
|
|
72
|
-
* scope ON — `project: "hardware"` (computeScore adds +
|
|
72
|
+
* scope ON — `project: "hardware"` (computeScore adds +scopeAffinity 0.15
|
|
73
73
|
* to hardware memories; marine gets 0; no hard filter).
|
|
74
74
|
*
|
|
75
75
|
* Metrics: contamination@5 (marine in top-5 / 5 — LOWER is better),
|
|
@@ -78,7 +78,12 @@
|
|
|
78
78
|
* `project` is the ONLY variable. Same invariants (noStrengthen, real
|
|
79
79
|
* embedder, uniform metadata, neverCalledEmbed self-check).
|
|
80
80
|
*
|
|
81
|
-
* Run: npm run eval:recall-sweep (== node dist/eval/recall-sweep.js)
|
|
81
|
+
* Run: npm run eval:recall-sweep [--now <ISO>] (== node dist/eval/recall-sweep.js)
|
|
82
|
+
*
|
|
83
|
+
* `--now <ISO>` (#458) pins the eval clock: the planted corpora's created_at
|
|
84
|
+
* AND every retrieve() call score against the SAME instant, so before/after
|
|
85
|
+
* runs are wall-clock-independent. Default: the live clock (the pre-#458
|
|
86
|
+
* behavior); the report header records which clock produced it.
|
|
82
87
|
*/
|
|
83
88
|
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
84
89
|
if (k2 === undefined) k2 = k;
|
|
@@ -119,6 +124,7 @@ const node_path_1 = require("node:path");
|
|
|
119
124
|
const node_fs_1 = require("node:fs");
|
|
120
125
|
const node_crypto_1 = require("node:crypto");
|
|
121
126
|
const db_js_1 = require("../db.js");
|
|
127
|
+
const eval_clock_js_1 = require("./eval-clock.js");
|
|
122
128
|
const storage = __importStar(require("../storage.js"));
|
|
123
129
|
const embedder_js_1 = require("../embedder.js");
|
|
124
130
|
const retrieval_js_1 = require("../retrieval.js");
|
|
@@ -424,7 +430,7 @@ function buildSessions() {
|
|
|
424
430
|
// ---------------------------------------------------------------------------
|
|
425
431
|
// DB construction
|
|
426
432
|
// ---------------------------------------------------------------------------
|
|
427
|
-
async function buildCorpusDb(dbPath) {
|
|
433
|
+
async function buildCorpusDb(dbPath, now) {
|
|
428
434
|
const db = (0, db_js_1.initDb)(dbPath);
|
|
429
435
|
const idToTopic = new Map();
|
|
430
436
|
for (const mem of CORPUS) {
|
|
@@ -433,6 +439,9 @@ async function buildCorpusDb(dbPath) {
|
|
|
433
439
|
sourceAgent: "eval-corpus",
|
|
434
440
|
memoryType: "experience",
|
|
435
441
|
baseStrength: 0.5, // uniform — strength is NOT a discriminator here
|
|
442
|
+
// #458: pinned created_at (the same instant retrieve() scores against)
|
|
443
|
+
// so the uniform-metadata invariant holds on a pinned clock too.
|
|
444
|
+
createdAt: (now ?? new Date()).toISOString(),
|
|
436
445
|
});
|
|
437
446
|
idToTopic.set(id, mem.topic);
|
|
438
447
|
}
|
|
@@ -447,7 +456,7 @@ async function buildCorpusDb(dbPath) {
|
|
|
447
456
|
*
|
|
448
457
|
* Returns `ids` in corpus order so SCOPE_QUERIES goldIndices map to ids.
|
|
449
458
|
*/
|
|
450
|
-
async function buildScopeDb(dbPath) {
|
|
459
|
+
async function buildScopeDb(dbPath, now) {
|
|
451
460
|
const db = (0, db_js_1.initDb)(dbPath);
|
|
452
461
|
const idToProject = new Map();
|
|
453
462
|
const ids = [];
|
|
@@ -458,6 +467,7 @@ async function buildScopeDb(dbPath) {
|
|
|
458
467
|
memoryType: "experience",
|
|
459
468
|
baseStrength: 0.5, // uniform — strength is NOT a discriminator here
|
|
460
469
|
project: mem.project, // #203 scope label
|
|
470
|
+
createdAt: (now ?? new Date()).toISOString(), // #458: pinned clock
|
|
461
471
|
});
|
|
462
472
|
idToProject.set(id, mem.project);
|
|
463
473
|
ids.push(id);
|
|
@@ -481,7 +491,7 @@ async function embedAllPrompts(sessions) {
|
|
|
481
491
|
* only the blend differs). A fresh SessionRecallRegistry per (weight,session)
|
|
482
492
|
* keeps each centroid seeded only by that session's own turns.
|
|
483
493
|
*/
|
|
484
|
-
async function runSweep(db, idToTopic, sessions) {
|
|
494
|
+
async function runSweep(db, idToTopic, sessions, now) {
|
|
485
495
|
const promptEmb = await embedAllPrompts(sessions);
|
|
486
496
|
// Fail-explicit embedFn: we ALWAYS supply queryEmbedding, so retrieve()
|
|
487
497
|
// must never call this. If it does, the sweep is measuring the wrong thing.
|
|
@@ -503,6 +513,7 @@ async function runSweep(db, idToTopic, sessions) {
|
|
|
503
513
|
limit: 5,
|
|
504
514
|
queryEmbedding: queryVec,
|
|
505
515
|
noStrengthen: true,
|
|
516
|
+
...(now ? { now } : {}), // #458: pinned clock (undefined = live)
|
|
506
517
|
});
|
|
507
518
|
const topIds = results.map((r) => r.id);
|
|
508
519
|
const onTopic = topIds.filter((id) => idToTopic.get(id) === turnDef.topic);
|
|
@@ -599,7 +610,8 @@ function renderReport(focused, shifts, records, sessions, meta) {
|
|
|
599
610
|
L.push("# Recall@k Sweep — session-intent keying (#192, 0867d6c)\n");
|
|
600
611
|
L.push(`Corpus: ${meta.memoryCount} memories across ${TOPICS.length} topics (5 each), embedded with the ` +
|
|
601
612
|
`real bge-small-en-v1.5 model. ${nFocused} focused scenarios x 4 turns + ${nShift} shift scenarios ` +
|
|
602
|
-
`x 4 turns (2 on A -> 2 on B)
|
|
613
|
+
`x 4 turns (2 on A -> 2 on B). Clock: ${meta.clock} (#458 — the instant the planted ` +
|
|
614
|
+
`created_at and every retrieve() scored against).\n`);
|
|
603
615
|
L.push("_Static DB across all retrieve() calls (`noStrengthen: true`); uniform memory metadata " +
|
|
604
616
|
"(base_strength=0.5, created_at~now, no links) so vector cosine + RRF is the only discriminator; " +
|
|
605
617
|
"cold-exposure slots are a no-op (all candidates equally cold). Centroid EMA at α=0.4. " +
|
|
@@ -784,7 +796,7 @@ function renderReport(focused, shifts, records, sessions, meta) {
|
|
|
784
796
|
* options. `queryEmbedding` is supplied (pure prompt), so retrieve() must never
|
|
785
797
|
* call the embedFn; the `neverCalledEmbed` self-check enforces that.
|
|
786
798
|
*/
|
|
787
|
-
async function runScopeSweep(db, idToProject, ids) {
|
|
799
|
+
async function runScopeSweep(db, idToProject, ids, now) {
|
|
788
800
|
// Precompute query embeddings once — pure prompt, no centroid (scope is the
|
|
789
801
|
// only variable; blend held at 0 to isolate it).
|
|
790
802
|
const queryEmb = new Map();
|
|
@@ -806,9 +818,10 @@ async function runScopeSweep(db, idToProject, ids) {
|
|
|
806
818
|
queryEmbedding: promptVec,
|
|
807
819
|
noStrengthen: true,
|
|
808
820
|
// OFF: omit project entirely (byte-identical to pre-#203).
|
|
809
|
-
// ON: send project — computeScore adds +
|
|
821
|
+
// ON: send project — computeScore adds +scopeAffinity (max() of the project signal → 0.15) to
|
|
810
822
|
// every hardware memory; marine memories get 0.
|
|
811
823
|
...(mode === "on" ? { project: SCOPE_PROJECT } : {}),
|
|
824
|
+
...(now ? { now } : {}), // #458: pinned clock (undefined = live)
|
|
812
825
|
});
|
|
813
826
|
const topIds = results.map((r) => r.id);
|
|
814
827
|
const marineCount = topIds.filter((id) => idToProject.get(id) === "marine").length;
|
|
@@ -842,7 +855,7 @@ function renderScopeSection(records) {
|
|
|
842
855
|
`(share a token — battery / drain / voltage / charge / temperature — with the hardware queries); ` +
|
|
843
856
|
`${SCOPE_CORPUS.filter((m) => m.project === "marine" && !m.seed).length} are marine-only fillers (no token overlap — control). ` +
|
|
844
857
|
`Each hardware query runs TWICE: scope OFF (no \`project\` sent — byte-identical to pre-#203) and scope ON (\`project: "${SCOPE_PROJECT}"\`). ` +
|
|
845
|
-
`The #203 affinity adds +
|
|
858
|
+
`The #203 affinity adds +scopeAffinity (max() of the project signal → 0.15) to hardware memories in computeScore; marine memories get 0. No hard filter anywhere.\n`);
|
|
846
859
|
L.push("_Same invariants as the blend sweep: static DB (`noStrengthen: true`), real bge-small-en-v1.5 embedder, " +
|
|
847
860
|
"uniform metadata (base_strength=0.5, created_at~now, no links). Blend weight held at 0 (pure prompt) — " +
|
|
848
861
|
"scope is orthogonal to session-intent keying. The `neverCalledEmbed` self-check still passes._\n");
|
|
@@ -926,6 +939,21 @@ async function main() {
|
|
|
926
939
|
(0, retrieval_js_1.configureDecay)();
|
|
927
940
|
(0, retrieval_js_1.configureRecall)();
|
|
928
941
|
(0, retrieval_js_1.configureSessionIntent)();
|
|
942
|
+
// Minimal flag parsing (the file's first flag): --now <ISO> pins the eval
|
|
943
|
+
// clock (#458). Invalid input fails explicit — never a silent live clock.
|
|
944
|
+
let now;
|
|
945
|
+
const argv = process.argv.slice(2);
|
|
946
|
+
const nowIdx = argv.indexOf("--now");
|
|
947
|
+
try {
|
|
948
|
+
now = (0, eval_clock_js_1.parsePinnedNow)(nowIdx !== -1 ? argv[nowIdx + 1] : undefined);
|
|
949
|
+
}
|
|
950
|
+
catch (err) {
|
|
951
|
+
console.error(`[recall-sweep] ${err instanceof Error ? err.message : err}`);
|
|
952
|
+
process.exitCode = 1;
|
|
953
|
+
return;
|
|
954
|
+
}
|
|
955
|
+
const nowArg = now ?? undefined;
|
|
956
|
+
console.log(`[recall-sweep] clock: ${(0, eval_clock_js_1.clockLabel)(now)}`);
|
|
929
957
|
const sessions = buildSessions();
|
|
930
958
|
const tmpDir = (0, node_path_1.join)((0, node_os_1.tmpdir)(), `hicortex-recall-sweep-${(0, node_crypto_1.randomUUID)().slice(0, 8)}`);
|
|
931
959
|
(0, node_fs_1.mkdirSync)(tmpDir, { recursive: true });
|
|
@@ -945,27 +973,28 @@ async function main() {
|
|
|
945
973
|
try {
|
|
946
974
|
console.log("[recall-sweep] building corpus (embedding 30 memories)...");
|
|
947
975
|
const t0 = Date.now();
|
|
948
|
-
const { db: opened, idToTopic } = await buildCorpusDb(dbPath);
|
|
976
|
+
const { db: opened, idToTopic } = await buildCorpusDb(dbPath, nowArg);
|
|
949
977
|
db = opened;
|
|
950
978
|
console.log(`[recall-sweep] corpus ready in ${Date.now() - t0}ms (${idToTopic.size} memories)`);
|
|
951
979
|
console.log("[recall-sweep] running sweep...");
|
|
952
980
|
const t1 = Date.now();
|
|
953
|
-
const records = await runSweep(db, idToTopic, sessions);
|
|
981
|
+
const records = await runSweep(db, idToTopic, sessions, nowArg);
|
|
954
982
|
console.log(`[recall-sweep] sweep done in ${Date.now() - t1}ms (${records.length} turn records)`);
|
|
955
983
|
const focused = summarizeFocused(records, sessions);
|
|
956
984
|
const shifts = summarizeShift(records);
|
|
957
985
|
let report = renderReport(focused, shifts, records, sessions, {
|
|
958
986
|
memoryCount: CORPUS.length,
|
|
987
|
+
clock: (0, eval_clock_js_1.clockLabel)(now),
|
|
959
988
|
});
|
|
960
989
|
// ---- SCOPE sweep (#203) ----
|
|
961
990
|
console.log(`[recall-sweep] building scope corpus (embedding ${SCOPE_CORPUS.length} project-labeled memories)...`);
|
|
962
991
|
const t2 = Date.now();
|
|
963
|
-
const scopeBuilt = await buildScopeDb(scopeDbPath);
|
|
992
|
+
const scopeBuilt = await buildScopeDb(scopeDbPath, nowArg);
|
|
964
993
|
scopeDb = scopeBuilt.db;
|
|
965
994
|
console.log(`[recall-sweep] scope corpus ready in ${Date.now() - t2}ms (${scopeBuilt.ids.length} memories)`);
|
|
966
995
|
console.log("[recall-sweep] running scope sweep (OFF vs ON)...");
|
|
967
996
|
const t3 = Date.now();
|
|
968
|
-
const scopeRecords = await runScopeSweep(scopeDb, scopeBuilt.idToProject, scopeBuilt.ids);
|
|
997
|
+
const scopeRecords = await runScopeSweep(scopeDb, scopeBuilt.idToProject, scopeBuilt.ids, nowArg);
|
|
969
998
|
console.log(`[recall-sweep] scope sweep done in ${Date.now() - t3}ms (${scopeRecords.length} turn records)`);
|
|
970
999
|
report += "\n\n" + renderScopeSection(scopeRecords);
|
|
971
1000
|
(0, node_fs_1.mkdirSync)(reportDir, { recursive: true });
|
|
@@ -59,6 +59,120 @@
|
|
|
59
59
|
* Run:
|
|
60
60
|
* npm run eval:relevance -- <snapshot.db> [prompts.json] [report.md] \
|
|
61
61
|
* [--judge-delay-ms=2000] [--max-calls=900] [--resume] \
|
|
62
|
-
* [--verdicts-json=path] [--verdicts-jsonl=path]
|
|
62
|
+
* [--verdicts-json=path] [--verdicts-jsonl=path] \
|
|
63
|
+
* [--now=<ISO>] [--runs=N]
|
|
64
|
+
*
|
|
65
|
+
* #458 clock pin + judge variance protocol:
|
|
66
|
+
* - `--now=<ISO>` pins the clock the retrieve sweep scores against (the
|
|
67
|
+
* retrieve() `now` seam) — before/after runs become wall-clock-independent
|
|
68
|
+
* (default: live clock, the pre-#458 behavior).
|
|
69
|
+
* - `--runs=N` (default 1) runs judge phases 1+2 N times over the SAME
|
|
70
|
+
* selections (one retrieve sweep) and reports the run-to-run noise floor
|
|
71
|
+
* (median + spread + selection-identical flip counts, §15b). Phases 3+4
|
|
72
|
+
* stay single-shot; the shared --max-calls budget spans all runs and the
|
|
73
|
+
* default scales ×N unless --max-calls was passed explicitly.
|
|
74
|
+
*/
|
|
75
|
+
type Verdict = "relevant" | "noise" | "stale";
|
|
76
|
+
interface JudgeVerdict {
|
|
77
|
+
/** Q1..Q8 label — mapped back to the real memory id via the caller's
|
|
78
|
+
* position-ordered topIds list. */
|
|
79
|
+
id: string;
|
|
80
|
+
verdict: Verdict;
|
|
81
|
+
reason: string;
|
|
82
|
+
}
|
|
83
|
+
type VerdictsOrError = JudgeVerdict[] | {
|
|
84
|
+
judgeError: string;
|
|
85
|
+
};
|
|
86
|
+
interface CliArgs {
|
|
87
|
+
positionals: string[];
|
|
88
|
+
judgeDelayMs: number;
|
|
89
|
+
maxCalls: number;
|
|
90
|
+
/** True when --max-calls was passed explicitly (#458) — an explicit budget
|
|
91
|
+
* was sized by the caller for the whole invocation and must NOT be scaled
|
|
92
|
+
* by --runs (see scaleMaxCalls). */
|
|
93
|
+
maxCallsExplicit: boolean;
|
|
94
|
+
resume: boolean;
|
|
95
|
+
/** Override the raw-verdicts JSON path (default VERDICTS_JSON_PATH) — added
|
|
96
|
+
* 2026-08-02 so a re-run against a NEW snapshot can never silently
|
|
97
|
+
* overwrite a baseline's raw verdicts. */
|
|
98
|
+
verdictsJsonPath: string;
|
|
99
|
+
/** Override the checkpoint sidecar / raw-verdicts JSONL path (default
|
|
100
|
+
* VERDICTS_JSONL_PATH) — same reason. */
|
|
101
|
+
verdictsJsonlPath: string;
|
|
102
|
+
/** #458: raw --now=<ISO> value — undefined = live clock. Validated in main
|
|
103
|
+
* via parsePinnedNow (invalid → usage-style error, exit 1). */
|
|
104
|
+
nowIso: string | undefined;
|
|
105
|
+
/** #458: judge runs over the fixed selections (phases 1+2 only). Default 1
|
|
106
|
+
* = the pre-#458 single-run behavior. */
|
|
107
|
+
runs: number;
|
|
108
|
+
}
|
|
109
|
+
/** Parsed and exported for tests (#458). Throws on an invalid --runs value
|
|
110
|
+
* (NaN or <1) — fail-explicit, never a silent default. */
|
|
111
|
+
export declare function parseArgs(argv: string[]): CliArgs;
|
|
112
|
+
/**
|
|
113
|
+
* #458 — resolve the effective --max-calls budget. The ONE shared budget
|
|
114
|
+
* spans ALL judge runs, so when --runs>1 and the caller did NOT pass
|
|
115
|
+
* --max-calls explicitly, the default scales ×N (each run re-judges every
|
|
116
|
+
* unit). An explicit budget always wins — the caller sized it for the whole
|
|
117
|
+
* invocation. Pure, exported for tests.
|
|
118
|
+
*/
|
|
119
|
+
export declare function scaleMaxCalls(runs: number, explicitMaxCalls: number | undefined, defaultMaxCalls: number): number;
|
|
120
|
+
/** Median of a numeric sample (even count → mean of the two middle values).
|
|
121
|
+
* Empty input → NaN, the mean() convention above. Does not mutate input. */
|
|
122
|
+
export declare function median(xs: number[]): number;
|
|
123
|
+
/** One judged unit's verdicts for variance accounting (#458): `key` is the
|
|
124
|
+
* stable (promptIdx|mode) unit identity, identical across runs because the
|
|
125
|
+
* retrieve sweep runs ONCE. */
|
|
126
|
+
export interface VarianceUnitVerdicts {
|
|
127
|
+
key: string;
|
|
128
|
+
verdicts: VerdictsOrError;
|
|
129
|
+
}
|
|
130
|
+
/** A headline metric measured once per run (e.g. precision@5 OFF). Values are
|
|
131
|
+
* in raw fractions; null = not computable that run (excluded from
|
|
132
|
+
* median/spread). */
|
|
133
|
+
export interface VarianceMetricSeries {
|
|
134
|
+
label: string;
|
|
135
|
+
values: Array<number | null>;
|
|
136
|
+
}
|
|
137
|
+
export interface VarianceMetricStats extends VarianceMetricSeries {
|
|
138
|
+
median: number | null;
|
|
139
|
+
/** (max − min) × 100, in points — the measured run-to-run noise floor. */
|
|
140
|
+
spreadPts: number | null;
|
|
141
|
+
}
|
|
142
|
+
/** Pairwise flip counts between runs i and j (1-based run numbers). */
|
|
143
|
+
export interface PairwiseFlips {
|
|
144
|
+
runA: number;
|
|
145
|
+
runB: number;
|
|
146
|
+
/** Verdict ROWS compared: (unit, Q-label) pairs present as parsed verdict
|
|
147
|
+
* arrays in BOTH runs. judge_error units and labels the judge omitted in
|
|
148
|
+
* one run are not comparable — excluded, never counted as flips. */
|
|
149
|
+
rowsCompared: number;
|
|
150
|
+
rowsFlipped: number;
|
|
151
|
+
unitsCompared: number;
|
|
152
|
+
/** A UNIT flips if any of its comparable rows differ. */
|
|
153
|
+
unitsFlipped: number;
|
|
154
|
+
}
|
|
155
|
+
export interface JudgeVariance {
|
|
156
|
+
runs: number;
|
|
157
|
+
metrics: VarianceMetricStats[];
|
|
158
|
+
flips: PairwiseFlips[];
|
|
159
|
+
maxRowsFlipped: number;
|
|
160
|
+
maxUnitsFlipped: number;
|
|
161
|
+
meanRowsFlipped: number;
|
|
162
|
+
meanUnitsFlipped: number;
|
|
163
|
+
/** Comparable rows of the first pair — selections are identical across
|
|
164
|
+
* runs, so only judge errors shrink this (reported per run below). */
|
|
165
|
+
totalComparableRows: number;
|
|
166
|
+
/** judge_error UNITS per run (a failed unit yields no parsed rows; it is
|
|
167
|
+
* excluded from every flip denominator). */
|
|
168
|
+
judgeErrorUnitsPerRun: number[];
|
|
169
|
+
}
|
|
170
|
+
/**
|
|
171
|
+
* Compute the #458 judge-variance account over N runs of the same selections:
|
|
172
|
+
* per-metric median + spread (the noise floor), and pairwise verdict flip
|
|
173
|
+
* counts (rows = one Q-label judgment on one (prompt,mode) unit; units flip
|
|
174
|
+
* if any row does). `perRunVerdicts[r]` is run r+1's units in a stable
|
|
175
|
+
* order/identity; `metricSeries[s].values[r]` is run r+1's headline value.
|
|
63
176
|
*/
|
|
177
|
+
export declare function computeJudgeVariance(perRunVerdicts: VarianceUnitVerdicts[][], metricSeries: VarianceMetricSeries[]): JudgeVariance;
|
|
64
178
|
export {};
|