@gamaze/hicortex 0.22.0 → 0.22.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -21,6 +21,11 @@
21
21
  * under the ranking work (no both-channel candidates exist, so the
22
22
  * boost can never fire — the case pins that the no-regression
23
23
  * guarantee is structural, not lucky).
24
+ * (c) linked_fallback — the #449 (items 1+3) connections fixture: a
25
+ * 16-link vs 20-link identical-content hub pair (the saturation tie)
26
+ * plus an unlinked higher-similarity row vs a 4-link lower-similarity
27
+ * row (the direction pin). Planted in its OWN throwaway DB — the 40
28
+ * link-neighbor rows would pollute the case-1/2 corpus.
24
29
  *
25
30
  * Honesty rules (the planted-pairs #393 A discipline):
26
31
  * - The texts are synthetic and generic (this package publishes to npm —
@@ -38,7 +43,7 @@
38
43
  * fixture exercises the same rows retrieval() will see.
39
44
  */
40
45
  Object.defineProperty(exports, "__esModule", { value: true });
41
- exports.REAL_SIRNAS_MEMORY_ID = exports.NO_MATCH_FIXTURE = exports.SIRNAS_FIXTURE = exports.RANKING_SOURCE_AGENT = exports.RANKING_PROJECT = void 0;
46
+ exports.LINKED_FALLBACK_FIXTURE = exports.REAL_SIRNAS_MEMORY_ID = exports.NO_MATCH_FIXTURE = exports.SIRNAS_FIXTURE = exports.RANKING_SOURCE_AGENT = exports.RANKING_PROJECT = void 0;
42
47
  exports.checkFixtureInvariants = checkFixtureInvariants;
43
48
  /** All planted rows share project + source_agent (metadata never refuses). */
44
49
  exports.RANKING_PROJECT = "ranking-eval";
@@ -163,6 +168,20 @@ exports.SIRNAS_FIXTURE = {
163
168
  query: "Vindarsvik",
164
169
  distinctiveToken: "vindarsvik",
165
170
  expectedTop1: "target_harbour",
171
+ // Undirected degrees (#449 invariant extension): the rivals' mutual-link
172
+ // cluster — fees/channel 4 each, maintenance 3, passage 1 (each linksTo
173
+ // entry is one link row; both directions of a mutual pair count).
174
+ expectedUndirectedDegree: {
175
+ rival_fees: 4,
176
+ rival_channel: 4,
177
+ rival_maintenance: 3,
178
+ rival_passage: 1,
179
+ target_harbour: 0,
180
+ fill_recipe: 0,
181
+ fill_books: 0,
182
+ fill_lang: 0,
183
+ fill_garden: 0,
184
+ },
166
185
  };
167
186
  // ---------------------------------------------------------------------------
168
187
  // Case 2 — no_match_control (AC2)
@@ -186,13 +205,213 @@ exports.NO_MATCH_FIXTURE = {
186
205
  // records the list; the gate compares runs).
187
206
  expectedTop1: "",
188
207
  };
189
- /** The battery runs against a real snapshot copy (ranking-eval.ts case 3). */
208
+ /** The battery runs against a real snapshot copy (ranking-eval.ts case 4). */
190
209
  exports.REAL_SIRNAS_MEMORY_ID = "12980c9d-0eba-4963-95eb-0548bc2aa93e";
191
210
  // ---------------------------------------------------------------------------
211
+ // Case 3 — linked_fallback (#449 items 1+3, PR E)
212
+ // ---------------------------------------------------------------------------
213
+ //
214
+ // The connections-term fixture for the log-saturation reshape. Two probes in
215
+ // ONE corpus (planted in its own throwaway DB by ranking-eval.ts — the 40
216
+ // link-neighbor rows stay out of the case-1/2 corpus):
217
+ //
218
+ // saturation pair — two hubs with IDENTICAL content strings and identical
219
+ // metadata (same baseStrength, accessCount, lastAccessedDaysAgo,
220
+ // createdAt, memoryType); hub A carries exactly 16 undirected links, hub B
221
+ // exactly 20, through disjoint off-domain neighbor rows. Declared
222
+ // expectation POST-change (Option A, K = 16): the two hubs' scores are
223
+ // byte-EQUAL — the term min(1, log1p(k)/log1p(16)) saturates at 1 for both.
224
+ // Under the pre-#449 linear term (connectionCount / maxConnections = k/20)
225
+ // they differ by 0.2 × 0.15 ≈ 0.03 composite — that red is the recorded
226
+ // before-picture; the harness does NOT make it pass.
227
+ //
228
+ // direction pair — an unlinked row carrying the query's distinctive proper
229
+ // noun (the highest measured similarity) vs a linked row (exactly 4 links)
230
+ // whose MEASURED similarity is engineered ~0.20 LOWER, with equal metadata
231
+ // otherwise (equal strength/dates). Declared expectation (Option A, owner
232
+ // decision 2026-09-17): the unlinked row ranks ABOVE the linked one — a
233
+ // 0.20 gap is worth 0.10 composite at the 0.50 similarity weight, more
234
+ // than the k=4 connections credit (0.0852 post-change). The gap must
235
+ // exceed 0.171 for the pin to be sound at k = 4; the wording is tuned
236
+ // until the real-embedder measurement (recorded in the photo/report)
237
+ // lands at ~0.20.
238
+ //
239
+ // Neighbors are DISTINCT filler rows, disjoint between the two hubs and from
240
+ // the direction pair's cluster, all off-domain from the query — they exist
241
+ // to carry the hub/dir-linked degrees and nothing else.
242
+ const HUB_CONTENT = "Community noticeboard archive: the autumn fair needs tombola volunteers, choir practice moves to " +
243
+ "Thursday evenings, and the book-swap shelf now carries the walking-map folder.";
244
+ /** Off-domain one-line filler rows (deterministic keys, uniform metadata). */
245
+ function fillerRows(prefix, contents) {
246
+ return contents.map((content, i) => ({
247
+ key: `${prefix}_n${i}`,
248
+ content,
249
+ createdAt: "2026-04-15T09:00:00.000Z",
250
+ memoryType: "experience",
251
+ baseStrength: 0.5,
252
+ accessCount: 0,
253
+ lastAccessedDaysAgo: 150,
254
+ linksTo: [],
255
+ }));
256
+ }
257
+ const HUB_A_NEIGHBOR_CONTENTS = [
258
+ "Slow-cooker beans: soak overnight, then four hours on low with a bay leaf and a piece of smoked pork.",
259
+ "The second detective novel in the series drags until the lighthouse keeper's ledger turns up.",
260
+ "German grammar drill: subordinate clauses shove the verb to the end, so read the whole sentence before translating.",
261
+ "Raised-bed rotation: legumes follow brassicas, and lime the bed the cabbage family just emptied.",
262
+ "October weather pattern: the first night frosts usually arrive right after the harvest moon.",
263
+ "Bike maintenance: true the wheel by tightening spokes a quarter turn at a time, listening for the ping.",
264
+ "Chess opening note: against the locked center, flank with pawns only after the pieces are developed.",
265
+ "The documentary about the alpine tunnels spent too long on the geologists' cafeteria.",
266
+ "Coffee grinding: a burr grind for the pour-over; the blade grinder is fine enough only for the French press.",
267
+ "Hiking the ridge trail: start before the thermals build, and carry an extra liter once the pass is in sight.",
268
+ "Piano practice: separate hands at half tempo until the fingering is automatic, then join them slowly.",
269
+ "Sewing patch pockets: interface the fabric, and topstitch the same distance from every edge.",
270
+ "Bird notes: the chiffchaff repeats its own name all day; the willow warbler slides down the scale.",
271
+ "Sourdough rhythm: feed the starter after work, shape in the morning, bake when the poke test springs back halfway.",
272
+ "Replacing the washing-machine hose: shut the valve, expect a cup of water, and check the jubilee clip.",
273
+ "Audiobook note: the narrator's accent makes every ship's name sound like a Hebridean island.",
274
+ ];
275
+ const HUB_B_NEIGHBOR_CONTENTS = [
276
+ "Winter tyres: swap before the first ice, and store the summer set flat in the dark basement.",
277
+ "The boardgame group settled on tiles-and-traders for the long evenings; the card game stays in the drawer.",
278
+ "Compost thermometer: the pile is cooking properly when it reads hotter than a warm bath for a week.",
279
+ "Frost-proof terracotta: empty the pots, stack them dry, and the succulents overwinter on the south sill.",
280
+ "The photography club theme this month is doors; shoot in morning shade and bracket the exposures.",
281
+ "Robot vacuum note: keep the charging dock's floor clear, or it nudges the rug and gives up mid-run.",
282
+ "Knife sharpening: the ceramic rod for touch-ups, the whetstone when the tomato skin resists.",
283
+ "The lecture on Roman concrete held the room until the slide about seawater pozzolana.",
284
+ "Necklace repair: jump rings open sideways, never apart, or the circle never closes again.",
285
+ "Ferry timetable change: the winter crossing departs an hour earlier and stops running on Sunday evenings.",
286
+ "Ski waxing: glide wax for the thaw, grip wax for the freeze, and never both in the same zone.",
287
+ "The archive boxes are labelled by decade now; the negatives live in the fire cupboard.",
288
+ "Beekeeping note: heft the hive — if one side lifts easily, the winter stores are short.",
289
+ "Jigsaw strategy: edges first, then the sky, and give the lamp a clean bulb for the blue hours.",
290
+ "The museum's clock gallery is closed for restoration until the pendulum clock returns.",
291
+ "Preserving basil: pesto freezes flat in bags; the leaves kept in water go black within a week.",
292
+ "Draft excluder: the brush strip on the letterbox cut the hallway draught more than the door seal did.",
293
+ "The organ recital opened with a toccata nobody knew and closed with everyone humming it.",
294
+ "Train-set wiring: one feeder per rail section, or the far loop slows to a crawl.",
295
+ "The pottery class moved on to wheel work; my first cylinder collapsed into a bowl with charm.",
296
+ ];
297
+ const DIR_NEIGHBOR_CONTENTS = [
298
+ "Vaccine appointment bookkeeping: the clinic sends both a text and a letter, and only the letter carries the time.",
299
+ "Reading glasses: the cheap pair from the station shop outlasted the optician's by two years.",
300
+ "The new debit card needs the reader for anything over the limit; the shop keeps the old machine as backup.",
301
+ "Laundering the down jacket: tennis balls in the dryer, low heat, and patience.",
302
+ ];
303
+ const hubNeighbors = (prefix, contents) => contents.map((_, i) => `${prefix}_n${i}`);
304
+ const LINKED_FALLBACK_ROWS = [
305
+ // --- saturation pair: identical content + metadata, degrees 16 vs 20 ---
306
+ {
307
+ key: "hub_sat_16",
308
+ content: HUB_CONTENT,
309
+ createdAt: "2026-05-10T09:00:00.000Z",
310
+ memoryType: "knowledge",
311
+ baseStrength: 0.6,
312
+ accessCount: 0,
313
+ lastAccessedDaysAgo: 120,
314
+ linksTo: hubNeighbors("huba", HUB_A_NEIGHBOR_CONTENTS),
315
+ },
316
+ {
317
+ key: "hub_sat_20",
318
+ content: HUB_CONTENT,
319
+ createdAt: "2026-05-10T09:00:00.000Z",
320
+ memoryType: "knowledge",
321
+ baseStrength: 0.6,
322
+ accessCount: 0,
323
+ lastAccessedDaysAgo: 120,
324
+ linksTo: hubNeighbors("hubb", HUB_B_NEIGHBOR_CONTENTS),
325
+ },
326
+ // --- direction pair: equal metadata; similarity gap engineered ~0.20 ---
327
+ // MEASURED (bge-small-en-v1.5, 2026-09-17): unlinked 0.8233 vs linked
328
+ // 0.6022 — gap 0.2212. The linked row deliberately carries the plural
329
+ // "guests" (an FTS-DISTINCT token from the query's "guest" under the
330
+ // unicode61 tokenizer) so it stays single-channel: the pin isolates
331
+ // similarity vs connections credit, not the both-channel boost.
332
+ {
333
+ key: "dir_unlinked",
334
+ content: "Home anchorage: Anchorholme, east of the skerries. The guest mooring lies by the old timber jetty — " +
335
+ "arrive at slack water and mind the ferry wake inside the bay.",
336
+ createdAt: "2026-07-01T09:00:00.000Z",
337
+ memoryType: "knowledge",
338
+ baseStrength: 0.6,
339
+ accessCount: 0,
340
+ lastAccessedDaysAgo: 90,
341
+ linksTo: [],
342
+ },
343
+ {
344
+ key: "dir_linked",
345
+ content: "Anchoring for the night behind the skerries: lay out chain at least four times the depth in the crowded " +
346
+ "bay, take one of the buoys kept for guests if the harbour has room, and check the swing radius around " +
347
+ "the mooring field before dark.",
348
+ createdAt: "2026-07-01T09:00:00.000Z",
349
+ memoryType: "knowledge",
350
+ baseStrength: 0.6,
351
+ accessCount: 0,
352
+ lastAccessedDaysAgo: 90,
353
+ linksTo: ["dirn_n0", "dirn_n1", "dirn_n2", "dirn_n3"],
354
+ },
355
+ // --- link carriers: 40 disjoint off-domain filler rows (degrees all 1) ---
356
+ ...fillerRows("huba", HUB_A_NEIGHBOR_CONTENTS),
357
+ ...fillerRows("hubb", HUB_B_NEIGHBOR_CONTENTS),
358
+ ...fillerRows("dirn", DIR_NEIGHBOR_CONTENTS),
359
+ ];
360
+ /** Declared degrees: hubs 16/20, direction-linked 4, every neighbor 1. */
361
+ function linkedFallbackDegrees() {
362
+ const degrees = {
363
+ hub_sat_16: 16,
364
+ hub_sat_20: 20,
365
+ dir_linked: 4,
366
+ dir_unlinked: 0,
367
+ };
368
+ for (const prefix of ["huba", "hubb", "dirn"]) {
369
+ const count = prefix === "huba"
370
+ ? HUB_A_NEIGHBOR_CONTENTS.length
371
+ : prefix === "hubb"
372
+ ? HUB_B_NEIGHBOR_CONTENTS.length
373
+ : DIR_NEIGHBOR_CONTENTS.length;
374
+ for (let i = 0; i < count; i++)
375
+ degrees[`${prefix}_n${i}`] = 1;
376
+ }
377
+ return degrees;
378
+ }
379
+ exports.LINKED_FALLBACK_FIXTURE = {
380
+ key: "linked_fallback",
381
+ description: "#449 (items 1+3): the connections reshape — a 16-link and a 20-link identical-content hub " +
382
+ "must score byte-equal post-change (both saturate at K=16), and an unlinked row ~0.20 more " +
383
+ "similar must outrank a 4-link rival (similarity dominates the saturated connections credit).",
384
+ rows: LINKED_FALLBACK_ROWS,
385
+ query: "Anchorholme guest anchorage",
386
+ distinctiveToken: "anchorholme",
387
+ expectedTop1: "dir_unlinked",
388
+ expectedUndirectedDegree: linkedFallbackDegrees(),
389
+ linkedExpectations: {
390
+ saturationPair: { hubA: "hub_sat_16", hubB: "hub_sat_20" },
391
+ directionPair: { unlinked: "dir_unlinked", linked: "dir_linked" },
392
+ },
393
+ };
394
+ // ---------------------------------------------------------------------------
192
395
  // Fixture invariants (checked by ranking-eval at run time AND pinned in
193
396
  // tests/ranking-eval-tooling.test.ts — a corpus that silently drifts from
194
397
  // its declared shape would fake a pass/fail)
195
398
  // ---------------------------------------------------------------------------
399
+ /**
400
+ * Undirected degree per row key from the declared linksTo arrays (#449):
401
+ * a row's own linksTo entries PLUS the times other rows reference it — each
402
+ * linksTo entry is one link row, so a mutual pair counts twice, mirroring
403
+ * storage.getLinks(id, "both").length.
404
+ */
405
+ function undirectedDegrees(rows) {
406
+ const degree = new Map(rows.map((r) => [r.key, 0]));
407
+ for (const r of rows) {
408
+ for (const to of r.linksTo) {
409
+ degree.set(r.key, (degree.get(r.key) ?? 0) + 1);
410
+ degree.set(to, (degree.get(to) ?? 0) + 1);
411
+ }
412
+ }
413
+ return degree;
414
+ }
196
415
  function checkFixtureInvariants(f) {
197
416
  const problems = [];
198
417
  const keys = new Set(f.rows.map((r) => r.key));
@@ -209,6 +428,46 @@ function checkFixtureInvariants(f) {
209
428
  if (f.expectedTop1 && !keys.has(f.expectedTop1)) {
210
429
  problems.push(`expectedTop1 ${f.expectedTop1} is not a row key`);
211
430
  }
431
+ if (f.linkedExpectations) {
432
+ const declared = [
433
+ ["hubA", f.linkedExpectations.saturationPair.hubA],
434
+ ["hubB", f.linkedExpectations.saturationPair.hubB],
435
+ ["unlinked", f.linkedExpectations.directionPair.unlinked],
436
+ ["linked", f.linkedExpectations.directionPair.linked],
437
+ ];
438
+ for (const [role, key] of declared) {
439
+ if (!keys.has(key)) {
440
+ problems.push(`linkedExpectations ${role} ${key} is not a row key`);
441
+ }
442
+ }
443
+ const hubs = f.rows.filter((r) => r.key === f.linkedExpectations.saturationPair.hubA ||
444
+ r.key === f.linkedExpectations.saturationPair.hubB);
445
+ if (hubs.length === 2) {
446
+ const [a, b] = hubs;
447
+ const metaEqual = a.content === b.content &&
448
+ a.createdAt === b.createdAt &&
449
+ a.memoryType === b.memoryType &&
450
+ a.baseStrength === b.baseStrength &&
451
+ a.accessCount === b.accessCount &&
452
+ a.lastAccessedDaysAgo === b.lastAccessedDaysAgo;
453
+ if (!metaEqual) {
454
+ problems.push("saturation-pair hubs differ beyond their links (content/metadata must be identical)");
455
+ }
456
+ }
457
+ }
458
+ if (f.expectedUndirectedDegree) {
459
+ const degree = undirectedDegrees(f.rows);
460
+ for (const [key, expected] of Object.entries(f.expectedUndirectedDegree)) {
461
+ if (!keys.has(key)) {
462
+ problems.push(`expectedUndirectedDegree names unknown key ${key}`);
463
+ continue;
464
+ }
465
+ const actual = degree.get(key) ?? 0;
466
+ if (actual !== expected) {
467
+ problems.push(`${key} undirected degree is ${actual}, declared ${expected}`);
468
+ }
469
+ }
470
+ }
212
471
  if (f.distinctiveToken) {
213
472
  const token = f.distinctiveToken.toLowerCase();
214
473
  const carriers = f.rows.filter((r) => r.content.toLowerCase().includes(token));
@@ -68,7 +68,7 @@
68
68
  *
69
69
  * Each hardware query runs TWICE on the same static DB:
70
70
  * scope OFF — no `project` sent (byte-identical to pre-#203).
71
- * scope ON — `project: "hardware"` (computeScore adds +projectAffinity 0.15
71
+ * scope ON — `project: "hardware"` (computeScore adds +scopeAffinity 0.15
72
72
  * to hardware memories; marine gets 0; no hard filter).
73
73
  *
74
74
  * Metrics: contamination@5 (marine in top-5 / 5 — LOWER is better),
@@ -77,6 +77,11 @@
77
77
  * `project` is the ONLY variable. Same invariants (noStrengthen, real
78
78
  * embedder, uniform metadata, neverCalledEmbed self-check).
79
79
  *
80
- * Run: npm run eval:recall-sweep (== node dist/eval/recall-sweep.js)
80
+ * Run: npm run eval:recall-sweep [--now <ISO>] (== node dist/eval/recall-sweep.js)
81
+ *
82
+ * `--now <ISO>` (#458) pins the eval clock: the planted corpora's created_at
83
+ * AND every retrieve() call score against the SAME instant, so before/after
84
+ * runs are wall-clock-independent. Default: the live clock (the pre-#458
85
+ * behavior); the report header records which clock produced it.
81
86
  */
82
87
  export {};
@@ -69,7 +69,7 @@
69
69
  *
70
70
  * Each hardware query runs TWICE on the same static DB:
71
71
  * scope OFF — no `project` sent (byte-identical to pre-#203).
72
- * scope ON — `project: "hardware"` (computeScore adds +projectAffinity 0.15
72
+ * scope ON — `project: "hardware"` (computeScore adds +scopeAffinity 0.15
73
73
  * to hardware memories; marine gets 0; no hard filter).
74
74
  *
75
75
  * Metrics: contamination@5 (marine in top-5 / 5 — LOWER is better),
@@ -78,7 +78,12 @@
78
78
  * `project` is the ONLY variable. Same invariants (noStrengthen, real
79
79
  * embedder, uniform metadata, neverCalledEmbed self-check).
80
80
  *
81
- * Run: npm run eval:recall-sweep (== node dist/eval/recall-sweep.js)
81
+ * Run: npm run eval:recall-sweep [--now <ISO>] (== node dist/eval/recall-sweep.js)
82
+ *
83
+ * `--now <ISO>` (#458) pins the eval clock: the planted corpora's created_at
84
+ * AND every retrieve() call score against the SAME instant, so before/after
85
+ * runs are wall-clock-independent. Default: the live clock (the pre-#458
86
+ * behavior); the report header records which clock produced it.
82
87
  */
83
88
  var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
84
89
  if (k2 === undefined) k2 = k;
@@ -119,6 +124,7 @@ const node_path_1 = require("node:path");
119
124
  const node_fs_1 = require("node:fs");
120
125
  const node_crypto_1 = require("node:crypto");
121
126
  const db_js_1 = require("../db.js");
127
+ const eval_clock_js_1 = require("./eval-clock.js");
122
128
  const storage = __importStar(require("../storage.js"));
123
129
  const embedder_js_1 = require("../embedder.js");
124
130
  const retrieval_js_1 = require("../retrieval.js");
@@ -424,7 +430,7 @@ function buildSessions() {
424
430
  // ---------------------------------------------------------------------------
425
431
  // DB construction
426
432
  // ---------------------------------------------------------------------------
427
- async function buildCorpusDb(dbPath) {
433
+ async function buildCorpusDb(dbPath, now) {
428
434
  const db = (0, db_js_1.initDb)(dbPath);
429
435
  const idToTopic = new Map();
430
436
  for (const mem of CORPUS) {
@@ -433,6 +439,9 @@ async function buildCorpusDb(dbPath) {
433
439
  sourceAgent: "eval-corpus",
434
440
  memoryType: "experience",
435
441
  baseStrength: 0.5, // uniform — strength is NOT a discriminator here
442
+ // #458: pinned created_at (the same instant retrieve() scores against)
443
+ // so the uniform-metadata invariant holds on a pinned clock too.
444
+ createdAt: (now ?? new Date()).toISOString(),
436
445
  });
437
446
  idToTopic.set(id, mem.topic);
438
447
  }
@@ -447,7 +456,7 @@ async function buildCorpusDb(dbPath) {
447
456
  *
448
457
  * Returns `ids` in corpus order so SCOPE_QUERIES goldIndices map to ids.
449
458
  */
450
- async function buildScopeDb(dbPath) {
459
+ async function buildScopeDb(dbPath, now) {
451
460
  const db = (0, db_js_1.initDb)(dbPath);
452
461
  const idToProject = new Map();
453
462
  const ids = [];
@@ -458,6 +467,7 @@ async function buildScopeDb(dbPath) {
458
467
  memoryType: "experience",
459
468
  baseStrength: 0.5, // uniform — strength is NOT a discriminator here
460
469
  project: mem.project, // #203 scope label
470
+ createdAt: (now ?? new Date()).toISOString(), // #458: pinned clock
461
471
  });
462
472
  idToProject.set(id, mem.project);
463
473
  ids.push(id);
@@ -481,7 +491,7 @@ async function embedAllPrompts(sessions) {
481
491
  * only the blend differs). A fresh SessionRecallRegistry per (weight,session)
482
492
  * keeps each centroid seeded only by that session's own turns.
483
493
  */
484
- async function runSweep(db, idToTopic, sessions) {
494
+ async function runSweep(db, idToTopic, sessions, now) {
485
495
  const promptEmb = await embedAllPrompts(sessions);
486
496
  // Fail-explicit embedFn: we ALWAYS supply queryEmbedding, so retrieve()
487
497
  // must never call this. If it does, the sweep is measuring the wrong thing.
@@ -503,6 +513,7 @@ async function runSweep(db, idToTopic, sessions) {
503
513
  limit: 5,
504
514
  queryEmbedding: queryVec,
505
515
  noStrengthen: true,
516
+ ...(now ? { now } : {}), // #458: pinned clock (undefined = live)
506
517
  });
507
518
  const topIds = results.map((r) => r.id);
508
519
  const onTopic = topIds.filter((id) => idToTopic.get(id) === turnDef.topic);
@@ -599,7 +610,8 @@ function renderReport(focused, shifts, records, sessions, meta) {
599
610
  L.push("# Recall@k Sweep — session-intent keying (#192, 0867d6c)\n");
600
611
  L.push(`Corpus: ${meta.memoryCount} memories across ${TOPICS.length} topics (5 each), embedded with the ` +
601
612
  `real bge-small-en-v1.5 model. ${nFocused} focused scenarios x 4 turns + ${nShift} shift scenarios ` +
602
- `x 4 turns (2 on A -> 2 on B).\n`);
613
+ `x 4 turns (2 on A -> 2 on B). Clock: ${meta.clock} (#458 — the instant the planted ` +
614
+ `created_at and every retrieve() scored against).\n`);
603
615
  L.push("_Static DB across all retrieve() calls (`noStrengthen: true`); uniform memory metadata " +
604
616
  "(base_strength=0.5, created_at~now, no links) so vector cosine + RRF is the only discriminator; " +
605
617
  "cold-exposure slots are a no-op (all candidates equally cold). Centroid EMA at α=0.4. " +
@@ -784,7 +796,7 @@ function renderReport(focused, shifts, records, sessions, meta) {
784
796
  * options. `queryEmbedding` is supplied (pure prompt), so retrieve() must never
785
797
  * call the embedFn; the `neverCalledEmbed` self-check enforces that.
786
798
  */
787
- async function runScopeSweep(db, idToProject, ids) {
799
+ async function runScopeSweep(db, idToProject, ids, now) {
788
800
  // Precompute query embeddings once — pure prompt, no centroid (scope is the
789
801
  // only variable; blend held at 0 to isolate it).
790
802
  const queryEmb = new Map();
@@ -806,9 +818,10 @@ async function runScopeSweep(db, idToProject, ids) {
806
818
  queryEmbedding: promptVec,
807
819
  noStrengthen: true,
808
820
  // OFF: omit project entirely (byte-identical to pre-#203).
809
- // ON: send project — computeScore adds +projectAffinity (0.15) to
821
+ // ON: send project — computeScore adds +scopeAffinity (max() of the project signal → 0.15) to
810
822
  // every hardware memory; marine memories get 0.
811
823
  ...(mode === "on" ? { project: SCOPE_PROJECT } : {}),
824
+ ...(now ? { now } : {}), // #458: pinned clock (undefined = live)
812
825
  });
813
826
  const topIds = results.map((r) => r.id);
814
827
  const marineCount = topIds.filter((id) => idToProject.get(id) === "marine").length;
@@ -842,7 +855,7 @@ function renderScopeSection(records) {
842
855
  `(share a token — battery / drain / voltage / charge / temperature — with the hardware queries); ` +
843
856
  `${SCOPE_CORPUS.filter((m) => m.project === "marine" && !m.seed).length} are marine-only fillers (no token overlap — control). ` +
844
857
  `Each hardware query runs TWICE: scope OFF (no \`project\` sent — byte-identical to pre-#203) and scope ON (\`project: "${SCOPE_PROJECT}"\`). ` +
845
- `The #203 affinity adds +projectAffinity (0.15) to hardware memories in computeScore; marine memories get 0. No hard filter anywhere.\n`);
858
+ `The #203 affinity adds +scopeAffinity (max() of the project signal → 0.15) to hardware memories in computeScore; marine memories get 0. No hard filter anywhere.\n`);
846
859
  L.push("_Same invariants as the blend sweep: static DB (`noStrengthen: true`), real bge-small-en-v1.5 embedder, " +
847
860
  "uniform metadata (base_strength=0.5, created_at~now, no links). Blend weight held at 0 (pure prompt) — " +
848
861
  "scope is orthogonal to session-intent keying. The `neverCalledEmbed` self-check still passes._\n");
@@ -926,6 +939,21 @@ async function main() {
926
939
  (0, retrieval_js_1.configureDecay)();
927
940
  (0, retrieval_js_1.configureRecall)();
928
941
  (0, retrieval_js_1.configureSessionIntent)();
942
+ // Minimal flag parsing (the file's first flag): --now <ISO> pins the eval
943
+ // clock (#458). Invalid input fails explicit — never a silent live clock.
944
+ let now;
945
+ const argv = process.argv.slice(2);
946
+ const nowIdx = argv.indexOf("--now");
947
+ try {
948
+ now = (0, eval_clock_js_1.parsePinnedNow)(nowIdx !== -1 ? argv[nowIdx + 1] : undefined);
949
+ }
950
+ catch (err) {
951
+ console.error(`[recall-sweep] ${err instanceof Error ? err.message : err}`);
952
+ process.exitCode = 1;
953
+ return;
954
+ }
955
+ const nowArg = now ?? undefined;
956
+ console.log(`[recall-sweep] clock: ${(0, eval_clock_js_1.clockLabel)(now)}`);
929
957
  const sessions = buildSessions();
930
958
  const tmpDir = (0, node_path_1.join)((0, node_os_1.tmpdir)(), `hicortex-recall-sweep-${(0, node_crypto_1.randomUUID)().slice(0, 8)}`);
931
959
  (0, node_fs_1.mkdirSync)(tmpDir, { recursive: true });
@@ -945,27 +973,28 @@ async function main() {
945
973
  try {
946
974
  console.log("[recall-sweep] building corpus (embedding 30 memories)...");
947
975
  const t0 = Date.now();
948
- const { db: opened, idToTopic } = await buildCorpusDb(dbPath);
976
+ const { db: opened, idToTopic } = await buildCorpusDb(dbPath, nowArg);
949
977
  db = opened;
950
978
  console.log(`[recall-sweep] corpus ready in ${Date.now() - t0}ms (${idToTopic.size} memories)`);
951
979
  console.log("[recall-sweep] running sweep...");
952
980
  const t1 = Date.now();
953
- const records = await runSweep(db, idToTopic, sessions);
981
+ const records = await runSweep(db, idToTopic, sessions, nowArg);
954
982
  console.log(`[recall-sweep] sweep done in ${Date.now() - t1}ms (${records.length} turn records)`);
955
983
  const focused = summarizeFocused(records, sessions);
956
984
  const shifts = summarizeShift(records);
957
985
  let report = renderReport(focused, shifts, records, sessions, {
958
986
  memoryCount: CORPUS.length,
987
+ clock: (0, eval_clock_js_1.clockLabel)(now),
959
988
  });
960
989
  // ---- SCOPE sweep (#203) ----
961
990
  console.log(`[recall-sweep] building scope corpus (embedding ${SCOPE_CORPUS.length} project-labeled memories)...`);
962
991
  const t2 = Date.now();
963
- const scopeBuilt = await buildScopeDb(scopeDbPath);
992
+ const scopeBuilt = await buildScopeDb(scopeDbPath, nowArg);
964
993
  scopeDb = scopeBuilt.db;
965
994
  console.log(`[recall-sweep] scope corpus ready in ${Date.now() - t2}ms (${scopeBuilt.ids.length} memories)`);
966
995
  console.log("[recall-sweep] running scope sweep (OFF vs ON)...");
967
996
  const t3 = Date.now();
968
- const scopeRecords = await runScopeSweep(scopeDb, scopeBuilt.idToProject, scopeBuilt.ids);
997
+ const scopeRecords = await runScopeSweep(scopeDb, scopeBuilt.idToProject, scopeBuilt.ids, nowArg);
969
998
  console.log(`[recall-sweep] scope sweep done in ${Date.now() - t3}ms (${scopeRecords.length} turn records)`);
970
999
  report += "\n\n" + renderScopeSection(scopeRecords);
971
1000
  (0, node_fs_1.mkdirSync)(reportDir, { recursive: true });
@@ -59,6 +59,120 @@
59
59
  * Run:
60
60
  * npm run eval:relevance -- <snapshot.db> [prompts.json] [report.md] \
61
61
  * [--judge-delay-ms=2000] [--max-calls=900] [--resume] \
62
- * [--verdicts-json=path] [--verdicts-jsonl=path]
62
+ * [--verdicts-json=path] [--verdicts-jsonl=path] \
63
+ * [--now=<ISO>] [--runs=N]
64
+ *
65
+ * #458 clock pin + judge variance protocol:
66
+ * - `--now=<ISO>` pins the clock the retrieve sweep scores against (the
67
+ * retrieve() `now` seam) — before/after runs become wall-clock-independent
68
+ * (default: live clock, the pre-#458 behavior).
69
+ * - `--runs=N` (default 1) runs judge phases 1+2 N times over the SAME
70
+ * selections (one retrieve sweep) and reports the run-to-run noise floor
71
+ * (median + spread + selection-identical flip counts, §15b). Phases 3+4
72
+ * stay single-shot; the shared --max-calls budget spans all runs and the
73
+ * default scales ×N unless --max-calls was passed explicitly.
74
+ */
75
+ type Verdict = "relevant" | "noise" | "stale";
76
+ interface JudgeVerdict {
77
+ /** Q1..Q8 label — mapped back to the real memory id via the caller's
78
+ * position-ordered topIds list. */
79
+ id: string;
80
+ verdict: Verdict;
81
+ reason: string;
82
+ }
83
+ type VerdictsOrError = JudgeVerdict[] | {
84
+ judgeError: string;
85
+ };
86
+ interface CliArgs {
87
+ positionals: string[];
88
+ judgeDelayMs: number;
89
+ maxCalls: number;
90
+ /** True when --max-calls was passed explicitly (#458) — an explicit budget
91
+ * was sized by the caller for the whole invocation and must NOT be scaled
92
+ * by --runs (see scaleMaxCalls). */
93
+ maxCallsExplicit: boolean;
94
+ resume: boolean;
95
+ /** Override the raw-verdicts JSON path (default VERDICTS_JSON_PATH) — added
96
+ * 2026-08-02 so a re-run against a NEW snapshot can never silently
97
+ * overwrite a baseline's raw verdicts. */
98
+ verdictsJsonPath: string;
99
+ /** Override the checkpoint sidecar / raw-verdicts JSONL path (default
100
+ * VERDICTS_JSONL_PATH) — same reason. */
101
+ verdictsJsonlPath: string;
102
+ /** #458: raw --now=<ISO> value — undefined = live clock. Validated in main
103
+ * via parsePinnedNow (invalid → usage-style error, exit 1). */
104
+ nowIso: string | undefined;
105
+ /** #458: judge runs over the fixed selections (phases 1+2 only). Default 1
106
+ * = the pre-#458 single-run behavior. */
107
+ runs: number;
108
+ }
109
+ /** Parsed and exported for tests (#458). Throws on an invalid --runs value
110
+ * (NaN or <1) — fail-explicit, never a silent default. */
111
+ export declare function parseArgs(argv: string[]): CliArgs;
112
+ /**
113
+ * #458 — resolve the effective --max-calls budget. The ONE shared budget
114
+ * spans ALL judge runs, so when --runs>1 and the caller did NOT pass
115
+ * --max-calls explicitly, the default scales ×N (each run re-judges every
116
+ * unit). An explicit budget always wins — the caller sized it for the whole
117
+ * invocation. Pure, exported for tests.
118
+ */
119
+ export declare function scaleMaxCalls(runs: number, explicitMaxCalls: number | undefined, defaultMaxCalls: number): number;
120
+ /** Median of a numeric sample (even count → mean of the two middle values).
121
+ * Empty input → NaN, the mean() convention above. Does not mutate input. */
122
+ export declare function median(xs: number[]): number;
123
+ /** One judged unit's verdicts for variance accounting (#458): `key` is the
124
+ * stable (promptIdx|mode) unit identity, identical across runs because the
125
+ * retrieve sweep runs ONCE. */
126
+ export interface VarianceUnitVerdicts {
127
+ key: string;
128
+ verdicts: VerdictsOrError;
129
+ }
130
+ /** A headline metric measured once per run (e.g. precision@5 OFF). Values are
131
+ * in raw fractions; null = not computable that run (excluded from
132
+ * median/spread). */
133
+ export interface VarianceMetricSeries {
134
+ label: string;
135
+ values: Array<number | null>;
136
+ }
137
+ export interface VarianceMetricStats extends VarianceMetricSeries {
138
+ median: number | null;
139
+ /** (max − min) × 100, in points — the measured run-to-run noise floor. */
140
+ spreadPts: number | null;
141
+ }
142
+ /** Pairwise flip counts between runs i and j (1-based run numbers). */
143
+ export interface PairwiseFlips {
144
+ runA: number;
145
+ runB: number;
146
+ /** Verdict ROWS compared: (unit, Q-label) pairs present as parsed verdict
147
+ * arrays in BOTH runs. judge_error units and labels the judge omitted in
148
+ * one run are not comparable — excluded, never counted as flips. */
149
+ rowsCompared: number;
150
+ rowsFlipped: number;
151
+ unitsCompared: number;
152
+ /** A UNIT flips if any of its comparable rows differ. */
153
+ unitsFlipped: number;
154
+ }
155
+ export interface JudgeVariance {
156
+ runs: number;
157
+ metrics: VarianceMetricStats[];
158
+ flips: PairwiseFlips[];
159
+ maxRowsFlipped: number;
160
+ maxUnitsFlipped: number;
161
+ meanRowsFlipped: number;
162
+ meanUnitsFlipped: number;
163
+ /** Comparable rows of the first pair — selections are identical across
164
+ * runs, so only judge errors shrink this (reported per run below). */
165
+ totalComparableRows: number;
166
+ /** judge_error UNITS per run (a failed unit yields no parsed rows; it is
167
+ * excluded from every flip denominator). */
168
+ judgeErrorUnitsPerRun: number[];
169
+ }
170
+ /**
171
+ * Compute the #458 judge-variance account over N runs of the same selections:
172
+ * per-metric median + spread (the noise floor), and pairwise verdict flip
173
+ * counts (rows = one Q-label judgment on one (prompt,mode) unit; units flip
174
+ * if any row does). `perRunVerdicts[r]` is run r+1's units in a stable
175
+ * order/identity; `metricSeries[s].values[r]` is run r+1's headline value.
63
176
  */
177
+ export declare function computeJudgeVariance(perRunVerdicts: VarianceUnitVerdicts[][], metricSeries: VarianceMetricSeries[]): JudgeVariance;
64
178
  export {};