@gamaze/hicortex 0.22.0 → 0.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +8 -2
- package/dist/calibration.d.ts +47 -12
- package/dist/calibration.js +52 -15
- package/dist/consolidate.js +14 -5
- package/dist/eval/decay-eval.d.ts +4 -2
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/eval-clock.d.ts +32 -0
- package/dist/eval/eval-clock.js +47 -0
- package/dist/eval/graph-eval.d.ts +15 -2
- package/dist/eval/graph-eval.js +51 -5
- package/dist/eval/planted-eval.d.ts +4 -0
- package/dist/eval/planted-eval.js +27 -2
- package/dist/eval/planted-harness.d.ts +7 -0
- package/dist/eval/planted-harness.js +2 -0
- package/dist/eval/ranking-battery.d.ts +49 -2
- package/dist/eval/ranking-battery.js +110 -2
- package/dist/eval/ranking-eval.d.ts +26 -6
- package/dist/eval/ranking-eval.js +197 -34
- package/dist/eval/ranking-fixtures.d.ts +41 -1
- package/dist/eval/ranking-fixtures.js +261 -2
- package/dist/eval/recall-sweep.d.ts +7 -2
- package/dist/eval/recall-sweep.js +42 -13
- package/dist/eval/relevance-eval.d.ts +115 -1
- package/dist/eval/relevance-eval.js +318 -32
- package/dist/eval/run-eval.d.ts +7 -4
- package/dist/eval/run-eval.js +36 -9
- package/dist/mcp-server.js +2 -1
- package/dist/retrieval.d.ts +34 -15
- package/dist/retrieval.js +132 -59
- package/package.json +1 -1
- package/server.json +2 -2
package/assets/dashboard.html
CHANGED
|
@@ -1391,9 +1391,15 @@ function fmtTokens(n){
|
|
|
1391
1391
|
return String(n);
|
|
1392
1392
|
}
|
|
1393
1393
|
/* deterministic per-id randomness — the same memory always lands in the
|
|
1394
|
-
same place across reloads (positions are content of the field, not noise)
|
|
1394
|
+
same place across reloads (positions are content of the field, not noise).
|
|
1395
|
+
#465: the salt hashes at the FRONT (hash32(salt+':'+s)) — FNV-1a does not
|
|
1396
|
+
diffuse a trailing-byte difference, and salts 1/2/3 differ only in their
|
|
1397
|
+
last character, so the old trailing form coupled angle to radius at
|
|
1398
|
+
Pearson r=0.953 (four spiral arms per domain). Salt-first measures
|
|
1399
|
+
r=0.019 (noise). Positions shift once (hash-derived content), then
|
|
1400
|
+
re-stabilize. */
|
|
1395
1401
|
function hash32(s){ let h=2166136261; for(let i=0;i<s.length;i++){ h^=s.charCodeAt(i); h=Math.imul(h,16777619);} return h>>>0; }
|
|
1396
|
-
function rand01(s,salt){ return hash32(
|
|
1402
|
+
function rand01(s,salt){ return hash32(salt+':'+s)/4294967296; }
|
|
1397
1403
|
|
|
1398
1404
|
/* ---------- token handling (harvested verbatim from the 0.20.4 console:
|
|
1399
1405
|
?token= handoff stripped from the URL, localStorage persistence, 401 →
|
package/dist/calibration.d.ts
CHANGED
|
@@ -115,23 +115,58 @@ export declare const SCORE_SIMILARITY_WEIGHT = 0.5;
|
|
|
115
115
|
export declare const SCORE_STRENGTH_WEIGHT = 0.2;
|
|
116
116
|
/** Graph-centrality share of the composite score (was 0.20; see above). */
|
|
117
117
|
export declare const SCORE_CONNECTIONS_WEIGHT = 0.15;
|
|
118
|
-
/**
|
|
118
|
+
/**
|
|
119
|
+
* Log-saturation degree K for the connections term (#449 PR E, items 1+3):
|
|
120
|
+
* the credit is min(1, log1p(k) / log1p(K)) × SCORE_CONNECTIONS_WEIGHT on
|
|
121
|
+
* the ABSOLUTE undirected-degree scale k — full at k = K, exactly +0 at
|
|
122
|
+
* k = 0 (unlinked rows score bit-identically to the linear term they
|
|
123
|
+
* replace). Provenance: p99 of the REAL undirected degree distribution
|
|
124
|
+
* (read-only audit 2026-09-17 of the canonical wave snapshot, 18,798
|
|
125
|
+
* memories / 14,600 linked: p50 = 4, p90 = 8, p95 = 9, p99 = 15, p99.9 =
|
|
126
|
+
* 28.4, max = 40; 135 memories ≥ 16 links = 0.92% of linked) — so the top
|
|
127
|
+
* ~1% of hubs tie at full credit and everything below differentiates
|
|
128
|
+
* modestly. Shape grounded in the #449 research base: ACT-R fan saturation
|
|
129
|
+
* (Anderson & Reder 1999 — activation falls with the LOG of fan), SAM's
|
|
130
|
+
* saturating returns, cue overload (Watkins & Watkins 1975). Replaces the
|
|
131
|
+
* candidate-set-relative normalization #449 deleted (k / the per-query max
|
|
132
|
+
* degree) — a row's score no longer depends on which other rows happened
|
|
133
|
+
* to match. Alternatives K = 8 (p90 — collapses all p90+ rows to
|
|
134
|
+
* full credit) and K = 32 (≈p99.9 — leaves the p99 hub at 0.12 of the
|
|
135
|
+
* term) were tabled in the #449 spec. RELEASE-MANAGED per this module's
|
|
136
|
+
* evolution contract — overridable only through the configureScoring seam.
|
|
137
|
+
*/
|
|
138
|
+
export declare const CONNECTIONS_SATURATION_DEGREE = 16;
|
|
139
|
+
/** Time-curve share of the composite score — the merged curve's blend weight
|
|
140
|
+
* AND slow-region amplitude (was 0.10; see above). The four blend weights
|
|
141
|
+
* still sum to 1.0; the head amplitude below is a documented overshoot, NOT
|
|
142
|
+
* a fifth blend weight (#430). Zero (via the seam) disables the whole term. */
|
|
119
143
|
export declare const SCORE_RECENCY_WEIGHT = 0.15;
|
|
120
|
-
/**
|
|
121
|
-
*
|
|
122
|
-
*
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
144
|
+
/** Hourly decay of the merged curve's slow region (#430) — promoted from the
|
|
145
|
+
* inline literal computeScore carried; half-life ≈ 57.7 days. At and beyond
|
|
146
|
+
* RECENCY_HEAD_DAYS the curve is bit-identical to the pre-#430 slow term
|
|
147
|
+
* (SCORE_RECENCY_WEIGHT × RECENCY_HOURLY_DECAY^hours). */
|
|
148
|
+
export declare const RECENCY_HOURLY_DECAY = 0.9995;
|
|
149
|
+
/** Merged time-curve head amplitude at age 0 (#430) — the pre-#430 slow
|
|
150
|
+
* weight 0.15 + fresh bonus 0.15: the freshness job is now the early steep
|
|
151
|
+
* part of ONE curve. The head joins value-continuously onto the slow region
|
|
152
|
+
* at RECENCY_HEAD_DAYS; its rate is DERIVED from that continuity, never a
|
|
153
|
+
* free constant. */
|
|
154
|
+
export declare const RECENCY_HEAD_WEIGHT = 0.3;
|
|
155
|
+
/** Join age in days of the merged time curve (#430) — the head window edge,
|
|
156
|
+
* carried over from the pre-#430 freshness window. Nightly capture means
|
|
157
|
+
* 1 day is the floor of "fresh" (#191 Phase B). */
|
|
158
|
+
export declare const RECENCY_HEAD_DAYS = 7;
|
|
126
159
|
/** Score multiplier for a memory a later decision superseded (0.15.2; the
|
|
127
160
|
* belief walk (#393 D) is the primary mechanism — this is the safety net
|
|
128
161
|
* for rows the walk does not reach). */
|
|
129
162
|
export declare const SUPERSEDED_DEMOTION = 0.5;
|
|
130
|
-
/** #
|
|
131
|
-
*
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
163
|
+
/** #430 merged scope affinity — ONE term replacing #203's two boosts (the
|
|
164
|
+
* project-affinity and domain-affinity constants) at their shared value.
|
|
165
|
+
* Boost = max(project-match indicator (1 on exact match), max overlapping
|
|
166
|
+
* domain-tag weight) × this weight: the strongest single scope signal counts
|
|
167
|
+
* once, never stacked. ADDITIVE, zero-boost neutral, never a penalty — an
|
|
168
|
+
* absent scope adds exactly 0; a foreign memory ranks equal, not lower. */
|
|
169
|
+
export declare const SCOPE_AFFINITY_WEIGHT = 0.15;
|
|
135
170
|
/** #205 RRF k parameter (1/(k+rank+1)) — matches the pre-#205 hardcoded 60
|
|
136
171
|
* so the no-config path was byte-identical to 0.15.3. */
|
|
137
172
|
export declare const RRF_K = 60;
|
package/dist/calibration.js
CHANGED
|
@@ -24,7 +24,8 @@
|
|
|
24
24
|
* the removal is never silent.
|
|
25
25
|
*/
|
|
26
26
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
27
|
-
exports.
|
|
27
|
+
exports.OLLAMA_FLUSH_WAIT_MS = exports.OLLAMA_FLUSH_EVERY = exports.NUM_CTX = exports.DEFAULT_FIRST_RUN_LOOKBACK_DAYS = exports.ENRICH_STRENGTH_DELTA = exports.STAGE_TRUTH_STRENGTH = exports.STAGE_BELIEF_STRENGTH = exports.STAGE_FADING_STRENGTH = exports.STAGE_FADING_DAYS = exports.WEAK_PRIMARY_FLOOR = exports.CORRECTION_REWRITE_MIN_CONFIDENCE = exports.CORRECTION_MIN_SIMILARITY = exports.SUPERSESSION_MIN_SIMILARITY = exports.DEDUP_AUTO_MERGE_THRESHOLD = exports.IMPORTANCE_CEILING = exports.BM25_WEIGHT_DOMAIN = exports.BM25_WEIGHT_PROJECT = exports.BM25_WEIGHT_BODY = exports.BOTH_CHANNEL_BOOST = exports.RRF_VECTOR_WEIGHT = exports.RRF_FTS_WEIGHT = exports.RRF_COMPOSITE_WEIGHT = exports.RRF_K = exports.SCOPE_AFFINITY_WEIGHT = exports.SUPERSEDED_DEMOTION = exports.RECENCY_HEAD_DAYS = exports.RECENCY_HEAD_WEIGHT = exports.RECENCY_HOURLY_DECAY = exports.SCORE_RECENCY_WEIGHT = exports.CONNECTIONS_SATURATION_DEGREE = exports.SCORE_CONNECTIONS_WEIGHT = exports.SCORE_STRENGTH_WEIGHT = exports.SCORE_SIMILARITY_WEIGHT = exports.RECALL_USES_AXIS_MAX = exports.RECALL_USES_NORMAL_MAX = exports.RECALL_USES_LOW_MAX = exports.RECALL_RESHOW_TURNS = exports.NOVELTY_FLOOR_SLOTS = exports.RECALL_TITLE_CHARS = exports.RECALL_MIN_PROMPT_CHARS = exports.RECALL_MAX_ITEMS = exports.RECALL_MIN_SIMILARITY = exports.SESSION_INTENT_WEIGHT = exports.COLD_EXPOSURE_SLOTS = exports.RECENT_WINDOW_DAYS = exports.RECENT_LIMIT = exports.SEARCH_LIMIT = exports.PROMOTION_STRENGTH_FLOOR = exports.PROMOTION_RATE = exports.DECAY_HALF_LIFE_DAYS = void 0;
|
|
28
|
+
exports.DIAGNOSTIC_ENV_TIER = void 0;
|
|
28
29
|
exports.resolveNumCtx = resolveNumCtx;
|
|
29
30
|
exports.resolveOllamaFlushEvery = resolveOllamaFlushEvery;
|
|
30
31
|
exports.resolveOllamaFlushWaitMs = resolveOllamaFlushWaitMs;
|
|
@@ -127,8 +128,9 @@ exports.RECALL_USES_NORMAL_MAX = 0.25;
|
|
|
127
128
|
* keeps a visible span, not a measured bound. */
|
|
128
129
|
exports.RECALL_USES_AXIS_MAX = 0.30;
|
|
129
130
|
// ---------------------------------------------------------------------------
|
|
130
|
-
// Composite ranking weights (was: score*Weight,
|
|
131
|
-
//
|
|
131
|
+
// Composite ranking weights (was: score*Weight, supersededDemotion,
|
|
132
|
+
// *AffinityWeight, rrf*; the pre-#430 freshness bonus constants merged into
|
|
133
|
+
// the RECENCY family below)
|
|
132
134
|
// ---------------------------------------------------------------------------
|
|
133
135
|
/** Semantic-similarity share of the composite score. 0.50 (raised from 0.40
|
|
134
136
|
* in the 0.15.2 rebalance, #191 Phase B): on the production corpus effective
|
|
@@ -140,23 +142,58 @@ exports.SCORE_SIMILARITY_WEIGHT = 0.50;
|
|
|
140
142
|
exports.SCORE_STRENGTH_WEIGHT = 0.20;
|
|
141
143
|
/** Graph-centrality share of the composite score (was 0.20; see above). */
|
|
142
144
|
exports.SCORE_CONNECTIONS_WEIGHT = 0.15;
|
|
143
|
-
/**
|
|
145
|
+
/**
|
|
146
|
+
* Log-saturation degree K for the connections term (#449 PR E, items 1+3):
|
|
147
|
+
* the credit is min(1, log1p(k) / log1p(K)) × SCORE_CONNECTIONS_WEIGHT on
|
|
148
|
+
* the ABSOLUTE undirected-degree scale k — full at k = K, exactly +0 at
|
|
149
|
+
* k = 0 (unlinked rows score bit-identically to the linear term they
|
|
150
|
+
* replace). Provenance: p99 of the REAL undirected degree distribution
|
|
151
|
+
* (read-only audit 2026-09-17 of the canonical wave snapshot, 18,798
|
|
152
|
+
* memories / 14,600 linked: p50 = 4, p90 = 8, p95 = 9, p99 = 15, p99.9 =
|
|
153
|
+
* 28.4, max = 40; 135 memories ≥ 16 links = 0.92% of linked) — so the top
|
|
154
|
+
* ~1% of hubs tie at full credit and everything below differentiates
|
|
155
|
+
* modestly. Shape grounded in the #449 research base: ACT-R fan saturation
|
|
156
|
+
* (Anderson & Reder 1999 — activation falls with the LOG of fan), SAM's
|
|
157
|
+
* saturating returns, cue overload (Watkins & Watkins 1975). Replaces the
|
|
158
|
+
* candidate-set-relative normalization #449 deleted (k / the per-query max
|
|
159
|
+
* degree) — a row's score no longer depends on which other rows happened
|
|
160
|
+
* to match. Alternatives K = 8 (p90 — collapses all p90+ rows to
|
|
161
|
+
* full credit) and K = 32 (≈p99.9 — leaves the p99 hub at 0.12 of the
|
|
162
|
+
* term) were tabled in the #449 spec. RELEASE-MANAGED per this module's
|
|
163
|
+
* evolution contract — overridable only through the configureScoring seam.
|
|
164
|
+
*/
|
|
165
|
+
exports.CONNECTIONS_SATURATION_DEGREE = 16;
|
|
166
|
+
/** Time-curve share of the composite score — the merged curve's blend weight
|
|
167
|
+
* AND slow-region amplitude (was 0.10; see above). The four blend weights
|
|
168
|
+
* still sum to 1.0; the head amplitude below is a documented overshoot, NOT
|
|
169
|
+
* a fifth blend weight (#430). Zero (via the seam) disables the whole term. */
|
|
144
170
|
exports.SCORE_RECENCY_WEIGHT = 0.15;
|
|
145
|
-
/**
|
|
146
|
-
*
|
|
147
|
-
*
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
171
|
+
/** Hourly decay of the merged curve's slow region (#430) — promoted from the
|
|
172
|
+
* inline literal computeScore carried; half-life ≈ 57.7 days. At and beyond
|
|
173
|
+
* RECENCY_HEAD_DAYS the curve is bit-identical to the pre-#430 slow term
|
|
174
|
+
* (SCORE_RECENCY_WEIGHT × RECENCY_HOURLY_DECAY^hours). */
|
|
175
|
+
exports.RECENCY_HOURLY_DECAY = 0.9995;
|
|
176
|
+
/** Merged time-curve head amplitude at age 0 (#430) — the pre-#430 slow
|
|
177
|
+
* weight 0.15 + fresh bonus 0.15: the freshness job is now the early steep
|
|
178
|
+
* part of ONE curve. The head joins value-continuously onto the slow region
|
|
179
|
+
* at RECENCY_HEAD_DAYS; its rate is DERIVED from that continuity, never a
|
|
180
|
+
* free constant. */
|
|
181
|
+
exports.RECENCY_HEAD_WEIGHT = 0.30;
|
|
182
|
+
/** Join age in days of the merged time curve (#430) — the head window edge,
|
|
183
|
+
* carried over from the pre-#430 freshness window. Nightly capture means
|
|
184
|
+
* 1 day is the floor of "fresh" (#191 Phase B). */
|
|
185
|
+
exports.RECENCY_HEAD_DAYS = 7;
|
|
151
186
|
/** Score multiplier for a memory a later decision superseded (0.15.2; the
|
|
152
187
|
* belief walk (#393 D) is the primary mechanism — this is the safety net
|
|
153
188
|
* for rows the walk does not reach). */
|
|
154
189
|
exports.SUPERSEDED_DEMOTION = 0.50;
|
|
155
|
-
/** #
|
|
156
|
-
*
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
190
|
+
/** #430 merged scope affinity — ONE term replacing #203's two boosts (the
|
|
191
|
+
* project-affinity and domain-affinity constants) at their shared value.
|
|
192
|
+
* Boost = max(project-match indicator (1 on exact match), max overlapping
|
|
193
|
+
* domain-tag weight) × this weight: the strongest single scope signal counts
|
|
194
|
+
* once, never stacked. ADDITIVE, zero-boost neutral, never a penalty — an
|
|
195
|
+
* absent scope adds exactly 0; a foreign memory ranks equal, not lower. */
|
|
196
|
+
exports.SCOPE_AFFINITY_WEIGHT = 0.15;
|
|
160
197
|
/** #205 RRF k parameter (1/(k+rank+1)) — matches the pre-#205 hardcoded 60
|
|
161
198
|
* so the no-config path was byte-identical to 0.15.3. */
|
|
162
199
|
exports.RRF_K = 60;
|
package/dist/consolidate.js
CHANGED
|
@@ -1376,6 +1376,7 @@ function stagePromotion(db, dryRun) {
|
|
|
1376
1376
|
const demoted = (0, retrieval_js_1.findDemotedIds)(db, rows.map((r) => r.id));
|
|
1377
1377
|
let promoted = 0;
|
|
1378
1378
|
let demotedSkipped = 0;
|
|
1379
|
+
let totalGain = 0;
|
|
1379
1380
|
const writes = [];
|
|
1380
1381
|
for (const row of rows) {
|
|
1381
1382
|
const accessCount = row.access_count ?? 0;
|
|
@@ -1389,18 +1390,25 @@ function stagePromotion(db, dryRun) {
|
|
|
1389
1390
|
}
|
|
1390
1391
|
// base_strength is NOT NULL after scoring; the `?? 0.5` mirrors
|
|
1391
1392
|
// stageDecayPrune's defensive default for unscored rows (inserts at 0.5).
|
|
1392
|
-
|
|
1393
|
+
const startStrength = row.base_strength ?? 0.5;
|
|
1394
|
+
let strength = startStrength;
|
|
1393
1395
|
for (let i = 0; i < row.delta; i++)
|
|
1394
1396
|
strength = applyStrengthPromotion(strength);
|
|
1397
|
+
totalGain += strength - startStrength;
|
|
1395
1398
|
promoted++;
|
|
1396
1399
|
writes.push({
|
|
1397
1400
|
id: row.id,
|
|
1398
1401
|
fields: { base_strength: strength, promotion_last_count: accessCount },
|
|
1399
1402
|
});
|
|
1400
1403
|
}
|
|
1404
|
+
// #459: the stage's one-line summary in the shared stage idiom (rows
|
|
1405
|
+
// examined / promoted / total gain, like the supersession summary) — the
|
|
1406
|
+
// stage writes its report in-memory only, so the log line is the soak-time
|
|
1407
|
+
// health signal. Zero examined rows stay silent (the supersession gate).
|
|
1401
1408
|
if (dryRun) {
|
|
1402
|
-
console.log(`[hicortex] Strength promotion (dry-run):
|
|
1403
|
-
|
|
1409
|
+
console.log(`[hicortex] Strength promotion (dry-run): ${rows.length} examined, would promote ` +
|
|
1410
|
+
`${promoted} (+${totalGain.toFixed(3)} total strength, ` +
|
|
1411
|
+
`${demotedSkipped} demotion-set rows advance their baseline only).`);
|
|
1404
1412
|
return { promoted, demoted_skipped: demotedSkipped };
|
|
1405
1413
|
}
|
|
1406
1414
|
// One transaction for the whole stage (the stageMemoryCapEviction pattern):
|
|
@@ -1411,8 +1419,9 @@ function stagePromotion(db, dryRun) {
|
|
|
1411
1419
|
storage.updateMemory(db, w.id, w.fields);
|
|
1412
1420
|
});
|
|
1413
1421
|
tx();
|
|
1414
|
-
console.log(`[hicortex] Strength promotion: promoted ${promoted}
|
|
1415
|
-
`(
|
|
1422
|
+
console.log(`[hicortex] Strength promotion: ${rows.length} examined, promoted ${promoted} ` +
|
|
1423
|
+
`(+${totalGain.toFixed(3)} total strength, ` +
|
|
1424
|
+
`${demotedSkipped} demotion-set rows advance their baseline only).`);
|
|
1416
1425
|
return { promoted, demoted_skipped: demotedSkipped };
|
|
1417
1426
|
}
|
|
1418
1427
|
// ---------------------------------------------------------------------------
|
|
@@ -70,8 +70,10 @@ export interface StructuralStrengthStats {
|
|
|
70
70
|
* "if nothing changes" projection), plus the asymptotic floor per memory.
|
|
71
71
|
* Requires `configureDecay` to already reflect the desired half-life (call
|
|
72
72
|
* `runPruneDryRun` first, or `configureDecay` directly, in the same process).
|
|
73
|
+
* `now` (#458): the instant "now" means — pinned by run-eval's --now for
|
|
74
|
+
* wall-clock-independent before/after audits; default the live clock.
|
|
73
75
|
*/
|
|
74
|
-
export declare function runStructuralStrengthStats(db: Database.Database): StructuralStrengthStats;
|
|
76
|
+
export declare function runStructuralStrengthStats(db: Database.Database, now?: Date): StructuralStrengthStats;
|
|
75
77
|
export interface AdoptionStats {
|
|
76
78
|
totalMemories: number;
|
|
77
79
|
shownCountHistogram: Record<string, number>;
|
|
@@ -106,4 +108,4 @@ export interface AdoptionStats {
|
|
|
106
108
|
* feature's youth as much as its effectiveness. Report callers should state
|
|
107
109
|
* that caveat alongside the numbers, not just the numbers.
|
|
108
110
|
*/
|
|
109
|
-
export declare function runAdoptionStats(db: Database.Database): AdoptionStats;
|
|
111
|
+
export declare function runAdoptionStats(db: Database.Database, now?: Date): AdoptionStats;
|
package/dist/eval/decay-eval.js
CHANGED
|
@@ -97,12 +97,13 @@ function histogram(values, edges = STRENGTH_BUCKET_EDGES) {
|
|
|
97
97
|
* "if nothing changes" projection), plus the asymptotic floor per memory.
|
|
98
98
|
* Requires `configureDecay` to already reflect the desired half-life (call
|
|
99
99
|
* `runPruneDryRun` first, or `configureDecay` directly, in the same process).
|
|
100
|
+
* `now` (#458): the instant "now" means — pinned by run-eval's --now for
|
|
101
|
+
* wall-clock-independent before/after audits; default the live clock.
|
|
100
102
|
*/
|
|
101
|
-
function runStructuralStrengthStats(db) {
|
|
103
|
+
function runStructuralStrengthStats(db, now = new Date()) {
|
|
102
104
|
const rows = db
|
|
103
105
|
.prepare("SELECT id, base_strength, last_accessed FROM memories")
|
|
104
106
|
.all();
|
|
105
|
-
const now = new Date();
|
|
106
107
|
const nowVals = [];
|
|
107
108
|
const at180 = [];
|
|
108
109
|
const at365 = [];
|
|
@@ -147,11 +148,10 @@ function ageBucketLabel(ageDays) {
|
|
|
147
148
|
* feature's youth as much as its effectiveness. Report callers should state
|
|
148
149
|
* that caveat alongside the numbers, not just the numbers.
|
|
149
150
|
*/
|
|
150
|
-
function runAdoptionStats(db) {
|
|
151
|
+
function runAdoptionStats(db, now = new Date()) {
|
|
151
152
|
const rows = db
|
|
152
153
|
.prepare("SELECT id, shown_count, access_count, source_agent, created_at FROM memories")
|
|
153
154
|
.all();
|
|
154
|
-
const now = new Date();
|
|
155
155
|
const shownVals = [];
|
|
156
156
|
const accessVals = [];
|
|
157
157
|
let totalShown = 0;
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* #458 — the shared pinned-clock helper for eval harnesses.
|
|
3
|
+
*
|
|
4
|
+
* The recall evals score against wall-clock time (recency + decay terms in
|
|
5
|
+
* computeScore/effectiveStrength), so a pre-photo and a post-run hours apart
|
|
6
|
+
* drift even on identical code — PR D evidenced mode-OFF units moving with no
|
|
7
|
+
* changed code in the path. The fix is a harness-visible pin: every eval that
|
|
8
|
+
* calls retrieve() (or the decay/adoption stats) accepts a `--now <ISO>` flag
|
|
9
|
+
* and threads the SAME instant into the retrieve() `now` seam (retrieval.ts —
|
|
10
|
+
* the injectable-clock idiom of run-deadline.ts), making before/after runs
|
|
11
|
+
* wall-clock-independent. Default stays the live clock — pinning is opt-in,
|
|
12
|
+
* for comparability.
|
|
13
|
+
*
|
|
14
|
+
* Fail-explicit by design (the repo convention): an invalid pin THROWS with a
|
|
15
|
+
* clear message rather than silently falling back to the live clock — a run
|
|
16
|
+
* that believed it was pinned but wasn't would silently reintroduce the drift
|
|
17
|
+
* class this helper exists to remove.
|
|
18
|
+
*/
|
|
19
|
+
/**
|
|
20
|
+
* Parse a `--now` flag value into the pinned instant.
|
|
21
|
+
*
|
|
22
|
+
* @param raw the raw flag value. undefined or empty/whitespace → null (the
|
|
23
|
+
* live clock — the flag was not passed). Anything else must parse as a
|
|
24
|
+
* Date; garbage or an invalid instant (e.g. month 13) throws.
|
|
25
|
+
* @returns the pinned Date, or null for the live clock.
|
|
26
|
+
*/
|
|
27
|
+
export declare function parsePinnedNow(raw: string | undefined): Date | null;
|
|
28
|
+
/**
|
|
29
|
+
* Human-readable clock mode for report headers / photo JSON: "pinned <ISO>"
|
|
30
|
+
* or "live" — so artifacts are self-describing about which clock produced them.
|
|
31
|
+
*/
|
|
32
|
+
export declare function clockLabel(now: Date | null): string;
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* #458 — the shared pinned-clock helper for eval harnesses.
|
|
4
|
+
*
|
|
5
|
+
* The recall evals score against wall-clock time (recency + decay terms in
|
|
6
|
+
* computeScore/effectiveStrength), so a pre-photo and a post-run hours apart
|
|
7
|
+
* drift even on identical code — PR D evidenced mode-OFF units moving with no
|
|
8
|
+
* changed code in the path. The fix is a harness-visible pin: every eval that
|
|
9
|
+
* calls retrieve() (or the decay/adoption stats) accepts a `--now <ISO>` flag
|
|
10
|
+
* and threads the SAME instant into the retrieve() `now` seam (retrieval.ts —
|
|
11
|
+
* the injectable-clock idiom of run-deadline.ts), making before/after runs
|
|
12
|
+
* wall-clock-independent. Default stays the live clock — pinning is opt-in,
|
|
13
|
+
* for comparability.
|
|
14
|
+
*
|
|
15
|
+
* Fail-explicit by design (the repo convention): an invalid pin THROWS with a
|
|
16
|
+
* clear message rather than silently falling back to the live clock — a run
|
|
17
|
+
* that believed it was pinned but wasn't would silently reintroduce the drift
|
|
18
|
+
* class this helper exists to remove.
|
|
19
|
+
*/
|
|
20
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
21
|
+
exports.parsePinnedNow = parsePinnedNow;
|
|
22
|
+
exports.clockLabel = clockLabel;
|
|
23
|
+
/**
|
|
24
|
+
* Parse a `--now` flag value into the pinned instant.
|
|
25
|
+
*
|
|
26
|
+
* @param raw the raw flag value. undefined or empty/whitespace → null (the
|
|
27
|
+
* live clock — the flag was not passed). Anything else must parse as a
|
|
28
|
+
* Date; garbage or an invalid instant (e.g. month 13) throws.
|
|
29
|
+
* @returns the pinned Date, or null for the live clock.
|
|
30
|
+
*/
|
|
31
|
+
function parsePinnedNow(raw) {
|
|
32
|
+
if (raw === undefined || raw.trim() === "")
|
|
33
|
+
return null;
|
|
34
|
+
const parsed = new Date(raw);
|
|
35
|
+
if (Number.isNaN(parsed.getTime())) {
|
|
36
|
+
throw new Error(`eval clock: invalid --now value "${raw}" — pass a full ISO 8601 instant ` +
|
|
37
|
+
`(e.g. 2026-09-17T09:00:00.000Z) or omit the flag to run on the live clock`);
|
|
38
|
+
}
|
|
39
|
+
return parsed;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Human-readable clock mode for report headers / photo JSON: "pinned <ISO>"
|
|
43
|
+
* or "live" — so artifacts are self-describing about which clock produced them.
|
|
44
|
+
*/
|
|
45
|
+
function clockLabel(now) {
|
|
46
|
+
return now === null ? "live" : `pinned ${now.toISOString()}`;
|
|
47
|
+
}
|
|
@@ -8,6 +8,13 @@
|
|
|
8
8
|
* corrected cosine formula, #145).
|
|
9
9
|
*/
|
|
10
10
|
import type Database from "better-sqlite3";
|
|
11
|
+
/**
|
|
12
|
+
* Deterministic sample of `rows` (#460): a seeded partial Fisher–Yates
|
|
13
|
+
* shuffles the LAST `size` slots from the back, and that shuffled tail is
|
|
14
|
+
* the sample. Rows must already be in a stable order (the caller sorts by
|
|
15
|
+
* PK) so the draw does not depend on SQLite's scan order.
|
|
16
|
+
*/
|
|
17
|
+
export declare function seededSample<T>(rows: T[], size: number, seed: number): T[];
|
|
11
18
|
export interface LinkRow {
|
|
12
19
|
source_id: string;
|
|
13
20
|
target_id: string;
|
|
@@ -50,8 +57,14 @@ export interface DriftReport {
|
|
|
50
57
|
maxAbsDrift: number;
|
|
51
58
|
skippedMissingEmbedding: number;
|
|
52
59
|
}
|
|
53
|
-
/**
|
|
54
|
-
|
|
60
|
+
/**
|
|
61
|
+
* Recompute cosine for a link sample and compare to stored `strength` (post
|
|
62
|
+
* migration-5 rescale, should track closely). The sample is deterministic
|
|
63
|
+
* (#460): all link rows in stable PK order, drawn by a seeded Fisher–Yates —
|
|
64
|
+
* two runs on the same build + snapshot produce an identical report, so
|
|
65
|
+
* before/after comparisons can require full-output identity.
|
|
66
|
+
*/
|
|
67
|
+
export declare function runDriftSample(db: Database.Database, embeddings: Map<string, Float32Array>, sampleSize?: number, seed?: number): DriftReport;
|
|
55
68
|
export interface PartitionStats {
|
|
56
69
|
linkCount: number;
|
|
57
70
|
byRelationship: Record<string, number>;
|
package/dist/eval/graph-eval.js
CHANGED
|
@@ -42,6 +42,7 @@ var __importStar = (this && this.__importStar) || (function () {
|
|
|
42
42
|
};
|
|
43
43
|
})();
|
|
44
44
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
45
|
+
exports.seededSample = seededSample;
|
|
45
46
|
exports.cosineBetween = cosineBetween;
|
|
46
47
|
exports.byRelationshipCounts = byRelationshipCounts;
|
|
47
48
|
exports.partitionByRelinkCursor = partitionByRelinkCursor;
|
|
@@ -55,8 +56,46 @@ const decay_eval_js_1 = require("./decay-eval.js");
|
|
|
55
56
|
/** Matches the middle D1 duplicate threshold — a link between near-dups is noise, not signal. */
|
|
56
57
|
const DUP_NOISE_THRESHOLD = 0.92;
|
|
57
58
|
const DRIFT_SAMPLE_SIZE = 500;
|
|
59
|
+
/**
|
|
60
|
+
* Seed for the drift sample's PRNG (#460). SQLite's `random()` cannot be
|
|
61
|
+
* seeded, so the draw happens client-side with a deterministic Fisher–Yates;
|
|
62
|
+
* this fixed seed makes the sample — and therefore the whole DriftReport —
|
|
63
|
+
* byte-identical across runs on the same build + snapshot, which lets
|
|
64
|
+
* before/after eval comparisons require full-output identity.
|
|
65
|
+
*/
|
|
66
|
+
const DRIFT_SAMPLE_SEED = 460;
|
|
58
67
|
const DEGREE_BUCKET_EDGES = [0, 1, 2, 3, 5, 10, 20, 50, 100];
|
|
59
68
|
const DRIFT_BUCKET_EDGES = [0, 0.01, 0.02, 0.05, 0.1, 0.2, 0.3, 0.5, 1.0];
|
|
69
|
+
/**
|
|
70
|
+
* mulberry32 — a 32-bit seeded PRNG (public-domain reference implementation).
|
|
71
|
+
* JS number ops are deterministic across platforms, so a fixed seed yields
|
|
72
|
+
* the same sequence everywhere.
|
|
73
|
+
*/
|
|
74
|
+
function mulberry32(seed) {
|
|
75
|
+
let a = seed >>> 0;
|
|
76
|
+
return () => {
|
|
77
|
+
a = (a + 0x6d2b79f5) | 0;
|
|
78
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
79
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
80
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
81
|
+
};
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* Deterministic sample of `rows` (#460): a seeded partial Fisher–Yates
|
|
85
|
+
* shuffles the LAST `size` slots from the back, and that shuffled tail is
|
|
86
|
+
* the sample. Rows must already be in a stable order (the caller sorts by
|
|
87
|
+
* PK) so the draw does not depend on SQLite's scan order.
|
|
88
|
+
*/
|
|
89
|
+
function seededSample(rows, size, seed) {
|
|
90
|
+
const rand = mulberry32(seed);
|
|
91
|
+
const arr = [...rows];
|
|
92
|
+
const n = Math.min(size, arr.length);
|
|
93
|
+
for (let i = arr.length - 1; i > arr.length - 1 - n; i--) {
|
|
94
|
+
const j = Math.floor(rand() * (i + 1));
|
|
95
|
+
[arr[i], arr[j]] = [arr[j], arr[i]];
|
|
96
|
+
}
|
|
97
|
+
return arr.slice(arr.length - n);
|
|
98
|
+
}
|
|
60
99
|
// ---------------------------------------------------------------------------
|
|
61
100
|
// Pure helpers (unit tested)
|
|
62
101
|
// ---------------------------------------------------------------------------
|
|
@@ -109,11 +148,18 @@ function runDegreeAudit(db, topN = 10) {
|
|
|
109
148
|
});
|
|
110
149
|
return { memoriesWithLinks: linkCounts.size, totalMemories, degreeHistogram, topHubs };
|
|
111
150
|
}
|
|
112
|
-
/**
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
151
|
+
/**
|
|
152
|
+
* Recompute cosine for a link sample and compare to stored `strength` (post
|
|
153
|
+
* migration-5 rescale, should track closely). The sample is deterministic
|
|
154
|
+
* (#460): all link rows in stable PK order, drawn by a seeded Fisher–Yates —
|
|
155
|
+
* two runs on the same build + snapshot produce an identical report, so
|
|
156
|
+
* before/after comparisons can require full-output identity.
|
|
157
|
+
*/
|
|
158
|
+
function runDriftSample(db, embeddings, sampleSize = DRIFT_SAMPLE_SIZE, seed = DRIFT_SAMPLE_SEED) {
|
|
159
|
+
const allLinks = db
|
|
160
|
+
.prepare("SELECT source_id, target_id, strength FROM memory_links ORDER BY source_id, target_id")
|
|
161
|
+
.all();
|
|
162
|
+
const sample = seededSample(allLinks, sampleSize, seed);
|
|
117
163
|
const drifts = [];
|
|
118
164
|
let skipped = 0;
|
|
119
165
|
for (const link of sample) {
|
|
@@ -20,6 +20,10 @@
|
|
|
20
20
|
* live-LLM real-corpus acceptance run is the release-soak step (refine Q3),
|
|
21
21
|
* not this one.
|
|
22
22
|
*
|
|
23
|
+
* `--now <ISO>` (#458) pins the clock the recall-surface retrieve() scores
|
|
24
|
+
* against — before/after gate runs become wall-clock-independent. Default:
|
|
25
|
+
* the live clock; the report records which clock produced it.
|
|
26
|
+
*
|
|
23
27
|
* Never touches ~/.hicortex: DB, state dir, and backups all live under a
|
|
24
28
|
* mkdtemp directory that is removed on exit.
|
|
25
29
|
*/
|
|
@@ -21,6 +21,10 @@
|
|
|
21
21
|
* live-LLM real-corpus acceptance run is the release-soak step (refine Q3),
|
|
22
22
|
* not this one.
|
|
23
23
|
*
|
|
24
|
+
* `--now <ISO>` (#458) pins the clock the recall-surface retrieve() scores
|
|
25
|
+
* against — before/after gate runs become wall-clock-independent. Default:
|
|
26
|
+
* the live clock; the report records which clock produced it.
|
|
27
|
+
*
|
|
24
28
|
* Never touches ~/.hicortex: DB, state dir, and backups all live under a
|
|
25
29
|
* mkdtemp directory that is removed on exit.
|
|
26
30
|
*/
|
|
@@ -32,12 +36,15 @@ const embedder_js_1 = require("../embedder.js");
|
|
|
32
36
|
const capture_js_1 = require("../capture.js");
|
|
33
37
|
const planted_fixtures_js_1 = require("./planted-fixtures.js");
|
|
34
38
|
const planted_harness_js_1 = require("./planted-harness.js");
|
|
39
|
+
const eval_clock_js_1 = require("./eval-clock.js");
|
|
35
40
|
const DEFAULT_REPORT_PATH = (0, node_path_1.join)(process.cwd(), "data", "planted-eval-report.md");
|
|
36
41
|
function usage() {
|
|
37
|
-
console.error("Usage: npm run eval:planted -- [report.md] [--snapshot <snapshot.db>]\n" +
|
|
42
|
+
console.error("Usage: npm run eval:planted -- [report.md] [--snapshot <snapshot.db>] [--now <ISO>]\n" +
|
|
38
43
|
" [report.md] output path (default data/planted-eval-report.md)\n" +
|
|
39
44
|
" --snapshot <db> plant into a COPY of a real snapshot (readonly source;\n" +
|
|
40
|
-
" the copy is writable, the snapshot is never touched)"
|
|
45
|
+
" the copy is writable, the snapshot is never touched)\n" +
|
|
46
|
+
" --now <ISO> #458: pin the eval clock (full ISO 8601 instant) the\n" +
|
|
47
|
+
" recall-surface retrieve() scores against. Default: live.");
|
|
41
48
|
process.exitCode = 1;
|
|
42
49
|
process.exit(1);
|
|
43
50
|
}
|
|
@@ -45,16 +52,32 @@ async function main() {
|
|
|
45
52
|
const argv = process.argv.slice(2);
|
|
46
53
|
let reportPath;
|
|
47
54
|
let snapshotPath;
|
|
55
|
+
let nowIso;
|
|
48
56
|
for (let i = 0; i < argv.length; i++) {
|
|
49
57
|
if (argv[i] === "--snapshot" && argv[i + 1]) {
|
|
50
58
|
snapshotPath = argv[++i];
|
|
51
59
|
continue;
|
|
52
60
|
}
|
|
61
|
+
if (argv[i] === "--now" && argv[i + 1]) {
|
|
62
|
+
nowIso = argv[++i];
|
|
63
|
+
continue;
|
|
64
|
+
}
|
|
53
65
|
if (!argv[i].startsWith("--"))
|
|
54
66
|
reportPath = argv[i];
|
|
55
67
|
}
|
|
56
68
|
if (argv.includes("--help") || argv.includes("-h"))
|
|
57
69
|
usage();
|
|
70
|
+
// #458: fail-explicit — an invalid pin is an error, never a live fallback.
|
|
71
|
+
let now;
|
|
72
|
+
try {
|
|
73
|
+
now = (0, eval_clock_js_1.parsePinnedNow)(nowIso);
|
|
74
|
+
}
|
|
75
|
+
catch (err) {
|
|
76
|
+
console.error(`[planted-eval] ${err instanceof Error ? err.message : err}`);
|
|
77
|
+
process.exitCode = 1;
|
|
78
|
+
return;
|
|
79
|
+
}
|
|
80
|
+
console.log(`[planted-eval] clock: ${(0, eval_clock_js_1.clockLabel)(now)}`);
|
|
58
81
|
const workDir = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-planted-eval-"));
|
|
59
82
|
try {
|
|
60
83
|
const dbPath = (0, node_path_1.join)(workDir, "planted.db");
|
|
@@ -70,6 +93,7 @@ async function main() {
|
|
|
70
93
|
stateDir,
|
|
71
94
|
embedFn: embedder_js_1.embed,
|
|
72
95
|
acquireLock: capture_js_1.acquireCaptureLock, // real lock, but on the temp stateDir only
|
|
96
|
+
now: now ?? undefined, // #458: pinned clock (undefined = live)
|
|
73
97
|
});
|
|
74
98
|
console.log(`[planted-eval] gate completed in ${Date.now() - t0}ms`);
|
|
75
99
|
const report = (0, planted_harness_js_1.renderPlantedReport)({
|
|
@@ -78,6 +102,7 @@ async function main() {
|
|
|
78
102
|
: "real-text planted corpus (synthetic texts, real embeddings)",
|
|
79
103
|
generatedAt: new Date().toISOString(),
|
|
80
104
|
embedderLabel: "real bge-small-en-v1.5 (Xenova, fp32) — cosines measured, never asserted",
|
|
105
|
+
clock: (0, eval_clock_js_1.clockLabel)(now),
|
|
81
106
|
result,
|
|
82
107
|
});
|
|
83
108
|
const outPath = reportPath ?? DEFAULT_REPORT_PATH;
|
|
@@ -162,6 +162,10 @@ export interface PlantedGateOptions {
|
|
|
162
162
|
acquireLock?: typeof acquireCaptureLock;
|
|
163
163
|
/** Zone ceiling. Default: the production default (0.92). */
|
|
164
164
|
threshold?: number;
|
|
165
|
+
/** #458: the pinned clock for the recall-surface retrieve() (undefined =
|
|
166
|
+
* live). Fixtures keep their absolute createdAt strings — only the
|
|
167
|
+
* scoring instant is pinned. */
|
|
168
|
+
now?: Date;
|
|
165
169
|
}
|
|
166
170
|
/**
|
|
167
171
|
* Build (or plant into) the DB, run the production stage, measure the gate.
|
|
@@ -172,5 +176,8 @@ export declare function renderPlantedReport(args: {
|
|
|
172
176
|
corpusLabel: string;
|
|
173
177
|
generatedAt: string;
|
|
174
178
|
embedderLabel: string;
|
|
179
|
+
/** #458: clock-mode label ("pinned <ISO>" | "live") — the instant the
|
|
180
|
+
* recall-surface retrieve() scored against. */
|
|
181
|
+
clock: string;
|
|
175
182
|
result: PlantedGateResult;
|
|
176
183
|
}): string;
|
|
@@ -521,6 +521,7 @@ async function runPlantedGate(corpus, opts) {
|
|
|
521
521
|
const recall = await (0, retrieval_js_1.retrieve)(db, opts.embedFn, queryContent, {
|
|
522
522
|
limit: 3,
|
|
523
523
|
noStrengthen: true,
|
|
524
|
+
...(opts.now ? { now: opts.now } : {}), // #458: pinned clock (undefined = live)
|
|
524
525
|
});
|
|
525
526
|
// Every surfaced row must be its own walk terminal — no superseded
|
|
526
527
|
// ancestor competes as a truth — and the linked chain's terminal
|
|
@@ -593,6 +594,7 @@ function renderPlantedReport(args) {
|
|
|
593
594
|
const L = [];
|
|
594
595
|
L.push("# Planted-Pairs Resolution Gate (#393 increment A)\n");
|
|
595
596
|
L.push(`Corpus: ${args.corpusLabel} \nGenerated: ${args.generatedAt} \nEmbedder: ${args.embedderLabel} \n` +
|
|
597
|
+
`Clock: ${args.clock} (#458) \n` +
|
|
596
598
|
`Judge: ground-truth stub (perfect-judge semantics — machinery isolation, per the refine Q3 ruling). \n` +
|
|
597
599
|
`Floors: correctionMinSimilarity 0.75 (judged-band floor) · dedupAutoMergeThreshold 0.92 (zone ceiling).\n`);
|
|
598
600
|
L.push("## Gate — per planted pair\n");
|