@gamaze/hicortex 0.23.1 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +64 -13
- package/dist/calibration.d.ts +31 -0
- package/dist/calibration.js +38 -1
- package/dist/capture.d.ts +7 -0
- package/dist/capture.js +10 -1
- package/dist/dashboard.d.ts +32 -7
- package/dist/dashboard.js +50 -10
- package/dist/db.js +45 -0
- package/dist/dedup.js +2 -2
- package/dist/distill-queue.d.ts +203 -0
- package/dist/distill-queue.js +440 -0
- package/dist/health.d.ts +13 -1
- package/dist/health.js +6 -1
- package/dist/hosted-boot.d.ts +1 -1
- package/dist/index.js +17 -2
- package/dist/learnings-identity.js +20 -1
- package/dist/localhost-bypass.js +1 -1
- package/dist/mcp-server.js +92 -7
- package/dist/nightly.js +88 -1
- package/dist/nofit.d.ts +1 -1
- package/dist/nofit.js +1 -1
- package/dist/schema-prototypes.d.ts +1 -1
- package/dist/schema-prototypes.js +1 -1
- package/dist/status.d.ts +11 -0
- package/dist/status.js +25 -0
- package/dist/types.d.ts +20 -0
- package/hermes-plugin/hicortex/README.md +13 -6
- package/hermes-plugin/hicortex/__init__.py +7 -0
- package/hermes-plugin/hicortex/client.py +64 -2
- package/hermes-plugin/hicortex/plugin.yaml +1 -1
- package/hermes-plugin/hicortex/provider.py +24 -1
- package/opencode-plugin/hicortex/index.ts +24 -1
- package/package.json +2 -1
- package/pi-extension/hicortex/index.ts +24 -1
- package/server.json +2 -2
- package/dist/eval/decay-eval.d.ts +0 -111
- package/dist/eval/decay-eval.js +0 -214
- package/dist/eval/dups.d.ts +0 -100
- package/dist/eval/dups.js +0 -174
- package/dist/eval/eval-clock.d.ts +0 -32
- package/dist/eval/eval-clock.js +0 -47
- package/dist/eval/eval-db.d.ts +0 -25
- package/dist/eval/eval-db.js +0 -67
- package/dist/eval/graph-eval.d.ts +0 -89
- package/dist/eval/graph-eval.js +0 -246
- package/dist/eval/importance-eval.d.ts +0 -85
- package/dist/eval/importance-eval.js +0 -286
- package/dist/eval/planted-eval.d.ts +0 -30
- package/dist/eval/planted-eval.js +0 -122
- package/dist/eval/planted-fixtures.d.ts +0 -107
- package/dist/eval/planted-fixtures.js +0 -283
- package/dist/eval/planted-harness.d.ts +0 -183
- package/dist/eval/planted-harness.js +0 -651
- package/dist/eval/ranking-battery.d.ts +0 -125
- package/dist/eval/ranking-battery.js +0 -289
- package/dist/eval/ranking-eval.d.ts +0 -61
- package/dist/eval/ranking-eval.js +0 -554
- package/dist/eval/ranking-fixtures.d.ts +0 -117
- package/dist/eval/ranking-fixtures.js +0 -485
- package/dist/eval/recall-sweep.d.ts +0 -87
- package/dist/eval/recall-sweep.js +0 -1030
- package/dist/eval/reflection-census.d.ts +0 -19
- package/dist/eval/reflection-census.js +0 -25
- package/dist/eval/relevance-eval.d.ts +0 -178
- package/dist/eval/relevance-eval.js +0 -2240
- package/dist/eval/run-eval.d.ts +0 -20
- package/dist/eval/run-eval.js +0 -299
package/dist/eval/graph-eval.js
DELETED
|
@@ -1,246 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
/**
|
|
3
|
-
* D6 — link-graph health audit (#191 mechanical baseline).
|
|
4
|
-
*
|
|
5
|
-
* Of the ~6.5k links: how many are meaningful vs near-duplicate noise, does
|
|
6
|
-
* the stored `memory_links.strength` still match a fresh cosine recompute
|
|
7
|
-
* (drift), and how do all these stats look before vs after the
|
|
8
|
-
* `relinkCursor` watermark (the resumable `hicortex relink` migration to the
|
|
9
|
-
* corrected cosine formula, #145).
|
|
10
|
-
*/
|
|
11
|
-
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
12
|
-
if (k2 === undefined) k2 = k;
|
|
13
|
-
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
14
|
-
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
15
|
-
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
16
|
-
}
|
|
17
|
-
Object.defineProperty(o, k2, desc);
|
|
18
|
-
}) : (function(o, m, k, k2) {
|
|
19
|
-
if (k2 === undefined) k2 = k;
|
|
20
|
-
o[k2] = m[k];
|
|
21
|
-
}));
|
|
22
|
-
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
23
|
-
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
24
|
-
}) : function(o, v) {
|
|
25
|
-
o["default"] = v;
|
|
26
|
-
});
|
|
27
|
-
var __importStar = (this && this.__importStar) || (function () {
|
|
28
|
-
var ownKeys = function(o) {
|
|
29
|
-
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
30
|
-
var ar = [];
|
|
31
|
-
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
32
|
-
return ar;
|
|
33
|
-
};
|
|
34
|
-
return ownKeys(o);
|
|
35
|
-
};
|
|
36
|
-
return function (mod) {
|
|
37
|
-
if (mod && mod.__esModule) return mod;
|
|
38
|
-
var result = {};
|
|
39
|
-
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
40
|
-
__setModuleDefault(result, mod);
|
|
41
|
-
return result;
|
|
42
|
-
};
|
|
43
|
-
})();
|
|
44
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
45
|
-
exports.seededSample = seededSample;
|
|
46
|
-
exports.cosineBetween = cosineBetween;
|
|
47
|
-
exports.byRelationshipCounts = byRelationshipCounts;
|
|
48
|
-
exports.partitionByRelinkCursor = partitionByRelinkCursor;
|
|
49
|
-
exports.runDegreeAudit = runDegreeAudit;
|
|
50
|
-
exports.runDriftSample = runDriftSample;
|
|
51
|
-
exports.runGraphAudit = runGraphAudit;
|
|
52
|
-
const eval_db_js_1 = require("./eval-db.js");
|
|
53
|
-
const storage = __importStar(require("../storage.js"));
|
|
54
|
-
const state_js_1 = require("../state.js");
|
|
55
|
-
const decay_eval_js_1 = require("./decay-eval.js");
|
|
56
|
-
/** Matches the middle D1 duplicate threshold — a link between near-dups is noise, not signal. */
|
|
57
|
-
const DUP_NOISE_THRESHOLD = 0.92;
|
|
58
|
-
const DRIFT_SAMPLE_SIZE = 500;
|
|
59
|
-
/**
|
|
60
|
-
* Seed for the drift sample's PRNG (#460). SQLite's `random()` cannot be
|
|
61
|
-
* seeded, so the draw happens client-side with a deterministic Fisher–Yates;
|
|
62
|
-
* this fixed seed makes the sample — and therefore the whole DriftReport —
|
|
63
|
-
* byte-identical across runs on the same build + snapshot, which lets
|
|
64
|
-
* before/after eval comparisons require full-output identity.
|
|
65
|
-
*/
|
|
66
|
-
const DRIFT_SAMPLE_SEED = 460;
|
|
67
|
-
const DEGREE_BUCKET_EDGES = [0, 1, 2, 3, 5, 10, 20, 50, 100];
|
|
68
|
-
const DRIFT_BUCKET_EDGES = [0, 0.01, 0.02, 0.05, 0.1, 0.2, 0.3, 0.5, 1.0];
|
|
69
|
-
/**
|
|
70
|
-
* mulberry32 — a 32-bit seeded PRNG (public-domain reference implementation).
|
|
71
|
-
* JS number ops are deterministic across platforms, so a fixed seed yields
|
|
72
|
-
* the same sequence everywhere.
|
|
73
|
-
*/
|
|
74
|
-
function mulberry32(seed) {
|
|
75
|
-
let a = seed >>> 0;
|
|
76
|
-
return () => {
|
|
77
|
-
a = (a + 0x6d2b79f5) | 0;
|
|
78
|
-
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
79
|
-
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
80
|
-
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
81
|
-
};
|
|
82
|
-
}
|
|
83
|
-
/**
|
|
84
|
-
* Deterministic sample of `rows` (#460): a seeded partial Fisher–Yates
|
|
85
|
-
* shuffles the LAST `size` slots from the back, and that shuffled tail is
|
|
86
|
-
* the sample. Rows must already be in a stable order (the caller sorts by
|
|
87
|
-
* PK) so the draw does not depend on SQLite's scan order.
|
|
88
|
-
*/
|
|
89
|
-
function seededSample(rows, size, seed) {
|
|
90
|
-
const rand = mulberry32(seed);
|
|
91
|
-
const arr = [...rows];
|
|
92
|
-
const n = Math.min(size, arr.length);
|
|
93
|
-
for (let i = arr.length - 1; i > arr.length - 1 - n; i--) {
|
|
94
|
-
const j = Math.floor(rand() * (i + 1));
|
|
95
|
-
[arr[i], arr[j]] = [arr[j], arr[i]];
|
|
96
|
-
}
|
|
97
|
-
return arr.slice(arr.length - n);
|
|
98
|
-
}
|
|
99
|
-
// ---------------------------------------------------------------------------
|
|
100
|
-
// Pure helpers (unit tested)
|
|
101
|
-
// ---------------------------------------------------------------------------
|
|
102
|
-
/** Cosine similarity between two L2-normalized embeddings (dot product). */
|
|
103
|
-
function cosineBetween(a, b) {
|
|
104
|
-
let dot = 0;
|
|
105
|
-
for (let i = 0; i < a.length; i++)
|
|
106
|
-
dot += a[i] * b[i];
|
|
107
|
-
return dot;
|
|
108
|
-
}
|
|
109
|
-
function byRelationshipCounts(links) {
|
|
110
|
-
const counts = {};
|
|
111
|
-
for (const l of links) {
|
|
112
|
-
counts[l.relationship] = (counts[l.relationship] ?? 0) + 1;
|
|
113
|
-
}
|
|
114
|
-
return counts;
|
|
115
|
-
}
|
|
116
|
-
/**
|
|
117
|
-
* Partition links by whether their SOURCE memory's rowid has been covered by
|
|
118
|
-
* the `hicortex relink` watermark. `relinkCursor` is the last fully
|
|
119
|
-
* committed rowid (relink.ts) — relink iterates memories by rowid and
|
|
120
|
-
* discovers/refreshes links FROM each one, so a link's source rowid <=
|
|
121
|
-
* cursor means a relink pass has already run for that source (current
|
|
122
|
-
* formula); null cursor (never run) puts everything in "notYetRelinked".
|
|
123
|
-
*/
|
|
124
|
-
function partitionByRelinkCursor(links, sourceRowid, relinkCursor) {
|
|
125
|
-
const relinked = [];
|
|
126
|
-
const notYetRelinked = [];
|
|
127
|
-
for (const l of links) {
|
|
128
|
-
const rowid = sourceRowid.get(l.source_id);
|
|
129
|
-
const covered = relinkCursor !== null && rowid !== undefined && rowid <= relinkCursor;
|
|
130
|
-
(covered ? relinked : notYetRelinked).push(l);
|
|
131
|
-
}
|
|
132
|
-
return { relinked, notYetRelinked };
|
|
133
|
-
}
|
|
134
|
-
function runDegreeAudit(db, topN = 10) {
|
|
135
|
-
const totalMemories = db.prepare("SELECT COUNT(*) AS c FROM memories").get().c;
|
|
136
|
-
const linkCounts = storage.getAllLinkCounts(db);
|
|
137
|
-
const degreeHistogram = (0, decay_eval_js_1.histogram)([...linkCounts.values()], DEGREE_BUCKET_EDGES);
|
|
138
|
-
const sortedIds = [...linkCounts.entries()].sort((a, b) => b[1] - a[1]).slice(0, topN);
|
|
139
|
-
const topHubs = sortedIds.map(([id, degree]) => {
|
|
140
|
-
const mem = storage.getMemory(db, id);
|
|
141
|
-
return {
|
|
142
|
-
id,
|
|
143
|
-
degree,
|
|
144
|
-
project: mem?.project ?? null,
|
|
145
|
-
domain: mem?.domain ?? null,
|
|
146
|
-
preview: (mem?.content ?? "").slice(0, 80),
|
|
147
|
-
};
|
|
148
|
-
});
|
|
149
|
-
return { memoriesWithLinks: linkCounts.size, totalMemories, degreeHistogram, topHubs };
|
|
150
|
-
}
|
|
151
|
-
/**
|
|
152
|
-
* Recompute cosine for a link sample and compare to stored `strength` (post
|
|
153
|
-
* migration-5 rescale, should track closely). The sample is deterministic
|
|
154
|
-
* (#460): all link rows in stable PK order, drawn by a seeded Fisher–Yates —
|
|
155
|
-
* two runs on the same build + snapshot produce an identical report, so
|
|
156
|
-
* before/after comparisons can require full-output identity.
|
|
157
|
-
*/
|
|
158
|
-
function runDriftSample(db, embeddings, sampleSize = DRIFT_SAMPLE_SIZE, seed = DRIFT_SAMPLE_SEED) {
|
|
159
|
-
const allLinks = db
|
|
160
|
-
.prepare("SELECT source_id, target_id, strength FROM memory_links ORDER BY source_id, target_id")
|
|
161
|
-
.all();
|
|
162
|
-
const sample = seededSample(allLinks, sampleSize, seed);
|
|
163
|
-
const drifts = [];
|
|
164
|
-
let skipped = 0;
|
|
165
|
-
for (const link of sample) {
|
|
166
|
-
const a = embeddings.get(link.source_id);
|
|
167
|
-
const b = embeddings.get(link.target_id);
|
|
168
|
-
if (!a || !b) {
|
|
169
|
-
skipped++;
|
|
170
|
-
continue;
|
|
171
|
-
}
|
|
172
|
-
const recomputed = cosineBetween(a, b);
|
|
173
|
-
drifts.push(Math.abs(recomputed - link.strength));
|
|
174
|
-
}
|
|
175
|
-
const meanAbsDrift = drifts.length > 0 ? drifts.reduce((s, d) => s + d, 0) / drifts.length : 0;
|
|
176
|
-
const maxAbsDrift = drifts.length > 0 ? Math.max(...drifts) : 0;
|
|
177
|
-
return {
|
|
178
|
-
sampleSize: drifts.length,
|
|
179
|
-
driftHistogram: (0, decay_eval_js_1.histogram)(drifts, DRIFT_BUCKET_EDGES),
|
|
180
|
-
meanAbsDrift,
|
|
181
|
-
maxAbsDrift,
|
|
182
|
-
skippedMissingEmbedding: skipped,
|
|
183
|
-
};
|
|
184
|
-
}
|
|
185
|
-
function statsFor(links, embeddings) {
|
|
186
|
-
let dupNoise = 0;
|
|
187
|
-
let measured = 0;
|
|
188
|
-
for (const l of links) {
|
|
189
|
-
const a = embeddings.get(l.source_id);
|
|
190
|
-
const b = embeddings.get(l.target_id);
|
|
191
|
-
if (!a || !b)
|
|
192
|
-
continue;
|
|
193
|
-
measured++;
|
|
194
|
-
if (cosineBetween(a, b) >= DUP_NOISE_THRESHOLD)
|
|
195
|
-
dupNoise++;
|
|
196
|
-
}
|
|
197
|
-
return {
|
|
198
|
-
linkCount: links.length,
|
|
199
|
-
byRelationship: byRelationshipCounts(links),
|
|
200
|
-
dupNoiseLinks: dupNoise,
|
|
201
|
-
dupNoiseShare: measured > 0 ? dupNoise / measured : 0,
|
|
202
|
-
};
|
|
203
|
-
}
|
|
204
|
-
/** Load every stored embedding once as id -> Float32Array. */
|
|
205
|
-
function loadEmbeddings(db) {
|
|
206
|
-
const rows = db.prepare("SELECT id, embedding FROM memory_vectors").all();
|
|
207
|
-
return new Map(rows.map((r) => [r.id, (0, eval_db_js_1.blobToEmbedding)(r.embedding)]));
|
|
208
|
-
}
|
|
209
|
-
function runGraphAudit(db, stateDir) {
|
|
210
|
-
const links = db
|
|
211
|
-
.prepare("SELECT source_id, target_id, relationship, strength FROM memory_links")
|
|
212
|
-
.all();
|
|
213
|
-
const embeddings = loadEmbeddings(db);
|
|
214
|
-
const degree = runDegreeAudit(db);
|
|
215
|
-
const drift = runDriftSample(db, embeddings);
|
|
216
|
-
let dupNoiseCount = 0;
|
|
217
|
-
let dupNoiseMeasured = 0;
|
|
218
|
-
for (const l of links) {
|
|
219
|
-
const a = embeddings.get(l.source_id);
|
|
220
|
-
const b = embeddings.get(l.target_id);
|
|
221
|
-
if (!a || !b)
|
|
222
|
-
continue;
|
|
223
|
-
dupNoiseMeasured++;
|
|
224
|
-
if (cosineBetween(a, b) >= DUP_NOISE_THRESHOLD)
|
|
225
|
-
dupNoiseCount++;
|
|
226
|
-
}
|
|
227
|
-
const relinkCursor = (0, state_js_1.loadState)(stateDir).relinkCursor ?? null;
|
|
228
|
-
const rowidRows = db.prepare("SELECT id, rowid AS rowid FROM memories").all();
|
|
229
|
-
const sourceRowid = new Map(rowidRows.map((r) => [r.id, r.rowid]));
|
|
230
|
-
const { relinked, notYetRelinked } = partitionByRelinkCursor(links, sourceRowid, relinkCursor);
|
|
231
|
-
return {
|
|
232
|
-
totalLinks: links.length,
|
|
233
|
-
byRelationship: byRelationshipCounts(links),
|
|
234
|
-
degree,
|
|
235
|
-
drift,
|
|
236
|
-
dupNoise: {
|
|
237
|
-
threshold: DUP_NOISE_THRESHOLD,
|
|
238
|
-
count: dupNoiseCount,
|
|
239
|
-
share: dupNoiseMeasured > 0 ? dupNoiseCount / dupNoiseMeasured : 0,
|
|
240
|
-
measured: dupNoiseMeasured,
|
|
241
|
-
},
|
|
242
|
-
relinkCursor,
|
|
243
|
-
relinkedPartition: statsFor(relinked, embeddings),
|
|
244
|
-
notYetRelinkedPartition: statsFor(notYetRelinked, embeddings),
|
|
245
|
-
};
|
|
246
|
-
}
|
|
@@ -1,85 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
/**
|
|
3
|
-
* Importance-scorer distribution eval (#425) — the AC4 measurement.
|
|
4
|
-
*
|
|
5
|
-
* npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
|
|
6
|
-
* [--llm-base-url U --llm-model M --llm-api-key K]
|
|
7
|
-
*
|
|
8
|
-
* Samples real memories from a READONLY snapshot (openSnapshot — never
|
|
9
|
-
* initDb), sends them through the PRODUCTION scoring path — the real
|
|
10
|
-
* `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
|
|
11
|
-
* call exactly like the nightly's stageImportance — and reports the score
|
|
12
|
-
* distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
|
|
13
|
-
* at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
|
|
14
|
-
* median 0.30-0.40, p90 <= 0.75, zero at 1.0).
|
|
15
|
-
*
|
|
16
|
-
* This is the before/after photo for the scorer recalibration: run it on
|
|
17
|
-
* the OLD prompt (the inflated-distribution red photo), then on the
|
|
18
|
-
* re-anchored prompt (AC4 gate).
|
|
19
|
-
*
|
|
20
|
-
* Determinism: the sample is stratified across created_at deciles and
|
|
21
|
-
* drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
|
|
22
|
-
* same sample + same batches + same prompts. Calls are SERIAL (one
|
|
23
|
-
* in-flight request, like the nightly).
|
|
24
|
-
*
|
|
25
|
-
* Honesty rules: no stub LLM, no fallback distribution, measured scores
|
|
26
|
-
* only; a mid-run endpoint error aborts with the partial histogram
|
|
27
|
-
* printed (never silently padded). The gateway endpoint is passed at RUN
|
|
28
|
-
* TIME via flags — nothing about any specific endpoint lives in this file.
|
|
29
|
-
*/
|
|
30
|
-
import { LlmClient } from "../llm.js";
|
|
31
|
-
export interface SampleRow {
|
|
32
|
-
id: string;
|
|
33
|
-
content: string;
|
|
34
|
-
created_at: string;
|
|
35
|
-
}
|
|
36
|
-
/**
|
|
37
|
-
* Deterministic decile-stratified sample: rows sorted by created_at, split
|
|
38
|
-
* into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
|
|
39
|
-
* (each decile shuffled by its own PRNG stream — index-independent). Same
|
|
40
|
-
* (rows, n, seed) always yields the same ordered selection.
|
|
41
|
-
*/
|
|
42
|
-
export declare function stratifiedSample(rows: SampleRow[], n: number, seed: number): SampleRow[];
|
|
43
|
-
export interface ScoreSummary {
|
|
44
|
-
n: number;
|
|
45
|
-
min: number;
|
|
46
|
-
p25: number;
|
|
47
|
-
median: number;
|
|
48
|
-
p75: number;
|
|
49
|
-
p90: number;
|
|
50
|
-
max: number;
|
|
51
|
-
atExactly1: number;
|
|
52
|
-
histogram: Record<string, number>;
|
|
53
|
-
}
|
|
54
|
-
export declare function summarizeScores(scores: number[]): ScoreSummary;
|
|
55
|
-
export interface ImportanceEvalOptions {
|
|
56
|
-
snapshotPath: string;
|
|
57
|
-
sample?: number;
|
|
58
|
-
seed?: number;
|
|
59
|
-
llm: LlmClient;
|
|
60
|
-
/** Where to write the report (default data/importance-eval-report.md). */
|
|
61
|
-
reportPath?: string;
|
|
62
|
-
}
|
|
63
|
-
export interface ImportanceEvalResult {
|
|
64
|
-
summary: ScoreSummary;
|
|
65
|
-
/** Per-batch measured scores, in call order (audit trail). */
|
|
66
|
-
batches: Array<{
|
|
67
|
-
sent: number;
|
|
68
|
-
scores: number[];
|
|
69
|
-
}>;
|
|
70
|
-
aborted: boolean;
|
|
71
|
-
error?: string;
|
|
72
|
-
}
|
|
73
|
-
/**
|
|
74
|
-
* Run the importance distribution measurement. Throws only on setup errors;
|
|
75
|
-
* an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
|
|
76
|
-
* the caller decides exit code, always loudly).
|
|
77
|
-
*/
|
|
78
|
-
export declare function runImportanceEval(opts: ImportanceEvalOptions): Promise<ImportanceEvalResult>;
|
|
79
|
-
export declare function renderImportanceReport(args: {
|
|
80
|
-
snapshotPath: string;
|
|
81
|
-
summary: ScoreSummary;
|
|
82
|
-
aborted: boolean;
|
|
83
|
-
error?: string;
|
|
84
|
-
model: string;
|
|
85
|
-
}): string;
|
|
@@ -1,286 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
"use strict";
|
|
3
|
-
/**
|
|
4
|
-
* Importance-scorer distribution eval (#425) — the AC4 measurement.
|
|
5
|
-
*
|
|
6
|
-
* npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
|
|
7
|
-
* [--llm-base-url U --llm-model M --llm-api-key K]
|
|
8
|
-
*
|
|
9
|
-
* Samples real memories from a READONLY snapshot (openSnapshot — never
|
|
10
|
-
* initDb), sends them through the PRODUCTION scoring path — the real
|
|
11
|
-
* `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
|
|
12
|
-
* call exactly like the nightly's stageImportance — and reports the score
|
|
13
|
-
* distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
|
|
14
|
-
* at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
|
|
15
|
-
* median 0.30-0.40, p90 <= 0.75, zero at 1.0).
|
|
16
|
-
*
|
|
17
|
-
* This is the before/after photo for the scorer recalibration: run it on
|
|
18
|
-
* the OLD prompt (the inflated-distribution red photo), then on the
|
|
19
|
-
* re-anchored prompt (AC4 gate).
|
|
20
|
-
*
|
|
21
|
-
* Determinism: the sample is stratified across created_at deciles and
|
|
22
|
-
* drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
|
|
23
|
-
* same sample + same batches + same prompts. Calls are SERIAL (one
|
|
24
|
-
* in-flight request, like the nightly).
|
|
25
|
-
*
|
|
26
|
-
* Honesty rules: no stub LLM, no fallback distribution, measured scores
|
|
27
|
-
* only; a mid-run endpoint error aborts with the partial histogram
|
|
28
|
-
* printed (never silently padded). The gateway endpoint is passed at RUN
|
|
29
|
-
* TIME via flags — nothing about any specific endpoint lives in this file.
|
|
30
|
-
*/
|
|
31
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
32
|
-
exports.stratifiedSample = stratifiedSample;
|
|
33
|
-
exports.summarizeScores = summarizeScores;
|
|
34
|
-
exports.runImportanceEval = runImportanceEval;
|
|
35
|
-
exports.renderImportanceReport = renderImportanceReport;
|
|
36
|
-
const node_fs_1 = require("node:fs");
|
|
37
|
-
const node_path_1 = require("node:path");
|
|
38
|
-
const eval_db_js_1 = require("./eval-db.js");
|
|
39
|
-
const llm_js_1 = require("../llm.js");
|
|
40
|
-
const consolidate_js_1 = require("../consolidate.js");
|
|
41
|
-
const prompts_js_1 = require("../prompts.js");
|
|
42
|
-
const DEFAULT_SAMPLE = 200;
|
|
43
|
-
const DEFAULT_SEED = 1;
|
|
44
|
-
/** The nightly's stageImportance batch size — matched exactly. */
|
|
45
|
-
const BATCH_SIZE = 10;
|
|
46
|
-
// ---------------------------------------------------------------------------
|
|
47
|
-
// Deterministic stratified sampler
|
|
48
|
-
// ---------------------------------------------------------------------------
|
|
49
|
-
/** mulberry32 — small, seedable, deterministic across Node versions. */
|
|
50
|
-
function mulberry32(seed) {
|
|
51
|
-
let a = seed >>> 0;
|
|
52
|
-
return () => {
|
|
53
|
-
a |= 0;
|
|
54
|
-
a = (a + 0x6d2b79f5) | 0;
|
|
55
|
-
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
56
|
-
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
57
|
-
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
58
|
-
};
|
|
59
|
-
}
|
|
60
|
-
/**
|
|
61
|
-
* Deterministic decile-stratified sample: rows sorted by created_at, split
|
|
62
|
-
* into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
|
|
63
|
-
* (each decile shuffled by its own PRNG stream — index-independent). Same
|
|
64
|
-
* (rows, n, seed) always yields the same ordered selection.
|
|
65
|
-
*/
|
|
66
|
-
function stratifiedSample(rows, n, seed) {
|
|
67
|
-
if (rows.length === 0 || n <= 0)
|
|
68
|
-
return [];
|
|
69
|
-
const sorted = [...rows].sort((a, b) => a.created_at === b.created_at
|
|
70
|
-
? a.id.localeCompare(b.id)
|
|
71
|
-
: a.created_at.localeCompare(b.created_at));
|
|
72
|
-
const decileSize = sorted.length / 10;
|
|
73
|
-
const perDecile = Math.max(1, Math.ceil(n / 10));
|
|
74
|
-
const out = [];
|
|
75
|
-
for (let d = 0; d < 10 && out.length < n; d++) {
|
|
76
|
-
const start = Math.floor(d * decileSize);
|
|
77
|
-
const end = d === 9 ? sorted.length : Math.floor((d + 1) * decileSize);
|
|
78
|
-
const decile = sorted.slice(start, end);
|
|
79
|
-
const rnd = mulberry32(seed * 31 + d);
|
|
80
|
-
// Fisher-Yates with the decile-local stream.
|
|
81
|
-
for (let i = decile.length - 1; i > 0; i--) {
|
|
82
|
-
const j = Math.floor(rnd() * (i + 1));
|
|
83
|
-
[decile[i], decile[j]] = [decile[j], decile[i]];
|
|
84
|
-
}
|
|
85
|
-
out.push(...decile.slice(0, Math.min(perDecile, n - out.length)));
|
|
86
|
-
}
|
|
87
|
-
return out;
|
|
88
|
-
}
|
|
89
|
-
/** Nearest-rank percentile on the ASCENDING-sorted values. */
|
|
90
|
-
function percentile(sorted, p) {
|
|
91
|
-
if (sorted.length === 0)
|
|
92
|
-
return Number.NaN;
|
|
93
|
-
const idx = Math.min(sorted.length - 1, Math.max(0, Math.ceil((p / 100) * sorted.length) - 1));
|
|
94
|
-
return sorted[idx];
|
|
95
|
-
}
|
|
96
|
-
function summarizeScores(scores) {
|
|
97
|
-
const sorted = [...scores].sort((a, b) => a - b);
|
|
98
|
-
const histogram = {};
|
|
99
|
-
for (let b = 0; b < 10; b++)
|
|
100
|
-
histogram[`${b}-${b + 1}`] = 0;
|
|
101
|
-
for (const s of scores) {
|
|
102
|
-
const b = Math.min(9, Math.max(0, Math.floor(s * 10)));
|
|
103
|
-
histogram[`${b}-${b + 1}`]++;
|
|
104
|
-
}
|
|
105
|
-
return {
|
|
106
|
-
n: scores.length,
|
|
107
|
-
min: percentile(sorted, 0),
|
|
108
|
-
p25: percentile(sorted, 25),
|
|
109
|
-
median: percentile(sorted, 50),
|
|
110
|
-
p75: percentile(sorted, 75),
|
|
111
|
-
p90: percentile(sorted, 90),
|
|
112
|
-
max: percentile(sorted, 100),
|
|
113
|
-
atExactly1: scores.filter((s) => s === 1).length,
|
|
114
|
-
histogram,
|
|
115
|
-
};
|
|
116
|
-
}
|
|
117
|
-
/**
|
|
118
|
-
* Run the importance distribution measurement. Throws only on setup errors;
|
|
119
|
-
* an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
|
|
120
|
-
* the caller decides exit code, always loudly).
|
|
121
|
-
*/
|
|
122
|
-
async function runImportanceEval(opts) {
|
|
123
|
-
const sampleSize = opts.sample ?? DEFAULT_SAMPLE;
|
|
124
|
-
const seed = opts.seed ?? DEFAULT_SEED;
|
|
125
|
-
const db = (0, eval_db_js_1.openSnapshot)(opts.snapshotPath);
|
|
126
|
-
let rows;
|
|
127
|
-
try {
|
|
128
|
-
rows = db
|
|
129
|
-
.prepare("SELECT id, content, created_at FROM memories WHERE COALESCE(status, '') != 'absorbed' ORDER BY id")
|
|
130
|
-
.all();
|
|
131
|
-
}
|
|
132
|
-
finally {
|
|
133
|
-
db.close();
|
|
134
|
-
}
|
|
135
|
-
if (rows.length === 0)
|
|
136
|
-
throw new Error("snapshot has no live memories");
|
|
137
|
-
const sample = stratifiedSample(rows, sampleSize, seed);
|
|
138
|
-
console.log(`[importance-eval] snapshot rows ${rows.length}, sampled ${sample.length} ` +
|
|
139
|
-
`(seed ${seed}, stratified over created_at deciles)`);
|
|
140
|
-
const batches = [];
|
|
141
|
-
const allScores = [];
|
|
142
|
-
for (let i = 0; i < sample.length; i += BATCH_SIZE) {
|
|
143
|
-
const batch = sample.slice(i, i + BATCH_SIZE);
|
|
144
|
-
const lines = batch.map((mem, idx) => `[${idx}] ${mem.content.slice(0, 500)}`);
|
|
145
|
-
const prompt = (0, prompts_js_1.importanceScoring)(lines.join("\n\n"));
|
|
146
|
-
try {
|
|
147
|
-
const r = await opts.llm.complete(prompt);
|
|
148
|
-
let scores = (0, consolidate_js_1.parseJsonLenient)(r.text, null);
|
|
149
|
-
if (!Array.isArray(scores))
|
|
150
|
-
scores = new Array(batch.length).fill(0.5);
|
|
151
|
-
while (scores.length < batch.length)
|
|
152
|
-
scores.push(0.5);
|
|
153
|
-
scores = scores.slice(0, batch.length);
|
|
154
|
-
const clamped = scores.map((s) => {
|
|
155
|
-
const v = Number(s);
|
|
156
|
-
return Number.isFinite(v) ? Math.max(0, Math.min(1, v)) : 0.5;
|
|
157
|
-
});
|
|
158
|
-
batches.push({ sent: batch.length, scores: clamped });
|
|
159
|
-
allScores.push(...clamped);
|
|
160
|
-
console.log(`[importance-eval] batch ${batches.length}: ${clamped.map((s) => s.toFixed(1)).join(", ")}`);
|
|
161
|
-
}
|
|
162
|
-
catch (err) {
|
|
163
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
164
|
-
console.error(`[importance-eval] endpoint error at batch ${Math.floor(i / BATCH_SIZE) + 1}: ${msg}`);
|
|
165
|
-
const summary = summarizeScores(allScores);
|
|
166
|
-
return { summary, batches, aborted: true, error: msg };
|
|
167
|
-
}
|
|
168
|
-
}
|
|
169
|
-
return { summary: summarizeScores(allScores), batches, aborted: false };
|
|
170
|
-
}
|
|
171
|
-
function renderImportanceReport(args) {
|
|
172
|
-
const s = args.summary;
|
|
173
|
-
const L = [];
|
|
174
|
-
L.push("# Importance-scorer distribution (#425 — AC4)\n");
|
|
175
|
-
L.push(`Snapshot: \`${args.snapshotPath}\` \nGenerated: ${new Date().toISOString()} \n` +
|
|
176
|
-
`Model: ${args.model} (the configured production endpoint) \n` +
|
|
177
|
-
`Prompt: the CURRENT production importanceScoring (whatever ships in this tree)\n`);
|
|
178
|
-
if (args.aborted) {
|
|
179
|
-
L.push(`**ABORTED MID-RUN** — ${args.error ?? "endpoint error"}; partial distribution over ${s.n} scores below.\n`);
|
|
180
|
-
}
|
|
181
|
-
L.push("## Distribution\n");
|
|
182
|
-
L.push("| stat | value |");
|
|
183
|
-
L.push("|---|---|");
|
|
184
|
-
L.push(`| n | ${s.n} |`);
|
|
185
|
-
L.push(`| min | ${s.min.toFixed(3)} |`);
|
|
186
|
-
L.push(`| p25 | ${s.p25.toFixed(3)} |`);
|
|
187
|
-
L.push(`| **median** | **${s.median.toFixed(3)}** |`);
|
|
188
|
-
L.push(`| p75 | ${s.p75.toFixed(3)} |`);
|
|
189
|
-
L.push(`| **p90** | **${s.p90.toFixed(3)}** |`);
|
|
190
|
-
L.push(`| max | ${s.max.toFixed(3)} |`);
|
|
191
|
-
L.push(`| at exactly 1.0 | ${s.atExactly1} |`);
|
|
192
|
-
L.push("");
|
|
193
|
-
L.push("| bucket | count |");
|
|
194
|
-
L.push("|---|---|");
|
|
195
|
-
for (const [bucket, count] of Object.entries(s.histogram)) {
|
|
196
|
-
L.push(`| ${bucket} | ${count} |`);
|
|
197
|
-
}
|
|
198
|
-
L.push("");
|
|
199
|
-
L.push(`Target band (owner D2, 2026-09-13): median 0.30-0.40 → ` +
|
|
200
|
-
`${s.median >= 0.3 && s.median <= 0.4 ? "**PASS**" : "**MISS**"} (${s.median.toFixed(3)}); ` +
|
|
201
|
-
`p90 <= 0.75 → ${s.p90 <= 0.75 ? "**PASS**" : "**MISS**"} (${s.p90.toFixed(3)}); ` +
|
|
202
|
-
`zero at 1.0 → ${s.atExactly1 === 0 ? "**PASS**" : "**MISS**"} (${s.atExactly1})\n`);
|
|
203
|
-
return L.join("\n");
|
|
204
|
-
}
|
|
205
|
-
// ---------------------------------------------------------------------------
|
|
206
|
-
// CLI
|
|
207
|
-
// ---------------------------------------------------------------------------
|
|
208
|
-
function usage() {
|
|
209
|
-
console.error("Usage: npm run eval:importance -- <snapshot.db> [--sample N=200] [--seed S=1]\n" +
|
|
210
|
-
" --llm-base-url U --llm-model M --llm-api-key K (required: the scoring endpoint)\n" +
|
|
211
|
-
" [--llm-thinking] (thinking ON; default OFF — production parity)\n" +
|
|
212
|
-
" [--report <out.md>]");
|
|
213
|
-
process.exit(1);
|
|
214
|
-
}
|
|
215
|
-
async function main() {
|
|
216
|
-
const argv = process.argv.slice(2);
|
|
217
|
-
const positional = [];
|
|
218
|
-
const flag = (name) => {
|
|
219
|
-
const i = argv.indexOf(name);
|
|
220
|
-
return i !== -1 && argv[i + 1] && !argv[i + 1].startsWith("--") ? argv[i + 1] : undefined;
|
|
221
|
-
};
|
|
222
|
-
for (let i = 0; i < argv.length; i++) {
|
|
223
|
-
if (!argv[i].startsWith("--"))
|
|
224
|
-
positional.push(argv[i]);
|
|
225
|
-
}
|
|
226
|
-
const snapshotPath = positional[0];
|
|
227
|
-
if (!snapshotPath || argv.includes("--help") || argv.includes("-h"))
|
|
228
|
-
usage();
|
|
229
|
-
const baseUrl = flag("--llm-base-url");
|
|
230
|
-
const model = flag("--llm-model");
|
|
231
|
-
const apiKey = flag("--llm-api-key");
|
|
232
|
-
if (!baseUrl || !model || !apiKey) {
|
|
233
|
-
console.error("[importance-eval] --llm-base-url, --llm-model and --llm-api-key are required (the production scoring endpoint)");
|
|
234
|
-
usage();
|
|
235
|
-
}
|
|
236
|
-
const sample = flag("--sample") ? parseInt(flag("--sample"), 10) : DEFAULT_SAMPLE;
|
|
237
|
-
const seed = flag("--seed") ? parseInt(flag("--seed"), 10) : DEFAULT_SEED;
|
|
238
|
-
if (!Number.isInteger(sample) || sample < 10 || sample > 5000) {
|
|
239
|
-
console.error("[importance-eval] --sample must be an integer in [10, 5000]");
|
|
240
|
-
process.exit(1);
|
|
241
|
-
}
|
|
242
|
-
if (!Number.isInteger(seed) || seed <= 0) {
|
|
243
|
-
console.error("[importance-eval] --seed must be a positive integer");
|
|
244
|
-
process.exit(1);
|
|
245
|
-
}
|
|
246
|
-
// OpenAI-compatible config built from flags ONLY (never hardcoded; the
|
|
247
|
-
// production endpoint is a runtime input, not a shipped value). Thinking
|
|
248
|
-
// defaults OFF (production parity: the scorer runs with thinking disabled —
|
|
249
|
-
// an unclosed think block can eat the whole output budget); opt in with
|
|
250
|
-
// --llm-thinking for a thinking-on endpoint.
|
|
251
|
-
const config = {
|
|
252
|
-
baseUrl,
|
|
253
|
-
model,
|
|
254
|
-
apiKey,
|
|
255
|
-
provider: "openai",
|
|
256
|
-
enableThinking: argv.includes("--llm-thinking") ? true : false,
|
|
257
|
-
};
|
|
258
|
-
const llm = new llm_js_1.LlmClient(config);
|
|
259
|
-
const t0 = Date.now();
|
|
260
|
-
const result = await runImportanceEval({
|
|
261
|
-
snapshotPath,
|
|
262
|
-
sample,
|
|
263
|
-
seed,
|
|
264
|
-
llm,
|
|
265
|
-
});
|
|
266
|
-
const report = renderImportanceReport({
|
|
267
|
-
snapshotPath,
|
|
268
|
-
summary: result.summary,
|
|
269
|
-
aborted: result.aborted,
|
|
270
|
-
error: result.error,
|
|
271
|
-
model,
|
|
272
|
-
});
|
|
273
|
-
const reportPath = flag("--report") ?? (0, node_path_1.join)(process.cwd(), "data", "importance-eval-report.md");
|
|
274
|
-
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(reportPath), { recursive: true });
|
|
275
|
-
(0, node_fs_1.writeFileSync)(reportPath, report, "utf-8");
|
|
276
|
-
console.log(`[importance-eval] report written: ${reportPath}`);
|
|
277
|
-
console.log(report);
|
|
278
|
-
console.log(`[importance-eval] ${result.aborted ? "ABORTED" : "complete"} in ${Math.round((Date.now() - t0) / 1000)}s`);
|
|
279
|
-
process.exitCode = result.aborted ? 1 : 0;
|
|
280
|
-
}
|
|
281
|
-
if (process.argv[1] && process.argv[1].endsWith("importance-eval.js")) {
|
|
282
|
-
main().catch((err) => {
|
|
283
|
-
console.error("[importance-eval] FAILED:", err instanceof Error ? err.stack : String(err));
|
|
284
|
-
process.exitCode = 1;
|
|
285
|
-
});
|
|
286
|
-
}
|
|
@@ -1,30 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
/**
|
|
3
|
-
* Planted-pairs resolution gate — the #393 increment A acceptance harness.
|
|
4
|
-
*
|
|
5
|
-
* npm run eval:planted -- [report.md] [--snapshot <db>]
|
|
6
|
-
*
|
|
7
|
-
* Builds the real-text planted corpus (planted-fixtures.ts) in a THROWAWAY
|
|
8
|
-
* writable DB, embeds it with the REAL bge-small-en-v1.5 embedder, runs the
|
|
9
|
-
* production resolution stage (`stageReconsolidation`, ground-truth stub
|
|
10
|
-
* judge — see planted-harness.ts), and writes the DETECTED / BOUND / RESOLVED
|
|
11
|
-
* gate table. This is the before/after photo every later #393 increment
|
|
12
|
-
* (scout, guard, walk) must re-run and beat.
|
|
13
|
-
*
|
|
14
|
-
* `--snapshot <db>` plants the same corpus into a COPY of a real snapshot
|
|
15
|
-
* instead of an empty DB (the real-corpus background). The copy is opened
|
|
16
|
-
* writable via initDb (migrations are idempotent) — the snapshot itself is
|
|
17
|
-
* never touched. With a fresh temp stateDir the stage re-scans the ENTIRE
|
|
18
|
-
* corpus (cursor 0) — on a 17k-memory snapshot that is a full pass with
|
|
19
|
-
* stub-verdict calls for every unlinked in-band pair (fast, no network); the
|
|
20
|
-
* live-LLM real-corpus acceptance run is the release-soak step (refine Q3),
|
|
21
|
-
* not this one.
|
|
22
|
-
*
|
|
23
|
-
* `--now <ISO>` (#458) pins the clock the recall-surface retrieve() scores
|
|
24
|
-
* against — before/after gate runs become wall-clock-independent. Default:
|
|
25
|
-
* the live clock; the report records which clock produced it.
|
|
26
|
-
*
|
|
27
|
-
* Never touches ~/.hicortex: DB, state dir, and backups all live under a
|
|
28
|
-
* mkdtemp directory that is removed on exit.
|
|
29
|
-
*/
|
|
30
|
-
export {};
|