@gamaze/hicortex 0.23.0 → 0.23.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +455 -153
- package/dist/dashboard.d.ts +18 -7
- package/dist/dashboard.js +42 -10
- package/dist/dedup.js +2 -2
- package/dist/index.js +17 -2
- package/dist/learnings-identity.js +20 -1
- package/dist/llm.d.ts +12 -1
- package/dist/llm.js +14 -3
- package/dist/mcp-server.js +6 -2
- package/dist/mcp-stdio.d.ts +61 -3
- package/dist/mcp-stdio.js +272 -51
- package/dist/nightly.js +1 -1
- package/hermes-plugin/hicortex/provider.py +23 -0
- package/opencode-plugin/hicortex/index.ts +24 -1
- package/package.json +2 -1
- package/pi-extension/hicortex/index.ts +24 -1
- package/server.json +2 -2
- package/dist/eval/decay-eval.d.ts +0 -111
- package/dist/eval/decay-eval.js +0 -214
- package/dist/eval/dups.d.ts +0 -100
- package/dist/eval/dups.js +0 -174
- package/dist/eval/eval-clock.d.ts +0 -32
- package/dist/eval/eval-clock.js +0 -47
- package/dist/eval/eval-db.d.ts +0 -25
- package/dist/eval/eval-db.js +0 -67
- package/dist/eval/graph-eval.d.ts +0 -89
- package/dist/eval/graph-eval.js +0 -246
- package/dist/eval/importance-eval.d.ts +0 -85
- package/dist/eval/importance-eval.js +0 -286
- package/dist/eval/planted-eval.d.ts +0 -30
- package/dist/eval/planted-eval.js +0 -122
- package/dist/eval/planted-fixtures.d.ts +0 -107
- package/dist/eval/planted-fixtures.js +0 -283
- package/dist/eval/planted-harness.d.ts +0 -183
- package/dist/eval/planted-harness.js +0 -651
- package/dist/eval/ranking-battery.d.ts +0 -125
- package/dist/eval/ranking-battery.js +0 -289
- package/dist/eval/ranking-eval.d.ts +0 -61
- package/dist/eval/ranking-eval.js +0 -554
- package/dist/eval/ranking-fixtures.d.ts +0 -117
- package/dist/eval/ranking-fixtures.js +0 -485
- package/dist/eval/recall-sweep.d.ts +0 -87
- package/dist/eval/recall-sweep.js +0 -1030
- package/dist/eval/reflection-census.d.ts +0 -19
- package/dist/eval/reflection-census.js +0 -25
- package/dist/eval/relevance-eval.d.ts +0 -178
- package/dist/eval/relevance-eval.js +0 -2240
- package/dist/eval/run-eval.d.ts +0 -20
- package/dist/eval/run-eval.js +0 -299
package/dist/eval/eval-db.d.ts
DELETED
|
@@ -1,25 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Read-only snapshot access for the #191 mechanical audit baseline.
|
|
3
|
-
*
|
|
4
|
-
* The eval NEVER touches a live database — it runs against a checkpointed
|
|
5
|
-
* copy (`data/audit-<date>/snapshot.db`, gitignored). This module opens that
|
|
6
|
-
* copy in better-sqlite3's `readonly` mode and loads the sqlite-vec
|
|
7
|
-
* extension the same way `db.ts#initDb` does, WITHOUT calling `initDb()`
|
|
8
|
-
* itself: `initDb` runs schema migrations, which write to the file. A
|
|
9
|
-
* snapshot is assumed to already be at the current schema version (verified
|
|
10
|
-
* by `assertReadonly`, which also proves no migration silently ran).
|
|
11
|
-
*/
|
|
12
|
-
import Database from "better-sqlite3";
|
|
13
|
-
/**
|
|
14
|
-
* Open a DB snapshot for read-only analysis.
|
|
15
|
-
*
|
|
16
|
-
* Throws if the path does not exist, or if a write attempt against the
|
|
17
|
-
* returned connection would (surprisingly) succeed — the second check is a
|
|
18
|
-
* belt-and-suspenders guard against a future better-sqlite3/OS combination
|
|
19
|
-
* where `readonly: true` is silently ignored (e.g. a non-standard
|
|
20
|
-
* filesystem), so a bug can never turn the audit into a mutation of
|
|
21
|
-
* production data.
|
|
22
|
-
*/
|
|
23
|
-
export declare function openSnapshot(dbPath: string): Database.Database;
|
|
24
|
-
/** Convert a sqlite-vec embedding BLOB (as read back from `memory_vectors`) to a Float32Array. */
|
|
25
|
-
export declare function blobToEmbedding(blob: Buffer): Float32Array;
|
package/dist/eval/eval-db.js
DELETED
|
@@ -1,67 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
/**
|
|
3
|
-
* Read-only snapshot access for the #191 mechanical audit baseline.
|
|
4
|
-
*
|
|
5
|
-
* The eval NEVER touches a live database — it runs against a checkpointed
|
|
6
|
-
* copy (`data/audit-<date>/snapshot.db`, gitignored). This module opens that
|
|
7
|
-
* copy in better-sqlite3's `readonly` mode and loads the sqlite-vec
|
|
8
|
-
* extension the same way `db.ts#initDb` does, WITHOUT calling `initDb()`
|
|
9
|
-
* itself: `initDb` runs schema migrations, which write to the file. A
|
|
10
|
-
* snapshot is assumed to already be at the current schema version (verified
|
|
11
|
-
* by `assertReadonly`, which also proves no migration silently ran).
|
|
12
|
-
*/
|
|
13
|
-
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
14
|
-
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
15
|
-
};
|
|
16
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
17
|
-
exports.openSnapshot = openSnapshot;
|
|
18
|
-
exports.blobToEmbedding = blobToEmbedding;
|
|
19
|
-
const better_sqlite3_1 = __importDefault(require("better-sqlite3"));
|
|
20
|
-
const node_fs_1 = require("node:fs");
|
|
21
|
-
/**
|
|
22
|
-
* Open a DB snapshot for read-only analysis.
|
|
23
|
-
*
|
|
24
|
-
* Throws if the path does not exist, or if a write attempt against the
|
|
25
|
-
* returned connection would (surprisingly) succeed — the second check is a
|
|
26
|
-
* belt-and-suspenders guard against a future better-sqlite3/OS combination
|
|
27
|
-
* where `readonly: true` is silently ignored (e.g. a non-standard
|
|
28
|
-
* filesystem), so a bug can never turn the audit into a mutation of
|
|
29
|
-
* production data.
|
|
30
|
-
*/
|
|
31
|
-
function openSnapshot(dbPath) {
|
|
32
|
-
if (!(0, node_fs_1.existsSync)(dbPath)) {
|
|
33
|
-
throw new Error(`Snapshot DB not found at ${dbPath}`);
|
|
34
|
-
}
|
|
35
|
-
const db = new better_sqlite3_1.default(dbPath, { readonly: true });
|
|
36
|
-
// eslint-disable-next-line @typescript-eslint/no-var-requires
|
|
37
|
-
const sqliteVec = require("sqlite-vec");
|
|
38
|
-
sqliteVec.load(db);
|
|
39
|
-
assertReadonly(db);
|
|
40
|
-
return db;
|
|
41
|
-
}
|
|
42
|
-
/**
|
|
43
|
-
* Verify the connection truly refuses writes. Uses a table guaranteed to
|
|
44
|
-
* exist in any migrated Hicortex DB (`schema_version`) and a no-op-shaped
|
|
45
|
-
* statement (touches a version number that cannot exist) so that even if the
|
|
46
|
-
* guard somehow failed open, the blast radius is a single junk row rather
|
|
47
|
-
* than corruption of real data.
|
|
48
|
-
*/
|
|
49
|
-
function assertReadonly(db) {
|
|
50
|
-
let wroteSuccessfully = false;
|
|
51
|
-
try {
|
|
52
|
-
db.prepare("INSERT INTO schema_version (version, name, applied_at) VALUES (-1, '__eval_readonly_probe__', '')").run();
|
|
53
|
-
wroteSuccessfully = true;
|
|
54
|
-
}
|
|
55
|
-
catch {
|
|
56
|
-
// Expected: SQLITE_READONLY. The connection is safe to use.
|
|
57
|
-
}
|
|
58
|
-
if (wroteSuccessfully) {
|
|
59
|
-
throw new Error("Snapshot DB accepted a write — refusing to run the eval against a " +
|
|
60
|
-
"connection that is not truly read-only. Check the better-sqlite3 " +
|
|
61
|
-
"readonly option and filesystem permissions.");
|
|
62
|
-
}
|
|
63
|
-
}
|
|
64
|
-
/** Convert a sqlite-vec embedding BLOB (as read back from `memory_vectors`) to a Float32Array. */
|
|
65
|
-
function blobToEmbedding(blob) {
|
|
66
|
-
return new Float32Array(blob.buffer, blob.byteOffset, blob.byteLength / 4);
|
|
67
|
-
}
|
|
@@ -1,89 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* D6 — link-graph health audit (#191 mechanical baseline).
|
|
3
|
-
*
|
|
4
|
-
* Of the ~6.5k links: how many are meaningful vs near-duplicate noise, does
|
|
5
|
-
* the stored `memory_links.strength` still match a fresh cosine recompute
|
|
6
|
-
* (drift), and how do all these stats look before vs after the
|
|
7
|
-
* `relinkCursor` watermark (the resumable `hicortex relink` migration to the
|
|
8
|
-
* corrected cosine formula, #145).
|
|
9
|
-
*/
|
|
10
|
-
import type Database from "better-sqlite3";
|
|
11
|
-
/**
|
|
12
|
-
* Deterministic sample of `rows` (#460): a seeded partial Fisher–Yates
|
|
13
|
-
* shuffles the LAST `size` slots from the back, and that shuffled tail is
|
|
14
|
-
* the sample. Rows must already be in a stable order (the caller sorts by
|
|
15
|
-
* PK) so the draw does not depend on SQLite's scan order.
|
|
16
|
-
*/
|
|
17
|
-
export declare function seededSample<T>(rows: T[], size: number, seed: number): T[];
|
|
18
|
-
export interface LinkRow {
|
|
19
|
-
source_id: string;
|
|
20
|
-
target_id: string;
|
|
21
|
-
relationship: string;
|
|
22
|
-
strength: number;
|
|
23
|
-
}
|
|
24
|
-
/** Cosine similarity between two L2-normalized embeddings (dot product). */
|
|
25
|
-
export declare function cosineBetween(a: Float32Array, b: Float32Array): number;
|
|
26
|
-
export declare function byRelationshipCounts(links: LinkRow[]): Record<string, number>;
|
|
27
|
-
/**
|
|
28
|
-
* Partition links by whether their SOURCE memory's rowid has been covered by
|
|
29
|
-
* the `hicortex relink` watermark. `relinkCursor` is the last fully
|
|
30
|
-
* committed rowid (relink.ts) — relink iterates memories by rowid and
|
|
31
|
-
* discovers/refreshes links FROM each one, so a link's source rowid <=
|
|
32
|
-
* cursor means a relink pass has already run for that source (current
|
|
33
|
-
* formula); null cursor (never run) puts everything in "notYetRelinked".
|
|
34
|
-
*/
|
|
35
|
-
export declare function partitionByRelinkCursor(links: LinkRow[], sourceRowid: Map<string, number>, relinkCursor: number | null): {
|
|
36
|
-
relinked: LinkRow[];
|
|
37
|
-
notYetRelinked: LinkRow[];
|
|
38
|
-
};
|
|
39
|
-
export interface HubEntry {
|
|
40
|
-
id: string;
|
|
41
|
-
degree: number;
|
|
42
|
-
project: string | null;
|
|
43
|
-
domain: string | null;
|
|
44
|
-
preview: string;
|
|
45
|
-
}
|
|
46
|
-
export interface DegreeReport {
|
|
47
|
-
memoriesWithLinks: number;
|
|
48
|
-
totalMemories: number;
|
|
49
|
-
degreeHistogram: Record<string, number>;
|
|
50
|
-
topHubs: HubEntry[];
|
|
51
|
-
}
|
|
52
|
-
export declare function runDegreeAudit(db: Database.Database, topN?: number): DegreeReport;
|
|
53
|
-
export interface DriftReport {
|
|
54
|
-
sampleSize: number;
|
|
55
|
-
driftHistogram: Record<string, number>;
|
|
56
|
-
meanAbsDrift: number;
|
|
57
|
-
maxAbsDrift: number;
|
|
58
|
-
skippedMissingEmbedding: number;
|
|
59
|
-
}
|
|
60
|
-
/**
|
|
61
|
-
* Recompute cosine for a link sample and compare to stored `strength` (post
|
|
62
|
-
* migration-5 rescale, should track closely). The sample is deterministic
|
|
63
|
-
* (#460): all link rows in stable PK order, drawn by a seeded Fisher–Yates —
|
|
64
|
-
* two runs on the same build + snapshot produce an identical report, so
|
|
65
|
-
* before/after comparisons can require full-output identity.
|
|
66
|
-
*/
|
|
67
|
-
export declare function runDriftSample(db: Database.Database, embeddings: Map<string, Float32Array>, sampleSize?: number, seed?: number): DriftReport;
|
|
68
|
-
export interface PartitionStats {
|
|
69
|
-
linkCount: number;
|
|
70
|
-
byRelationship: Record<string, number>;
|
|
71
|
-
dupNoiseLinks: number;
|
|
72
|
-
dupNoiseShare: number;
|
|
73
|
-
}
|
|
74
|
-
export interface GraphReport {
|
|
75
|
-
totalLinks: number;
|
|
76
|
-
byRelationship: Record<string, number>;
|
|
77
|
-
degree: DegreeReport;
|
|
78
|
-
drift: DriftReport;
|
|
79
|
-
dupNoise: {
|
|
80
|
-
threshold: number;
|
|
81
|
-
count: number;
|
|
82
|
-
share: number;
|
|
83
|
-
measured: number;
|
|
84
|
-
};
|
|
85
|
-
relinkCursor: number | null;
|
|
86
|
-
relinkedPartition: PartitionStats;
|
|
87
|
-
notYetRelinkedPartition: PartitionStats;
|
|
88
|
-
}
|
|
89
|
-
export declare function runGraphAudit(db: Database.Database, stateDir?: string): GraphReport;
|
package/dist/eval/graph-eval.js
DELETED
|
@@ -1,246 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
/**
|
|
3
|
-
* D6 — link-graph health audit (#191 mechanical baseline).
|
|
4
|
-
*
|
|
5
|
-
* Of the ~6.5k links: how many are meaningful vs near-duplicate noise, does
|
|
6
|
-
* the stored `memory_links.strength` still match a fresh cosine recompute
|
|
7
|
-
* (drift), and how do all these stats look before vs after the
|
|
8
|
-
* `relinkCursor` watermark (the resumable `hicortex relink` migration to the
|
|
9
|
-
* corrected cosine formula, #145).
|
|
10
|
-
*/
|
|
11
|
-
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
12
|
-
if (k2 === undefined) k2 = k;
|
|
13
|
-
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
14
|
-
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
15
|
-
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
16
|
-
}
|
|
17
|
-
Object.defineProperty(o, k2, desc);
|
|
18
|
-
}) : (function(o, m, k, k2) {
|
|
19
|
-
if (k2 === undefined) k2 = k;
|
|
20
|
-
o[k2] = m[k];
|
|
21
|
-
}));
|
|
22
|
-
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
23
|
-
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
24
|
-
}) : function(o, v) {
|
|
25
|
-
o["default"] = v;
|
|
26
|
-
});
|
|
27
|
-
var __importStar = (this && this.__importStar) || (function () {
|
|
28
|
-
var ownKeys = function(o) {
|
|
29
|
-
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
30
|
-
var ar = [];
|
|
31
|
-
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
32
|
-
return ar;
|
|
33
|
-
};
|
|
34
|
-
return ownKeys(o);
|
|
35
|
-
};
|
|
36
|
-
return function (mod) {
|
|
37
|
-
if (mod && mod.__esModule) return mod;
|
|
38
|
-
var result = {};
|
|
39
|
-
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
40
|
-
__setModuleDefault(result, mod);
|
|
41
|
-
return result;
|
|
42
|
-
};
|
|
43
|
-
})();
|
|
44
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
45
|
-
exports.seededSample = seededSample;
|
|
46
|
-
exports.cosineBetween = cosineBetween;
|
|
47
|
-
exports.byRelationshipCounts = byRelationshipCounts;
|
|
48
|
-
exports.partitionByRelinkCursor = partitionByRelinkCursor;
|
|
49
|
-
exports.runDegreeAudit = runDegreeAudit;
|
|
50
|
-
exports.runDriftSample = runDriftSample;
|
|
51
|
-
exports.runGraphAudit = runGraphAudit;
|
|
52
|
-
const eval_db_js_1 = require("./eval-db.js");
|
|
53
|
-
const storage = __importStar(require("../storage.js"));
|
|
54
|
-
const state_js_1 = require("../state.js");
|
|
55
|
-
const decay_eval_js_1 = require("./decay-eval.js");
|
|
56
|
-
/** Matches the middle D1 duplicate threshold — a link between near-dups is noise, not signal. */
|
|
57
|
-
const DUP_NOISE_THRESHOLD = 0.92;
|
|
58
|
-
const DRIFT_SAMPLE_SIZE = 500;
|
|
59
|
-
/**
|
|
60
|
-
* Seed for the drift sample's PRNG (#460). SQLite's `random()` cannot be
|
|
61
|
-
* seeded, so the draw happens client-side with a deterministic Fisher–Yates;
|
|
62
|
-
* this fixed seed makes the sample — and therefore the whole DriftReport —
|
|
63
|
-
* byte-identical across runs on the same build + snapshot, which lets
|
|
64
|
-
* before/after eval comparisons require full-output identity.
|
|
65
|
-
*/
|
|
66
|
-
const DRIFT_SAMPLE_SEED = 460;
|
|
67
|
-
const DEGREE_BUCKET_EDGES = [0, 1, 2, 3, 5, 10, 20, 50, 100];
|
|
68
|
-
const DRIFT_BUCKET_EDGES = [0, 0.01, 0.02, 0.05, 0.1, 0.2, 0.3, 0.5, 1.0];
|
|
69
|
-
/**
|
|
70
|
-
* mulberry32 — a 32-bit seeded PRNG (public-domain reference implementation).
|
|
71
|
-
* JS number ops are deterministic across platforms, so a fixed seed yields
|
|
72
|
-
* the same sequence everywhere.
|
|
73
|
-
*/
|
|
74
|
-
function mulberry32(seed) {
|
|
75
|
-
let a = seed >>> 0;
|
|
76
|
-
return () => {
|
|
77
|
-
a = (a + 0x6d2b79f5) | 0;
|
|
78
|
-
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
79
|
-
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
80
|
-
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
81
|
-
};
|
|
82
|
-
}
|
|
83
|
-
/**
|
|
84
|
-
* Deterministic sample of `rows` (#460): a seeded partial Fisher–Yates
|
|
85
|
-
* shuffles the LAST `size` slots from the back, and that shuffled tail is
|
|
86
|
-
* the sample. Rows must already be in a stable order (the caller sorts by
|
|
87
|
-
* PK) so the draw does not depend on SQLite's scan order.
|
|
88
|
-
*/
|
|
89
|
-
function seededSample(rows, size, seed) {
|
|
90
|
-
const rand = mulberry32(seed);
|
|
91
|
-
const arr = [...rows];
|
|
92
|
-
const n = Math.min(size, arr.length);
|
|
93
|
-
for (let i = arr.length - 1; i > arr.length - 1 - n; i--) {
|
|
94
|
-
const j = Math.floor(rand() * (i + 1));
|
|
95
|
-
[arr[i], arr[j]] = [arr[j], arr[i]];
|
|
96
|
-
}
|
|
97
|
-
return arr.slice(arr.length - n);
|
|
98
|
-
}
|
|
99
|
-
// ---------------------------------------------------------------------------
|
|
100
|
-
// Pure helpers (unit tested)
|
|
101
|
-
// ---------------------------------------------------------------------------
|
|
102
|
-
/** Cosine similarity between two L2-normalized embeddings (dot product). */
|
|
103
|
-
function cosineBetween(a, b) {
|
|
104
|
-
let dot = 0;
|
|
105
|
-
for (let i = 0; i < a.length; i++)
|
|
106
|
-
dot += a[i] * b[i];
|
|
107
|
-
return dot;
|
|
108
|
-
}
|
|
109
|
-
function byRelationshipCounts(links) {
|
|
110
|
-
const counts = {};
|
|
111
|
-
for (const l of links) {
|
|
112
|
-
counts[l.relationship] = (counts[l.relationship] ?? 0) + 1;
|
|
113
|
-
}
|
|
114
|
-
return counts;
|
|
115
|
-
}
|
|
116
|
-
/**
|
|
117
|
-
* Partition links by whether their SOURCE memory's rowid has been covered by
|
|
118
|
-
* the `hicortex relink` watermark. `relinkCursor` is the last fully
|
|
119
|
-
* committed rowid (relink.ts) — relink iterates memories by rowid and
|
|
120
|
-
* discovers/refreshes links FROM each one, so a link's source rowid <=
|
|
121
|
-
* cursor means a relink pass has already run for that source (current
|
|
122
|
-
* formula); null cursor (never run) puts everything in "notYetRelinked".
|
|
123
|
-
*/
|
|
124
|
-
function partitionByRelinkCursor(links, sourceRowid, relinkCursor) {
|
|
125
|
-
const relinked = [];
|
|
126
|
-
const notYetRelinked = [];
|
|
127
|
-
for (const l of links) {
|
|
128
|
-
const rowid = sourceRowid.get(l.source_id);
|
|
129
|
-
const covered = relinkCursor !== null && rowid !== undefined && rowid <= relinkCursor;
|
|
130
|
-
(covered ? relinked : notYetRelinked).push(l);
|
|
131
|
-
}
|
|
132
|
-
return { relinked, notYetRelinked };
|
|
133
|
-
}
|
|
134
|
-
function runDegreeAudit(db, topN = 10) {
|
|
135
|
-
const totalMemories = db.prepare("SELECT COUNT(*) AS c FROM memories").get().c;
|
|
136
|
-
const linkCounts = storage.getAllLinkCounts(db);
|
|
137
|
-
const degreeHistogram = (0, decay_eval_js_1.histogram)([...linkCounts.values()], DEGREE_BUCKET_EDGES);
|
|
138
|
-
const sortedIds = [...linkCounts.entries()].sort((a, b) => b[1] - a[1]).slice(0, topN);
|
|
139
|
-
const topHubs = sortedIds.map(([id, degree]) => {
|
|
140
|
-
const mem = storage.getMemory(db, id);
|
|
141
|
-
return {
|
|
142
|
-
id,
|
|
143
|
-
degree,
|
|
144
|
-
project: mem?.project ?? null,
|
|
145
|
-
domain: mem?.domain ?? null,
|
|
146
|
-
preview: (mem?.content ?? "").slice(0, 80),
|
|
147
|
-
};
|
|
148
|
-
});
|
|
149
|
-
return { memoriesWithLinks: linkCounts.size, totalMemories, degreeHistogram, topHubs };
|
|
150
|
-
}
|
|
151
|
-
/**
|
|
152
|
-
* Recompute cosine for a link sample and compare to stored `strength` (post
|
|
153
|
-
* migration-5 rescale, should track closely). The sample is deterministic
|
|
154
|
-
* (#460): all link rows in stable PK order, drawn by a seeded Fisher–Yates —
|
|
155
|
-
* two runs on the same build + snapshot produce an identical report, so
|
|
156
|
-
* before/after comparisons can require full-output identity.
|
|
157
|
-
*/
|
|
158
|
-
function runDriftSample(db, embeddings, sampleSize = DRIFT_SAMPLE_SIZE, seed = DRIFT_SAMPLE_SEED) {
|
|
159
|
-
const allLinks = db
|
|
160
|
-
.prepare("SELECT source_id, target_id, strength FROM memory_links ORDER BY source_id, target_id")
|
|
161
|
-
.all();
|
|
162
|
-
const sample = seededSample(allLinks, sampleSize, seed);
|
|
163
|
-
const drifts = [];
|
|
164
|
-
let skipped = 0;
|
|
165
|
-
for (const link of sample) {
|
|
166
|
-
const a = embeddings.get(link.source_id);
|
|
167
|
-
const b = embeddings.get(link.target_id);
|
|
168
|
-
if (!a || !b) {
|
|
169
|
-
skipped++;
|
|
170
|
-
continue;
|
|
171
|
-
}
|
|
172
|
-
const recomputed = cosineBetween(a, b);
|
|
173
|
-
drifts.push(Math.abs(recomputed - link.strength));
|
|
174
|
-
}
|
|
175
|
-
const meanAbsDrift = drifts.length > 0 ? drifts.reduce((s, d) => s + d, 0) / drifts.length : 0;
|
|
176
|
-
const maxAbsDrift = drifts.length > 0 ? Math.max(...drifts) : 0;
|
|
177
|
-
return {
|
|
178
|
-
sampleSize: drifts.length,
|
|
179
|
-
driftHistogram: (0, decay_eval_js_1.histogram)(drifts, DRIFT_BUCKET_EDGES),
|
|
180
|
-
meanAbsDrift,
|
|
181
|
-
maxAbsDrift,
|
|
182
|
-
skippedMissingEmbedding: skipped,
|
|
183
|
-
};
|
|
184
|
-
}
|
|
185
|
-
function statsFor(links, embeddings) {
|
|
186
|
-
let dupNoise = 0;
|
|
187
|
-
let measured = 0;
|
|
188
|
-
for (const l of links) {
|
|
189
|
-
const a = embeddings.get(l.source_id);
|
|
190
|
-
const b = embeddings.get(l.target_id);
|
|
191
|
-
if (!a || !b)
|
|
192
|
-
continue;
|
|
193
|
-
measured++;
|
|
194
|
-
if (cosineBetween(a, b) >= DUP_NOISE_THRESHOLD)
|
|
195
|
-
dupNoise++;
|
|
196
|
-
}
|
|
197
|
-
return {
|
|
198
|
-
linkCount: links.length,
|
|
199
|
-
byRelationship: byRelationshipCounts(links),
|
|
200
|
-
dupNoiseLinks: dupNoise,
|
|
201
|
-
dupNoiseShare: measured > 0 ? dupNoise / measured : 0,
|
|
202
|
-
};
|
|
203
|
-
}
|
|
204
|
-
/** Load every stored embedding once as id -> Float32Array. */
|
|
205
|
-
function loadEmbeddings(db) {
|
|
206
|
-
const rows = db.prepare("SELECT id, embedding FROM memory_vectors").all();
|
|
207
|
-
return new Map(rows.map((r) => [r.id, (0, eval_db_js_1.blobToEmbedding)(r.embedding)]));
|
|
208
|
-
}
|
|
209
|
-
function runGraphAudit(db, stateDir) {
|
|
210
|
-
const links = db
|
|
211
|
-
.prepare("SELECT source_id, target_id, relationship, strength FROM memory_links")
|
|
212
|
-
.all();
|
|
213
|
-
const embeddings = loadEmbeddings(db);
|
|
214
|
-
const degree = runDegreeAudit(db);
|
|
215
|
-
const drift = runDriftSample(db, embeddings);
|
|
216
|
-
let dupNoiseCount = 0;
|
|
217
|
-
let dupNoiseMeasured = 0;
|
|
218
|
-
for (const l of links) {
|
|
219
|
-
const a = embeddings.get(l.source_id);
|
|
220
|
-
const b = embeddings.get(l.target_id);
|
|
221
|
-
if (!a || !b)
|
|
222
|
-
continue;
|
|
223
|
-
dupNoiseMeasured++;
|
|
224
|
-
if (cosineBetween(a, b) >= DUP_NOISE_THRESHOLD)
|
|
225
|
-
dupNoiseCount++;
|
|
226
|
-
}
|
|
227
|
-
const relinkCursor = (0, state_js_1.loadState)(stateDir).relinkCursor ?? null;
|
|
228
|
-
const rowidRows = db.prepare("SELECT id, rowid AS rowid FROM memories").all();
|
|
229
|
-
const sourceRowid = new Map(rowidRows.map((r) => [r.id, r.rowid]));
|
|
230
|
-
const { relinked, notYetRelinked } = partitionByRelinkCursor(links, sourceRowid, relinkCursor);
|
|
231
|
-
return {
|
|
232
|
-
totalLinks: links.length,
|
|
233
|
-
byRelationship: byRelationshipCounts(links),
|
|
234
|
-
degree,
|
|
235
|
-
drift,
|
|
236
|
-
dupNoise: {
|
|
237
|
-
threshold: DUP_NOISE_THRESHOLD,
|
|
238
|
-
count: dupNoiseCount,
|
|
239
|
-
share: dupNoiseMeasured > 0 ? dupNoiseCount / dupNoiseMeasured : 0,
|
|
240
|
-
measured: dupNoiseMeasured,
|
|
241
|
-
},
|
|
242
|
-
relinkCursor,
|
|
243
|
-
relinkedPartition: statsFor(relinked, embeddings),
|
|
244
|
-
notYetRelinkedPartition: statsFor(notYetRelinked, embeddings),
|
|
245
|
-
};
|
|
246
|
-
}
|
|
@@ -1,85 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
/**
|
|
3
|
-
* Importance-scorer distribution eval (#425) — the AC4 measurement.
|
|
4
|
-
*
|
|
5
|
-
* npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
|
|
6
|
-
* [--llm-base-url U --llm-model M --llm-api-key K]
|
|
7
|
-
*
|
|
8
|
-
* Samples real memories from a READONLY snapshot (openSnapshot — never
|
|
9
|
-
* initDb), sends them through the PRODUCTION scoring path — the real
|
|
10
|
-
* `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
|
|
11
|
-
* call exactly like the nightly's stageImportance — and reports the score
|
|
12
|
-
* distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
|
|
13
|
-
* at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
|
|
14
|
-
* median 0.30-0.40, p90 <= 0.75, zero at 1.0).
|
|
15
|
-
*
|
|
16
|
-
* This is the before/after photo for the scorer recalibration: run it on
|
|
17
|
-
* the OLD prompt (the inflated-distribution red photo), then on the
|
|
18
|
-
* re-anchored prompt (AC4 gate).
|
|
19
|
-
*
|
|
20
|
-
* Determinism: the sample is stratified across created_at deciles and
|
|
21
|
-
* drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
|
|
22
|
-
* same sample + same batches + same prompts. Calls are SERIAL (one
|
|
23
|
-
* in-flight request, like the nightly).
|
|
24
|
-
*
|
|
25
|
-
* Honesty rules: no stub LLM, no fallback distribution, measured scores
|
|
26
|
-
* only; a mid-run endpoint error aborts with the partial histogram
|
|
27
|
-
* printed (never silently padded). The gateway endpoint is passed at RUN
|
|
28
|
-
* TIME via flags — nothing about any specific endpoint lives in this file.
|
|
29
|
-
*/
|
|
30
|
-
import { LlmClient } from "../llm.js";
|
|
31
|
-
export interface SampleRow {
|
|
32
|
-
id: string;
|
|
33
|
-
content: string;
|
|
34
|
-
created_at: string;
|
|
35
|
-
}
|
|
36
|
-
/**
|
|
37
|
-
* Deterministic decile-stratified sample: rows sorted by created_at, split
|
|
38
|
-
* into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
|
|
39
|
-
* (each decile shuffled by its own PRNG stream — index-independent). Same
|
|
40
|
-
* (rows, n, seed) always yields the same ordered selection.
|
|
41
|
-
*/
|
|
42
|
-
export declare function stratifiedSample(rows: SampleRow[], n: number, seed: number): SampleRow[];
|
|
43
|
-
export interface ScoreSummary {
|
|
44
|
-
n: number;
|
|
45
|
-
min: number;
|
|
46
|
-
p25: number;
|
|
47
|
-
median: number;
|
|
48
|
-
p75: number;
|
|
49
|
-
p90: number;
|
|
50
|
-
max: number;
|
|
51
|
-
atExactly1: number;
|
|
52
|
-
histogram: Record<string, number>;
|
|
53
|
-
}
|
|
54
|
-
export declare function summarizeScores(scores: number[]): ScoreSummary;
|
|
55
|
-
export interface ImportanceEvalOptions {
|
|
56
|
-
snapshotPath: string;
|
|
57
|
-
sample?: number;
|
|
58
|
-
seed?: number;
|
|
59
|
-
llm: LlmClient;
|
|
60
|
-
/** Where to write the report (default data/importance-eval-report.md). */
|
|
61
|
-
reportPath?: string;
|
|
62
|
-
}
|
|
63
|
-
export interface ImportanceEvalResult {
|
|
64
|
-
summary: ScoreSummary;
|
|
65
|
-
/** Per-batch measured scores, in call order (audit trail). */
|
|
66
|
-
batches: Array<{
|
|
67
|
-
sent: number;
|
|
68
|
-
scores: number[];
|
|
69
|
-
}>;
|
|
70
|
-
aborted: boolean;
|
|
71
|
-
error?: string;
|
|
72
|
-
}
|
|
73
|
-
/**
|
|
74
|
-
* Run the importance distribution measurement. Throws only on setup errors;
|
|
75
|
-
* an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
|
|
76
|
-
* the caller decides exit code, always loudly).
|
|
77
|
-
*/
|
|
78
|
-
export declare function runImportanceEval(opts: ImportanceEvalOptions): Promise<ImportanceEvalResult>;
|
|
79
|
-
export declare function renderImportanceReport(args: {
|
|
80
|
-
snapshotPath: string;
|
|
81
|
-
summary: ScoreSummary;
|
|
82
|
-
aborted: boolean;
|
|
83
|
-
error?: string;
|
|
84
|
-
model: string;
|
|
85
|
-
}): string;
|