@gamaze/hicortex 0.20.9 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -0
- package/assets/dashboard.html +4174 -835
- package/dist/calibration.d.ts +119 -0
- package/dist/calibration.js +149 -1
- package/dist/capture-health.d.ts +87 -0
- package/dist/capture-health.js +106 -0
- package/dist/capture-pause.d.ts +86 -0
- package/dist/capture-pause.js +127 -0
- package/dist/capture.d.ts +9 -0
- package/dist/capture.js +2 -1
- package/dist/cli.js +36 -0
- package/dist/consolidate.d.ts +35 -0
- package/dist/consolidate.js +85 -9
- package/dist/dashboard.d.ts +322 -3
- package/dist/dashboard.js +592 -7
- package/dist/db.js +105 -0
- package/dist/eval/importance-eval.d.ts +85 -0
- package/dist/eval/importance-eval.js +286 -0
- package/dist/eval/planted-fixtures.d.ts +1 -1
- package/dist/eval/ranking-battery.d.ts +78 -0
- package/dist/eval/ranking-battery.js +181 -0
- package/dist/eval/ranking-eval.d.ts +41 -0
- package/dist/eval/ranking-eval.js +391 -0
- package/dist/eval/ranking-fixtures.d.ts +77 -0
- package/dist/eval/ranking-fixtures.js +226 -0
- package/dist/identity-store.d.ts +21 -0
- package/dist/identity-store.js +49 -0
- package/dist/init.d.ts +14 -0
- package/dist/init.js +32 -0
- package/dist/mcp-server.d.ts +12 -0
- package/dist/mcp-server.js +184 -3
- package/dist/nightly.d.ts +9 -1
- package/dist/nightly.js +59 -7
- package/dist/prompts.d.ts +10 -0
- package/dist/prompts.js +28 -5
- package/dist/reconsolidation.d.ts +59 -30
- package/dist/reconsolidation.js +526 -296
- package/dist/rescore-importance.d.ts +80 -0
- package/dist/rescore-importance.js +236 -0
- package/dist/retrieval.d.ts +12 -0
- package/dist/retrieval.js +30 -1
- package/dist/stages.d.ts +37 -0
- package/dist/stages.js +51 -0
- package/dist/state.d.ts +32 -6
- package/dist/storage.d.ts +34 -2
- package/dist/storage.js +63 -6
- package/dist/types.d.ts +48 -0
- package/package.json +3 -1
package/dist/db.js
CHANGED
|
@@ -547,6 +547,111 @@ const MIGRATIONS = [
|
|
|
547
547
|
db.exec("CREATE INDEX IF NOT EXISTS idx_memory_history_memory ON memory_history(memory_id)");
|
|
548
548
|
},
|
|
549
549
|
},
|
|
550
|
+
{
|
|
551
|
+
version: 15,
|
|
552
|
+
name: "source_machine",
|
|
553
|
+
up: (db) => {
|
|
554
|
+
// #421 owner direction (machine × harness identity): nullable, NO
|
|
555
|
+
// backfill — honesty over guesses. Rows written before this column
|
|
556
|
+
// stay NULL and group under "earlier captures" in the console; that
|
|
557
|
+
// bucket shrinks as nights accumulate stamped captures. Stamped by the
|
|
558
|
+
// nightly capture path (config `machineName` ?? os.hostname()) and
|
|
559
|
+
// accepted optionally by /distill + /ingest. Idempotent via hasColumn.
|
|
560
|
+
if (!hasColumn(db, "memories", "source_machine")) {
|
|
561
|
+
db.exec("ALTER TABLE memories ADD COLUMN source_machine TEXT");
|
|
562
|
+
}
|
|
563
|
+
},
|
|
564
|
+
},
|
|
565
|
+
{
|
|
566
|
+
version: 16,
|
|
567
|
+
name: "distill_activity",
|
|
568
|
+
up: (db) => {
|
|
569
|
+
// #422 Phase 2 — /distill capture-health accounting. One row per POST
|
|
570
|
+
// (every outcome incl. held/skipped), written by
|
|
571
|
+
// capture-health.ts:recordDistillActivity from the /distill handler's
|
|
572
|
+
// exits. This is OPERATIONS telemetry, not memory data: rows prune
|
|
573
|
+
// after 7 days (in-module, once per process per UTC day), so the table
|
|
574
|
+
// stays bounded while giving the console's capture-health card a real
|
|
575
|
+
// posts/sessions/bytes/held picture per machine × agent. `retried` is
|
|
576
|
+
// computed at insert (an earlier row with the same session_id +
|
|
577
|
+
// segment_id means this POST is a client retry of a cursor-held
|
|
578
|
+
// segment) — never updated afterwards. Sidecar table (no memories FK):
|
|
579
|
+
// the memory rows of a failed POST may never exist; the activity row
|
|
580
|
+
// must record the attempt anyway. Idempotent: IF NOT EXISTS everywhere.
|
|
581
|
+
db.exec(`
|
|
582
|
+
CREATE TABLE IF NOT EXISTS distill_activity (
|
|
583
|
+
ts TEXT NOT NULL,
|
|
584
|
+
day TEXT NOT NULL,
|
|
585
|
+
machine TEXT NOT NULL DEFAULT '',
|
|
586
|
+
agent TEXT NOT NULL DEFAULT '',
|
|
587
|
+
session_id TEXT,
|
|
588
|
+
segment_id TEXT,
|
|
589
|
+
bytes INTEGER NOT NULL DEFAULT 0,
|
|
590
|
+
outcome TEXT NOT NULL,
|
|
591
|
+
retried INTEGER NOT NULL DEFAULT 0
|
|
592
|
+
)
|
|
593
|
+
`);
|
|
594
|
+
db.exec("CREATE INDEX IF NOT EXISTS idx_distill_activity_day ON distill_activity(day)");
|
|
595
|
+
db.exec("CREATE INDEX IF NOT EXISTS idx_distill_activity_session_segment ON distill_activity(session_id, segment_id)");
|
|
596
|
+
},
|
|
597
|
+
},
|
|
598
|
+
{
|
|
599
|
+
version: 17,
|
|
600
|
+
name: "memory_corroboration",
|
|
601
|
+
up: (db) => {
|
|
602
|
+
// #423 phase 3 — explicit owner corroboration trail (POST /enrich).
|
|
603
|
+
// DEFAULT 0 with NO backfill: a pre-v17 row was never enriched and 0 is
|
|
604
|
+
// the honest count. /memory rides it via SELECT * (the console detail's
|
|
605
|
+
// "corroborated × N"); the enrich itself writes base_strength — it
|
|
606
|
+
// never fakes access/shown counts (those are the recall-adoption
|
|
607
|
+
// signal). Idempotent via hasColumn (the v15 pattern).
|
|
608
|
+
if (!hasColumn(db, "memories", "corroboration_count")) {
|
|
609
|
+
db.exec("ALTER TABLE memories ADD COLUMN corroboration_count INTEGER NOT NULL DEFAULT 0");
|
|
610
|
+
}
|
|
611
|
+
},
|
|
612
|
+
},
|
|
613
|
+
{
|
|
614
|
+
version: 18,
|
|
615
|
+
name: "capture_pauses",
|
|
616
|
+
up: (db) => {
|
|
617
|
+
// #423 phase 3, D3 — server-side 200-skip for a paused machine ×
|
|
618
|
+
// harness bundle. A row EXISTS = paused; absence = capturing (no
|
|
619
|
+
// "paused" flag to keep honest). Deliberately NO retention/pruning: the
|
|
620
|
+
// rows are few and operator-owned, and pruning one would silently
|
|
621
|
+
// resume capture the operator meant to hold. Idempotent: IF NOT EXISTS.
|
|
622
|
+
db.exec(`
|
|
623
|
+
CREATE TABLE IF NOT EXISTS capture_pauses (
|
|
624
|
+
machine TEXT NOT NULL DEFAULT '',
|
|
625
|
+
harness TEXT NOT NULL,
|
|
626
|
+
paused_at TEXT NOT NULL,
|
|
627
|
+
PRIMARY KEY(machine, harness)
|
|
628
|
+
)
|
|
629
|
+
`);
|
|
630
|
+
},
|
|
631
|
+
},
|
|
632
|
+
{
|
|
633
|
+
version: 19,
|
|
634
|
+
name: "add_importance_scored_at",
|
|
635
|
+
up: (db) => {
|
|
636
|
+
// #425 — scored-at watermark for importance scoring. getUnscoredMemories
|
|
637
|
+
// keys on importance_scored_at IS NULL (v19+), replacing the old
|
|
638
|
+
// base_strength = 0.5 sentinel, which re-rolled every row the model
|
|
639
|
+
// genuinely scored 0.5 every night. Backfill: existing rows that are NOT
|
|
640
|
+
// at the 0.5 sentinel are stamped "settled" (COALESCE(updated_at,
|
|
641
|
+
// ingested_at, created_at)) — they carry a real historical score and
|
|
642
|
+
// leave the nightly pool. Rows AT the sentinel stay NULL so the next
|
|
643
|
+
// nightly scores them ONCE under the new rubric (bounded — the watermark
|
|
644
|
+
// write in stageImportance then takes them out of the pool). The
|
|
645
|
+
// rescore-importance backfill CLI re-judges settled rows wholesale under
|
|
646
|
+
// its own cursor; this migration only makes the NIGHTLY pool honest.
|
|
647
|
+
// Idempotent via hasColumn (the v8 pattern).
|
|
648
|
+
if (!hasColumn(db, "memories", "importance_scored_at")) {
|
|
649
|
+
db.exec("ALTER TABLE memories ADD COLUMN importance_scored_at TEXT");
|
|
650
|
+
}
|
|
651
|
+
db.exec(`UPDATE memories SET importance_scored_at = COALESCE(updated_at, ingested_at, created_at)
|
|
652
|
+
WHERE importance_scored_at IS NULL AND base_strength != 0.5`);
|
|
653
|
+
},
|
|
654
|
+
},
|
|
550
655
|
];
|
|
551
656
|
/**
|
|
552
657
|
* Run all pending migrations against the database.
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Importance-scorer distribution eval (#425) — the AC4 measurement.
|
|
4
|
+
*
|
|
5
|
+
* npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
|
|
6
|
+
* [--llm-base-url U --llm-model M --llm-api-key K]
|
|
7
|
+
*
|
|
8
|
+
* Samples real memories from a READONLY snapshot (openSnapshot — never
|
|
9
|
+
* initDb), sends them through the PRODUCTION scoring path — the real
|
|
10
|
+
* `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
|
|
11
|
+
* call exactly like the nightly's stageImportance — and reports the score
|
|
12
|
+
* distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
|
|
13
|
+
* at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
|
|
14
|
+
* median 0.30-0.40, p90 <= 0.75, zero at 1.0).
|
|
15
|
+
*
|
|
16
|
+
* This is the before/after photo for the scorer recalibration: run it on
|
|
17
|
+
* the OLD prompt (the inflated-distribution red photo), then on the
|
|
18
|
+
* re-anchored prompt (AC4 gate).
|
|
19
|
+
*
|
|
20
|
+
* Determinism: the sample is stratified across created_at deciles and
|
|
21
|
+
* drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
|
|
22
|
+
* same sample + same batches + same prompts. Calls are SERIAL (one
|
|
23
|
+
* in-flight request, like the nightly).
|
|
24
|
+
*
|
|
25
|
+
* Honesty rules: no stub LLM, no fallback distribution, measured scores
|
|
26
|
+
* only; a mid-run endpoint error aborts with the partial histogram
|
|
27
|
+
* printed (never silently padded). The gateway endpoint is passed at RUN
|
|
28
|
+
* TIME via flags — nothing about any specific endpoint lives in this file.
|
|
29
|
+
*/
|
|
30
|
+
import { LlmClient } from "../llm.js";
|
|
31
|
+
export interface SampleRow {
|
|
32
|
+
id: string;
|
|
33
|
+
content: string;
|
|
34
|
+
created_at: string;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Deterministic decile-stratified sample: rows sorted by created_at, split
|
|
38
|
+
* into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
|
|
39
|
+
* (each decile shuffled by its own PRNG stream — index-independent). Same
|
|
40
|
+
* (rows, n, seed) always yields the same ordered selection.
|
|
41
|
+
*/
|
|
42
|
+
export declare function stratifiedSample(rows: SampleRow[], n: number, seed: number): SampleRow[];
|
|
43
|
+
export interface ScoreSummary {
|
|
44
|
+
n: number;
|
|
45
|
+
min: number;
|
|
46
|
+
p25: number;
|
|
47
|
+
median: number;
|
|
48
|
+
p75: number;
|
|
49
|
+
p90: number;
|
|
50
|
+
max: number;
|
|
51
|
+
atExactly1: number;
|
|
52
|
+
histogram: Record<string, number>;
|
|
53
|
+
}
|
|
54
|
+
export declare function summarizeScores(scores: number[]): ScoreSummary;
|
|
55
|
+
export interface ImportanceEvalOptions {
|
|
56
|
+
snapshotPath: string;
|
|
57
|
+
sample?: number;
|
|
58
|
+
seed?: number;
|
|
59
|
+
llm: LlmClient;
|
|
60
|
+
/** Where to write the report (default data/importance-eval-report.md). */
|
|
61
|
+
reportPath?: string;
|
|
62
|
+
}
|
|
63
|
+
export interface ImportanceEvalResult {
|
|
64
|
+
summary: ScoreSummary;
|
|
65
|
+
/** Per-batch measured scores, in call order (audit trail). */
|
|
66
|
+
batches: Array<{
|
|
67
|
+
sent: number;
|
|
68
|
+
scores: number[];
|
|
69
|
+
}>;
|
|
70
|
+
aborted: boolean;
|
|
71
|
+
error?: string;
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Run the importance distribution measurement. Throws only on setup errors;
|
|
75
|
+
* an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
|
|
76
|
+
* the caller decides exit code, always loudly).
|
|
77
|
+
*/
|
|
78
|
+
export declare function runImportanceEval(opts: ImportanceEvalOptions): Promise<ImportanceEvalResult>;
|
|
79
|
+
export declare function renderImportanceReport(args: {
|
|
80
|
+
snapshotPath: string;
|
|
81
|
+
summary: ScoreSummary;
|
|
82
|
+
aborted: boolean;
|
|
83
|
+
error?: string;
|
|
84
|
+
model: string;
|
|
85
|
+
}): string;
|
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
"use strict";
|
|
3
|
+
/**
|
|
4
|
+
* Importance-scorer distribution eval (#425) — the AC4 measurement.
|
|
5
|
+
*
|
|
6
|
+
* npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
|
|
7
|
+
* [--llm-base-url U --llm-model M --llm-api-key K]
|
|
8
|
+
*
|
|
9
|
+
* Samples real memories from a READONLY snapshot (openSnapshot — never
|
|
10
|
+
* initDb), sends them through the PRODUCTION scoring path — the real
|
|
11
|
+
* `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
|
|
12
|
+
* call exactly like the nightly's stageImportance — and reports the score
|
|
13
|
+
* distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
|
|
14
|
+
* at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
|
|
15
|
+
* median 0.30-0.40, p90 <= 0.75, zero at 1.0).
|
|
16
|
+
*
|
|
17
|
+
* This is the before/after photo for the scorer recalibration: run it on
|
|
18
|
+
* the OLD prompt (the inflated-distribution red photo), then on the
|
|
19
|
+
* re-anchored prompt (AC4 gate).
|
|
20
|
+
*
|
|
21
|
+
* Determinism: the sample is stratified across created_at deciles and
|
|
22
|
+
* drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
|
|
23
|
+
* same sample + same batches + same prompts. Calls are SERIAL (one
|
|
24
|
+
* in-flight request, like the nightly).
|
|
25
|
+
*
|
|
26
|
+
* Honesty rules: no stub LLM, no fallback distribution, measured scores
|
|
27
|
+
* only; a mid-run endpoint error aborts with the partial histogram
|
|
28
|
+
* printed (never silently padded). The gateway endpoint is passed at RUN
|
|
29
|
+
* TIME via flags — nothing about any specific endpoint lives in this file.
|
|
30
|
+
*/
|
|
31
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
32
|
+
exports.stratifiedSample = stratifiedSample;
|
|
33
|
+
exports.summarizeScores = summarizeScores;
|
|
34
|
+
exports.runImportanceEval = runImportanceEval;
|
|
35
|
+
exports.renderImportanceReport = renderImportanceReport;
|
|
36
|
+
const node_fs_1 = require("node:fs");
|
|
37
|
+
const node_path_1 = require("node:path");
|
|
38
|
+
const eval_db_js_1 = require("./eval-db.js");
|
|
39
|
+
const llm_js_1 = require("../llm.js");
|
|
40
|
+
const consolidate_js_1 = require("../consolidate.js");
|
|
41
|
+
const prompts_js_1 = require("../prompts.js");
|
|
42
|
+
const DEFAULT_SAMPLE = 200;
|
|
43
|
+
const DEFAULT_SEED = 1;
|
|
44
|
+
/** The nightly's stageImportance batch size — matched exactly. */
|
|
45
|
+
const BATCH_SIZE = 10;
|
|
46
|
+
// ---------------------------------------------------------------------------
|
|
47
|
+
// Deterministic stratified sampler
|
|
48
|
+
// ---------------------------------------------------------------------------
|
|
49
|
+
/** mulberry32 — small, seedable, deterministic across Node versions. */
|
|
50
|
+
function mulberry32(seed) {
|
|
51
|
+
let a = seed >>> 0;
|
|
52
|
+
return () => {
|
|
53
|
+
a |= 0;
|
|
54
|
+
a = (a + 0x6d2b79f5) | 0;
|
|
55
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
56
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
57
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* Deterministic decile-stratified sample: rows sorted by created_at, split
|
|
62
|
+
* into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
|
|
63
|
+
* (each decile shuffled by its own PRNG stream — index-independent). Same
|
|
64
|
+
* (rows, n, seed) always yields the same ordered selection.
|
|
65
|
+
*/
|
|
66
|
+
function stratifiedSample(rows, n, seed) {
|
|
67
|
+
if (rows.length === 0 || n <= 0)
|
|
68
|
+
return [];
|
|
69
|
+
const sorted = [...rows].sort((a, b) => a.created_at === b.created_at
|
|
70
|
+
? a.id.localeCompare(b.id)
|
|
71
|
+
: a.created_at.localeCompare(b.created_at));
|
|
72
|
+
const decileSize = sorted.length / 10;
|
|
73
|
+
const perDecile = Math.max(1, Math.ceil(n / 10));
|
|
74
|
+
const out = [];
|
|
75
|
+
for (let d = 0; d < 10 && out.length < n; d++) {
|
|
76
|
+
const start = Math.floor(d * decileSize);
|
|
77
|
+
const end = d === 9 ? sorted.length : Math.floor((d + 1) * decileSize);
|
|
78
|
+
const decile = sorted.slice(start, end);
|
|
79
|
+
const rnd = mulberry32(seed * 31 + d);
|
|
80
|
+
// Fisher-Yates with the decile-local stream.
|
|
81
|
+
for (let i = decile.length - 1; i > 0; i--) {
|
|
82
|
+
const j = Math.floor(rnd() * (i + 1));
|
|
83
|
+
[decile[i], decile[j]] = [decile[j], decile[i]];
|
|
84
|
+
}
|
|
85
|
+
out.push(...decile.slice(0, Math.min(perDecile, n - out.length)));
|
|
86
|
+
}
|
|
87
|
+
return out;
|
|
88
|
+
}
|
|
89
|
+
/** Nearest-rank percentile on the ASCENDING-sorted values. */
|
|
90
|
+
function percentile(sorted, p) {
|
|
91
|
+
if (sorted.length === 0)
|
|
92
|
+
return Number.NaN;
|
|
93
|
+
const idx = Math.min(sorted.length - 1, Math.max(0, Math.ceil((p / 100) * sorted.length) - 1));
|
|
94
|
+
return sorted[idx];
|
|
95
|
+
}
|
|
96
|
+
function summarizeScores(scores) {
|
|
97
|
+
const sorted = [...scores].sort((a, b) => a - b);
|
|
98
|
+
const histogram = {};
|
|
99
|
+
for (let b = 0; b < 10; b++)
|
|
100
|
+
histogram[`${b}-${b + 1}`] = 0;
|
|
101
|
+
for (const s of scores) {
|
|
102
|
+
const b = Math.min(9, Math.max(0, Math.floor(s * 10)));
|
|
103
|
+
histogram[`${b}-${b + 1}`]++;
|
|
104
|
+
}
|
|
105
|
+
return {
|
|
106
|
+
n: scores.length,
|
|
107
|
+
min: percentile(sorted, 0),
|
|
108
|
+
p25: percentile(sorted, 25),
|
|
109
|
+
median: percentile(sorted, 50),
|
|
110
|
+
p75: percentile(sorted, 75),
|
|
111
|
+
p90: percentile(sorted, 90),
|
|
112
|
+
max: percentile(sorted, 100),
|
|
113
|
+
atExactly1: scores.filter((s) => s === 1).length,
|
|
114
|
+
histogram,
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Run the importance distribution measurement. Throws only on setup errors;
|
|
119
|
+
* an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
|
|
120
|
+
* the caller decides exit code, always loudly).
|
|
121
|
+
*/
|
|
122
|
+
async function runImportanceEval(opts) {
|
|
123
|
+
const sampleSize = opts.sample ?? DEFAULT_SAMPLE;
|
|
124
|
+
const seed = opts.seed ?? DEFAULT_SEED;
|
|
125
|
+
const db = (0, eval_db_js_1.openSnapshot)(opts.snapshotPath);
|
|
126
|
+
let rows;
|
|
127
|
+
try {
|
|
128
|
+
rows = db
|
|
129
|
+
.prepare("SELECT id, content, created_at FROM memories WHERE COALESCE(status, '') != 'absorbed' ORDER BY id")
|
|
130
|
+
.all();
|
|
131
|
+
}
|
|
132
|
+
finally {
|
|
133
|
+
db.close();
|
|
134
|
+
}
|
|
135
|
+
if (rows.length === 0)
|
|
136
|
+
throw new Error("snapshot has no live memories");
|
|
137
|
+
const sample = stratifiedSample(rows, sampleSize, seed);
|
|
138
|
+
console.log(`[importance-eval] snapshot rows ${rows.length}, sampled ${sample.length} ` +
|
|
139
|
+
`(seed ${seed}, stratified over created_at deciles)`);
|
|
140
|
+
const batches = [];
|
|
141
|
+
const allScores = [];
|
|
142
|
+
for (let i = 0; i < sample.length; i += BATCH_SIZE) {
|
|
143
|
+
const batch = sample.slice(i, i + BATCH_SIZE);
|
|
144
|
+
const lines = batch.map((mem, idx) => `[${idx}] ${mem.content.slice(0, 500)}`);
|
|
145
|
+
const prompt = (0, prompts_js_1.importanceScoring)(lines.join("\n\n"));
|
|
146
|
+
try {
|
|
147
|
+
const r = await opts.llm.complete(prompt);
|
|
148
|
+
let scores = (0, consolidate_js_1.parseJsonLenient)(r.text, null);
|
|
149
|
+
if (!Array.isArray(scores))
|
|
150
|
+
scores = new Array(batch.length).fill(0.5);
|
|
151
|
+
while (scores.length < batch.length)
|
|
152
|
+
scores.push(0.5);
|
|
153
|
+
scores = scores.slice(0, batch.length);
|
|
154
|
+
const clamped = scores.map((s) => {
|
|
155
|
+
const v = Number(s);
|
|
156
|
+
return Number.isFinite(v) ? Math.max(0, Math.min(1, v)) : 0.5;
|
|
157
|
+
});
|
|
158
|
+
batches.push({ sent: batch.length, scores: clamped });
|
|
159
|
+
allScores.push(...clamped);
|
|
160
|
+
console.log(`[importance-eval] batch ${batches.length}: ${clamped.map((s) => s.toFixed(1)).join(", ")}`);
|
|
161
|
+
}
|
|
162
|
+
catch (err) {
|
|
163
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
164
|
+
console.error(`[importance-eval] endpoint error at batch ${Math.floor(i / BATCH_SIZE) + 1}: ${msg}`);
|
|
165
|
+
const summary = summarizeScores(allScores);
|
|
166
|
+
return { summary, batches, aborted: true, error: msg };
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
return { summary: summarizeScores(allScores), batches, aborted: false };
|
|
170
|
+
}
|
|
171
|
+
function renderImportanceReport(args) {
|
|
172
|
+
const s = args.summary;
|
|
173
|
+
const L = [];
|
|
174
|
+
L.push("# Importance-scorer distribution (#425 — AC4)\n");
|
|
175
|
+
L.push(`Snapshot: \`${args.snapshotPath}\` \nGenerated: ${new Date().toISOString()} \n` +
|
|
176
|
+
`Model: ${args.model} (the configured production endpoint) \n` +
|
|
177
|
+
`Prompt: the CURRENT production importanceScoring (whatever ships in this tree)\n`);
|
|
178
|
+
if (args.aborted) {
|
|
179
|
+
L.push(`**ABORTED MID-RUN** — ${args.error ?? "endpoint error"}; partial distribution over ${s.n} scores below.\n`);
|
|
180
|
+
}
|
|
181
|
+
L.push("## Distribution\n");
|
|
182
|
+
L.push("| stat | value |");
|
|
183
|
+
L.push("|---|---|");
|
|
184
|
+
L.push(`| n | ${s.n} |`);
|
|
185
|
+
L.push(`| min | ${s.min.toFixed(3)} |`);
|
|
186
|
+
L.push(`| p25 | ${s.p25.toFixed(3)} |`);
|
|
187
|
+
L.push(`| **median** | **${s.median.toFixed(3)}** |`);
|
|
188
|
+
L.push(`| p75 | ${s.p75.toFixed(3)} |`);
|
|
189
|
+
L.push(`| **p90** | **${s.p90.toFixed(3)}** |`);
|
|
190
|
+
L.push(`| max | ${s.max.toFixed(3)} |`);
|
|
191
|
+
L.push(`| at exactly 1.0 | ${s.atExactly1} |`);
|
|
192
|
+
L.push("");
|
|
193
|
+
L.push("| bucket | count |");
|
|
194
|
+
L.push("|---|---|");
|
|
195
|
+
for (const [bucket, count] of Object.entries(s.histogram)) {
|
|
196
|
+
L.push(`| ${bucket} | ${count} |`);
|
|
197
|
+
}
|
|
198
|
+
L.push("");
|
|
199
|
+
L.push(`Target band (owner D2, 2026-09-13): median 0.30-0.40 → ` +
|
|
200
|
+
`${s.median >= 0.3 && s.median <= 0.4 ? "**PASS**" : "**MISS**"} (${s.median.toFixed(3)}); ` +
|
|
201
|
+
`p90 <= 0.75 → ${s.p90 <= 0.75 ? "**PASS**" : "**MISS**"} (${s.p90.toFixed(3)}); ` +
|
|
202
|
+
`zero at 1.0 → ${s.atExactly1 === 0 ? "**PASS**" : "**MISS**"} (${s.atExactly1})\n`);
|
|
203
|
+
return L.join("\n");
|
|
204
|
+
}
|
|
205
|
+
// ---------------------------------------------------------------------------
|
|
206
|
+
// CLI
|
|
207
|
+
// ---------------------------------------------------------------------------
|
|
208
|
+
function usage() {
|
|
209
|
+
console.error("Usage: npm run eval:importance -- <snapshot.db> [--sample N=200] [--seed S=1]\n" +
|
|
210
|
+
" --llm-base-url U --llm-model M --llm-api-key K (required: the scoring endpoint)\n" +
|
|
211
|
+
" [--llm-thinking] (thinking ON; default OFF — production parity)\n" +
|
|
212
|
+
" [--report <out.md>]");
|
|
213
|
+
process.exit(1);
|
|
214
|
+
}
|
|
215
|
+
async function main() {
|
|
216
|
+
const argv = process.argv.slice(2);
|
|
217
|
+
const positional = [];
|
|
218
|
+
const flag = (name) => {
|
|
219
|
+
const i = argv.indexOf(name);
|
|
220
|
+
return i !== -1 && argv[i + 1] && !argv[i + 1].startsWith("--") ? argv[i + 1] : undefined;
|
|
221
|
+
};
|
|
222
|
+
for (let i = 0; i < argv.length; i++) {
|
|
223
|
+
if (!argv[i].startsWith("--"))
|
|
224
|
+
positional.push(argv[i]);
|
|
225
|
+
}
|
|
226
|
+
const snapshotPath = positional[0];
|
|
227
|
+
if (!snapshotPath || argv.includes("--help") || argv.includes("-h"))
|
|
228
|
+
usage();
|
|
229
|
+
const baseUrl = flag("--llm-base-url");
|
|
230
|
+
const model = flag("--llm-model");
|
|
231
|
+
const apiKey = flag("--llm-api-key");
|
|
232
|
+
if (!baseUrl || !model || !apiKey) {
|
|
233
|
+
console.error("[importance-eval] --llm-base-url, --llm-model and --llm-api-key are required (the production scoring endpoint)");
|
|
234
|
+
usage();
|
|
235
|
+
}
|
|
236
|
+
const sample = flag("--sample") ? parseInt(flag("--sample"), 10) : DEFAULT_SAMPLE;
|
|
237
|
+
const seed = flag("--seed") ? parseInt(flag("--seed"), 10) : DEFAULT_SEED;
|
|
238
|
+
if (!Number.isInteger(sample) || sample < 10 || sample > 5000) {
|
|
239
|
+
console.error("[importance-eval] --sample must be an integer in [10, 5000]");
|
|
240
|
+
process.exit(1);
|
|
241
|
+
}
|
|
242
|
+
if (!Number.isInteger(seed) || seed <= 0) {
|
|
243
|
+
console.error("[importance-eval] --seed must be a positive integer");
|
|
244
|
+
process.exit(1);
|
|
245
|
+
}
|
|
246
|
+
// OpenAI-compatible config built from flags ONLY (never hardcoded; the
|
|
247
|
+
// production endpoint is a runtime input, not a shipped value). Thinking
|
|
248
|
+
// defaults OFF (production parity: the scorer runs with thinking disabled —
|
|
249
|
+
// an unclosed think block can eat the whole output budget); opt in with
|
|
250
|
+
// --llm-thinking for a thinking-on endpoint.
|
|
251
|
+
const config = {
|
|
252
|
+
baseUrl,
|
|
253
|
+
model,
|
|
254
|
+
apiKey,
|
|
255
|
+
provider: "openai",
|
|
256
|
+
enableThinking: argv.includes("--llm-thinking") ? true : false,
|
|
257
|
+
};
|
|
258
|
+
const llm = new llm_js_1.LlmClient(config);
|
|
259
|
+
const t0 = Date.now();
|
|
260
|
+
const result = await runImportanceEval({
|
|
261
|
+
snapshotPath,
|
|
262
|
+
sample,
|
|
263
|
+
seed,
|
|
264
|
+
llm,
|
|
265
|
+
});
|
|
266
|
+
const report = renderImportanceReport({
|
|
267
|
+
snapshotPath,
|
|
268
|
+
summary: result.summary,
|
|
269
|
+
aborted: result.aborted,
|
|
270
|
+
error: result.error,
|
|
271
|
+
model,
|
|
272
|
+
});
|
|
273
|
+
const reportPath = flag("--report") ?? (0, node_path_1.join)(process.cwd(), "data", "importance-eval-report.md");
|
|
274
|
+
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(reportPath), { recursive: true });
|
|
275
|
+
(0, node_fs_1.writeFileSync)(reportPath, report, "utf-8");
|
|
276
|
+
console.log(`[importance-eval] report written: ${reportPath}`);
|
|
277
|
+
console.log(report);
|
|
278
|
+
console.log(`[importance-eval] ${result.aborted ? "ABORTED" : "complete"} in ${Math.round((Date.now() - t0) / 1000)}s`);
|
|
279
|
+
process.exitCode = result.aborted ? 1 : 0;
|
|
280
|
+
}
|
|
281
|
+
if (process.argv[1] && process.argv[1].endsWith("importance-eval.js")) {
|
|
282
|
+
main().catch((err) => {
|
|
283
|
+
console.error("[importance-eval] FAILED:", err instanceof Error ? err.stack : String(err));
|
|
284
|
+
process.exitCode = 1;
|
|
285
|
+
});
|
|
286
|
+
}
|
|
@@ -65,7 +65,7 @@ export interface PlantedPair {
|
|
|
65
65
|
/**
|
|
66
66
|
* For `corrects` pairs: the scripted corrected text the ground-truth judge
|
|
67
67
|
* returns from the rewrite contract (REQUIRED for corrects — the harness
|
|
68
|
-
* throws if a corrects verdict ever reaches
|
|
68
|
+
* throws if a corrects verdict ever reaches its rewrite call without one).
|
|
69
69
|
*/
|
|
70
70
|
rewritten?: string;
|
|
71
71
|
/**
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The real-query battery for the #425 ranking gate — the DETERMINISTIC,
|
|
3
|
+
* dependency-free half (kept out of ranking-eval.ts so vitest can import
|
|
4
|
+
* it without pulling the ONNX embedder; same split as planted-fixtures vs
|
|
5
|
+
* planted-eval).
|
|
6
|
+
*
|
|
7
|
+
* The battery is a stable set of ~30 queries derived from the SNAPSHOT
|
|
8
|
+
* CORPUS ITSELF: fixed df windows over content tokens, fixed tie-breaks —
|
|
9
|
+
* same snapshot, same queries, every run. Bands: 10 high-df (50-400 docs),
|
|
10
|
+
* 10 mid-df (10-50), 5 rare-df (3-10), 4 mid-sentence capitalized proper
|
|
11
|
+
* nouns (df 2-8), plus the fixed "Sirnäs" query (the owner's live case).
|
|
12
|
+
*
|
|
13
|
+
* Also carries the photo-comparison gates the sweep protocol defines:
|
|
14
|
+
* - case 2 byte-stability (no-match queries must not drift),
|
|
15
|
+
* - battery top-1 stability >= 90%,
|
|
16
|
+
* - no query losing a both-channel exact match from its top-3.
|
|
17
|
+
*/
|
|
18
|
+
export declare const SIRNAS_QUERY = "Sirn\u00E4s";
|
|
19
|
+
/** Compact English stopword block (battery derivation only, not product). */
|
|
20
|
+
export declare const STOPWORDS: Set<string>;
|
|
21
|
+
export interface BatteryQuery {
|
|
22
|
+
q: string;
|
|
23
|
+
band: "proper-noun-fixed" | "high-df" | "mid-df" | "rare-df" | "proper-noun";
|
|
24
|
+
}
|
|
25
|
+
export interface BatteryResult {
|
|
26
|
+
q: string;
|
|
27
|
+
band: string;
|
|
28
|
+
top8: string[];
|
|
29
|
+
/** Per-result provenance (ranking-eval fills it; the gates read it). */
|
|
30
|
+
meta?: Array<{
|
|
31
|
+
id: string;
|
|
32
|
+
source: string | null;
|
|
33
|
+
term: boolean | null;
|
|
34
|
+
}>;
|
|
35
|
+
}
|
|
36
|
+
export interface BatteryPhoto {
|
|
37
|
+
queries: BatteryResult[];
|
|
38
|
+
/** Rank (1-based) of the real Sirnäs memory for the fixed query; null = not in top-8. */
|
|
39
|
+
sirnasRank: number | null;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Derive the battery from live corpus contents — DETERMINISTIC given the
|
|
43
|
+
* snapshot: fixed stopword list, fixed df windows, fixed tie-breaks
|
|
44
|
+
* (df desc, then token asc).
|
|
45
|
+
*/
|
|
46
|
+
export declare function deriveBatteryQueries(rows: Array<{
|
|
47
|
+
id: string;
|
|
48
|
+
content: string;
|
|
49
|
+
}>): BatteryQuery[];
|
|
50
|
+
/** Case 2 gate: the FULL returned list must be byte-identical (ids + order). */
|
|
51
|
+
export declare function compareCase2(baseline: string[], current: string[]): boolean;
|
|
52
|
+
export interface BatteryComparison {
|
|
53
|
+
top1Stability: number;
|
|
54
|
+
/** Stability with intended D3 promotions excluded: old top-1 was NOT a
|
|
55
|
+
* both-channel exact match and the new top-1 IS one (the boost doing its
|
|
56
|
+
* job on real queries — reported, not gated). */
|
|
57
|
+
top1StabilityExPromotions: number;
|
|
58
|
+
changedTop1: string[];
|
|
59
|
+
/** Queries whose top-3 LOST every both-channel exact match (the gate:
|
|
60
|
+
* must be empty). */
|
|
61
|
+
lostBothChannelExact: string[];
|
|
62
|
+
/** Per-id exact-match (term-carrying) rows displaced from top-3 — reported
|
|
63
|
+
* for the sweep record, not gated (multiple genuine matches make per-id
|
|
64
|
+
* membership arbitrary). */
|
|
65
|
+
lostExactMatches: string[];
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* Battery gates vs a recorded baseline photo. `contentById` maps memory id →
|
|
69
|
+
* content for the corpus BOTH photos were recorded against.
|
|
70
|
+
*
|
|
71
|
+
* Gates: (a) any query whose baseline top-3 held a BOTH-CHANNEL exact match
|
|
72
|
+
* (vector+FTS agreeing on a term-carrying row) must still hold one — the
|
|
73
|
+
* genuine-match guarantee never regresses; (b) top-1 stability. Intended D3
|
|
74
|
+
* promotions (a both-channel exact match taking top-1 from a non-exact or
|
|
75
|
+
* single-channel row) are counted separately — they are the feature firing,
|
|
76
|
+
* and the sweep record shows both numbers.
|
|
77
|
+
*/
|
|
78
|
+
export declare function batteryComparison(baseline: BatteryPhoto, current: BatteryPhoto, contentById: Map<string, string>): BatteryComparison;
|