@gamaze/hicortex 0.20.9 → 0.20.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +8 -0
  2. package/assets/dashboard.html +3989 -836
  3. package/dist/calibration.d.ts +119 -0
  4. package/dist/calibration.js +149 -1
  5. package/dist/capture-health.d.ts +87 -0
  6. package/dist/capture-health.js +106 -0
  7. package/dist/capture-pause.d.ts +86 -0
  8. package/dist/capture-pause.js +127 -0
  9. package/dist/capture.d.ts +9 -0
  10. package/dist/capture.js +2 -1
  11. package/dist/cli.js +36 -0
  12. package/dist/consolidate.d.ts +35 -0
  13. package/dist/consolidate.js +85 -9
  14. package/dist/dashboard.d.ts +322 -3
  15. package/dist/dashboard.js +592 -7
  16. package/dist/db.js +105 -0
  17. package/dist/eval/importance-eval.d.ts +85 -0
  18. package/dist/eval/importance-eval.js +286 -0
  19. package/dist/eval/planted-fixtures.d.ts +1 -1
  20. package/dist/eval/ranking-battery.d.ts +78 -0
  21. package/dist/eval/ranking-battery.js +181 -0
  22. package/dist/eval/ranking-eval.d.ts +41 -0
  23. package/dist/eval/ranking-eval.js +391 -0
  24. package/dist/eval/ranking-fixtures.d.ts +77 -0
  25. package/dist/eval/ranking-fixtures.js +226 -0
  26. package/dist/identity-store.d.ts +21 -0
  27. package/dist/identity-store.js +49 -0
  28. package/dist/init.d.ts +14 -0
  29. package/dist/init.js +32 -0
  30. package/dist/mcp-server.d.ts +12 -0
  31. package/dist/mcp-server.js +184 -3
  32. package/dist/nightly.d.ts +9 -1
  33. package/dist/nightly.js +59 -7
  34. package/dist/prompts.d.ts +10 -0
  35. package/dist/prompts.js +28 -5
  36. package/dist/reconsolidation.d.ts +59 -30
  37. package/dist/reconsolidation.js +526 -296
  38. package/dist/rescore-importance.d.ts +80 -0
  39. package/dist/rescore-importance.js +236 -0
  40. package/dist/retrieval.d.ts +12 -0
  41. package/dist/retrieval.js +30 -1
  42. package/dist/stages.d.ts +37 -0
  43. package/dist/stages.js +51 -0
  44. package/dist/state.d.ts +32 -6
  45. package/dist/storage.d.ts +34 -2
  46. package/dist/storage.js +63 -6
  47. package/dist/types.d.ts +48 -0
  48. package/package.json +3 -1
package/dist/db.js CHANGED
@@ -547,6 +547,111 @@ const MIGRATIONS = [
547
547
  db.exec("CREATE INDEX IF NOT EXISTS idx_memory_history_memory ON memory_history(memory_id)");
548
548
  },
549
549
  },
550
+ {
551
+ version: 15,
552
+ name: "source_machine",
553
+ up: (db) => {
554
+ // #421 owner direction (machine × harness identity): nullable, NO
555
+ // backfill — honesty over guesses. Rows written before this column
556
+ // stay NULL and group under "earlier captures" in the console; that
557
+ // bucket shrinks as nights accumulate stamped captures. Stamped by the
558
+ // nightly capture path (config `machineName` ?? os.hostname()) and
559
+ // accepted optionally by /distill + /ingest. Idempotent via hasColumn.
560
+ if (!hasColumn(db, "memories", "source_machine")) {
561
+ db.exec("ALTER TABLE memories ADD COLUMN source_machine TEXT");
562
+ }
563
+ },
564
+ },
565
+ {
566
+ version: 16,
567
+ name: "distill_activity",
568
+ up: (db) => {
569
+ // #422 Phase 2 — /distill capture-health accounting. One row per POST
570
+ // (every outcome incl. held/skipped), written by
571
+ // capture-health.ts:recordDistillActivity from the /distill handler's
572
+ // exits. This is OPERATIONS telemetry, not memory data: rows prune
573
+ // after 7 days (in-module, once per process per UTC day), so the table
574
+ // stays bounded while giving the console's capture-health card a real
575
+ // posts/sessions/bytes/held picture per machine × agent. `retried` is
576
+ // computed at insert (an earlier row with the same session_id +
577
+ // segment_id means this POST is a client retry of a cursor-held
578
+ // segment) — never updated afterwards. Sidecar table (no memories FK):
579
+ // the memory rows of a failed POST may never exist; the activity row
580
+ // must record the attempt anyway. Idempotent: IF NOT EXISTS everywhere.
581
+ db.exec(`
582
+ CREATE TABLE IF NOT EXISTS distill_activity (
583
+ ts TEXT NOT NULL,
584
+ day TEXT NOT NULL,
585
+ machine TEXT NOT NULL DEFAULT '',
586
+ agent TEXT NOT NULL DEFAULT '',
587
+ session_id TEXT,
588
+ segment_id TEXT,
589
+ bytes INTEGER NOT NULL DEFAULT 0,
590
+ outcome TEXT NOT NULL,
591
+ retried INTEGER NOT NULL DEFAULT 0
592
+ )
593
+ `);
594
+ db.exec("CREATE INDEX IF NOT EXISTS idx_distill_activity_day ON distill_activity(day)");
595
+ db.exec("CREATE INDEX IF NOT EXISTS idx_distill_activity_session_segment ON distill_activity(session_id, segment_id)");
596
+ },
597
+ },
598
+ {
599
+ version: 17,
600
+ name: "memory_corroboration",
601
+ up: (db) => {
602
+ // #423 phase 3 — explicit owner corroboration trail (POST /enrich).
603
+ // DEFAULT 0 with NO backfill: a pre-v17 row was never enriched and 0 is
604
+ // the honest count. /memory rides it via SELECT * (the console detail's
605
+ // "corroborated × N"); the enrich itself writes base_strength — it
606
+ // never fakes access/shown counts (those are the recall-adoption
607
+ // signal). Idempotent via hasColumn (the v15 pattern).
608
+ if (!hasColumn(db, "memories", "corroboration_count")) {
609
+ db.exec("ALTER TABLE memories ADD COLUMN corroboration_count INTEGER NOT NULL DEFAULT 0");
610
+ }
611
+ },
612
+ },
613
+ {
614
+ version: 18,
615
+ name: "capture_pauses",
616
+ up: (db) => {
617
+ // #423 phase 3, D3 — server-side 200-skip for a paused machine ×
618
+ // harness bundle. A row EXISTS = paused; absence = capturing (no
619
+ // "paused" flag to keep honest). Deliberately NO retention/pruning: the
620
+ // rows are few and operator-owned, and pruning one would silently
621
+ // resume capture the operator meant to hold. Idempotent: IF NOT EXISTS.
622
+ db.exec(`
623
+ CREATE TABLE IF NOT EXISTS capture_pauses (
624
+ machine TEXT NOT NULL DEFAULT '',
625
+ harness TEXT NOT NULL,
626
+ paused_at TEXT NOT NULL,
627
+ PRIMARY KEY(machine, harness)
628
+ )
629
+ `);
630
+ },
631
+ },
632
+ {
633
+ version: 19,
634
+ name: "add_importance_scored_at",
635
+ up: (db) => {
636
+ // #425 — scored-at watermark for importance scoring. getUnscoredMemories
637
+ // keys on importance_scored_at IS NULL (v19+), replacing the old
638
+ // base_strength = 0.5 sentinel, which re-rolled every row the model
639
+ // genuinely scored 0.5 every night. Backfill: existing rows that are NOT
640
+ // at the 0.5 sentinel are stamped "settled" (COALESCE(updated_at,
641
+ // ingested_at, created_at)) — they carry a real historical score and
642
+ // leave the nightly pool. Rows AT the sentinel stay NULL so the next
643
+ // nightly scores them ONCE under the new rubric (bounded — the watermark
644
+ // write in stageImportance then takes them out of the pool). The
645
+ // rescore-importance backfill CLI re-judges settled rows wholesale under
646
+ // its own cursor; this migration only makes the NIGHTLY pool honest.
647
+ // Idempotent via hasColumn (the v8 pattern).
648
+ if (!hasColumn(db, "memories", "importance_scored_at")) {
649
+ db.exec("ALTER TABLE memories ADD COLUMN importance_scored_at TEXT");
650
+ }
651
+ db.exec(`UPDATE memories SET importance_scored_at = COALESCE(updated_at, ingested_at, created_at)
652
+ WHERE importance_scored_at IS NULL AND base_strength != 0.5`);
653
+ },
654
+ },
550
655
  ];
551
656
  /**
552
657
  * Run all pending migrations against the database.
@@ -0,0 +1,85 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Importance-scorer distribution eval (#425) — the AC4 measurement.
4
+ *
5
+ * npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
6
+ * [--llm-base-url U --llm-model M --llm-api-key K]
7
+ *
8
+ * Samples real memories from a READONLY snapshot (openSnapshot — never
9
+ * initDb), sends them through the PRODUCTION scoring path — the real
10
+ * `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
11
+ * call exactly like the nightly's stageImportance — and reports the score
12
+ * distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
13
+ * at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
14
+ * median 0.30-0.40, p90 <= 0.75, zero at 1.0).
15
+ *
16
+ * This is the before/after photo for the scorer recalibration: run it on
17
+ * the OLD prompt (the inflated-distribution red photo), then on the
18
+ * re-anchored prompt (AC4 gate).
19
+ *
20
+ * Determinism: the sample is stratified across created_at deciles and
21
+ * drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
22
+ * same sample + same batches + same prompts. Calls are SERIAL (one
23
+ * in-flight request, like the nightly).
24
+ *
25
+ * Honesty rules: no stub LLM, no fallback distribution, measured scores
26
+ * only; a mid-run endpoint error aborts with the partial histogram
27
+ * printed (never silently padded). The gateway endpoint is passed at RUN
28
+ * TIME via flags — nothing about any specific endpoint lives in this file.
29
+ */
30
+ import { LlmClient } from "../llm.js";
31
+ export interface SampleRow {
32
+ id: string;
33
+ content: string;
34
+ created_at: string;
35
+ }
36
+ /**
37
+ * Deterministic decile-stratified sample: rows sorted by created_at, split
38
+ * into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
39
+ * (each decile shuffled by its own PRNG stream — index-independent). Same
40
+ * (rows, n, seed) always yields the same ordered selection.
41
+ */
42
+ export declare function stratifiedSample(rows: SampleRow[], n: number, seed: number): SampleRow[];
43
+ export interface ScoreSummary {
44
+ n: number;
45
+ min: number;
46
+ p25: number;
47
+ median: number;
48
+ p75: number;
49
+ p90: number;
50
+ max: number;
51
+ atExactly1: number;
52
+ histogram: Record<string, number>;
53
+ }
54
+ export declare function summarizeScores(scores: number[]): ScoreSummary;
55
+ export interface ImportanceEvalOptions {
56
+ snapshotPath: string;
57
+ sample?: number;
58
+ seed?: number;
59
+ llm: LlmClient;
60
+ /** Where to write the report (default data/importance-eval-report.md). */
61
+ reportPath?: string;
62
+ }
63
+ export interface ImportanceEvalResult {
64
+ summary: ScoreSummary;
65
+ /** Per-batch measured scores, in call order (audit trail). */
66
+ batches: Array<{
67
+ sent: number;
68
+ scores: number[];
69
+ }>;
70
+ aborted: boolean;
71
+ error?: string;
72
+ }
73
+ /**
74
+ * Run the importance distribution measurement. Throws only on setup errors;
75
+ * an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
76
+ * the caller decides exit code, always loudly).
77
+ */
78
+ export declare function runImportanceEval(opts: ImportanceEvalOptions): Promise<ImportanceEvalResult>;
79
+ export declare function renderImportanceReport(args: {
80
+ snapshotPath: string;
81
+ summary: ScoreSummary;
82
+ aborted: boolean;
83
+ error?: string;
84
+ model: string;
85
+ }): string;
@@ -0,0 +1,286 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * Importance-scorer distribution eval (#425) — the AC4 measurement.
5
+ *
6
+ * npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
7
+ * [--llm-base-url U --llm-model M --llm-api-key K]
8
+ *
9
+ * Samples real memories from a READONLY snapshot (openSnapshot — never
10
+ * initDb), sends them through the PRODUCTION scoring path — the real
11
+ * `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
12
+ * call exactly like the nightly's stageImportance — and reports the score
13
+ * distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
14
+ * at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
15
+ * median 0.30-0.40, p90 <= 0.75, zero at 1.0).
16
+ *
17
+ * This is the before/after photo for the scorer recalibration: run it on
18
+ * the OLD prompt (the inflated-distribution red photo), then on the
19
+ * re-anchored prompt (AC4 gate).
20
+ *
21
+ * Determinism: the sample is stratified across created_at deciles and
22
+ * drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
23
+ * same sample + same batches + same prompts. Calls are SERIAL (one
24
+ * in-flight request, like the nightly).
25
+ *
26
+ * Honesty rules: no stub LLM, no fallback distribution, measured scores
27
+ * only; a mid-run endpoint error aborts with the partial histogram
28
+ * printed (never silently padded). The gateway endpoint is passed at RUN
29
+ * TIME via flags — nothing about any specific endpoint lives in this file.
30
+ */
31
+ Object.defineProperty(exports, "__esModule", { value: true });
32
+ exports.stratifiedSample = stratifiedSample;
33
+ exports.summarizeScores = summarizeScores;
34
+ exports.runImportanceEval = runImportanceEval;
35
+ exports.renderImportanceReport = renderImportanceReport;
36
+ const node_fs_1 = require("node:fs");
37
+ const node_path_1 = require("node:path");
38
+ const eval_db_js_1 = require("./eval-db.js");
39
+ const llm_js_1 = require("../llm.js");
40
+ const consolidate_js_1 = require("../consolidate.js");
41
+ const prompts_js_1 = require("../prompts.js");
42
+ const DEFAULT_SAMPLE = 200;
43
+ const DEFAULT_SEED = 1;
44
+ /** The nightly's stageImportance batch size — matched exactly. */
45
+ const BATCH_SIZE = 10;
46
+ // ---------------------------------------------------------------------------
47
+ // Deterministic stratified sampler
48
+ // ---------------------------------------------------------------------------
49
+ /** mulberry32 — small, seedable, deterministic across Node versions. */
50
+ function mulberry32(seed) {
51
+ let a = seed >>> 0;
52
+ return () => {
53
+ a |= 0;
54
+ a = (a + 0x6d2b79f5) | 0;
55
+ let t = Math.imul(a ^ (a >>> 15), 1 | a);
56
+ t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
57
+ return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
58
+ };
59
+ }
60
+ /**
61
+ * Deterministic decile-stratified sample: rows sorted by created_at, split
62
+ * into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
63
+ * (each decile shuffled by its own PRNG stream — index-independent). Same
64
+ * (rows, n, seed) always yields the same ordered selection.
65
+ */
66
+ function stratifiedSample(rows, n, seed) {
67
+ if (rows.length === 0 || n <= 0)
68
+ return [];
69
+ const sorted = [...rows].sort((a, b) => a.created_at === b.created_at
70
+ ? a.id.localeCompare(b.id)
71
+ : a.created_at.localeCompare(b.created_at));
72
+ const decileSize = sorted.length / 10;
73
+ const perDecile = Math.max(1, Math.ceil(n / 10));
74
+ const out = [];
75
+ for (let d = 0; d < 10 && out.length < n; d++) {
76
+ const start = Math.floor(d * decileSize);
77
+ const end = d === 9 ? sorted.length : Math.floor((d + 1) * decileSize);
78
+ const decile = sorted.slice(start, end);
79
+ const rnd = mulberry32(seed * 31 + d);
80
+ // Fisher-Yates with the decile-local stream.
81
+ for (let i = decile.length - 1; i > 0; i--) {
82
+ const j = Math.floor(rnd() * (i + 1));
83
+ [decile[i], decile[j]] = [decile[j], decile[i]];
84
+ }
85
+ out.push(...decile.slice(0, Math.min(perDecile, n - out.length)));
86
+ }
87
+ return out;
88
+ }
89
+ /** Nearest-rank percentile on the ASCENDING-sorted values. */
90
+ function percentile(sorted, p) {
91
+ if (sorted.length === 0)
92
+ return Number.NaN;
93
+ const idx = Math.min(sorted.length - 1, Math.max(0, Math.ceil((p / 100) * sorted.length) - 1));
94
+ return sorted[idx];
95
+ }
96
+ function summarizeScores(scores) {
97
+ const sorted = [...scores].sort((a, b) => a - b);
98
+ const histogram = {};
99
+ for (let b = 0; b < 10; b++)
100
+ histogram[`${b}-${b + 1}`] = 0;
101
+ for (const s of scores) {
102
+ const b = Math.min(9, Math.max(0, Math.floor(s * 10)));
103
+ histogram[`${b}-${b + 1}`]++;
104
+ }
105
+ return {
106
+ n: scores.length,
107
+ min: percentile(sorted, 0),
108
+ p25: percentile(sorted, 25),
109
+ median: percentile(sorted, 50),
110
+ p75: percentile(sorted, 75),
111
+ p90: percentile(sorted, 90),
112
+ max: percentile(sorted, 100),
113
+ atExactly1: scores.filter((s) => s === 1).length,
114
+ histogram,
115
+ };
116
+ }
117
+ /**
118
+ * Run the importance distribution measurement. Throws only on setup errors;
119
+ * an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
120
+ * the caller decides exit code, always loudly).
121
+ */
122
+ async function runImportanceEval(opts) {
123
+ const sampleSize = opts.sample ?? DEFAULT_SAMPLE;
124
+ const seed = opts.seed ?? DEFAULT_SEED;
125
+ const db = (0, eval_db_js_1.openSnapshot)(opts.snapshotPath);
126
+ let rows;
127
+ try {
128
+ rows = db
129
+ .prepare("SELECT id, content, created_at FROM memories WHERE COALESCE(status, '') != 'absorbed' ORDER BY id")
130
+ .all();
131
+ }
132
+ finally {
133
+ db.close();
134
+ }
135
+ if (rows.length === 0)
136
+ throw new Error("snapshot has no live memories");
137
+ const sample = stratifiedSample(rows, sampleSize, seed);
138
+ console.log(`[importance-eval] snapshot rows ${rows.length}, sampled ${sample.length} ` +
139
+ `(seed ${seed}, stratified over created_at deciles)`);
140
+ const batches = [];
141
+ const allScores = [];
142
+ for (let i = 0; i < sample.length; i += BATCH_SIZE) {
143
+ const batch = sample.slice(i, i + BATCH_SIZE);
144
+ const lines = batch.map((mem, idx) => `[${idx}] ${mem.content.slice(0, 500)}`);
145
+ const prompt = (0, prompts_js_1.importanceScoring)(lines.join("\n\n"));
146
+ try {
147
+ const r = await opts.llm.complete(prompt);
148
+ let scores = (0, consolidate_js_1.parseJsonLenient)(r.text, null);
149
+ if (!Array.isArray(scores))
150
+ scores = new Array(batch.length).fill(0.5);
151
+ while (scores.length < batch.length)
152
+ scores.push(0.5);
153
+ scores = scores.slice(0, batch.length);
154
+ const clamped = scores.map((s) => {
155
+ const v = Number(s);
156
+ return Number.isFinite(v) ? Math.max(0, Math.min(1, v)) : 0.5;
157
+ });
158
+ batches.push({ sent: batch.length, scores: clamped });
159
+ allScores.push(...clamped);
160
+ console.log(`[importance-eval] batch ${batches.length}: ${clamped.map((s) => s.toFixed(1)).join(", ")}`);
161
+ }
162
+ catch (err) {
163
+ const msg = err instanceof Error ? err.message : String(err);
164
+ console.error(`[importance-eval] endpoint error at batch ${Math.floor(i / BATCH_SIZE) + 1}: ${msg}`);
165
+ const summary = summarizeScores(allScores);
166
+ return { summary, batches, aborted: true, error: msg };
167
+ }
168
+ }
169
+ return { summary: summarizeScores(allScores), batches, aborted: false };
170
+ }
171
+ function renderImportanceReport(args) {
172
+ const s = args.summary;
173
+ const L = [];
174
+ L.push("# Importance-scorer distribution (#425 — AC4)\n");
175
+ L.push(`Snapshot: \`${args.snapshotPath}\` \nGenerated: ${new Date().toISOString()} \n` +
176
+ `Model: ${args.model} (the configured production endpoint) \n` +
177
+ `Prompt: the CURRENT production importanceScoring (whatever ships in this tree)\n`);
178
+ if (args.aborted) {
179
+ L.push(`**ABORTED MID-RUN** — ${args.error ?? "endpoint error"}; partial distribution over ${s.n} scores below.\n`);
180
+ }
181
+ L.push("## Distribution\n");
182
+ L.push("| stat | value |");
183
+ L.push("|---|---|");
184
+ L.push(`| n | ${s.n} |`);
185
+ L.push(`| min | ${s.min.toFixed(3)} |`);
186
+ L.push(`| p25 | ${s.p25.toFixed(3)} |`);
187
+ L.push(`| **median** | **${s.median.toFixed(3)}** |`);
188
+ L.push(`| p75 | ${s.p75.toFixed(3)} |`);
189
+ L.push(`| **p90** | **${s.p90.toFixed(3)}** |`);
190
+ L.push(`| max | ${s.max.toFixed(3)} |`);
191
+ L.push(`| at exactly 1.0 | ${s.atExactly1} |`);
192
+ L.push("");
193
+ L.push("| bucket | count |");
194
+ L.push("|---|---|");
195
+ for (const [bucket, count] of Object.entries(s.histogram)) {
196
+ L.push(`| ${bucket} | ${count} |`);
197
+ }
198
+ L.push("");
199
+ L.push(`Target band (owner D2, 2026-09-13): median 0.30-0.40 → ` +
200
+ `${s.median >= 0.3 && s.median <= 0.4 ? "**PASS**" : "**MISS**"} (${s.median.toFixed(3)}); ` +
201
+ `p90 <= 0.75 → ${s.p90 <= 0.75 ? "**PASS**" : "**MISS**"} (${s.p90.toFixed(3)}); ` +
202
+ `zero at 1.0 → ${s.atExactly1 === 0 ? "**PASS**" : "**MISS**"} (${s.atExactly1})\n`);
203
+ return L.join("\n");
204
+ }
205
+ // ---------------------------------------------------------------------------
206
+ // CLI
207
+ // ---------------------------------------------------------------------------
208
+ function usage() {
209
+ console.error("Usage: npm run eval:importance -- <snapshot.db> [--sample N=200] [--seed S=1]\n" +
210
+ " --llm-base-url U --llm-model M --llm-api-key K (required: the scoring endpoint)\n" +
211
+ " [--llm-thinking] (thinking ON; default OFF — production parity)\n" +
212
+ " [--report <out.md>]");
213
+ process.exit(1);
214
+ }
215
+ async function main() {
216
+ const argv = process.argv.slice(2);
217
+ const positional = [];
218
+ const flag = (name) => {
219
+ const i = argv.indexOf(name);
220
+ return i !== -1 && argv[i + 1] && !argv[i + 1].startsWith("--") ? argv[i + 1] : undefined;
221
+ };
222
+ for (let i = 0; i < argv.length; i++) {
223
+ if (!argv[i].startsWith("--"))
224
+ positional.push(argv[i]);
225
+ }
226
+ const snapshotPath = positional[0];
227
+ if (!snapshotPath || argv.includes("--help") || argv.includes("-h"))
228
+ usage();
229
+ const baseUrl = flag("--llm-base-url");
230
+ const model = flag("--llm-model");
231
+ const apiKey = flag("--llm-api-key");
232
+ if (!baseUrl || !model || !apiKey) {
233
+ console.error("[importance-eval] --llm-base-url, --llm-model and --llm-api-key are required (the production scoring endpoint)");
234
+ usage();
235
+ }
236
+ const sample = flag("--sample") ? parseInt(flag("--sample"), 10) : DEFAULT_SAMPLE;
237
+ const seed = flag("--seed") ? parseInt(flag("--seed"), 10) : DEFAULT_SEED;
238
+ if (!Number.isInteger(sample) || sample < 10 || sample > 5000) {
239
+ console.error("[importance-eval] --sample must be an integer in [10, 5000]");
240
+ process.exit(1);
241
+ }
242
+ if (!Number.isInteger(seed) || seed <= 0) {
243
+ console.error("[importance-eval] --seed must be a positive integer");
244
+ process.exit(1);
245
+ }
246
+ // OpenAI-compatible config built from flags ONLY (never hardcoded; the
247
+ // production endpoint is a runtime input, not a shipped value). Thinking
248
+ // defaults OFF (production parity: the scorer runs with thinking disabled —
249
+ // an unclosed think block can eat the whole output budget); opt in with
250
+ // --llm-thinking for a thinking-on endpoint.
251
+ const config = {
252
+ baseUrl,
253
+ model,
254
+ apiKey,
255
+ provider: "openai",
256
+ enableThinking: argv.includes("--llm-thinking") ? true : false,
257
+ };
258
+ const llm = new llm_js_1.LlmClient(config);
259
+ const t0 = Date.now();
260
+ const result = await runImportanceEval({
261
+ snapshotPath,
262
+ sample,
263
+ seed,
264
+ llm,
265
+ });
266
+ const report = renderImportanceReport({
267
+ snapshotPath,
268
+ summary: result.summary,
269
+ aborted: result.aborted,
270
+ error: result.error,
271
+ model,
272
+ });
273
+ const reportPath = flag("--report") ?? (0, node_path_1.join)(process.cwd(), "data", "importance-eval-report.md");
274
+ (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(reportPath), { recursive: true });
275
+ (0, node_fs_1.writeFileSync)(reportPath, report, "utf-8");
276
+ console.log(`[importance-eval] report written: ${reportPath}`);
277
+ console.log(report);
278
+ console.log(`[importance-eval] ${result.aborted ? "ABORTED" : "complete"} in ${Math.round((Date.now() - t0) / 1000)}s`);
279
+ process.exitCode = result.aborted ? 1 : 0;
280
+ }
281
+ if (process.argv[1] && process.argv[1].endsWith("importance-eval.js")) {
282
+ main().catch((err) => {
283
+ console.error("[importance-eval] FAILED:", err instanceof Error ? err.stack : String(err));
284
+ process.exitCode = 1;
285
+ });
286
+ }
@@ -65,7 +65,7 @@ export interface PlantedPair {
65
65
  /**
66
66
  * For `corrects` pairs: the scripted corrected text the ground-truth judge
67
67
  * returns from the rewrite contract (REQUIRED for corrects — the harness
68
- * throws if a corrects verdict ever reaches the rewrite phase without one).
68
+ * throws if a corrects verdict ever reaches its rewrite call without one).
69
69
  */
70
70
  rewritten?: string;
71
71
  /**
@@ -0,0 +1,78 @@
1
+ /**
2
+ * The real-query battery for the #425 ranking gate — the DETERMINISTIC,
3
+ * dependency-free half (kept out of ranking-eval.ts so vitest can import
4
+ * it without pulling the ONNX embedder; same split as planted-fixtures vs
5
+ * planted-eval).
6
+ *
7
+ * The battery is a stable set of ~30 queries derived from the SNAPSHOT
8
+ * CORPUS ITSELF: fixed df windows over content tokens, fixed tie-breaks —
9
+ * same snapshot, same queries, every run. Bands: 10 high-df (50-400 docs),
10
+ * 10 mid-df (10-50), 5 rare-df (3-10), 4 mid-sentence capitalized proper
11
+ * nouns (df 2-8), plus the fixed "Sirnäs" query (the owner's live case).
12
+ *
13
+ * Also carries the photo-comparison gates the sweep protocol defines:
14
+ * - case 2 byte-stability (no-match queries must not drift),
15
+ * - battery top-1 stability >= 90%,
16
+ * - no query losing a both-channel exact match from its top-3.
17
+ */
18
+ export declare const SIRNAS_QUERY = "Sirn\u00E4s";
19
+ /** Compact English stopword block (battery derivation only, not product). */
20
+ export declare const STOPWORDS: Set<string>;
21
+ export interface BatteryQuery {
22
+ q: string;
23
+ band: "proper-noun-fixed" | "high-df" | "mid-df" | "rare-df" | "proper-noun";
24
+ }
25
+ export interface BatteryResult {
26
+ q: string;
27
+ band: string;
28
+ top8: string[];
29
+ /** Per-result provenance (ranking-eval fills it; the gates read it). */
30
+ meta?: Array<{
31
+ id: string;
32
+ source: string | null;
33
+ term: boolean | null;
34
+ }>;
35
+ }
36
+ export interface BatteryPhoto {
37
+ queries: BatteryResult[];
38
+ /** Rank (1-based) of the real Sirnäs memory for the fixed query; null = not in top-8. */
39
+ sirnasRank: number | null;
40
+ }
41
+ /**
42
+ * Derive the battery from live corpus contents — DETERMINISTIC given the
43
+ * snapshot: fixed stopword list, fixed df windows, fixed tie-breaks
44
+ * (df desc, then token asc).
45
+ */
46
+ export declare function deriveBatteryQueries(rows: Array<{
47
+ id: string;
48
+ content: string;
49
+ }>): BatteryQuery[];
50
+ /** Case 2 gate: the FULL returned list must be byte-identical (ids + order). */
51
+ export declare function compareCase2(baseline: string[], current: string[]): boolean;
52
+ export interface BatteryComparison {
53
+ top1Stability: number;
54
+ /** Stability with intended D3 promotions excluded: old top-1 was NOT a
55
+ * both-channel exact match and the new top-1 IS one (the boost doing its
56
+ * job on real queries — reported, not gated). */
57
+ top1StabilityExPromotions: number;
58
+ changedTop1: string[];
59
+ /** Queries whose top-3 LOST every both-channel exact match (the gate:
60
+ * must be empty). */
61
+ lostBothChannelExact: string[];
62
+ /** Per-id exact-match (term-carrying) rows displaced from top-3 — reported
63
+ * for the sweep record, not gated (multiple genuine matches make per-id
64
+ * membership arbitrary). */
65
+ lostExactMatches: string[];
66
+ }
67
+ /**
68
+ * Battery gates vs a recorded baseline photo. `contentById` maps memory id →
69
+ * content for the corpus BOTH photos were recorded against.
70
+ *
71
+ * Gates: (a) any query whose baseline top-3 held a BOTH-CHANNEL exact match
72
+ * (vector+FTS agreeing on a term-carrying row) must still hold one — the
73
+ * genuine-match guarantee never regresses; (b) top-1 stability. Intended D3
74
+ * promotions (a both-channel exact match taking top-1 from a non-exact or
75
+ * single-channel row) are counted separately — they are the feature firing,
76
+ * and the sweep record shows both numbers.
77
+ */
78
+ export declare function batteryComparison(baseline: BatteryPhoto, current: BatteryPhoto, contentById: Map<string, string>): BatteryComparison;