@gamaze/hicortex 0.20.7 → 0.20.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +18 -41
  2. package/assets/dashboard.html +3989 -836
  3. package/dist/calibration.d.ts +293 -0
  4. package/dist/calibration.js +379 -0
  5. package/dist/capture-health.d.ts +87 -0
  6. package/dist/capture-health.js +106 -0
  7. package/dist/capture-pause.d.ts +86 -0
  8. package/dist/capture-pause.js +127 -0
  9. package/dist/capture.d.ts +24 -3
  10. package/dist/capture.js +11 -1
  11. package/dist/classify-domains.d.ts +6 -0
  12. package/dist/classify-domains.js +7 -1
  13. package/dist/cli.js +38 -3
  14. package/dist/config-read.d.ts +1 -1
  15. package/dist/config-read.js +96 -9
  16. package/dist/consolidate.d.ts +114 -68
  17. package/dist/consolidate.js +302 -182
  18. package/dist/dashboard.d.ts +326 -6
  19. package/dist/dashboard.js +592 -7
  20. package/dist/db.js +105 -0
  21. package/dist/dedup.d.ts +34 -26
  22. package/dist/dedup.js +91 -57
  23. package/dist/distiller.js +1 -1
  24. package/dist/domain-classify.d.ts +7 -6
  25. package/dist/domain-classify.js +12 -10
  26. package/dist/eval/decay-eval.d.ts +3 -3
  27. package/dist/eval/decay-eval.js +4 -4
  28. package/dist/eval/importance-eval.d.ts +85 -0
  29. package/dist/eval/importance-eval.js +286 -0
  30. package/dist/eval/planted-eval.d.ts +26 -0
  31. package/dist/eval/planted-eval.js +97 -0
  32. package/dist/eval/planted-fixtures.d.ts +107 -0
  33. package/dist/eval/planted-fixtures.js +283 -0
  34. package/dist/eval/planted-harness.d.ts +176 -0
  35. package/dist/eval/planted-harness.js +649 -0
  36. package/dist/eval/ranking-battery.d.ts +78 -0
  37. package/dist/eval/ranking-battery.js +181 -0
  38. package/dist/eval/ranking-eval.d.ts +41 -0
  39. package/dist/eval/ranking-eval.js +391 -0
  40. package/dist/eval/ranking-fixtures.d.ts +77 -0
  41. package/dist/eval/ranking-fixtures.js +226 -0
  42. package/dist/identity-store.d.ts +21 -0
  43. package/dist/identity-store.js +49 -0
  44. package/dist/index.js +4 -3
  45. package/dist/init.d.ts +23 -3
  46. package/dist/init.js +84 -9
  47. package/dist/llm.d.ts +43 -58
  48. package/dist/llm.js +87 -101
  49. package/dist/mcp-server.d.ts +12 -0
  50. package/dist/mcp-server.js +213 -32
  51. package/dist/nightly.d.ts +9 -1
  52. package/dist/nightly.js +164 -110
  53. package/dist/nofit.d.ts +4 -11
  54. package/dist/nofit.js +6 -23
  55. package/dist/prompts.d.ts +10 -0
  56. package/dist/prompts.js +28 -5
  57. package/dist/recall-index.d.ts +30 -28
  58. package/dist/recall-index.js +21 -18
  59. package/dist/recall-registry.d.ts +2 -1
  60. package/dist/recall-registry.js +35 -1
  61. package/dist/reconsolidation.d.ts +168 -87
  62. package/dist/reconsolidation.js +818 -377
  63. package/dist/relink.js +3 -4
  64. package/dist/rescore-importance.d.ts +80 -0
  65. package/dist/rescore-importance.js +236 -0
  66. package/dist/retrieval.d.ts +80 -35
  67. package/dist/retrieval.js +322 -105
  68. package/dist/run-deadline.d.ts +62 -0
  69. package/dist/run-deadline.js +73 -0
  70. package/dist/schema-prototypes.d.ts +3 -3
  71. package/dist/schema-prototypes.js +3 -3
  72. package/dist/stages.d.ts +37 -0
  73. package/dist/stages.js +51 -0
  74. package/dist/state.d.ts +34 -9
  75. package/dist/storage.d.ts +50 -18
  76. package/dist/storage.js +125 -30
  77. package/dist/telemetry.d.ts +8 -7
  78. package/dist/token-budget.js +3 -4
  79. package/dist/type-classify.js +4 -4
  80. package/dist/types.d.ts +143 -155
  81. package/domains.example.json +4 -5
  82. package/hermes-plugin/hicortex/README.md +2 -2
  83. package/openclaw.plugin.json +1 -1
  84. package/package.json +4 -1
  85. package/pi-extension/hicortex/README.md +1 -1
  86. package/server.json +3 -3
@@ -89,12 +89,12 @@ function runDomainBacklogAudit(db, stateDir) {
89
89
  /**
90
90
  * Run the actual `stageDecayPrune` (imported from consolidate.ts, `dryRun:
91
91
  * true`) against the snapshot. Configures the decay clock to the given
92
- * half-life first (bedrock has no `decayHalfLifeDays` override, so the
93
- * caller should pass the shipped default — see run-eval.ts) so the eval and
94
- * production score with the same clock.
92
+ * half-life first via the eval seam (#408: the half-life is a calibration
93
+ * constant — the caller should pass the shipped default, see run-eval.ts) so
94
+ * the eval and production score with the same clock.
95
95
  */
96
96
  function runPruneDryRun(db, decayHalfLifeDays = retrieval_js_1.DEFAULT_DECAY_HALF_LIFE_DAYS) {
97
- (0, retrieval_js_1.configureDecay)({ halfLifeDays: decayHalfLifeDays });
97
+ (0, retrieval_js_1.configureDecay)(decayHalfLifeDays);
98
98
  const result = (0, consolidate_js_1.stageDecayPrune)(db, true);
99
99
  return { ...result, decayHalfLifeDaysUsed: decayHalfLifeDays };
100
100
  }
@@ -0,0 +1,85 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Importance-scorer distribution eval (#425) — the AC4 measurement.
4
+ *
5
+ * npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
6
+ * [--llm-base-url U --llm-model M --llm-api-key K]
7
+ *
8
+ * Samples real memories from a READONLY snapshot (openSnapshot — never
9
+ * initDb), sends them through the PRODUCTION scoring path — the real
10
+ * `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
11
+ * call exactly like the nightly's stageImportance — and reports the score
12
+ * distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
13
+ * at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
14
+ * median 0.30-0.40, p90 <= 0.75, zero at 1.0).
15
+ *
16
+ * This is the before/after photo for the scorer recalibration: run it on
17
+ * the OLD prompt (the inflated-distribution red photo), then on the
18
+ * re-anchored prompt (AC4 gate).
19
+ *
20
+ * Determinism: the sample is stratified across created_at deciles and
21
+ * drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
22
+ * same sample + same batches + same prompts. Calls are SERIAL (one
23
+ * in-flight request, like the nightly).
24
+ *
25
+ * Honesty rules: no stub LLM, no fallback distribution, measured scores
26
+ * only; a mid-run endpoint error aborts with the partial histogram
27
+ * printed (never silently padded). The gateway endpoint is passed at RUN
28
+ * TIME via flags — nothing about any specific endpoint lives in this file.
29
+ */
30
+ import { LlmClient } from "../llm.js";
31
+ export interface SampleRow {
32
+ id: string;
33
+ content: string;
34
+ created_at: string;
35
+ }
36
+ /**
37
+ * Deterministic decile-stratified sample: rows sorted by created_at, split
38
+ * into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
39
+ * (each decile shuffled by its own PRNG stream — index-independent). Same
40
+ * (rows, n, seed) always yields the same ordered selection.
41
+ */
42
+ export declare function stratifiedSample(rows: SampleRow[], n: number, seed: number): SampleRow[];
43
+ export interface ScoreSummary {
44
+ n: number;
45
+ min: number;
46
+ p25: number;
47
+ median: number;
48
+ p75: number;
49
+ p90: number;
50
+ max: number;
51
+ atExactly1: number;
52
+ histogram: Record<string, number>;
53
+ }
54
+ export declare function summarizeScores(scores: number[]): ScoreSummary;
55
+ export interface ImportanceEvalOptions {
56
+ snapshotPath: string;
57
+ sample?: number;
58
+ seed?: number;
59
+ llm: LlmClient;
60
+ /** Where to write the report (default data/importance-eval-report.md). */
61
+ reportPath?: string;
62
+ }
63
+ export interface ImportanceEvalResult {
64
+ summary: ScoreSummary;
65
+ /** Per-batch measured scores, in call order (audit trail). */
66
+ batches: Array<{
67
+ sent: number;
68
+ scores: number[];
69
+ }>;
70
+ aborted: boolean;
71
+ error?: string;
72
+ }
73
+ /**
74
+ * Run the importance distribution measurement. Throws only on setup errors;
75
+ * an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
76
+ * the caller decides exit code, always loudly).
77
+ */
78
+ export declare function runImportanceEval(opts: ImportanceEvalOptions): Promise<ImportanceEvalResult>;
79
+ export declare function renderImportanceReport(args: {
80
+ snapshotPath: string;
81
+ summary: ScoreSummary;
82
+ aborted: boolean;
83
+ error?: string;
84
+ model: string;
85
+ }): string;
@@ -0,0 +1,286 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * Importance-scorer distribution eval (#425) — the AC4 measurement.
5
+ *
6
+ * npm run eval:importance -- <snapshot.db> [--sample N] [--seed S]
7
+ * [--llm-base-url U --llm-model M --llm-api-key K]
8
+ *
9
+ * Samples real memories from a READONLY snapshot (openSnapshot — never
10
+ * initDb), sends them through the PRODUCTION scoring path — the real
11
+ * `LlmClient` and the CURRENT `importanceScoring` prompt, batched 10 per
12
+ * call exactly like the nightly's stageImportance — and reports the score
13
+ * distribution: histogram (0.1 buckets), min/p25/median/p75/p90/max, count
14
+ * at exactly 1.0, and the owner-approved target band (D2, 2026-09-13:
15
+ * median 0.30-0.40, p90 <= 0.75, zero at 1.0).
16
+ *
17
+ * This is the before/after photo for the scorer recalibration: run it on
18
+ * the OLD prompt (the inflated-distribution red photo), then on the
19
+ * re-anchored prompt (AC4 gate).
20
+ *
21
+ * Determinism: the sample is stratified across created_at deciles and
22
+ * drawn with a SEEDED PRNG (mulberry32) — same snapshot + same seed =
23
+ * same sample + same batches + same prompts. Calls are SERIAL (one
24
+ * in-flight request, like the nightly).
25
+ *
26
+ * Honesty rules: no stub LLM, no fallback distribution, measured scores
27
+ * only; a mid-run endpoint error aborts with the partial histogram
28
+ * printed (never silently padded). The gateway endpoint is passed at RUN
29
+ * TIME via flags — nothing about any specific endpoint lives in this file.
30
+ */
31
+ Object.defineProperty(exports, "__esModule", { value: true });
32
+ exports.stratifiedSample = stratifiedSample;
33
+ exports.summarizeScores = summarizeScores;
34
+ exports.runImportanceEval = runImportanceEval;
35
+ exports.renderImportanceReport = renderImportanceReport;
36
+ const node_fs_1 = require("node:fs");
37
+ const node_path_1 = require("node:path");
38
+ const eval_db_js_1 = require("./eval-db.js");
39
+ const llm_js_1 = require("../llm.js");
40
+ const consolidate_js_1 = require("../consolidate.js");
41
+ const prompts_js_1 = require("../prompts.js");
42
+ const DEFAULT_SAMPLE = 200;
43
+ const DEFAULT_SEED = 1;
44
+ /** The nightly's stageImportance batch size — matched exactly. */
45
+ const BATCH_SIZE = 10;
46
+ // ---------------------------------------------------------------------------
47
+ // Deterministic stratified sampler
48
+ // ---------------------------------------------------------------------------
49
+ /** mulberry32 — small, seedable, deterministic across Node versions. */
50
+ function mulberry32(seed) {
51
+ let a = seed >>> 0;
52
+ return () => {
53
+ a |= 0;
54
+ a = (a + 0x6d2b79f5) | 0;
55
+ let t = Math.imul(a ^ (a >>> 15), 1 | a);
56
+ t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
57
+ return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
58
+ };
59
+ }
60
+ /**
61
+ * Deterministic decile-stratified sample: rows sorted by created_at, split
62
+ * into 10 equal deciles, `ceil(n/10)` drawn from each via the seeded PRNG
63
+ * (each decile shuffled by its own PRNG stream — index-independent). Same
64
+ * (rows, n, seed) always yields the same ordered selection.
65
+ */
66
+ function stratifiedSample(rows, n, seed) {
67
+ if (rows.length === 0 || n <= 0)
68
+ return [];
69
+ const sorted = [...rows].sort((a, b) => a.created_at === b.created_at
70
+ ? a.id.localeCompare(b.id)
71
+ : a.created_at.localeCompare(b.created_at));
72
+ const decileSize = sorted.length / 10;
73
+ const perDecile = Math.max(1, Math.ceil(n / 10));
74
+ const out = [];
75
+ for (let d = 0; d < 10 && out.length < n; d++) {
76
+ const start = Math.floor(d * decileSize);
77
+ const end = d === 9 ? sorted.length : Math.floor((d + 1) * decileSize);
78
+ const decile = sorted.slice(start, end);
79
+ const rnd = mulberry32(seed * 31 + d);
80
+ // Fisher-Yates with the decile-local stream.
81
+ for (let i = decile.length - 1; i > 0; i--) {
82
+ const j = Math.floor(rnd() * (i + 1));
83
+ [decile[i], decile[j]] = [decile[j], decile[i]];
84
+ }
85
+ out.push(...decile.slice(0, Math.min(perDecile, n - out.length)));
86
+ }
87
+ return out;
88
+ }
89
+ /** Nearest-rank percentile on the ASCENDING-sorted values. */
90
+ function percentile(sorted, p) {
91
+ if (sorted.length === 0)
92
+ return Number.NaN;
93
+ const idx = Math.min(sorted.length - 1, Math.max(0, Math.ceil((p / 100) * sorted.length) - 1));
94
+ return sorted[idx];
95
+ }
96
+ function summarizeScores(scores) {
97
+ const sorted = [...scores].sort((a, b) => a - b);
98
+ const histogram = {};
99
+ for (let b = 0; b < 10; b++)
100
+ histogram[`${b}-${b + 1}`] = 0;
101
+ for (const s of scores) {
102
+ const b = Math.min(9, Math.max(0, Math.floor(s * 10)));
103
+ histogram[`${b}-${b + 1}`]++;
104
+ }
105
+ return {
106
+ n: scores.length,
107
+ min: percentile(sorted, 0),
108
+ p25: percentile(sorted, 25),
109
+ median: percentile(sorted, 50),
110
+ p75: percentile(sorted, 75),
111
+ p90: percentile(sorted, 90),
112
+ max: percentile(sorted, 100),
113
+ atExactly1: scores.filter((s) => s === 1).length,
114
+ histogram,
115
+ };
116
+ }
117
+ /**
118
+ * Run the importance distribution measurement. Throws only on setup errors;
119
+ * an LLM/endpoint failure mid-run sets `aborted` (partial results returned —
120
+ * the caller decides exit code, always loudly).
121
+ */
122
+ async function runImportanceEval(opts) {
123
+ const sampleSize = opts.sample ?? DEFAULT_SAMPLE;
124
+ const seed = opts.seed ?? DEFAULT_SEED;
125
+ const db = (0, eval_db_js_1.openSnapshot)(opts.snapshotPath);
126
+ let rows;
127
+ try {
128
+ rows = db
129
+ .prepare("SELECT id, content, created_at FROM memories WHERE COALESCE(status, '') != 'absorbed' ORDER BY id")
130
+ .all();
131
+ }
132
+ finally {
133
+ db.close();
134
+ }
135
+ if (rows.length === 0)
136
+ throw new Error("snapshot has no live memories");
137
+ const sample = stratifiedSample(rows, sampleSize, seed);
138
+ console.log(`[importance-eval] snapshot rows ${rows.length}, sampled ${sample.length} ` +
139
+ `(seed ${seed}, stratified over created_at deciles)`);
140
+ const batches = [];
141
+ const allScores = [];
142
+ for (let i = 0; i < sample.length; i += BATCH_SIZE) {
143
+ const batch = sample.slice(i, i + BATCH_SIZE);
144
+ const lines = batch.map((mem, idx) => `[${idx}] ${mem.content.slice(0, 500)}`);
145
+ const prompt = (0, prompts_js_1.importanceScoring)(lines.join("\n\n"));
146
+ try {
147
+ const r = await opts.llm.complete(prompt);
148
+ let scores = (0, consolidate_js_1.parseJsonLenient)(r.text, null);
149
+ if (!Array.isArray(scores))
150
+ scores = new Array(batch.length).fill(0.5);
151
+ while (scores.length < batch.length)
152
+ scores.push(0.5);
153
+ scores = scores.slice(0, batch.length);
154
+ const clamped = scores.map((s) => {
155
+ const v = Number(s);
156
+ return Number.isFinite(v) ? Math.max(0, Math.min(1, v)) : 0.5;
157
+ });
158
+ batches.push({ sent: batch.length, scores: clamped });
159
+ allScores.push(...clamped);
160
+ console.log(`[importance-eval] batch ${batches.length}: ${clamped.map((s) => s.toFixed(1)).join(", ")}`);
161
+ }
162
+ catch (err) {
163
+ const msg = err instanceof Error ? err.message : String(err);
164
+ console.error(`[importance-eval] endpoint error at batch ${Math.floor(i / BATCH_SIZE) + 1}: ${msg}`);
165
+ const summary = summarizeScores(allScores);
166
+ return { summary, batches, aborted: true, error: msg };
167
+ }
168
+ }
169
+ return { summary: summarizeScores(allScores), batches, aborted: false };
170
+ }
171
+ function renderImportanceReport(args) {
172
+ const s = args.summary;
173
+ const L = [];
174
+ L.push("# Importance-scorer distribution (#425 — AC4)\n");
175
+ L.push(`Snapshot: \`${args.snapshotPath}\` \nGenerated: ${new Date().toISOString()} \n` +
176
+ `Model: ${args.model} (the configured production endpoint) \n` +
177
+ `Prompt: the CURRENT production importanceScoring (whatever ships in this tree)\n`);
178
+ if (args.aborted) {
179
+ L.push(`**ABORTED MID-RUN** — ${args.error ?? "endpoint error"}; partial distribution over ${s.n} scores below.\n`);
180
+ }
181
+ L.push("## Distribution\n");
182
+ L.push("| stat | value |");
183
+ L.push("|---|---|");
184
+ L.push(`| n | ${s.n} |`);
185
+ L.push(`| min | ${s.min.toFixed(3)} |`);
186
+ L.push(`| p25 | ${s.p25.toFixed(3)} |`);
187
+ L.push(`| **median** | **${s.median.toFixed(3)}** |`);
188
+ L.push(`| p75 | ${s.p75.toFixed(3)} |`);
189
+ L.push(`| **p90** | **${s.p90.toFixed(3)}** |`);
190
+ L.push(`| max | ${s.max.toFixed(3)} |`);
191
+ L.push(`| at exactly 1.0 | ${s.atExactly1} |`);
192
+ L.push("");
193
+ L.push("| bucket | count |");
194
+ L.push("|---|---|");
195
+ for (const [bucket, count] of Object.entries(s.histogram)) {
196
+ L.push(`| ${bucket} | ${count} |`);
197
+ }
198
+ L.push("");
199
+ L.push(`Target band (owner D2, 2026-09-13): median 0.30-0.40 → ` +
200
+ `${s.median >= 0.3 && s.median <= 0.4 ? "**PASS**" : "**MISS**"} (${s.median.toFixed(3)}); ` +
201
+ `p90 <= 0.75 → ${s.p90 <= 0.75 ? "**PASS**" : "**MISS**"} (${s.p90.toFixed(3)}); ` +
202
+ `zero at 1.0 → ${s.atExactly1 === 0 ? "**PASS**" : "**MISS**"} (${s.atExactly1})\n`);
203
+ return L.join("\n");
204
+ }
205
+ // ---------------------------------------------------------------------------
206
+ // CLI
207
+ // ---------------------------------------------------------------------------
208
+ function usage() {
209
+ console.error("Usage: npm run eval:importance -- <snapshot.db> [--sample N=200] [--seed S=1]\n" +
210
+ " --llm-base-url U --llm-model M --llm-api-key K (required: the scoring endpoint)\n" +
211
+ " [--llm-thinking] (thinking ON; default OFF — production parity)\n" +
212
+ " [--report <out.md>]");
213
+ process.exit(1);
214
+ }
215
+ async function main() {
216
+ const argv = process.argv.slice(2);
217
+ const positional = [];
218
+ const flag = (name) => {
219
+ const i = argv.indexOf(name);
220
+ return i !== -1 && argv[i + 1] && !argv[i + 1].startsWith("--") ? argv[i + 1] : undefined;
221
+ };
222
+ for (let i = 0; i < argv.length; i++) {
223
+ if (!argv[i].startsWith("--"))
224
+ positional.push(argv[i]);
225
+ }
226
+ const snapshotPath = positional[0];
227
+ if (!snapshotPath || argv.includes("--help") || argv.includes("-h"))
228
+ usage();
229
+ const baseUrl = flag("--llm-base-url");
230
+ const model = flag("--llm-model");
231
+ const apiKey = flag("--llm-api-key");
232
+ if (!baseUrl || !model || !apiKey) {
233
+ console.error("[importance-eval] --llm-base-url, --llm-model and --llm-api-key are required (the production scoring endpoint)");
234
+ usage();
235
+ }
236
+ const sample = flag("--sample") ? parseInt(flag("--sample"), 10) : DEFAULT_SAMPLE;
237
+ const seed = flag("--seed") ? parseInt(flag("--seed"), 10) : DEFAULT_SEED;
238
+ if (!Number.isInteger(sample) || sample < 10 || sample > 5000) {
239
+ console.error("[importance-eval] --sample must be an integer in [10, 5000]");
240
+ process.exit(1);
241
+ }
242
+ if (!Number.isInteger(seed) || seed <= 0) {
243
+ console.error("[importance-eval] --seed must be a positive integer");
244
+ process.exit(1);
245
+ }
246
+ // OpenAI-compatible config built from flags ONLY (never hardcoded; the
247
+ // production endpoint is a runtime input, not a shipped value). Thinking
248
+ // defaults OFF (production parity: the scorer runs with thinking disabled —
249
+ // an unclosed think block can eat the whole output budget); opt in with
250
+ // --llm-thinking for a thinking-on endpoint.
251
+ const config = {
252
+ baseUrl,
253
+ model,
254
+ apiKey,
255
+ provider: "openai",
256
+ enableThinking: argv.includes("--llm-thinking") ? true : false,
257
+ };
258
+ const llm = new llm_js_1.LlmClient(config);
259
+ const t0 = Date.now();
260
+ const result = await runImportanceEval({
261
+ snapshotPath,
262
+ sample,
263
+ seed,
264
+ llm,
265
+ });
266
+ const report = renderImportanceReport({
267
+ snapshotPath,
268
+ summary: result.summary,
269
+ aborted: result.aborted,
270
+ error: result.error,
271
+ model,
272
+ });
273
+ const reportPath = flag("--report") ?? (0, node_path_1.join)(process.cwd(), "data", "importance-eval-report.md");
274
+ (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(reportPath), { recursive: true });
275
+ (0, node_fs_1.writeFileSync)(reportPath, report, "utf-8");
276
+ console.log(`[importance-eval] report written: ${reportPath}`);
277
+ console.log(report);
278
+ console.log(`[importance-eval] ${result.aborted ? "ABORTED" : "complete"} in ${Math.round((Date.now() - t0) / 1000)}s`);
279
+ process.exitCode = result.aborted ? 1 : 0;
280
+ }
281
+ if (process.argv[1] && process.argv[1].endsWith("importance-eval.js")) {
282
+ main().catch((err) => {
283
+ console.error("[importance-eval] FAILED:", err instanceof Error ? err.stack : String(err));
284
+ process.exitCode = 1;
285
+ });
286
+ }
@@ -0,0 +1,26 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Planted-pairs resolution gate — the #393 increment A acceptance harness.
4
+ *
5
+ * npm run eval:planted -- [report.md] [--snapshot <db>]
6
+ *
7
+ * Builds the real-text planted corpus (planted-fixtures.ts) in a THROWAWAY
8
+ * writable DB, embeds it with the REAL bge-small-en-v1.5 embedder, runs the
9
+ * production resolution stage (`stageReconsolidation`, ground-truth stub
10
+ * judge — see planted-harness.ts), and writes the DETECTED / BOUND / RESOLVED
11
+ * gate table. This is the before/after photo every later #393 increment
12
+ * (scout, guard, walk) must re-run and beat.
13
+ *
14
+ * `--snapshot <db>` plants the same corpus into a COPY of a real snapshot
15
+ * instead of an empty DB (the real-corpus background). The copy is opened
16
+ * writable via initDb (migrations are idempotent) — the snapshot itself is
17
+ * never touched. With a fresh temp stateDir the stage re-scans the ENTIRE
18
+ * corpus (cursor 0) — on a 17k-memory snapshot that is a full pass with
19
+ * stub-verdict calls for every unlinked in-band pair (fast, no network); the
20
+ * live-LLM real-corpus acceptance run is the release-soak step (refine Q3),
21
+ * not this one.
22
+ *
23
+ * Never touches ~/.hicortex: DB, state dir, and backups all live under a
24
+ * mkdtemp directory that is removed on exit.
25
+ */
26
+ export {};
@@ -0,0 +1,97 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * Planted-pairs resolution gate — the #393 increment A acceptance harness.
5
+ *
6
+ * npm run eval:planted -- [report.md] [--snapshot <db>]
7
+ *
8
+ * Builds the real-text planted corpus (planted-fixtures.ts) in a THROWAWAY
9
+ * writable DB, embeds it with the REAL bge-small-en-v1.5 embedder, runs the
10
+ * production resolution stage (`stageReconsolidation`, ground-truth stub
11
+ * judge — see planted-harness.ts), and writes the DETECTED / BOUND / RESOLVED
12
+ * gate table. This is the before/after photo every later #393 increment
13
+ * (scout, guard, walk) must re-run and beat.
14
+ *
15
+ * `--snapshot <db>` plants the same corpus into a COPY of a real snapshot
16
+ * instead of an empty DB (the real-corpus background). The copy is opened
17
+ * writable via initDb (migrations are idempotent) — the snapshot itself is
18
+ * never touched. With a fresh temp stateDir the stage re-scans the ENTIRE
19
+ * corpus (cursor 0) — on a 17k-memory snapshot that is a full pass with
20
+ * stub-verdict calls for every unlinked in-band pair (fast, no network); the
21
+ * live-LLM real-corpus acceptance run is the release-soak step (refine Q3),
22
+ * not this one.
23
+ *
24
+ * Never touches ~/.hicortex: DB, state dir, and backups all live under a
25
+ * mkdtemp directory that is removed on exit.
26
+ */
27
+ Object.defineProperty(exports, "__esModule", { value: true });
28
+ const node_fs_1 = require("node:fs");
29
+ const node_os_1 = require("node:os");
30
+ const node_path_1 = require("node:path");
31
+ const embedder_js_1 = require("../embedder.js");
32
+ const capture_js_1 = require("../capture.js");
33
+ const planted_fixtures_js_1 = require("./planted-fixtures.js");
34
+ const planted_harness_js_1 = require("./planted-harness.js");
35
+ const DEFAULT_REPORT_PATH = (0, node_path_1.join)(process.cwd(), "data", "planted-eval-report.md");
36
+ function usage() {
37
+ console.error("Usage: npm run eval:planted -- [report.md] [--snapshot <snapshot.db>]\n" +
38
+ " [report.md] output path (default data/planted-eval-report.md)\n" +
39
+ " --snapshot <db> plant into a COPY of a real snapshot (readonly source;\n" +
40
+ " the copy is writable, the snapshot is never touched)");
41
+ process.exitCode = 1;
42
+ process.exit(1);
43
+ }
44
+ async function main() {
45
+ const argv = process.argv.slice(2);
46
+ let reportPath;
47
+ let snapshotPath;
48
+ for (let i = 0; i < argv.length; i++) {
49
+ if (argv[i] === "--snapshot" && argv[i + 1]) {
50
+ snapshotPath = argv[++i];
51
+ continue;
52
+ }
53
+ if (!argv[i].startsWith("--"))
54
+ reportPath = argv[i];
55
+ }
56
+ if (argv.includes("--help") || argv.includes("-h"))
57
+ usage();
58
+ const workDir = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-planted-eval-"));
59
+ try {
60
+ const dbPath = (0, node_path_1.join)(workDir, "planted.db");
61
+ if (snapshotPath) {
62
+ (0, node_fs_1.copyFileSync)(snapshotPath, dbPath);
63
+ console.log(`[planted-eval] snapshot copied (readonly source): ${snapshotPath}`);
64
+ }
65
+ const stateDir = (0, node_path_1.join)(workDir, "state");
66
+ console.log("[planted-eval] loading real bge-small-en-v1.5 embedder...");
67
+ const t0 = Date.now();
68
+ const result = await (0, planted_harness_js_1.runPlantedGate)(planted_fixtures_js_1.REAL_TEXT_CORPUS, {
69
+ dbPath,
70
+ stateDir,
71
+ embedFn: embedder_js_1.embed,
72
+ acquireLock: capture_js_1.acquireCaptureLock, // real lock, but on the temp stateDir only
73
+ });
74
+ console.log(`[planted-eval] gate completed in ${Date.now() - t0}ms`);
75
+ const report = (0, planted_harness_js_1.renderPlantedReport)({
76
+ corpusLabel: snapshotPath
77
+ ? `real-text planted corpus ON snapshot copy \`${snapshotPath}\``
78
+ : "real-text planted corpus (synthetic texts, real embeddings)",
79
+ generatedAt: new Date().toISOString(),
80
+ embedderLabel: "real bge-small-en-v1.5 (Xenova, fp32) — cosines measured, never asserted",
81
+ result,
82
+ });
83
+ const outPath = reportPath ?? DEFAULT_REPORT_PATH;
84
+ (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(outPath), { recursive: true });
85
+ (0, node_fs_1.writeFileSync)(outPath, report, "utf-8");
86
+ console.log(`[planted-eval] report written: ${outPath}\n`);
87
+ // Stdout mirror of the gate tables (the issue-comment shaped summary).
88
+ console.log(report);
89
+ }
90
+ finally {
91
+ (0, node_fs_1.rmSync)(workDir, { recursive: true, force: true });
92
+ }
93
+ }
94
+ main().catch((err) => {
95
+ console.error("[planted-eval] FAILED:", err instanceof Error ? err.stack : String(err));
96
+ process.exitCode = 1;
97
+ });
@@ -0,0 +1,107 @@
1
+ /**
2
+ * Planted-pair fixture corpus for the resolution acceptance gate (#393
3
+ * increment A — the eval-first harness).
4
+ *
5
+ * Five failure classes, each encoded as planted memory pairs with a DECLARED
6
+ * ground-truth relation and a DECLARED expected cosine band:
7
+ *
8
+ * (a) cross_topic_correction — the correction rides inside a topically
9
+ * unrelated experience memory; measured cosine ~0.5-0.6, BELOW the
10
+ * `correctionMinSimilarity` floor (0.75) — the documented field hole
11
+ * (0 `corrects` verdicts in 246 on the 17k snapshot).
12
+ * (b) source_conflict — two live records that are both true AS RECORDS (each
13
+ * faithfully reports a different source) at >= the deterministic-zone
14
+ * ceiling (0.92). Guard-C closes the hole: the scout flags the newer
15
+ * record contradiction-shaped, the judge answers `conflicts`, the link
16
+ * guards the zone, and both stay live.
17
+ * (c) version_chain — v1 -> v2 -> v3 of a revised decision record: the
18
+ * 0.78-0.96 stale-churn class (#392 owner calibration). v1-v2 sits in
19
+ * the judged band; v2-v3 (numeric status churn inside a long shared
20
+ * record) sits above the ceiling, where the zone blends the NEWEST
21
+ * version into the STALE one (canonical = oldest created_at).
22
+ * (d) same_fact_paraphrase — one fact, two wordings: should merge.
23
+ * (e) complementary_facets — same subject, different facets: should keep
24
+ * both (judge verdict `none`).
25
+ *
26
+ * Honesty rules:
27
+ * - The texts are synthetic and generic (this package publishes to npm — no
28
+ * real infrastructure, people, or fleet detail), but SHAPED like the field
29
+ * failures: corrections ride in experience memories, conflicts share long
30
+ * context tails, churn edits one field of a long record.
31
+ * - Cosines are MEASURED at run time with the real bge-small-en-v1.5 embedder
32
+ * (planted-eval.ts) — never asserted into existence. The declared bands
33
+ * below were calibrated against that embedder (see the band table in the
34
+ * report); a pair landing outside its band renders as a loud OUT-OF-BAND
35
+ * flag rather than a hard failure, so an embedder change is visible, not
36
+ * silent (same posture as run-eval).
37
+ * - Neutral filler rows keep the KNN neighborhoods non-trivial and give the
38
+ * collider analysis something to check (any non-ground-truth pair >= 0.70
39
+ * is reported).
40
+ */
41
+ /** The five failure classes the gate measures (one row per planted pair). */
42
+ export type FixtureClass = "cross_topic_correction" | "source_conflict" | "version_chain" | "same_fact_paraphrase" | "complementary_facets";
43
+ /**
44
+ * Ground truth for a planted pair — what a PERFECT judge would verdict
45
+ * (`none` here means "related but distinct — keep both"; every other
46
+ * relation is a verdict action since guard-C added `conflicts`).
47
+ */
48
+ export type GroundTruthRelation = "corrects" | "supersedes" | "merge" | "conflict" | "none";
49
+ export interface PlantedRow {
50
+ /** Stable identifier used by pairs, reports, and the ground-truth judge. */
51
+ key: string;
52
+ content: string;
53
+ /** ISO timestamp. Ordering is LOAD-BEARING: detection only pairs older -> newer. */
54
+ createdAt: string;
55
+ /** knowledge | experience | decision — drives the rewrite/mark-only fork. */
56
+ memoryType: string;
57
+ }
58
+ export interface PlantedPair {
59
+ class: FixtureClass;
60
+ olderKey: string;
61
+ newerKey: string;
62
+ relation: GroundTruthRelation;
63
+ /** Expected measured-cosine band [lo, hi] (inclusive). */
64
+ band: [number, number];
65
+ /**
66
+ * For `corrects` pairs: the scripted corrected text the ground-truth judge
67
+ * returns from the rewrite contract (REQUIRED for corrects — the harness
68
+ * throws if a corrects verdict ever reaches its rewrite call without one).
69
+ */
70
+ rewritten?: string;
71
+ /**
72
+ * For resolution-shaped pairs (`corrects`, `supersedes`, AND `conflict`):
73
+ * the distinctive terms of the referenced OLD claim, as a well-behaved
74
+ * scout LLM would extract them from the NEWER memory (REQUIRED for those
75
+ * relations — the ground-truth judge throws if the shape call reaches a
76
+ * resolution-shaped pair without one; guard-C added `conflict` because two
77
+ * records disagreeing on one quantity share MORE wording than a
78
+ * cross-topic correction does). Honesty rule: every term must appear
79
+ * VERBATIM in the older row (that is the field-failure mechanism — the
80
+ * correction contains the words of what it corrects; the harness verifies
81
+ * it in tests/planted-pairs.test.ts).
82
+ */
83
+ references?: string;
84
+ }
85
+ export interface PlantedCorpus {
86
+ /** Insertion order — keep it createdAt-ascending so rowid order matches age. */
87
+ rows: PlantedRow[];
88
+ pairs: PlantedPair[];
89
+ }
90
+ /** All planted rows share project + source_agent so the metadata rails never refuse a fixture merge. */
91
+ export declare const PLANTED_PROJECT = "planted-eval";
92
+ export declare const PLANTED_SOURCE_AGENT = "planted-eval";
93
+ /**
94
+ * Measured with the real bge-small-en-v1.5 embedder at fixture-authoring time
95
+ * (2026-09-12): cross-topic 0.625, conflict 0.993, chain v1-v2 0.788 (judged
96
+ * band), v2-v3 0.998 (zone band), paraphrase 0.985, facets 0.774; no
97
+ * non-ground-truth pair >= 0.70. Bands are set with margin around those
98
+ * values — the report re-measures and flags drift.
99
+ */
100
+ export declare const REAL_TEXT_CORPUS: PlantedCorpus;
101
+ /**
102
+ * The version_chain class is measured BOTH per-pair and as a class: the
103
+ * desired end state is the chain's TERMINAL (newest member) live with every
104
+ * non-terminal demoted or absorbed. Members are derived from the declared
105
+ * pairs, ordered by createdAt.
106
+ */
107
+ export declare function chainMembers(corpus: PlantedCorpus): string[];