@gamaze/hicortex 0.20.7 → 0.20.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/README.md +10 -41
  2. package/dist/calibration.d.ts +174 -0
  3. package/dist/calibration.js +231 -0
  4. package/dist/capture.d.ts +15 -3
  5. package/dist/capture.js +10 -1
  6. package/dist/classify-domains.d.ts +6 -0
  7. package/dist/classify-domains.js +7 -1
  8. package/dist/cli.js +2 -3
  9. package/dist/config-read.d.ts +1 -1
  10. package/dist/config-read.js +96 -9
  11. package/dist/consolidate.d.ts +79 -68
  12. package/dist/consolidate.js +218 -174
  13. package/dist/dashboard.d.ts +4 -3
  14. package/dist/dedup.d.ts +34 -26
  15. package/dist/dedup.js +91 -57
  16. package/dist/distiller.js +1 -1
  17. package/dist/domain-classify.d.ts +7 -6
  18. package/dist/domain-classify.js +12 -10
  19. package/dist/eval/decay-eval.d.ts +3 -3
  20. package/dist/eval/decay-eval.js +4 -4
  21. package/dist/eval/planted-eval.d.ts +26 -0
  22. package/dist/eval/planted-eval.js +97 -0
  23. package/dist/eval/planted-fixtures.d.ts +107 -0
  24. package/dist/eval/planted-fixtures.js +283 -0
  25. package/dist/eval/planted-harness.d.ts +176 -0
  26. package/dist/eval/planted-harness.js +649 -0
  27. package/dist/index.js +4 -3
  28. package/dist/init.d.ts +9 -3
  29. package/dist/init.js +52 -9
  30. package/dist/llm.d.ts +43 -58
  31. package/dist/llm.js +87 -101
  32. package/dist/mcp-server.js +29 -29
  33. package/dist/nightly.js +105 -103
  34. package/dist/nofit.d.ts +4 -11
  35. package/dist/nofit.js +6 -23
  36. package/dist/recall-index.d.ts +30 -28
  37. package/dist/recall-index.js +21 -18
  38. package/dist/recall-registry.d.ts +2 -1
  39. package/dist/recall-registry.js +35 -1
  40. package/dist/reconsolidation.d.ts +124 -72
  41. package/dist/reconsolidation.js +359 -148
  42. package/dist/relink.js +3 -4
  43. package/dist/retrieval.d.ts +68 -35
  44. package/dist/retrieval.js +292 -104
  45. package/dist/run-deadline.d.ts +62 -0
  46. package/dist/run-deadline.js +73 -0
  47. package/dist/schema-prototypes.d.ts +3 -3
  48. package/dist/schema-prototypes.js +3 -3
  49. package/dist/state.d.ts +2 -3
  50. package/dist/storage.d.ts +16 -16
  51. package/dist/storage.js +62 -24
  52. package/dist/telemetry.d.ts +8 -7
  53. package/dist/token-budget.js +3 -4
  54. package/dist/type-classify.js +4 -4
  55. package/dist/types.d.ts +95 -155
  56. package/domains.example.json +4 -5
  57. package/hermes-plugin/hicortex/README.md +2 -2
  58. package/openclaw.plugin.json +1 -1
  59. package/package.json +2 -1
  60. package/pi-extension/hicortex/README.md +1 -1
  61. package/server.json +3 -3
@@ -0,0 +1,97 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * Planted-pairs resolution gate — the #393 increment A acceptance harness.
5
+ *
6
+ * npm run eval:planted -- [report.md] [--snapshot <db>]
7
+ *
8
+ * Builds the real-text planted corpus (planted-fixtures.ts) in a THROWAWAY
9
+ * writable DB, embeds it with the REAL bge-small-en-v1.5 embedder, runs the
10
+ * production resolution stage (`stageReconsolidation`, ground-truth stub
11
+ * judge — see planted-harness.ts), and writes the DETECTED / BOUND / RESOLVED
12
+ * gate table. This is the before/after photo every later #393 increment
13
+ * (scout, guard, walk) must re-run and beat.
14
+ *
15
+ * `--snapshot <db>` plants the same corpus into a COPY of a real snapshot
16
+ * instead of an empty DB (the real-corpus background). The copy is opened
17
+ * writable via initDb (migrations are idempotent) — the snapshot itself is
18
+ * never touched. With a fresh temp stateDir the stage re-scans the ENTIRE
19
+ * corpus (cursor 0) — on a 17k-memory snapshot that is a full pass with
20
+ * stub-verdict calls for every unlinked in-band pair (fast, no network); the
21
+ * live-LLM real-corpus acceptance run is the release-soak step (refine Q3),
22
+ * not this one.
23
+ *
24
+ * Never touches ~/.hicortex: DB, state dir, and backups all live under a
25
+ * mkdtemp directory that is removed on exit.
26
+ */
27
+ Object.defineProperty(exports, "__esModule", { value: true });
28
+ const node_fs_1 = require("node:fs");
29
+ const node_os_1 = require("node:os");
30
+ const node_path_1 = require("node:path");
31
+ const embedder_js_1 = require("../embedder.js");
32
+ const capture_js_1 = require("../capture.js");
33
+ const planted_fixtures_js_1 = require("./planted-fixtures.js");
34
+ const planted_harness_js_1 = require("./planted-harness.js");
35
+ const DEFAULT_REPORT_PATH = (0, node_path_1.join)(process.cwd(), "data", "planted-eval-report.md");
36
+ function usage() {
37
+ console.error("Usage: npm run eval:planted -- [report.md] [--snapshot <snapshot.db>]\n" +
38
+ " [report.md] output path (default data/planted-eval-report.md)\n" +
39
+ " --snapshot <db> plant into a COPY of a real snapshot (readonly source;\n" +
40
+ " the copy is writable, the snapshot is never touched)");
41
+ process.exitCode = 1;
42
+ process.exit(1);
43
+ }
44
+ async function main() {
45
+ const argv = process.argv.slice(2);
46
+ let reportPath;
47
+ let snapshotPath;
48
+ for (let i = 0; i < argv.length; i++) {
49
+ if (argv[i] === "--snapshot" && argv[i + 1]) {
50
+ snapshotPath = argv[++i];
51
+ continue;
52
+ }
53
+ if (!argv[i].startsWith("--"))
54
+ reportPath = argv[i];
55
+ }
56
+ if (argv.includes("--help") || argv.includes("-h"))
57
+ usage();
58
+ const workDir = (0, node_fs_1.mkdtempSync)((0, node_path_1.join)((0, node_os_1.tmpdir)(), "hicortex-planted-eval-"));
59
+ try {
60
+ const dbPath = (0, node_path_1.join)(workDir, "planted.db");
61
+ if (snapshotPath) {
62
+ (0, node_fs_1.copyFileSync)(snapshotPath, dbPath);
63
+ console.log(`[planted-eval] snapshot copied (readonly source): ${snapshotPath}`);
64
+ }
65
+ const stateDir = (0, node_path_1.join)(workDir, "state");
66
+ console.log("[planted-eval] loading real bge-small-en-v1.5 embedder...");
67
+ const t0 = Date.now();
68
+ const result = await (0, planted_harness_js_1.runPlantedGate)(planted_fixtures_js_1.REAL_TEXT_CORPUS, {
69
+ dbPath,
70
+ stateDir,
71
+ embedFn: embedder_js_1.embed,
72
+ acquireLock: capture_js_1.acquireCaptureLock, // real lock, but on the temp stateDir only
73
+ });
74
+ console.log(`[planted-eval] gate completed in ${Date.now() - t0}ms`);
75
+ const report = (0, planted_harness_js_1.renderPlantedReport)({
76
+ corpusLabel: snapshotPath
77
+ ? `real-text planted corpus ON snapshot copy \`${snapshotPath}\``
78
+ : "real-text planted corpus (synthetic texts, real embeddings)",
79
+ generatedAt: new Date().toISOString(),
80
+ embedderLabel: "real bge-small-en-v1.5 (Xenova, fp32) — cosines measured, never asserted",
81
+ result,
82
+ });
83
+ const outPath = reportPath ?? DEFAULT_REPORT_PATH;
84
+ (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(outPath), { recursive: true });
85
+ (0, node_fs_1.writeFileSync)(outPath, report, "utf-8");
86
+ console.log(`[planted-eval] report written: ${outPath}\n`);
87
+ // Stdout mirror of the gate tables (the issue-comment shaped summary).
88
+ console.log(report);
89
+ }
90
+ finally {
91
+ (0, node_fs_1.rmSync)(workDir, { recursive: true, force: true });
92
+ }
93
+ }
94
+ main().catch((err) => {
95
+ console.error("[planted-eval] FAILED:", err instanceof Error ? err.stack : String(err));
96
+ process.exitCode = 1;
97
+ });
@@ -0,0 +1,107 @@
1
+ /**
2
+ * Planted-pair fixture corpus for the resolution acceptance gate (#393
3
+ * increment A — the eval-first harness).
4
+ *
5
+ * Five failure classes, each encoded as planted memory pairs with a DECLARED
6
+ * ground-truth relation and a DECLARED expected cosine band:
7
+ *
8
+ * (a) cross_topic_correction — the correction rides inside a topically
9
+ * unrelated experience memory; measured cosine ~0.5-0.6, BELOW the
10
+ * `correctionMinSimilarity` floor (0.75) — the documented field hole
11
+ * (0 `corrects` verdicts in 246 on the 17k snapshot).
12
+ * (b) source_conflict — two live records that are both true AS RECORDS (each
13
+ * faithfully reports a different source) at >= the deterministic-zone
14
+ * ceiling (0.92). Guard-C closes the hole: the scout flags the newer
15
+ * record contradiction-shaped, the judge answers `conflicts`, the link
16
+ * guards the zone, and both stay live.
17
+ * (c) version_chain — v1 -> v2 -> v3 of a revised decision record: the
18
+ * 0.78-0.96 stale-churn class (#392 owner calibration). v1-v2 sits in
19
+ * the judged band; v2-v3 (numeric status churn inside a long shared
20
+ * record) sits above the ceiling, where the zone blends the NEWEST
21
+ * version into the STALE one (canonical = oldest created_at).
22
+ * (d) same_fact_paraphrase — one fact, two wordings: should merge.
23
+ * (e) complementary_facets — same subject, different facets: should keep
24
+ * both (judge verdict `none`).
25
+ *
26
+ * Honesty rules:
27
+ * - The texts are synthetic and generic (this package publishes to npm — no
28
+ * real infrastructure, people, or fleet detail), but SHAPED like the field
29
+ * failures: corrections ride in experience memories, conflicts share long
30
+ * context tails, churn edits one field of a long record.
31
+ * - Cosines are MEASURED at run time with the real bge-small-en-v1.5 embedder
32
+ * (planted-eval.ts) — never asserted into existence. The declared bands
33
+ * below were calibrated against that embedder (see the band table in the
34
+ * report); a pair landing outside its band renders as a loud OUT-OF-BAND
35
+ * flag rather than a hard failure, so an embedder change is visible, not
36
+ * silent (same posture as run-eval).
37
+ * - Neutral filler rows keep the KNN neighborhoods non-trivial and give the
38
+ * collider analysis something to check (any non-ground-truth pair >= 0.70
39
+ * is reported).
40
+ */
41
+ /** The five failure classes the gate measures (one row per planted pair). */
42
+ export type FixtureClass = "cross_topic_correction" | "source_conflict" | "version_chain" | "same_fact_paraphrase" | "complementary_facets";
43
+ /**
44
+ * Ground truth for a planted pair — what a PERFECT judge would verdict
45
+ * (`none` here means "related but distinct — keep both"; every other
46
+ * relation is a verdict action since guard-C added `conflicts`).
47
+ */
48
+ export type GroundTruthRelation = "corrects" | "supersedes" | "merge" | "conflict" | "none";
49
+ export interface PlantedRow {
50
+ /** Stable identifier used by pairs, reports, and the ground-truth judge. */
51
+ key: string;
52
+ content: string;
53
+ /** ISO timestamp. Ordering is LOAD-BEARING: detection only pairs older -> newer. */
54
+ createdAt: string;
55
+ /** knowledge | experience | decision — drives the rewrite/mark-only fork. */
56
+ memoryType: string;
57
+ }
58
+ export interface PlantedPair {
59
+ class: FixtureClass;
60
+ olderKey: string;
61
+ newerKey: string;
62
+ relation: GroundTruthRelation;
63
+ /** Expected measured-cosine band [lo, hi] (inclusive). */
64
+ band: [number, number];
65
+ /**
66
+ * For `corrects` pairs: the scripted corrected text the ground-truth judge
67
+ * returns from the rewrite contract (REQUIRED for corrects — the harness
68
+ * throws if a corrects verdict ever reaches the rewrite phase without one).
69
+ */
70
+ rewritten?: string;
71
+ /**
72
+ * For resolution-shaped pairs (`corrects`, `supersedes`, AND `conflict`):
73
+ * the distinctive terms of the referenced OLD claim, as a well-behaved
74
+ * scout LLM would extract them from the NEWER memory (REQUIRED for those
75
+ * relations — the ground-truth judge throws if the shape call reaches a
76
+ * resolution-shaped pair without one; guard-C added `conflict` because two
77
+ * records disagreeing on one quantity share MORE wording than a
78
+ * cross-topic correction does). Honesty rule: every term must appear
79
+ * VERBATIM in the older row (that is the field-failure mechanism — the
80
+ * correction contains the words of what it corrects; the harness verifies
81
+ * it in tests/planted-pairs.test.ts).
82
+ */
83
+ references?: string;
84
+ }
85
+ export interface PlantedCorpus {
86
+ /** Insertion order — keep it createdAt-ascending so rowid order matches age. */
87
+ rows: PlantedRow[];
88
+ pairs: PlantedPair[];
89
+ }
90
+ /** All planted rows share project + source_agent so the metadata rails never refuse a fixture merge. */
91
+ export declare const PLANTED_PROJECT = "planted-eval";
92
+ export declare const PLANTED_SOURCE_AGENT = "planted-eval";
93
+ /**
94
+ * Measured with the real bge-small-en-v1.5 embedder at fixture-authoring time
95
+ * (2026-09-12): cross-topic 0.625, conflict 0.993, chain v1-v2 0.788 (judged
96
+ * band), v2-v3 0.998 (zone band), paraphrase 0.985, facets 0.774; no
97
+ * non-ground-truth pair >= 0.70. Bands are set with margin around those
98
+ * values — the report re-measures and flags drift.
99
+ */
100
+ export declare const REAL_TEXT_CORPUS: PlantedCorpus;
101
+ /**
102
+ * The version_chain class is measured BOTH per-pair and as a class: the
103
+ * desired end state is the chain's TERMINAL (newest member) live with every
104
+ * non-terminal demoted or absorbed. Members are derived from the declared
105
+ * pairs, ordered by createdAt.
106
+ */
107
+ export declare function chainMembers(corpus: PlantedCorpus): string[];
@@ -0,0 +1,283 @@
1
+ "use strict";
2
+ /**
3
+ * Planted-pair fixture corpus for the resolution acceptance gate (#393
4
+ * increment A — the eval-first harness).
5
+ *
6
+ * Five failure classes, each encoded as planted memory pairs with a DECLARED
7
+ * ground-truth relation and a DECLARED expected cosine band:
8
+ *
9
+ * (a) cross_topic_correction — the correction rides inside a topically
10
+ * unrelated experience memory; measured cosine ~0.5-0.6, BELOW the
11
+ * `correctionMinSimilarity` floor (0.75) — the documented field hole
12
+ * (0 `corrects` verdicts in 246 on the 17k snapshot).
13
+ * (b) source_conflict — two live records that are both true AS RECORDS (each
14
+ * faithfully reports a different source) at >= the deterministic-zone
15
+ * ceiling (0.92). Guard-C closes the hole: the scout flags the newer
16
+ * record contradiction-shaped, the judge answers `conflicts`, the link
17
+ * guards the zone, and both stay live.
18
+ * (c) version_chain — v1 -> v2 -> v3 of a revised decision record: the
19
+ * 0.78-0.96 stale-churn class (#392 owner calibration). v1-v2 sits in
20
+ * the judged band; v2-v3 (numeric status churn inside a long shared
21
+ * record) sits above the ceiling, where the zone blends the NEWEST
22
+ * version into the STALE one (canonical = oldest created_at).
23
+ * (d) same_fact_paraphrase — one fact, two wordings: should merge.
24
+ * (e) complementary_facets — same subject, different facets: should keep
25
+ * both (judge verdict `none`).
26
+ *
27
+ * Honesty rules:
28
+ * - The texts are synthetic and generic (this package publishes to npm — no
29
+ * real infrastructure, people, or fleet detail), but SHAPED like the field
30
+ * failures: corrections ride in experience memories, conflicts share long
31
+ * context tails, churn edits one field of a long record.
32
+ * - Cosines are MEASURED at run time with the real bge-small-en-v1.5 embedder
33
+ * (planted-eval.ts) — never asserted into existence. The declared bands
34
+ * below were calibrated against that embedder (see the band table in the
35
+ * report); a pair landing outside its band renders as a loud OUT-OF-BAND
36
+ * flag rather than a hard failure, so an embedder change is visible, not
37
+ * silent (same posture as run-eval).
38
+ * - Neutral filler rows keep the KNN neighborhoods non-trivial and give the
39
+ * collider analysis something to check (any non-ground-truth pair >= 0.70
40
+ * is reported).
41
+ */
42
+ Object.defineProperty(exports, "__esModule", { value: true });
43
+ exports.REAL_TEXT_CORPUS = exports.PLANTED_SOURCE_AGENT = exports.PLANTED_PROJECT = void 0;
44
+ exports.chainMembers = chainMembers;
45
+ /** All planted rows share project + source_agent so the metadata rails never refuse a fixture merge. */
46
+ exports.PLANTED_PROJECT = "planted-eval";
47
+ exports.PLANTED_SOURCE_AGENT = "planted-eval";
48
+ // ---------------------------------------------------------------------------
49
+ // The real-text corpus (embedded with the REAL embedder by planted-eval.ts)
50
+ // ---------------------------------------------------------------------------
51
+ const FILLER = [
52
+ [
53
+ "fill_recipe",
54
+ "Batch-cooked a tomato and butter sauce from the Rome cookbook: San Marzano tomatoes, one whole onion halved, and far more butter than seems reasonable. Simmered 45 minutes. The whole family preferred it to last month's pesto attempt.",
55
+ "2026-06-06T09:00:00.000Z",
56
+ ],
57
+ [
58
+ "fill_run",
59
+ "Half-marathon training week 6: three easy runs at conversational pace, one interval session on the track, and a 16-kilometer long run on Sunday. Left knee complains on downhills; foam rolling after every session keeps it quiet.",
60
+ "2026-06-07T09:00:00.000Z",
61
+ ],
62
+ [
63
+ "fill_books",
64
+ "Finished the third novel in the frontier trilogy. The middle book dragged through the mining-town chapters, but the finale pays everything off when the surveyor returns to the flooded valley and finally reads her grandmother's letters.",
65
+ "2026-06-08T09:00:00.000Z",
66
+ ],
67
+ [
68
+ "fill_travel",
69
+ "Train trip through the coastal towns in autumn: two nights in the fishing village with the breakwater walk, then the mountain line with the switchback tunnels. Pack the warm layer even when the departure platform is sunny.",
70
+ "2026-06-09T09:00:00.000Z",
71
+ ],
72
+ [
73
+ "fill_garden",
74
+ "The raised beds need refreshing before spring: compost the spent tomato vines, rotate the legume row to where the squash was, and prune the apple espalier before the buds swell.",
75
+ "2026-06-10T09:00:00.000Z",
76
+ ],
77
+ [
78
+ "fill_lang",
79
+ "Language study notes: the dative prepositions finally clicked after drilling them as a sung list. Irregular verbs still need spaced repetition; the flashcard app's new algorithm keeps resurfacing the ones I already know.",
80
+ "2026-06-21T09:00:00.000Z",
81
+ ],
82
+ [
83
+ "fill_music",
84
+ "Piano practice log: the Chopin nocturne's middle section is still uneven at tempo. Slow practice hands-separately for the polyrhythm bars, then glue the phrases at 80 percent speed before next week's lesson.",
85
+ "2026-06-22T09:00:00.000Z",
86
+ ],
87
+ [
88
+ "fill_car",
89
+ "Car maintenance records: switched the hatchback to the synthetic blend at the 90,000-kilometer service; the mechanic noted a slow weep on the rear shock that we watch for now instead of replacing immediately.",
90
+ "2026-06-23T09:00:00.000Z",
91
+ ],
92
+ ];
93
+ /**
94
+ * Measured with the real bge-small-en-v1.5 embedder at fixture-authoring time
95
+ * (2026-09-12): cross-topic 0.625, conflict 0.993, chain v1-v2 0.788 (judged
96
+ * band), v2-v3 0.998 (zone band), paraphrase 0.985, facets 0.774; no
97
+ * non-ground-truth pair >= 0.70. Bands are set with margin around those
98
+ * values — the report re-measures and flags drift.
99
+ */
100
+ exports.REAL_TEXT_CORPUS = {
101
+ rows: [
102
+ // --- old side (+ first filler block) ---
103
+ {
104
+ key: "xt_old",
105
+ content: "The analytics dashboard refreshes its data every 15 minutes through the scheduled ingestion job.",
106
+ createdAt: "2026-06-01T09:00:00.000Z",
107
+ memoryType: "knowledge",
108
+ },
109
+ {
110
+ key: "conflict_a",
111
+ content: "Per the provider's official pricing page, the vendor API rate limit is 60 requests per minute per account. " +
112
+ "Verified while sizing the sync worker retries; bursts above the limit return HTTP 429 with a Retry-After header.",
113
+ createdAt: "2026-06-02T09:00:00.000Z",
114
+ memoryType: "knowledge",
115
+ },
116
+ {
117
+ key: "chain_v1",
118
+ content: "Deployment strategy (decision record 14): all releases ship through the blue-green pipeline with an atomic " +
119
+ "traffic flip between two identical production environments. Rollback means flipping traffic back to the " +
120
+ "previous environment. Owner: platform team. Migration status: not started. Reviewed quarterly.",
121
+ createdAt: "2026-06-03T09:00:00.000Z",
122
+ memoryType: "decision",
123
+ },
124
+ {
125
+ key: "para_a",
126
+ content: "The team standup happens every weekday morning at 09:30 and never runs longer than fifteen minutes.",
127
+ createdAt: "2026-06-04T09:00:00.000Z",
128
+ memoryType: "knowledge",
129
+ },
130
+ {
131
+ key: "facet_a",
132
+ content: "The team wiki runs on a self-hosted instance behind the company VPN; only the ops group holds " +
133
+ "administrator rights on the wiki instance.",
134
+ createdAt: "2026-06-05T09:00:00.000Z",
135
+ memoryType: "knowledge",
136
+ },
137
+ ...FILLER.slice(0, 5).map(([key, content, createdAt]) => ({
138
+ key,
139
+ content,
140
+ createdAt,
141
+ memoryType: "experience",
142
+ })),
143
+ // --- new side (+ second filler block) ---
144
+ {
145
+ key: "xt_new",
146
+ content: "On-call shift recap: we chased a payment webhook outage for two hours; the root cause was a stale " +
147
+ "connection pool starving the workers under load. While verifying configs afterwards we also found the " +
148
+ "dashboard ingestion schedule quietly moved from every 15 minutes to hourly — the old 15-minute figure " +
149
+ "is outdated.",
150
+ createdAt: "2026-06-15T09:00:00.000Z",
151
+ memoryType: "experience",
152
+ },
153
+ {
154
+ key: "conflict_b",
155
+ content: "Per the provider's engineering wiki, the vendor API rate limit is 100 requests per minute per account. " +
156
+ "Verified while sizing the sync worker retries; bursts above the limit return HTTP 429 with a Retry-After header.",
157
+ createdAt: "2026-06-16T09:00:00.000Z",
158
+ memoryType: "knowledge",
159
+ },
160
+ {
161
+ key: "chain_v2",
162
+ content: "Deployment strategy (decision record 14): all releases now roll out through staged rollouts with automatic " +
163
+ "rollback on error-budget burn. Rollback means halting the rollout and reverting the release. " +
164
+ "Owner: platform team. Migration status: stage 2 of 4 complete. Reviewed quarterly.",
165
+ createdAt: "2026-06-17T09:00:00.000Z",
166
+ memoryType: "decision",
167
+ },
168
+ {
169
+ key: "chain_v3",
170
+ content: "Deployment strategy (decision record 14): all releases now roll out through staged rollouts with automatic " +
171
+ "rollback on error-budget burn. Rollback means halting the rollout and reverting the release. " +
172
+ "Owner: platform team. Migration status: stage 3 of 4 complete. Reviewed quarterly.",
173
+ createdAt: "2026-06-18T09:00:00.000Z",
174
+ memoryType: "decision",
175
+ },
176
+ {
177
+ key: "para_b",
178
+ content: "The team standup is every weekday at 09:30 in the morning; it never runs longer than fifteen minutes.",
179
+ createdAt: "2026-06-19T09:00:00.000Z",
180
+ memoryType: "knowledge",
181
+ },
182
+ {
183
+ key: "facet_b",
184
+ content: "The team wiki instance behind the company VPN is slow to load for read-only users because the " +
185
+ "self-hosted server has no page cache; ops plans to add one.",
186
+ createdAt: "2026-06-20T09:00:00.000Z",
187
+ memoryType: "knowledge",
188
+ },
189
+ ...FILLER.slice(5).map(([key, content, createdAt]) => ({
190
+ key,
191
+ content,
192
+ createdAt,
193
+ memoryType: "experience",
194
+ })),
195
+ ],
196
+ pairs: [
197
+ {
198
+ class: "cross_topic_correction",
199
+ olderKey: "xt_old",
200
+ newerKey: "xt_new",
201
+ relation: "corrects",
202
+ band: [0.4, 0.68],
203
+ rewritten: "The analytics dashboard refreshes its data hourly through the scheduled ingestion job; the refresh " +
204
+ "cadence was reduced from every 15 minutes during the June platform review.",
205
+ // xt_new says "the dashboard ingestion schedule ... moved from every 15
206
+ // minutes to hourly" — these five terms all appear verbatim in xt_old.
207
+ references: "dashboard ingestion every 15 minutes",
208
+ },
209
+ {
210
+ class: "source_conflict",
211
+ olderKey: "conflict_a",
212
+ newerKey: "conflict_b",
213
+ relation: "conflict",
214
+ band: [0.92, 1.0],
215
+ // conflict_b contradicts conflict_a on one quantity while sharing its
216
+ // whole context tail — every token appears verbatim in conflict_a (the
217
+ // guard-C fixture mechanism: the scout's FTS finds the old record, the
218
+ // re-tag rule lets the judge see a >=0.92 pair).
219
+ references: "vendor API rate limit requests per minute per account",
220
+ },
221
+ {
222
+ class: "version_chain",
223
+ olderKey: "chain_v1",
224
+ newerKey: "chain_v2",
225
+ relation: "supersedes",
226
+ band: [0.75, 0.91],
227
+ references: "deployment strategy decision record 14",
228
+ },
229
+ {
230
+ class: "version_chain",
231
+ olderKey: "chain_v2",
232
+ newerKey: "chain_v3",
233
+ relation: "supersedes",
234
+ band: [0.92, 1.0],
235
+ references: "deployment strategy decision record 14",
236
+ },
237
+ {
238
+ class: "version_chain",
239
+ olderKey: "chain_v1",
240
+ newerKey: "chain_v3",
241
+ relation: "supersedes",
242
+ band: [0.75, 0.91],
243
+ references: "deployment strategy decision record 14",
244
+ },
245
+ {
246
+ class: "same_fact_paraphrase",
247
+ olderKey: "para_a",
248
+ newerKey: "para_b",
249
+ relation: "merge",
250
+ band: [0.9, 1.0],
251
+ },
252
+ {
253
+ class: "complementary_facets",
254
+ olderKey: "facet_a",
255
+ newerKey: "facet_b",
256
+ relation: "none",
257
+ band: [0.7, 0.9],
258
+ },
259
+ ],
260
+ };
261
+ /**
262
+ * The version_chain class is measured BOTH per-pair and as a class: the
263
+ * desired end state is the chain's TERMINAL (newest member) live with every
264
+ * non-terminal demoted or absorbed. Members are derived from the declared
265
+ * pairs, ordered by createdAt.
266
+ */
267
+ function chainMembers(corpus) {
268
+ const members = new Set();
269
+ for (const p of corpus.pairs) {
270
+ if (p.class !== "version_chain")
271
+ continue;
272
+ members.add(p.olderKey);
273
+ members.add(p.newerKey);
274
+ }
275
+ const byKey = new Map(corpus.rows.map((r) => [r.key, r]));
276
+ return [...members].sort((a, b) => {
277
+ const ra = byKey.get(a);
278
+ const rb = byKey.get(b);
279
+ const ca = ra?.createdAt ?? "";
280
+ const cb = rb?.createdAt ?? "";
281
+ return ca === cb ? a.localeCompare(b) : ca.localeCompare(cb);
282
+ });
283
+ }
@@ -0,0 +1,176 @@
1
+ /**
2
+ * The planted-pairs acceptance gate (#393 increment A) — machinery shared by
3
+ * the CLI runner (planted-eval.ts, REAL embedder + real-text corpus) and the
4
+ * vitest suite (tests/planted-pairs.test.ts, controlled synthetic vectors).
5
+ *
6
+ * What it measures, per planted pair, before/after running the REAL
7
+ * production resolution stage (`stageReconsolidation`, which internally runs
8
+ * the deterministic merge zone LAST — guard-C: judgment outranks the sweep):
9
+ *
10
+ * DETECTED — the pair entered the resolution machinery as a candidate: a
11
+ * verdict call was made on it (judged band, logged by the ground-truth
12
+ * judge) OR it co-clustered in the pre-stage `planDedup` discovery at the
13
+ * zone ceiling. Tagged by source ("judge" = KNN/similarity, "scout" = the
14
+ * #393 B reference-extraction source — attribution rule: a verdict call
15
+ * on a below-floor pair can only be scout-sourced, "zone" = the
16
+ * deterministic ceiling).
17
+ * BOUND — a resolution structure reflecting the ground truth now connects
18
+ * the pair: `corrected_by` / `superseded_by` / `conflicts` link, or a
19
+ * merge (`dedup_log` row). Guard-C added the `conflicts` bind (the judge
20
+ * flags a genuine conflict instead of leaving it inexpressible).
21
+ * RESOLVED — the class-specific desired END STATE holds in the store:
22
+ * corrects = correction effective (older demoted/absorbed/rewritten) and
23
+ * the correction information live;
24
+ * supersedes= older demoted-or-absorbed and newer live in recall;
25
+ * merge = exactly one live row remains, the other absorbed via dedup;
26
+ * conflict = both live AND not blended AND the conflicts link exists
27
+ * (guard-C's end state — the consumer sees both truths);
28
+ * none = both live, not blended, no resolution link between them.
29
+ *
30
+ * The version_chain CLASS end-state is measured on the RECALL surface, not
31
+ * the store (#393 D): a retrieval-only change cannot move a store probe, so
32
+ * post-stage the harness runs the production retrieve() (read-only: limit 3,
33
+ * below the cold-exposure threshold, noStrengthen) with the OLDEST chain
34
+ * member's content as query and requires the belief walk's guarantee — no
35
+ * superseded ancestor surfaces as a competing truth, and the linked chain's
36
+ * terminal surfaces. Store facts (terminal live, non-terminals gone) remain
37
+ * in the class note as corroboration.
38
+ *
39
+ * The judge is a GROUND-TRUTH-STUBBED LlmClient (the refine addendum's Q3
40
+ * recommendation): it answers each planted pair's declared relation and
41
+ * defaults unknown/collider pairs to `none`. This isolates the MACHINERY
42
+ * (detection sources, binding actions, merge guard, walk) from judge quality —
43
+ * the harness answers "would the pipeline resolve a KNOWN correction if the
44
+ * judge were perfect?". Live-LLM verdict quality is the real-corpus soak run.
45
+ *
46
+ * Read/write discipline: the gate NEVER touches ~/.hicortex — the fixture DB
47
+ * and the stage's stateDir are caller-supplied (temp) paths. A real snapshot
48
+ * is only ever PLANTED INTO A COPY (planted-eval.ts --snapshot).
49
+ */
50
+ import type Database from "better-sqlite3";
51
+ import type { LlmClient } from "../llm.js";
52
+ import { type EmbedFn } from "../retrieval.js";
53
+ import type { acquireCaptureLock } from "../capture.js";
54
+ import { type ReconsolidationStageResult } from "../reconsolidation.js";
55
+ import { type PlanDedupResult } from "../dedup.js";
56
+ import { type PlantedCorpus, type FixtureClass, type GroundTruthRelation } from "./planted-fixtures.js";
57
+ /** Insert the corpus into a writable DB (fresh, or a copy of a snapshot). */
58
+ export declare function plantCorpus(db: Database.Database, corpus: PlantedCorpus, embedFn: EmbedFn): Promise<Map<string, string>>;
59
+ export interface VerdictCall {
60
+ olderKey: string;
61
+ newerKey: string;
62
+ /** True when the pair matched a declared ground-truth pair; false = default-none fallback. */
63
+ grounded: boolean;
64
+ }
65
+ export interface RewriteCall {
66
+ targetKey: string;
67
+ triggerKeys: string[];
68
+ }
69
+ /** A scout correction-shape call (#393 B) on one memory, ground-truth answered. */
70
+ export interface ScoutCall {
71
+ key: string;
72
+ correction: boolean;
73
+ }
74
+ export interface GroundTruthJudge {
75
+ llm: LlmClient;
76
+ verdictCalls: VerdictCall[];
77
+ rewriteCalls: RewriteCall[];
78
+ scoutCalls: ScoutCall[];
79
+ }
80
+ /**
81
+ * Deterministic, ground-truth-keyed judge. Parses each prompt back into its
82
+ * (older, newer) contents — the exact texts `buildCorrectionVerdictPrompt` /
83
+ * `buildRewritePrompt` embed — and answers from the corpus's declared
84
+ * relations:
85
+ * corrects -> {"action":"corrects","confidence":0.95} (above the gate)
86
+ * supersedes-> {"action":"supersedes","confidence":0.9}
87
+ * merge -> {"action":"merge","confidence":0.95} (above the gate)
88
+ * conflict -> {"action":"conflicts","confidence":0.9} (guard-C: flag,
89
+ * both live, never blended)
90
+ * none -> {"action":"none","confidence":0.9}
91
+ * Unknown pairs (colliders, filler cross-pairs) default to none.
92
+ *
93
+ * #393 B: the judge also answers the scout's per-memory shape calls — a
94
+ * memory is correction-shaped iff it is the newer side of a declared
95
+ * corrects/supersedes/conflict pair (declared `references` returned verbatim,
96
+ * throwing if a resolution-shaped pair never declared them; guard-C added
97
+ * `conflict` to the shaped set — two records disagreeing on one quantity
98
+ * share their wording); everything else answers correction=false. Shape calls
99
+ * are logged in scoutCalls.
100
+ */
101
+ export declare function makeGroundTruthJudge(corpus: PlantedCorpus, idByKey: Map<string, string>): GroundTruthJudge;
102
+ export interface PlantedPairResult {
103
+ class: FixtureClass;
104
+ olderKey: string;
105
+ newerKey: string;
106
+ relation: GroundTruthRelation;
107
+ cosine: number;
108
+ inBand: boolean;
109
+ detected: boolean;
110
+ /**
111
+ * Which source(s) made the pair a candidate: "judge" (verdict call via the
112
+ * KNN similarity source), "scout" (verdict call on a BELOW-floor pair — only
113
+ * the scout can source one; #393 B), "zone" (co-clustered at the ceiling).
114
+ * In-band pairs the KNN source could have found are attributed "judge" even
115
+ * when the scout also found them — the attribution answers "what made this
116
+ * pair reachable", and for in-band pairs that is the similarity floor.
117
+ */
118
+ detectionSource: "" | "judge" | "zone" | "judge+zone" | "scout" | "scout+zone";
119
+ bound: boolean | null;
120
+ boundHow: string;
121
+ resolved: boolean;
122
+ notes: string;
123
+ }
124
+ export interface PlantedClassResult {
125
+ class: FixtureClass;
126
+ pairs: number;
127
+ detected: number;
128
+ bound: number;
129
+ resolved: number;
130
+ /** Class-level end state (version_chain: terminal live + non-terminals gone). */
131
+ classResolved: boolean;
132
+ notes: string;
133
+ }
134
+ export interface ColliderRow {
135
+ aKey: string;
136
+ bKey: string;
137
+ cosine: number;
138
+ verdictCalled: boolean;
139
+ merged: boolean;
140
+ bothLive: boolean;
141
+ }
142
+ export interface PlantedGateResult {
143
+ pairResults: PlantedPairResult[];
144
+ classResults: PlantedClassResult[];
145
+ colliders: ColliderRow[];
146
+ stageReport: ReconsolidationStageResult;
147
+ zonePlanBefore: PlanDedupResult;
148
+ /** Pre-run zone-blend predictions: ground-truth pair -> surviving (canonical) key. */
149
+ blendPredictions: Array<{
150
+ pair: string;
151
+ canonicalKey: string;
152
+ loserKey: string;
153
+ }>;
154
+ }
155
+ export interface PlantedGateOptions {
156
+ /** Writable fixture DB path — created fresh, or a snapshot COPY to plant into. */
157
+ dbPath: string;
158
+ /** Temp state dir for the stage (state.json cursor, backups, band stats). */
159
+ stateDir: string;
160
+ embedFn: EmbedFn;
161
+ /** Lock acquirer for the zone/merge windows. Default: always-acquire (hermetic). */
162
+ acquireLock?: typeof acquireCaptureLock;
163
+ /** Zone ceiling. Default: the production default (0.92). */
164
+ threshold?: number;
165
+ }
166
+ /**
167
+ * Build (or plant into) the DB, run the production stage, measure the gate.
168
+ * This is the ONE orchestration shared by the CLI runner and the vitest suite.
169
+ */
170
+ export declare function runPlantedGate(corpus: PlantedCorpus, opts: PlantedGateOptions): Promise<PlantedGateResult>;
171
+ export declare function renderPlantedReport(args: {
172
+ corpusLabel: string;
173
+ generatedAt: string;
174
+ embedderLabel: string;
175
+ result: PlantedGateResult;
176
+ }): string;