@gamaze/hicortex 0.20.7 → 0.20.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -41
- package/assets/dashboard.html +3989 -836
- package/dist/calibration.d.ts +293 -0
- package/dist/calibration.js +379 -0
- package/dist/capture-health.d.ts +87 -0
- package/dist/capture-health.js +106 -0
- package/dist/capture-pause.d.ts +86 -0
- package/dist/capture-pause.js +127 -0
- package/dist/capture.d.ts +24 -3
- package/dist/capture.js +11 -1
- package/dist/classify-domains.d.ts +6 -0
- package/dist/classify-domains.js +7 -1
- package/dist/cli.js +38 -3
- package/dist/config-read.d.ts +1 -1
- package/dist/config-read.js +96 -9
- package/dist/consolidate.d.ts +114 -68
- package/dist/consolidate.js +302 -182
- package/dist/dashboard.d.ts +326 -6
- package/dist/dashboard.js +592 -7
- package/dist/db.js +105 -0
- package/dist/dedup.d.ts +34 -26
- package/dist/dedup.js +91 -57
- package/dist/distiller.js +1 -1
- package/dist/domain-classify.d.ts +7 -6
- package/dist/domain-classify.js +12 -10
- package/dist/eval/decay-eval.d.ts +3 -3
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/importance-eval.d.ts +85 -0
- package/dist/eval/importance-eval.js +286 -0
- package/dist/eval/planted-eval.d.ts +26 -0
- package/dist/eval/planted-eval.js +97 -0
- package/dist/eval/planted-fixtures.d.ts +107 -0
- package/dist/eval/planted-fixtures.js +283 -0
- package/dist/eval/planted-harness.d.ts +176 -0
- package/dist/eval/planted-harness.js +649 -0
- package/dist/eval/ranking-battery.d.ts +78 -0
- package/dist/eval/ranking-battery.js +181 -0
- package/dist/eval/ranking-eval.d.ts +41 -0
- package/dist/eval/ranking-eval.js +391 -0
- package/dist/eval/ranking-fixtures.d.ts +77 -0
- package/dist/eval/ranking-fixtures.js +226 -0
- package/dist/identity-store.d.ts +21 -0
- package/dist/identity-store.js +49 -0
- package/dist/index.js +4 -3
- package/dist/init.d.ts +23 -3
- package/dist/init.js +84 -9
- package/dist/llm.d.ts +43 -58
- package/dist/llm.js +87 -101
- package/dist/mcp-server.d.ts +12 -0
- package/dist/mcp-server.js +213 -32
- package/dist/nightly.d.ts +9 -1
- package/dist/nightly.js +164 -110
- package/dist/nofit.d.ts +4 -11
- package/dist/nofit.js +6 -23
- package/dist/prompts.d.ts +10 -0
- package/dist/prompts.js +28 -5
- package/dist/recall-index.d.ts +30 -28
- package/dist/recall-index.js +21 -18
- package/dist/recall-registry.d.ts +2 -1
- package/dist/recall-registry.js +35 -1
- package/dist/reconsolidation.d.ts +168 -87
- package/dist/reconsolidation.js +818 -377
- package/dist/relink.js +3 -4
- package/dist/rescore-importance.d.ts +80 -0
- package/dist/rescore-importance.js +236 -0
- package/dist/retrieval.d.ts +80 -35
- package/dist/retrieval.js +322 -105
- package/dist/run-deadline.d.ts +62 -0
- package/dist/run-deadline.js +73 -0
- package/dist/schema-prototypes.d.ts +3 -3
- package/dist/schema-prototypes.js +3 -3
- package/dist/stages.d.ts +37 -0
- package/dist/stages.js +51 -0
- package/dist/state.d.ts +34 -9
- package/dist/storage.d.ts +50 -18
- package/dist/storage.js +125 -30
- package/dist/telemetry.d.ts +8 -7
- package/dist/token-budget.js +3 -4
- package/dist/type-classify.js +4 -4
- package/dist/types.d.ts +143 -155
- package/domains.example.json +4 -5
- package/hermes-plugin/hicortex/README.md +2 -2
- package/openclaw.plugin.json +1 -1
- package/package.json +4 -1
- package/pi-extension/hicortex/README.md +1 -1
- package/server.json +3 -3
|
@@ -0,0 +1,283 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Planted-pair fixture corpus for the resolution acceptance gate (#393
|
|
4
|
+
* increment A — the eval-first harness).
|
|
5
|
+
*
|
|
6
|
+
* Five failure classes, each encoded as planted memory pairs with a DECLARED
|
|
7
|
+
* ground-truth relation and a DECLARED expected cosine band:
|
|
8
|
+
*
|
|
9
|
+
* (a) cross_topic_correction — the correction rides inside a topically
|
|
10
|
+
* unrelated experience memory; measured cosine ~0.5-0.6, BELOW the
|
|
11
|
+
* `correctionMinSimilarity` floor (0.75) — the documented field hole
|
|
12
|
+
* (0 `corrects` verdicts in 246 on the 17k snapshot).
|
|
13
|
+
* (b) source_conflict — two live records that are both true AS RECORDS (each
|
|
14
|
+
* faithfully reports a different source) at >= the deterministic-zone
|
|
15
|
+
* ceiling (0.92). Guard-C closes the hole: the scout flags the newer
|
|
16
|
+
* record contradiction-shaped, the judge answers `conflicts`, the link
|
|
17
|
+
* guards the zone, and both stay live.
|
|
18
|
+
* (c) version_chain — v1 -> v2 -> v3 of a revised decision record: the
|
|
19
|
+
* 0.78-0.96 stale-churn class (#392 owner calibration). v1-v2 sits in
|
|
20
|
+
* the judged band; v2-v3 (numeric status churn inside a long shared
|
|
21
|
+
* record) sits above the ceiling, where the zone blends the NEWEST
|
|
22
|
+
* version into the STALE one (canonical = oldest created_at).
|
|
23
|
+
* (d) same_fact_paraphrase — one fact, two wordings: should merge.
|
|
24
|
+
* (e) complementary_facets — same subject, different facets: should keep
|
|
25
|
+
* both (judge verdict `none`).
|
|
26
|
+
*
|
|
27
|
+
* Honesty rules:
|
|
28
|
+
* - The texts are synthetic and generic (this package publishes to npm — no
|
|
29
|
+
* real infrastructure, people, or fleet detail), but SHAPED like the field
|
|
30
|
+
* failures: corrections ride in experience memories, conflicts share long
|
|
31
|
+
* context tails, churn edits one field of a long record.
|
|
32
|
+
* - Cosines are MEASURED at run time with the real bge-small-en-v1.5 embedder
|
|
33
|
+
* (planted-eval.ts) — never asserted into existence. The declared bands
|
|
34
|
+
* below were calibrated against that embedder (see the band table in the
|
|
35
|
+
* report); a pair landing outside its band renders as a loud OUT-OF-BAND
|
|
36
|
+
* flag rather than a hard failure, so an embedder change is visible, not
|
|
37
|
+
* silent (same posture as run-eval).
|
|
38
|
+
* - Neutral filler rows keep the KNN neighborhoods non-trivial and give the
|
|
39
|
+
* collider analysis something to check (any non-ground-truth pair >= 0.70
|
|
40
|
+
* is reported).
|
|
41
|
+
*/
|
|
42
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
43
|
+
exports.REAL_TEXT_CORPUS = exports.PLANTED_SOURCE_AGENT = exports.PLANTED_PROJECT = void 0;
|
|
44
|
+
exports.chainMembers = chainMembers;
|
|
45
|
+
/** All planted rows share project + source_agent so the metadata rails never refuse a fixture merge. */
|
|
46
|
+
exports.PLANTED_PROJECT = "planted-eval";
|
|
47
|
+
exports.PLANTED_SOURCE_AGENT = "planted-eval";
|
|
48
|
+
// ---------------------------------------------------------------------------
|
|
49
|
+
// The real-text corpus (embedded with the REAL embedder by planted-eval.ts)
|
|
50
|
+
// ---------------------------------------------------------------------------
|
|
51
|
+
const FILLER = [
|
|
52
|
+
[
|
|
53
|
+
"fill_recipe",
|
|
54
|
+
"Batch-cooked a tomato and butter sauce from the Rome cookbook: San Marzano tomatoes, one whole onion halved, and far more butter than seems reasonable. Simmered 45 minutes. The whole family preferred it to last month's pesto attempt.",
|
|
55
|
+
"2026-06-06T09:00:00.000Z",
|
|
56
|
+
],
|
|
57
|
+
[
|
|
58
|
+
"fill_run",
|
|
59
|
+
"Half-marathon training week 6: three easy runs at conversational pace, one interval session on the track, and a 16-kilometer long run on Sunday. Left knee complains on downhills; foam rolling after every session keeps it quiet.",
|
|
60
|
+
"2026-06-07T09:00:00.000Z",
|
|
61
|
+
],
|
|
62
|
+
[
|
|
63
|
+
"fill_books",
|
|
64
|
+
"Finished the third novel in the frontier trilogy. The middle book dragged through the mining-town chapters, but the finale pays everything off when the surveyor returns to the flooded valley and finally reads her grandmother's letters.",
|
|
65
|
+
"2026-06-08T09:00:00.000Z",
|
|
66
|
+
],
|
|
67
|
+
[
|
|
68
|
+
"fill_travel",
|
|
69
|
+
"Train trip through the coastal towns in autumn: two nights in the fishing village with the breakwater walk, then the mountain line with the switchback tunnels. Pack the warm layer even when the departure platform is sunny.",
|
|
70
|
+
"2026-06-09T09:00:00.000Z",
|
|
71
|
+
],
|
|
72
|
+
[
|
|
73
|
+
"fill_garden",
|
|
74
|
+
"The raised beds need refreshing before spring: compost the spent tomato vines, rotate the legume row to where the squash was, and prune the apple espalier before the buds swell.",
|
|
75
|
+
"2026-06-10T09:00:00.000Z",
|
|
76
|
+
],
|
|
77
|
+
[
|
|
78
|
+
"fill_lang",
|
|
79
|
+
"Language study notes: the dative prepositions finally clicked after drilling them as a sung list. Irregular verbs still need spaced repetition; the flashcard app's new algorithm keeps resurfacing the ones I already know.",
|
|
80
|
+
"2026-06-21T09:00:00.000Z",
|
|
81
|
+
],
|
|
82
|
+
[
|
|
83
|
+
"fill_music",
|
|
84
|
+
"Piano practice log: the Chopin nocturne's middle section is still uneven at tempo. Slow practice hands-separately for the polyrhythm bars, then glue the phrases at 80 percent speed before next week's lesson.",
|
|
85
|
+
"2026-06-22T09:00:00.000Z",
|
|
86
|
+
],
|
|
87
|
+
[
|
|
88
|
+
"fill_car",
|
|
89
|
+
"Car maintenance records: switched the hatchback to the synthetic blend at the 90,000-kilometer service; the mechanic noted a slow weep on the rear shock that we watch for now instead of replacing immediately.",
|
|
90
|
+
"2026-06-23T09:00:00.000Z",
|
|
91
|
+
],
|
|
92
|
+
];
|
|
93
|
+
/**
|
|
94
|
+
* Measured with the real bge-small-en-v1.5 embedder at fixture-authoring time
|
|
95
|
+
* (2026-09-12): cross-topic 0.625, conflict 0.993, chain v1-v2 0.788 (judged
|
|
96
|
+
* band), v2-v3 0.998 (zone band), paraphrase 0.985, facets 0.774; no
|
|
97
|
+
* non-ground-truth pair >= 0.70. Bands are set with margin around those
|
|
98
|
+
* values — the report re-measures and flags drift.
|
|
99
|
+
*/
|
|
100
|
+
exports.REAL_TEXT_CORPUS = {
|
|
101
|
+
rows: [
|
|
102
|
+
// --- old side (+ first filler block) ---
|
|
103
|
+
{
|
|
104
|
+
key: "xt_old",
|
|
105
|
+
content: "The analytics dashboard refreshes its data every 15 minutes through the scheduled ingestion job.",
|
|
106
|
+
createdAt: "2026-06-01T09:00:00.000Z",
|
|
107
|
+
memoryType: "knowledge",
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
key: "conflict_a",
|
|
111
|
+
content: "Per the provider's official pricing page, the vendor API rate limit is 60 requests per minute per account. " +
|
|
112
|
+
"Verified while sizing the sync worker retries; bursts above the limit return HTTP 429 with a Retry-After header.",
|
|
113
|
+
createdAt: "2026-06-02T09:00:00.000Z",
|
|
114
|
+
memoryType: "knowledge",
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
key: "chain_v1",
|
|
118
|
+
content: "Deployment strategy (decision record 14): all releases ship through the blue-green pipeline with an atomic " +
|
|
119
|
+
"traffic flip between two identical production environments. Rollback means flipping traffic back to the " +
|
|
120
|
+
"previous environment. Owner: platform team. Migration status: not started. Reviewed quarterly.",
|
|
121
|
+
createdAt: "2026-06-03T09:00:00.000Z",
|
|
122
|
+
memoryType: "decision",
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
key: "para_a",
|
|
126
|
+
content: "The team standup happens every weekday morning at 09:30 and never runs longer than fifteen minutes.",
|
|
127
|
+
createdAt: "2026-06-04T09:00:00.000Z",
|
|
128
|
+
memoryType: "knowledge",
|
|
129
|
+
},
|
|
130
|
+
{
|
|
131
|
+
key: "facet_a",
|
|
132
|
+
content: "The team wiki runs on a self-hosted instance behind the company VPN; only the ops group holds " +
|
|
133
|
+
"administrator rights on the wiki instance.",
|
|
134
|
+
createdAt: "2026-06-05T09:00:00.000Z",
|
|
135
|
+
memoryType: "knowledge",
|
|
136
|
+
},
|
|
137
|
+
...FILLER.slice(0, 5).map(([key, content, createdAt]) => ({
|
|
138
|
+
key,
|
|
139
|
+
content,
|
|
140
|
+
createdAt,
|
|
141
|
+
memoryType: "experience",
|
|
142
|
+
})),
|
|
143
|
+
// --- new side (+ second filler block) ---
|
|
144
|
+
{
|
|
145
|
+
key: "xt_new",
|
|
146
|
+
content: "On-call shift recap: we chased a payment webhook outage for two hours; the root cause was a stale " +
|
|
147
|
+
"connection pool starving the workers under load. While verifying configs afterwards we also found the " +
|
|
148
|
+
"dashboard ingestion schedule quietly moved from every 15 minutes to hourly — the old 15-minute figure " +
|
|
149
|
+
"is outdated.",
|
|
150
|
+
createdAt: "2026-06-15T09:00:00.000Z",
|
|
151
|
+
memoryType: "experience",
|
|
152
|
+
},
|
|
153
|
+
{
|
|
154
|
+
key: "conflict_b",
|
|
155
|
+
content: "Per the provider's engineering wiki, the vendor API rate limit is 100 requests per minute per account. " +
|
|
156
|
+
"Verified while sizing the sync worker retries; bursts above the limit return HTTP 429 with a Retry-After header.",
|
|
157
|
+
createdAt: "2026-06-16T09:00:00.000Z",
|
|
158
|
+
memoryType: "knowledge",
|
|
159
|
+
},
|
|
160
|
+
{
|
|
161
|
+
key: "chain_v2",
|
|
162
|
+
content: "Deployment strategy (decision record 14): all releases now roll out through staged rollouts with automatic " +
|
|
163
|
+
"rollback on error-budget burn. Rollback means halting the rollout and reverting the release. " +
|
|
164
|
+
"Owner: platform team. Migration status: stage 2 of 4 complete. Reviewed quarterly.",
|
|
165
|
+
createdAt: "2026-06-17T09:00:00.000Z",
|
|
166
|
+
memoryType: "decision",
|
|
167
|
+
},
|
|
168
|
+
{
|
|
169
|
+
key: "chain_v3",
|
|
170
|
+
content: "Deployment strategy (decision record 14): all releases now roll out through staged rollouts with automatic " +
|
|
171
|
+
"rollback on error-budget burn. Rollback means halting the rollout and reverting the release. " +
|
|
172
|
+
"Owner: platform team. Migration status: stage 3 of 4 complete. Reviewed quarterly.",
|
|
173
|
+
createdAt: "2026-06-18T09:00:00.000Z",
|
|
174
|
+
memoryType: "decision",
|
|
175
|
+
},
|
|
176
|
+
{
|
|
177
|
+
key: "para_b",
|
|
178
|
+
content: "The team standup is every weekday at 09:30 in the morning; it never runs longer than fifteen minutes.",
|
|
179
|
+
createdAt: "2026-06-19T09:00:00.000Z",
|
|
180
|
+
memoryType: "knowledge",
|
|
181
|
+
},
|
|
182
|
+
{
|
|
183
|
+
key: "facet_b",
|
|
184
|
+
content: "The team wiki instance behind the company VPN is slow to load for read-only users because the " +
|
|
185
|
+
"self-hosted server has no page cache; ops plans to add one.",
|
|
186
|
+
createdAt: "2026-06-20T09:00:00.000Z",
|
|
187
|
+
memoryType: "knowledge",
|
|
188
|
+
},
|
|
189
|
+
...FILLER.slice(5).map(([key, content, createdAt]) => ({
|
|
190
|
+
key,
|
|
191
|
+
content,
|
|
192
|
+
createdAt,
|
|
193
|
+
memoryType: "experience",
|
|
194
|
+
})),
|
|
195
|
+
],
|
|
196
|
+
pairs: [
|
|
197
|
+
{
|
|
198
|
+
class: "cross_topic_correction",
|
|
199
|
+
olderKey: "xt_old",
|
|
200
|
+
newerKey: "xt_new",
|
|
201
|
+
relation: "corrects",
|
|
202
|
+
band: [0.4, 0.68],
|
|
203
|
+
rewritten: "The analytics dashboard refreshes its data hourly through the scheduled ingestion job; the refresh " +
|
|
204
|
+
"cadence was reduced from every 15 minutes during the June platform review.",
|
|
205
|
+
// xt_new says "the dashboard ingestion schedule ... moved from every 15
|
|
206
|
+
// minutes to hourly" — these five terms all appear verbatim in xt_old.
|
|
207
|
+
references: "dashboard ingestion every 15 minutes",
|
|
208
|
+
},
|
|
209
|
+
{
|
|
210
|
+
class: "source_conflict",
|
|
211
|
+
olderKey: "conflict_a",
|
|
212
|
+
newerKey: "conflict_b",
|
|
213
|
+
relation: "conflict",
|
|
214
|
+
band: [0.92, 1.0],
|
|
215
|
+
// conflict_b contradicts conflict_a on one quantity while sharing its
|
|
216
|
+
// whole context tail — every token appears verbatim in conflict_a (the
|
|
217
|
+
// guard-C fixture mechanism: the scout's FTS finds the old record, the
|
|
218
|
+
// re-tag rule lets the judge see a >=0.92 pair).
|
|
219
|
+
references: "vendor API rate limit requests per minute per account",
|
|
220
|
+
},
|
|
221
|
+
{
|
|
222
|
+
class: "version_chain",
|
|
223
|
+
olderKey: "chain_v1",
|
|
224
|
+
newerKey: "chain_v2",
|
|
225
|
+
relation: "supersedes",
|
|
226
|
+
band: [0.75, 0.91],
|
|
227
|
+
references: "deployment strategy decision record 14",
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
class: "version_chain",
|
|
231
|
+
olderKey: "chain_v2",
|
|
232
|
+
newerKey: "chain_v3",
|
|
233
|
+
relation: "supersedes",
|
|
234
|
+
band: [0.92, 1.0],
|
|
235
|
+
references: "deployment strategy decision record 14",
|
|
236
|
+
},
|
|
237
|
+
{
|
|
238
|
+
class: "version_chain",
|
|
239
|
+
olderKey: "chain_v1",
|
|
240
|
+
newerKey: "chain_v3",
|
|
241
|
+
relation: "supersedes",
|
|
242
|
+
band: [0.75, 0.91],
|
|
243
|
+
references: "deployment strategy decision record 14",
|
|
244
|
+
},
|
|
245
|
+
{
|
|
246
|
+
class: "same_fact_paraphrase",
|
|
247
|
+
olderKey: "para_a",
|
|
248
|
+
newerKey: "para_b",
|
|
249
|
+
relation: "merge",
|
|
250
|
+
band: [0.9, 1.0],
|
|
251
|
+
},
|
|
252
|
+
{
|
|
253
|
+
class: "complementary_facets",
|
|
254
|
+
olderKey: "facet_a",
|
|
255
|
+
newerKey: "facet_b",
|
|
256
|
+
relation: "none",
|
|
257
|
+
band: [0.7, 0.9],
|
|
258
|
+
},
|
|
259
|
+
],
|
|
260
|
+
};
|
|
261
|
+
/**
|
|
262
|
+
* The version_chain class is measured BOTH per-pair and as a class: the
|
|
263
|
+
* desired end state is the chain's TERMINAL (newest member) live with every
|
|
264
|
+
* non-terminal demoted or absorbed. Members are derived from the declared
|
|
265
|
+
* pairs, ordered by createdAt.
|
|
266
|
+
*/
|
|
267
|
+
function chainMembers(corpus) {
|
|
268
|
+
const members = new Set();
|
|
269
|
+
for (const p of corpus.pairs) {
|
|
270
|
+
if (p.class !== "version_chain")
|
|
271
|
+
continue;
|
|
272
|
+
members.add(p.olderKey);
|
|
273
|
+
members.add(p.newerKey);
|
|
274
|
+
}
|
|
275
|
+
const byKey = new Map(corpus.rows.map((r) => [r.key, r]));
|
|
276
|
+
return [...members].sort((a, b) => {
|
|
277
|
+
const ra = byKey.get(a);
|
|
278
|
+
const rb = byKey.get(b);
|
|
279
|
+
const ca = ra?.createdAt ?? "";
|
|
280
|
+
const cb = rb?.createdAt ?? "";
|
|
281
|
+
return ca === cb ? a.localeCompare(b) : ca.localeCompare(cb);
|
|
282
|
+
});
|
|
283
|
+
}
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The planted-pairs acceptance gate (#393 increment A) — machinery shared by
|
|
3
|
+
* the CLI runner (planted-eval.ts, REAL embedder + real-text corpus) and the
|
|
4
|
+
* vitest suite (tests/planted-pairs.test.ts, controlled synthetic vectors).
|
|
5
|
+
*
|
|
6
|
+
* What it measures, per planted pair, before/after running the REAL
|
|
7
|
+
* production resolution stage (`stageReconsolidation`, which internally runs
|
|
8
|
+
* the deterministic merge zone LAST — guard-C: judgment outranks the sweep):
|
|
9
|
+
*
|
|
10
|
+
* DETECTED — the pair entered the resolution machinery as a candidate: a
|
|
11
|
+
* verdict call was made on it (judged band, logged by the ground-truth
|
|
12
|
+
* judge) OR it co-clustered in the pre-stage `planDedup` discovery at the
|
|
13
|
+
* zone ceiling. Tagged by source ("judge" = KNN/similarity, "scout" = the
|
|
14
|
+
* #393 B reference-extraction source — attribution rule: a verdict call
|
|
15
|
+
* on a below-floor pair can only be scout-sourced, "zone" = the
|
|
16
|
+
* deterministic ceiling).
|
|
17
|
+
* BOUND — a resolution structure reflecting the ground truth now connects
|
|
18
|
+
* the pair: `corrected_by` / `superseded_by` / `conflicts` link, or a
|
|
19
|
+
* merge (`dedup_log` row). Guard-C added the `conflicts` bind (the judge
|
|
20
|
+
* flags a genuine conflict instead of leaving it inexpressible).
|
|
21
|
+
* RESOLVED — the class-specific desired END STATE holds in the store:
|
|
22
|
+
* corrects = correction effective (older demoted/absorbed/rewritten) and
|
|
23
|
+
* the correction information live;
|
|
24
|
+
* supersedes= older demoted-or-absorbed and newer live in recall;
|
|
25
|
+
* merge = exactly one live row remains, the other absorbed via dedup;
|
|
26
|
+
* conflict = both live AND not blended AND the conflicts link exists
|
|
27
|
+
* (guard-C's end state — the consumer sees both truths);
|
|
28
|
+
* none = both live, not blended, no resolution link between them.
|
|
29
|
+
*
|
|
30
|
+
* The version_chain CLASS end-state is measured on the RECALL surface, not
|
|
31
|
+
* the store (#393 D): a retrieval-only change cannot move a store probe, so
|
|
32
|
+
* post-stage the harness runs the production retrieve() (read-only: limit 3,
|
|
33
|
+
* below the cold-exposure threshold, noStrengthen) with the OLDEST chain
|
|
34
|
+
* member's content as query and requires the belief walk's guarantee — no
|
|
35
|
+
* superseded ancestor surfaces as a competing truth, and the linked chain's
|
|
36
|
+
* terminal surfaces. Store facts (terminal live, non-terminals gone) remain
|
|
37
|
+
* in the class note as corroboration.
|
|
38
|
+
*
|
|
39
|
+
* The judge is a GROUND-TRUTH-STUBBED LlmClient (the refine addendum's Q3
|
|
40
|
+
* recommendation): it answers each planted pair's declared relation and
|
|
41
|
+
* defaults unknown/collider pairs to `none`. This isolates the MACHINERY
|
|
42
|
+
* (detection sources, binding actions, merge guard, walk) from judge quality —
|
|
43
|
+
* the harness answers "would the pipeline resolve a KNOWN correction if the
|
|
44
|
+
* judge were perfect?". Live-LLM verdict quality is the real-corpus soak run.
|
|
45
|
+
*
|
|
46
|
+
* Read/write discipline: the gate NEVER touches ~/.hicortex — the fixture DB
|
|
47
|
+
* and the stage's stateDir are caller-supplied (temp) paths. A real snapshot
|
|
48
|
+
* is only ever PLANTED INTO A COPY (planted-eval.ts --snapshot).
|
|
49
|
+
*/
|
|
50
|
+
import type Database from "better-sqlite3";
|
|
51
|
+
import type { LlmClient } from "../llm.js";
|
|
52
|
+
import { type EmbedFn } from "../retrieval.js";
|
|
53
|
+
import type { acquireCaptureLock } from "../capture.js";
|
|
54
|
+
import { type ReconsolidationStageResult } from "../reconsolidation.js";
|
|
55
|
+
import { type PlanDedupResult } from "../dedup.js";
|
|
56
|
+
import { type PlantedCorpus, type FixtureClass, type GroundTruthRelation } from "./planted-fixtures.js";
|
|
57
|
+
/** Insert the corpus into a writable DB (fresh, or a copy of a snapshot). */
|
|
58
|
+
export declare function plantCorpus(db: Database.Database, corpus: PlantedCorpus, embedFn: EmbedFn): Promise<Map<string, string>>;
|
|
59
|
+
export interface VerdictCall {
|
|
60
|
+
olderKey: string;
|
|
61
|
+
newerKey: string;
|
|
62
|
+
/** True when the pair matched a declared ground-truth pair; false = default-none fallback. */
|
|
63
|
+
grounded: boolean;
|
|
64
|
+
}
|
|
65
|
+
export interface RewriteCall {
|
|
66
|
+
targetKey: string;
|
|
67
|
+
triggerKeys: string[];
|
|
68
|
+
}
|
|
69
|
+
/** A scout correction-shape call (#393 B) on one memory, ground-truth answered. */
|
|
70
|
+
export interface ScoutCall {
|
|
71
|
+
key: string;
|
|
72
|
+
correction: boolean;
|
|
73
|
+
}
|
|
74
|
+
export interface GroundTruthJudge {
|
|
75
|
+
llm: LlmClient;
|
|
76
|
+
verdictCalls: VerdictCall[];
|
|
77
|
+
rewriteCalls: RewriteCall[];
|
|
78
|
+
scoutCalls: ScoutCall[];
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Deterministic, ground-truth-keyed judge. Parses each prompt back into its
|
|
82
|
+
* (older, newer) contents — the exact texts `buildCorrectionVerdictPrompt` /
|
|
83
|
+
* `buildRewritePrompt` embed — and answers from the corpus's declared
|
|
84
|
+
* relations:
|
|
85
|
+
* corrects -> {"action":"corrects","confidence":0.95} (above the gate)
|
|
86
|
+
* supersedes-> {"action":"supersedes","confidence":0.9}
|
|
87
|
+
* merge -> {"action":"merge","confidence":0.95} (above the gate)
|
|
88
|
+
* conflict -> {"action":"conflicts","confidence":0.9} (guard-C: flag,
|
|
89
|
+
* both live, never blended)
|
|
90
|
+
* none -> {"action":"none","confidence":0.9}
|
|
91
|
+
* Unknown pairs (colliders, filler cross-pairs) default to none.
|
|
92
|
+
*
|
|
93
|
+
* #393 B: the judge also answers the scout's per-memory shape calls — a
|
|
94
|
+
* memory is correction-shaped iff it is the newer side of a declared
|
|
95
|
+
* corrects/supersedes/conflict pair (declared `references` returned verbatim,
|
|
96
|
+
* throwing if a resolution-shaped pair never declared them; guard-C added
|
|
97
|
+
* `conflict` to the shaped set — two records disagreeing on one quantity
|
|
98
|
+
* share their wording); everything else answers correction=false. Shape calls
|
|
99
|
+
* are logged in scoutCalls.
|
|
100
|
+
*/
|
|
101
|
+
export declare function makeGroundTruthJudge(corpus: PlantedCorpus, idByKey: Map<string, string>): GroundTruthJudge;
|
|
102
|
+
export interface PlantedPairResult {
|
|
103
|
+
class: FixtureClass;
|
|
104
|
+
olderKey: string;
|
|
105
|
+
newerKey: string;
|
|
106
|
+
relation: GroundTruthRelation;
|
|
107
|
+
cosine: number;
|
|
108
|
+
inBand: boolean;
|
|
109
|
+
detected: boolean;
|
|
110
|
+
/**
|
|
111
|
+
* Which source(s) made the pair a candidate: "judge" (verdict call via the
|
|
112
|
+
* KNN similarity source), "scout" (verdict call on a BELOW-floor pair — only
|
|
113
|
+
* the scout can source one; #393 B), "zone" (co-clustered at the ceiling).
|
|
114
|
+
* In-band pairs the KNN source could have found are attributed "judge" even
|
|
115
|
+
* when the scout also found them — the attribution answers "what made this
|
|
116
|
+
* pair reachable", and for in-band pairs that is the similarity floor.
|
|
117
|
+
*/
|
|
118
|
+
detectionSource: "" | "judge" | "zone" | "judge+zone" | "scout" | "scout+zone";
|
|
119
|
+
bound: boolean | null;
|
|
120
|
+
boundHow: string;
|
|
121
|
+
resolved: boolean;
|
|
122
|
+
notes: string;
|
|
123
|
+
}
|
|
124
|
+
export interface PlantedClassResult {
|
|
125
|
+
class: FixtureClass;
|
|
126
|
+
pairs: number;
|
|
127
|
+
detected: number;
|
|
128
|
+
bound: number;
|
|
129
|
+
resolved: number;
|
|
130
|
+
/** Class-level end state (version_chain: terminal live + non-terminals gone). */
|
|
131
|
+
classResolved: boolean;
|
|
132
|
+
notes: string;
|
|
133
|
+
}
|
|
134
|
+
export interface ColliderRow {
|
|
135
|
+
aKey: string;
|
|
136
|
+
bKey: string;
|
|
137
|
+
cosine: number;
|
|
138
|
+
verdictCalled: boolean;
|
|
139
|
+
merged: boolean;
|
|
140
|
+
bothLive: boolean;
|
|
141
|
+
}
|
|
142
|
+
export interface PlantedGateResult {
|
|
143
|
+
pairResults: PlantedPairResult[];
|
|
144
|
+
classResults: PlantedClassResult[];
|
|
145
|
+
colliders: ColliderRow[];
|
|
146
|
+
stageReport: ReconsolidationStageResult;
|
|
147
|
+
zonePlanBefore: PlanDedupResult;
|
|
148
|
+
/** Pre-run zone-blend predictions: ground-truth pair -> surviving (canonical) key. */
|
|
149
|
+
blendPredictions: Array<{
|
|
150
|
+
pair: string;
|
|
151
|
+
canonicalKey: string;
|
|
152
|
+
loserKey: string;
|
|
153
|
+
}>;
|
|
154
|
+
}
|
|
155
|
+
export interface PlantedGateOptions {
|
|
156
|
+
/** Writable fixture DB path — created fresh, or a snapshot COPY to plant into. */
|
|
157
|
+
dbPath: string;
|
|
158
|
+
/** Temp state dir for the stage (state.json cursor, backups, band stats). */
|
|
159
|
+
stateDir: string;
|
|
160
|
+
embedFn: EmbedFn;
|
|
161
|
+
/** Lock acquirer for the zone/merge windows. Default: always-acquire (hermetic). */
|
|
162
|
+
acquireLock?: typeof acquireCaptureLock;
|
|
163
|
+
/** Zone ceiling. Default: the production default (0.92). */
|
|
164
|
+
threshold?: number;
|
|
165
|
+
}
|
|
166
|
+
/**
|
|
167
|
+
* Build (or plant into) the DB, run the production stage, measure the gate.
|
|
168
|
+
* This is the ONE orchestration shared by the CLI runner and the vitest suite.
|
|
169
|
+
*/
|
|
170
|
+
export declare function runPlantedGate(corpus: PlantedCorpus, opts: PlantedGateOptions): Promise<PlantedGateResult>;
|
|
171
|
+
export declare function renderPlantedReport(args: {
|
|
172
|
+
corpusLabel: string;
|
|
173
|
+
generatedAt: string;
|
|
174
|
+
embedderLabel: string;
|
|
175
|
+
result: PlantedGateResult;
|
|
176
|
+
}): string;
|