@gamaze/hicortex 0.20.7 → 0.20.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -41
- package/assets/dashboard.html +3989 -836
- package/dist/calibration.d.ts +293 -0
- package/dist/calibration.js +379 -0
- package/dist/capture-health.d.ts +87 -0
- package/dist/capture-health.js +106 -0
- package/dist/capture-pause.d.ts +86 -0
- package/dist/capture-pause.js +127 -0
- package/dist/capture.d.ts +24 -3
- package/dist/capture.js +11 -1
- package/dist/classify-domains.d.ts +6 -0
- package/dist/classify-domains.js +7 -1
- package/dist/cli.js +38 -3
- package/dist/config-read.d.ts +1 -1
- package/dist/config-read.js +96 -9
- package/dist/consolidate.d.ts +114 -68
- package/dist/consolidate.js +302 -182
- package/dist/dashboard.d.ts +326 -6
- package/dist/dashboard.js +592 -7
- package/dist/db.js +105 -0
- package/dist/dedup.d.ts +34 -26
- package/dist/dedup.js +91 -57
- package/dist/distiller.js +1 -1
- package/dist/domain-classify.d.ts +7 -6
- package/dist/domain-classify.js +12 -10
- package/dist/eval/decay-eval.d.ts +3 -3
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/importance-eval.d.ts +85 -0
- package/dist/eval/importance-eval.js +286 -0
- package/dist/eval/planted-eval.d.ts +26 -0
- package/dist/eval/planted-eval.js +97 -0
- package/dist/eval/planted-fixtures.d.ts +107 -0
- package/dist/eval/planted-fixtures.js +283 -0
- package/dist/eval/planted-harness.d.ts +176 -0
- package/dist/eval/planted-harness.js +649 -0
- package/dist/eval/ranking-battery.d.ts +78 -0
- package/dist/eval/ranking-battery.js +181 -0
- package/dist/eval/ranking-eval.d.ts +41 -0
- package/dist/eval/ranking-eval.js +391 -0
- package/dist/eval/ranking-fixtures.d.ts +77 -0
- package/dist/eval/ranking-fixtures.js +226 -0
- package/dist/identity-store.d.ts +21 -0
- package/dist/identity-store.js +49 -0
- package/dist/index.js +4 -3
- package/dist/init.d.ts +23 -3
- package/dist/init.js +84 -9
- package/dist/llm.d.ts +43 -58
- package/dist/llm.js +87 -101
- package/dist/mcp-server.d.ts +12 -0
- package/dist/mcp-server.js +213 -32
- package/dist/nightly.d.ts +9 -1
- package/dist/nightly.js +164 -110
- package/dist/nofit.d.ts +4 -11
- package/dist/nofit.js +6 -23
- package/dist/prompts.d.ts +10 -0
- package/dist/prompts.js +28 -5
- package/dist/recall-index.d.ts +30 -28
- package/dist/recall-index.js +21 -18
- package/dist/recall-registry.d.ts +2 -1
- package/dist/recall-registry.js +35 -1
- package/dist/reconsolidation.d.ts +168 -87
- package/dist/reconsolidation.js +818 -377
- package/dist/relink.js +3 -4
- package/dist/rescore-importance.d.ts +80 -0
- package/dist/rescore-importance.js +236 -0
- package/dist/retrieval.d.ts +80 -35
- package/dist/retrieval.js +322 -105
- package/dist/run-deadline.d.ts +62 -0
- package/dist/run-deadline.js +73 -0
- package/dist/schema-prototypes.d.ts +3 -3
- package/dist/schema-prototypes.js +3 -3
- package/dist/stages.d.ts +37 -0
- package/dist/stages.js +51 -0
- package/dist/state.d.ts +34 -9
- package/dist/storage.d.ts +50 -18
- package/dist/storage.js +125 -30
- package/dist/telemetry.d.ts +8 -7
- package/dist/token-budget.js +3 -4
- package/dist/type-classify.js +4 -4
- package/dist/types.d.ts +143 -155
- package/domains.example.json +4 -5
- package/hermes-plugin/hicortex/README.md +2 -2
- package/openclaw.plugin.json +1 -1
- package/package.json +4 -1
- package/pi-extension/hicortex/README.md +1 -1
- package/server.json +3 -3
package/dist/prompts.js
CHANGED
|
@@ -16,23 +16,46 @@ exports.distillation = distillation;
|
|
|
16
16
|
exports.domainCuration = domainCuration;
|
|
17
17
|
/**
|
|
18
18
|
* Importance scoring prompt. Takes a {memories_block} with indexed memories.
|
|
19
|
+
*
|
|
20
|
+
* RE-ANCHORED (#425): the pre-fix anchors put "useful context" at 0.3-0.5 —
|
|
21
|
+
* but the distiller's ephemera gate already removes trivia before anything
|
|
22
|
+
* reaches this scorer, so the model only ever saw curated material and the
|
|
23
|
+
* distribution compressed upward (measured on the production snapshot via
|
|
24
|
+
* eval:importance: median base 0.8, ~11% at exactly 1.0, which the decay
|
|
25
|
+
* model never forgets). The anchors now place routine-but-curated content
|
|
26
|
+
* LOW and add an explicit distribution instruction (owner decision D2,
|
|
27
|
+
* 2026-09-13: target median 0.30-0.40, p90 <= 0.75; 1.0 is never a valid
|
|
28
|
+
* score). The strict JSON-array response contract is unchanged.
|
|
19
29
|
*/
|
|
20
30
|
function importanceScoring(memoriesBlock) {
|
|
21
31
|
return `You are a memory importance scorer. Rate each memory's long-term value.
|
|
22
32
|
|
|
23
33
|
Score each memory from 0.0 (trivial/ephemeral) to 1.0 (critical/foundational).
|
|
34
|
+
These memories are pre-filtered for durability, so ordinary competent work is
|
|
35
|
+
the NORMAL case — use the full scale and keep the bulk of a batch in the lower
|
|
36
|
+
half. When torn between two bands, choose the LOWER.
|
|
24
37
|
|
|
25
38
|
Scoring guide:
|
|
26
|
-
- 0.0-0.2:
|
|
27
|
-
- 0.
|
|
28
|
-
|
|
29
|
-
- 0.
|
|
39
|
+
- 0.0-0.2: Ephemeral state, routine actions, one-off fixes
|
|
40
|
+
- 0.2-0.4: Useful context, ordinary decisions, standard patterns — THE DEFAULT
|
|
41
|
+
BAND; most memories belong here
|
|
42
|
+
- 0.4-0.6: Notable decisions, recurring patterns, project-shaping context —
|
|
43
|
+
only rows that clearly stand above ordinary work
|
|
44
|
+
- 0.6-0.8: Important decisions, debugging breakthroughs, architectural
|
|
45
|
+
choices — rare; at most one or two in a typical batch
|
|
46
|
+
- 0.8-0.95: ONLY genuinely foundational principles, critical constraints, core
|
|
47
|
+
identity facts — material that would be serious to lose. Never score 1.0.
|
|
48
|
+
|
|
49
|
+
Distribution: a typical batch should have a median around 0.35 — half the
|
|
50
|
+
batch sits at 0.2-0.4. A score of 0.5 or more says "among the more important
|
|
51
|
+
quarter of everything in long-term memory"; a score above 0.8 must be rare
|
|
52
|
+
and immediately defensible as costly to lose.
|
|
30
53
|
|
|
31
54
|
MEMORIES:
|
|
32
55
|
${memoriesBlock}
|
|
33
56
|
|
|
34
57
|
Respond with ONLY a JSON array of scores in the same order, e.g.:
|
|
35
|
-
[0.3, 0.
|
|
58
|
+
[0.3, 0.4, 0.2, 0.7]
|
|
36
59
|
|
|
37
60
|
No explanations. Just the JSON array.`;
|
|
38
61
|
}
|
package/dist/recall-index.d.ts
CHANGED
|
@@ -51,9 +51,10 @@ import type { MemorySearchResult } from "./types.js";
|
|
|
51
51
|
import * as storage from "./storage.js";
|
|
52
52
|
import { SessionRecallRegistry } from "./recall-registry.js";
|
|
53
53
|
export interface RecallIndexOptions {
|
|
54
|
-
/** Minimum measured cosine for vector-only candidates (
|
|
55
|
-
*
|
|
56
|
-
*
|
|
54
|
+
/** Minimum measured cosine for vector-only candidates (release-managed
|
|
55
|
+
* since #408 — calibration.ts RECALL_MIN_SIMILARITY; this field is the
|
|
56
|
+
* eval/test seam). FTS-matched candidates pass regardless — a BM25
|
|
57
|
+
* text match is direct evidence of relevance. 0.62 (raised from 0.55
|
|
57
58
|
* on 2026-08-03 per a 0.01-step floor sweep on the rewritten corpus): steady
|
|
58
59
|
* ~3:1 noise:signal removal with no knee; 0.62 = +2.2pts precision, 10/98
|
|
59
60
|
* prompts silent, sits below the 0.63 local pessimum. The floor is a noise
|
|
@@ -61,41 +62,42 @@ export interface RecallIndexOptions {
|
|
|
61
62
|
* comes with ~1.5 wrongly-silenced (real signal); a non-cosine gate is the
|
|
62
63
|
* real silence fix (eval #3 §4). */
|
|
63
64
|
minSimilarity?: number;
|
|
64
|
-
/** Max index lines per response (
|
|
65
|
-
* (lowered from 6 on 2026-08-03). Per-slot
|
|
66
|
-
* slot 6 gives NO prompt its first relevant
|
|
67
|
-
* robust, prompt-set-independent finding, and
|
|
68
|
-
* is monotone (precision@4 33.7% > @6 30.6% >
|
|
69
|
-
* lower-noise — but the 4-vs-5 distinction rests on 5
|
|
70
|
-
* overfitting-fragile (K and the floor were tuned on
|
|
71
|
-
* with coverage at modest cost.
|
|
65
|
+
/** Max index lines per response (release-managed — calibration.ts
|
|
66
|
+
* RECALL_MAX_ITEMS; seam only). 5 (lowered from 6 on 2026-08-03). Per-slot
|
|
67
|
+
* decomposition at floor 0.62: slot 6 gives NO prompt its first relevant
|
|
68
|
+
* memory — "6 is wrong" is the robust, prompt-set-independent finding, and
|
|
69
|
+
* 5 captures it. The K-sweep is monotone (precision@4 33.7% > @6 30.6% >
|
|
70
|
+
* @8 28.3%), so 4 is lower-noise — but the 4-vs-5 distinction rests on 5
|
|
71
|
+
* of 98 prompts and is overfitting-fragile (K and the floor were tuned on
|
|
72
|
+
* the same set); 5 hedges with coverage at modest cost. */
|
|
72
73
|
maxItems?: number;
|
|
73
74
|
/** Prompts shorter than this are skipped (continuations, "yes", "do it"). */
|
|
74
75
|
minPromptLength?: number;
|
|
75
|
-
/** Max chars of the memory's first line shown in an index entry
|
|
76
|
-
*
|
|
77
|
-
*
|
|
78
|
-
* identical (0.6pts apart, N=40,
|
|
79
|
-
* per block. */
|
|
76
|
+
/** Max chars of the memory's first line shown in an index entry
|
|
77
|
+
* (release-managed — calibration.ts RECALL_TITLE_CHARS; seam only).
|
|
78
|
+
* 100 (reverted from 150 on 2026-08-03): the full-corpus relevance eval
|
|
79
|
+
* (#3, §5) found 100 vs 150 statistically identical (0.6pts apart, N=40,
|
|
80
|
+
* full CI overlap); 100 saves ~13% tokens per block. */
|
|
80
81
|
titleChars?: number;
|
|
81
82
|
/** Slots of `maxItems` guaranteed to the pure-prompt (unblended) search's
|
|
82
|
-
* top passing hit(s) — the #324 novelty floor.
|
|
83
|
-
*
|
|
84
|
-
* takeover
|
|
85
|
-
* Clamped to [0, maxItems]. */
|
|
83
|
+
* top passing hit(s) — the #324 novelty floor. Release-managed
|
|
84
|
+
* (calibration.ts NOVELTY_FLOOR_SLOTS; seam only); 2 mirrors
|
|
85
|
+
* coldExposureSlots sizing: small, a floor not a takeover. 0 disables the
|
|
86
|
+
* pure-prompt search entirely (the kill-switch). Clamped to [0, maxItems]. */
|
|
86
87
|
noveltyFloorSlots?: number;
|
|
87
88
|
}
|
|
88
|
-
/** Default #324 novelty-floor slots (
|
|
89
|
-
* coldExposureSlots sizing — enough to
|
|
90
|
-
* plus a runner-up, never a takeover of
|
|
91
|
-
* slots when a pure-prompt hit differs
|
|
92
|
-
* switch); continuing-intent sessions pay
|
|
93
|
-
* log's knob line (mcp-server
|
|
89
|
+
/** Default #324 novelty-floor slots (release-managed — calibration.ts
|
|
90
|
+
* NOVELTY_FLOOR_SLOTS). 2 mirrors coldExposureSlots sizing — enough to
|
|
91
|
+
* guarantee the pure-prompt top hit plus a runner-up, never a takeover of
|
|
92
|
+
* the index. The floor only SPENDS slots when a pure-prompt hit differs
|
|
93
|
+
* from the blended picks (topic switch); continuing-intent sessions pay
|
|
94
|
+
* nothing. Exported for the boot log's knob line (mcp-server prints the
|
|
95
|
+
* calibration constant here, once). */
|
|
94
96
|
export declare const DEFAULT_NOVELTY_FLOOR_SLOTS = 2;
|
|
95
97
|
/** Resolve the EFFECTIVE novelty floor (raw ?? default, clamped to
|
|
96
98
|
* [0, maxItems]) — one definition shared by the handler and the boot knob
|
|
97
99
|
* line so the logged value is what handleRecallIndex actually uses.
|
|
98
|
-
* maxItems may be the handler's already-resolved number OR raw
|
|
100
|
+
* maxItems may be the handler's already-resolved number OR raw/undefined
|
|
99
101
|
* (boot-log site) — raw is resolved with the handler's exact constants. */
|
|
100
102
|
export declare function resolveNoveltyFloorSlots(rawSlots: unknown, rawMaxItems: unknown): number;
|
|
101
103
|
export interface RecallIndexResult {
|
|
@@ -118,7 +120,7 @@ export declare function memoryTitle(content: string, maxLen?: number): string;
|
|
|
118
120
|
* Render one production index line. Exported (2026-08-02, relevance eval #v2)
|
|
119
121
|
* so the eval can measure the REAL rendered surface instead of reimplementing
|
|
120
122
|
* it — `maxLen` threads through to `memoryTitle` unchanged (default
|
|
121
|
-
* DEFAULT_TITLE_CHARS = 100,
|
|
123
|
+
* DEFAULT_TITLE_CHARS = 100, release-managed since #408) so the eval's snippet-length
|
|
122
124
|
* sweep (spec §4.2) can call this SAME function at 100/150/title1sent without
|
|
123
125
|
* duplicating the date/scope/agent/type meta-line logic.
|
|
124
126
|
*/
|
package/dist/recall-index.js
CHANGED
|
@@ -94,28 +94,31 @@ exports.formatMemoryGetText = formatMemoryGetText;
|
|
|
94
94
|
const storage = __importStar(require("./storage.js"));
|
|
95
95
|
const type_labels_js_1 = require("./type-labels.js");
|
|
96
96
|
const retrieval_js_1 = require("./retrieval.js");
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
*
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
97
|
+
const CALIBRATION = __importStar(require("./calibration.js"));
|
|
98
|
+
/** Relevance-gate floor for vector-only candidates (release-managed —
|
|
99
|
+
* calibration.ts RECALL_MIN_SIMILARITY). 0.62 (was 0.55; raised 2026-08-03
|
|
100
|
+
* on the fine-grain floor sweep — see the minSimilarity doc above). */
|
|
101
|
+
const DEFAULT_MIN_SIMILARITY = CALIBRATION.RECALL_MIN_SIMILARITY;
|
|
102
|
+
/** Max index lines per pushed recall block (release-managed — calibration.ts
|
|
103
|
+
* RECALL_MAX_ITEMS). 5 (was 6; lowered 2026-08-03 — slot 6 is pure padding
|
|
104
|
+
* at floor 0.62). */
|
|
105
|
+
const DEFAULT_MAX_ITEMS = CALIBRATION.RECALL_MAX_ITEMS;
|
|
106
|
+
const DEFAULT_MIN_PROMPT_LENGTH = CALIBRATION.RECALL_MIN_PROMPT_CHARS;
|
|
105
107
|
/** Default index-line title length. 100 (reverted from 150 on 2026-08-03:
|
|
106
108
|
* eval #3 §5 showed 100 vs 150 statistically identical; 100 saves ~13% tokens). */
|
|
107
|
-
const DEFAULT_TITLE_CHARS =
|
|
108
|
-
/** Default #324 novelty-floor slots (
|
|
109
|
-
* coldExposureSlots sizing — enough to
|
|
110
|
-
* plus a runner-up, never a takeover of
|
|
111
|
-
* slots when a pure-prompt hit differs
|
|
112
|
-
* switch); continuing-intent sessions pay
|
|
113
|
-
* log's knob line (mcp-server
|
|
114
|
-
|
|
109
|
+
const DEFAULT_TITLE_CHARS = CALIBRATION.RECALL_TITLE_CHARS;
|
|
110
|
+
/** Default #324 novelty-floor slots (release-managed — calibration.ts
|
|
111
|
+
* NOVELTY_FLOOR_SLOTS). 2 mirrors coldExposureSlots sizing — enough to
|
|
112
|
+
* guarantee the pure-prompt top hit plus a runner-up, never a takeover of
|
|
113
|
+
* the index. The floor only SPENDS slots when a pure-prompt hit differs
|
|
114
|
+
* from the blended picks (topic switch); continuing-intent sessions pay
|
|
115
|
+
* nothing. Exported for the boot log's knob line (mcp-server prints the
|
|
116
|
+
* calibration constant here, once). */
|
|
117
|
+
exports.DEFAULT_NOVELTY_FLOOR_SLOTS = CALIBRATION.NOVELTY_FLOOR_SLOTS;
|
|
115
118
|
/** Resolve the EFFECTIVE novelty floor (raw ?? default, clamped to
|
|
116
119
|
* [0, maxItems]) — one definition shared by the handler and the boot knob
|
|
117
120
|
* line so the logged value is what handleRecallIndex actually uses.
|
|
118
|
-
* maxItems may be the handler's already-resolved number OR raw
|
|
121
|
+
* maxItems may be the handler's already-resolved number OR raw/undefined
|
|
119
122
|
* (boot-log site) — raw is resolved with the handler's exact constants. */
|
|
120
123
|
function resolveNoveltyFloorSlots(rawSlots, rawMaxItems) {
|
|
121
124
|
const maxItems = typeof rawMaxItems === "number"
|
|
@@ -164,7 +167,7 @@ function formatDate(iso) {
|
|
|
164
167
|
* Render one production index line. Exported (2026-08-02, relevance eval #v2)
|
|
165
168
|
* so the eval can measure the REAL rendered surface instead of reimplementing
|
|
166
169
|
* it — `maxLen` threads through to `memoryTitle` unchanged (default
|
|
167
|
-
* DEFAULT_TITLE_CHARS = 100,
|
|
170
|
+
* DEFAULT_TITLE_CHARS = 100, release-managed since #408) so the eval's snippet-length
|
|
168
171
|
* sweep (spec §4.2) can call this SAME function at 100/150/title1sent without
|
|
169
172
|
* duplicating the date/scope/agent/type meta-line logic.
|
|
170
173
|
*/
|
|
@@ -36,7 +36,8 @@
|
|
|
36
36
|
* LRU beyond maxSessions so long-running servers don't accumulate state.
|
|
37
37
|
*/
|
|
38
38
|
export interface RecallRegistryOptions {
|
|
39
|
-
/** Turns a shown id stays suppressed.
|
|
39
|
+
/** Turns a shown id stays suppressed. Release-managed since #408 —
|
|
40
|
+
* calibration.ts RECALL_RESHOW_TURNS (30); this field is the eval/test seam. */
|
|
40
41
|
reshowTurns?: number;
|
|
41
42
|
/** Max tracked sessions before LRU eviction. */
|
|
42
43
|
maxSessions?: number;
|
package/dist/recall-registry.js
CHANGED
|
@@ -36,10 +36,44 @@
|
|
|
36
36
|
* early re-shows (~15 tokens each) — harmless by design. Sessions are pruned
|
|
37
37
|
* LRU beyond maxSessions so long-running servers don't accumulate state.
|
|
38
38
|
*/
|
|
39
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
40
|
+
if (k2 === undefined) k2 = k;
|
|
41
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
42
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
43
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
44
|
+
}
|
|
45
|
+
Object.defineProperty(o, k2, desc);
|
|
46
|
+
}) : (function(o, m, k, k2) {
|
|
47
|
+
if (k2 === undefined) k2 = k;
|
|
48
|
+
o[k2] = m[k];
|
|
49
|
+
}));
|
|
50
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
51
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
52
|
+
}) : function(o, v) {
|
|
53
|
+
o["default"] = v;
|
|
54
|
+
});
|
|
55
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
56
|
+
var ownKeys = function(o) {
|
|
57
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
58
|
+
var ar = [];
|
|
59
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
60
|
+
return ar;
|
|
61
|
+
};
|
|
62
|
+
return ownKeys(o);
|
|
63
|
+
};
|
|
64
|
+
return function (mod) {
|
|
65
|
+
if (mod && mod.__esModule) return mod;
|
|
66
|
+
var result = {};
|
|
67
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
68
|
+
__setModuleDefault(result, mod);
|
|
69
|
+
return result;
|
|
70
|
+
};
|
|
71
|
+
})();
|
|
39
72
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
40
73
|
exports.SessionRecallRegistry = exports.DEFAULT_RESHOW_TURNS = void 0;
|
|
41
74
|
const schema_prototypes_js_1 = require("./schema-prototypes.js");
|
|
42
|
-
|
|
75
|
+
const CALIBRATION = __importStar(require("./calibration.js"));
|
|
76
|
+
exports.DEFAULT_RESHOW_TURNS = CALIBRATION.RECALL_RESHOW_TURNS;
|
|
43
77
|
const DEFAULT_MAX_SESSIONS = 500;
|
|
44
78
|
class SessionRecallRegistry {
|
|
45
79
|
reshowTurns;
|
|
@@ -15,11 +15,58 @@
|
|
|
15
15
|
* #392 — one zone system, ONE verdict per pair: below `correctionMinSimilarity`
|
|
16
16
|
* (floor, 0.75) pairs are not candidates; in [floor, `dedupAutoMergeThreshold`)
|
|
17
17
|
* (ceiling, 0.92) each unlinked pair gets ONE verdict call whose action is
|
|
18
|
-
* `merge` | `corrects` | `supersedes` | `none`; at/above the
|
|
19
|
-
* deterministic merge zone (dedup.ts runDeterministicMergeZone —
|
|
20
|
-
* budget-free) owns the pair. The merge disposition reuses the
|
|
21
|
-
* execution (canonical pick, link re-point, dedup_log, metadata
|
|
22
|
-
* merge verdict below `correctionRewriteMinConfidence` keeps both
|
|
18
|
+
* `merge` | `corrects` | `supersedes` | `conflicts` | `none`; at/above the
|
|
19
|
+
* ceiling the deterministic merge zone (dedup.ts runDeterministicMergeZone —
|
|
20
|
+
* LLM-free, budget-free) owns the pair. The merge disposition reuses the
|
|
21
|
+
* dedup core's execution (canonical pick, link re-point, dedup_log, metadata
|
|
22
|
+
* rails); a merge verdict below `correctionRewriteMinConfidence` keeps both
|
|
23
|
+
* memories.
|
|
24
|
+
*
|
|
25
|
+
* #393 increment B — the SCOUT, a second detection source with the SAME
|
|
26
|
+
* judge: the similarity floor is structurally blind to corrections riding
|
|
27
|
+
* inside topically unrelated memories (the field failure — cosine ~0.5-0.6 to
|
|
28
|
+
* their target, zero `corrects` verdicts in the whole corpus baseline), so
|
|
29
|
+
* per NEW memory ONE classify-tier shape call asks whether it corrects/
|
|
30
|
+
* retracts/supersedes/CONTRADICTS something previously recorded (guard-C
|
|
31
|
+
* extended the question); correction-shaped
|
|
32
|
+
* memories FTS the corpus with the referenced claim's distinctive terms (the
|
|
33
|
+
* correction CONTAINS the words of what it corrects) and the hits become
|
|
34
|
+
* candidate pairs in the SAME verdict loop — no similarity gate for this
|
|
35
|
+
* source: cosine is a ranker/link strength, never a blocker. Per-source
|
|
36
|
+
* counters (scout_scanned / scout_correction_shaped / scout_candidates_found)
|
|
37
|
+
* ride the stage report; cosine band stats stay similarity-source-only.
|
|
38
|
+
*
|
|
39
|
+
* #393 guard-C — the conflicts flag + the zone-runs-last order: judgment
|
|
40
|
+
* OUTRANKS the deterministic sweep. A `conflicts` verdict writes a symmetric
|
|
41
|
+
* `conflicts` link (the pair genuinely disagrees — cannot both be true) and
|
|
42
|
+
* NOTHING else: no status change, no rewrite, no merge queue; both records
|
|
43
|
+
* stay live so the consumer sees both truths. Both merge paths (the zone's
|
|
44
|
+
* planDedup and the judged mergeMemoryIds) refuse to blend a conflicts-linked
|
|
45
|
+
* pair, counted as conflict_skipped. The zone therefore runs AFTER the scan —
|
|
46
|
+
* with the zone first, a >=0.92 conflict pair was blended
|
|
47
|
+
* before the judge ever saw it (the planted-eval harm: canonical=older, the
|
|
48
|
+
* newer truth erased); running it last means verdicts/marks/binds land first
|
|
49
|
+
* and the zone merges only what no verdict claimed — a conflicts bind set by
|
|
50
|
+
* this run's scan guards the SAME run's zone.
|
|
51
|
+
*
|
|
52
|
+
* #439 apply-on-confirm — confirmed merges and rewrite groups apply at the
|
|
53
|
+
* candidate BOUNDARY (the end of the scan iteration that confirmed them), not
|
|
54
|
+
* in post-scan phases. The old end-of-run batch was a completion assumption
|
|
55
|
+
* written when nightlies finished in an hour; under #405 budget pressure it
|
|
56
|
+
* became a days-long queue where confirmed work never landed and every night
|
|
57
|
+
* re-paid the judgment cost (cursor held below un-applied groups, pairs
|
|
58
|
+
* re-detected, re-judged). Now each judged-merge pair applies via
|
|
59
|
+
* mergeMemoryIds in its OWN transaction at confirmation time, each rewrite
|
|
60
|
+
* group via its own rewrite call + applyRewriteGroup transaction; the cursor
|
|
61
|
+
* advances per APPLIED candidate, so a deferral holds it below exactly ONE
|
|
62
|
+
* candidate's pairs. One pre-merge backup per run (lazy, before the first
|
|
63
|
+
* application); the capture lock is taken per boundary batch with a same-run
|
|
64
|
+
* retry list + a final drain. A shared trigger IS the current candidate, so
|
|
65
|
+
* the multi-target keep rule resolves across the boundary's groups (any keep
|
|
66
|
+
* keeps). A target corrected by two different candidates takes two sequential
|
|
67
|
+
* rewrites — the second composes the already-corrected story — instead of one
|
|
68
|
+
* grouped call (the ONE-call grouping was a cost optimization, not a
|
|
69
|
+
* correctness invariant; accepted semantics change).
|
|
23
70
|
*
|
|
24
71
|
* Status vocabulary (code-defined, extensible — deliberately NOT config):
|
|
25
72
|
* NULL/'active' default | 'superseded' + 'retracted' demote in ranking |
|
|
@@ -41,40 +88,25 @@ import type { LlmClient, LlmUsage } from "./llm.js";
|
|
|
41
88
|
import type { ConsolidationReport } from "./types.js";
|
|
42
89
|
import type { EmbedFn } from "./retrieval.js";
|
|
43
90
|
import { acquireCaptureLock } from "./capture.js";
|
|
91
|
+
import type { RunDeadline } from "./run-deadline.js";
|
|
44
92
|
/** Stage label used for every budget.use()/recordUsage() call (#384). */
|
|
45
93
|
export declare const RECONSOLIDATION_STAGE_LABEL = "reconsolidation";
|
|
46
94
|
/**
|
|
47
|
-
* Default minimum COSINE similarity for a correction candidate pair
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
*
|
|
51
|
-
*
|
|
95
|
+
* Default minimum COSINE similarity for a correction candidate pair —
|
|
96
|
+
* RELEASE-MANAGED since #408 (calibration.ts CORRECTION_MIN_SIMILARITY;
|
|
97
|
+
* provenance there). Lower than the supersession stage's 0.80 on purpose: a
|
|
98
|
+
* retraction often rides inside an otherwise unrelated memory (the field
|
|
99
|
+
* failure that opened this issue), so the neighborhood gate must be a touch
|
|
100
|
+
* wider while the LLM verdict + confidence gate carry the precision load.
|
|
52
101
|
*/
|
|
53
102
|
export declare const DEFAULT_CORRECTION_MIN_SIMILARITY = 0.75;
|
|
54
103
|
/**
|
|
55
|
-
* Default minimum verdict confidence for the REWRITE fork
|
|
104
|
+
* Default minimum verdict confidence for the REWRITE fork — release-managed
|
|
105
|
+
* (calibration.ts CORRECTION_REWRITE_MIN_CONFIDENCE). Below this a
|
|
56
106
|
* `corrects` verdict degrades to mark-only — a weak mark is recoverable, a
|
|
57
107
|
* weak rewrite is corruption.
|
|
58
108
|
*/
|
|
59
109
|
export declare const DEFAULT_CORRECTION_REWRITE_MIN_CONFIDENCE = 0.8;
|
|
60
|
-
/**
|
|
61
|
-
* Default wall-clock bound for the stage, in minutes (#401). Checked at the
|
|
62
|
-
* top of the candidate scan loop (and before each rewrite contract call); on
|
|
63
|
-
* expiry the scan breaks cleanly at the last fully-considered candidate and
|
|
64
|
-
* the next run resumes from the persisted cursor. 120 sits safely under any
|
|
65
|
-
* sane process-level nightly timeout. 0 disables the bound. Invalid →
|
|
66
|
-
* default.
|
|
67
|
-
*/
|
|
68
|
-
export declare const DEFAULT_RECONSOLIDATION_MAX_MINUTES = 120;
|
|
69
|
-
/**
|
|
70
|
-
* Default per-run classify-call ceiling for the stage (#401) — the
|
|
71
|
-
* supersessionMaxCalls pattern with a NON-ZERO default ON PURPOSE: that
|
|
72
|
-
* knob's 0=unlimited default is what let the first full-corpus pass grow
|
|
73
|
-
* unbounded. Counts EVERY classify-tier call the stage makes (mark
|
|
74
|
-
* verifications, pair verdicts, rewrite contracts). 0 disables the cap.
|
|
75
|
-
* Invalid → default.
|
|
76
|
-
*/
|
|
77
|
-
export declare const DEFAULT_RECONSOLIDATION_MAX_CALLS = 600;
|
|
78
110
|
/** Head of the old content quoted in the provenance footer. */
|
|
79
111
|
export declare const FOOTER_HEAD_MAX_CHARS = 160;
|
|
80
112
|
/** The code-defined status vocabulary (see module doc). Not user-configurable. */
|
|
@@ -87,50 +119,38 @@ export declare const DEMOTED_STATUSES: readonly [MemoryStatus, MemoryStatus];
|
|
|
87
119
|
/** structural subset of consolidate.BudgetTracker (avoids an import cycle). */
|
|
88
120
|
export interface StageBudget {
|
|
89
121
|
readonly exhausted: boolean;
|
|
90
|
-
use(stage: string
|
|
122
|
+
use(stage: string): boolean;
|
|
91
123
|
recordUsage(stage: string, usage: LlmUsage | undefined): void;
|
|
92
124
|
}
|
|
93
125
|
export interface ReconsolidationOptions {
|
|
94
|
-
/**
|
|
126
|
+
/** Correction-pair cosine floor. Release-managed default (calibration.ts
|
|
127
|
+
* CORRECTION_MIN_SIMILARITY, 0.75); this field is the eval/test seam.
|
|
128
|
+
* Invalid → default. */
|
|
95
129
|
minSimilarity?: number;
|
|
96
|
-
/**
|
|
130
|
+
/** Rewrite-fork confidence floor. Release-managed default (calibration.ts
|
|
131
|
+
* CORRECTION_REWRITE_MIN_CONFIDENCE, 0.80); seam only. Invalid → default. */
|
|
97
132
|
rewriteMinConfidence?: number;
|
|
98
133
|
/**
|
|
99
|
-
*
|
|
100
|
-
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
* Invalid → default.
|
|
134
|
+
* The deterministic/LLM boundary of the unified resolution pass (#392):
|
|
135
|
+
* pairs at/above it merge via the LLM-free zone, pairs in [floor, ceiling)
|
|
136
|
+
* get the verdict. Release-managed default (calibration.ts
|
|
137
|
+
* DEDUP_AUTO_MERGE_THRESHOLD, 0.92); seam only. Invalid → default.
|
|
104
138
|
*/
|
|
105
139
|
autoMergeThreshold?: number;
|
|
106
140
|
/**
|
|
107
|
-
*
|
|
108
|
-
*
|
|
109
|
-
*
|
|
141
|
+
* The run-wide pipeline deadline (#405 — successor of the stage-local
|
|
142
|
+
* reconsolidationMaxMinutes clock, #401): created at nightly start, shared
|
|
143
|
+
* with capture and every other stage, threaded here by runConsolidation.
|
|
144
|
+
* Checked at the top of the candidate scan loop and before each rewrite
|
|
145
|
+
* contract call; on expiry the scan breaks cleanly — the cursor already
|
|
146
|
+
* points at the last fully-considered candidate, so the run ends
|
|
147
|
+
* consistent and the next nightly resumes from it.
|
|
110
148
|
*/
|
|
111
|
-
|
|
149
|
+
deadline?: RunDeadline;
|
|
112
150
|
/**
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
*
|
|
116
|
-
* expiry the scan breaks cleanly — the cursor already points at the last
|
|
117
|
-
* fully-considered candidate, so the run ends consistent and the next
|
|
118
|
-
* nightly resumes from it. Invalid → default.
|
|
119
|
-
*/
|
|
120
|
-
maxMinutes?: number;
|
|
121
|
-
/**
|
|
122
|
-
* reconsolidationMaxCalls (config; default 600; 0 disables) — per-run
|
|
123
|
-
* ceiling on classify-tier calls for the stage (#401), the
|
|
124
|
-
* supersessionMaxCalls pattern with a NON-ZERO default (the 0=unlimited
|
|
125
|
-
* default there is what removed the last per-stage bound). Exhaustion
|
|
126
|
-
* mid-neighbor-loop or mid-rewrite-phase stops/defers cleanly at the
|
|
127
|
-
* current candidate boundary. Invalid → default.
|
|
128
|
-
*/
|
|
129
|
-
maxCalls?: number;
|
|
130
|
-
/**
|
|
131
|
-
* Capture-lock acquirer override (tests) — the deterministic zone and the
|
|
132
|
-
* judged-merge phase each hold a short lock window. Defaults to the real
|
|
133
|
-
* capture.ts lock. DedupOptions.acquireLock pattern.
|
|
151
|
+
* Capture-lock acquirer override (tests) — the deterministic zone and each
|
|
152
|
+
* #439 boundary's judged-merge batch hold a short lock window. Defaults to
|
|
153
|
+
* the real capture.ts lock. DedupOptions.acquireLock pattern.
|
|
134
154
|
*/
|
|
135
155
|
acquireLock?: typeof acquireCaptureLock;
|
|
136
156
|
}
|
|
@@ -146,12 +166,50 @@ export declare function isFactShapedTarget(mem: {
|
|
|
146
166
|
content: string;
|
|
147
167
|
}): boolean;
|
|
148
168
|
/**
|
|
149
|
-
* The unified resolution verdict (#392): ONE call per unlinked
|
|
150
|
-
* how the newer memory relates to the older — merge (same
|
|
151
|
-
* fact/verdict, differing in wording/qualifiers), corrects,
|
|
152
|
-
*
|
|
169
|
+
* The unified resolution verdict (#392, #393 guard-C): ONE call per unlinked
|
|
170
|
+
* pair decides how the newer memory relates to the older — merge (same
|
|
171
|
+
* underlying fact/verdict, differing in wording/qualifiers), corrects,
|
|
172
|
+
* supersedes, conflicts (genuine disagreement — cannot both be true; flag,
|
|
173
|
+
* keep both, never blend), or none (related but distinct).
|
|
174
|
+
*/
|
|
175
|
+
export type ResolutionAction = "merge" | "corrects" | "supersedes" | "conflicts" | "none";
|
|
176
|
+
/**
|
|
177
|
+
* The scout's correction-shape answer (#393 B, guard-C): does this NEW memory
|
|
178
|
+
* correct, retract, supersede, or CONTRADICT something previously recorded —
|
|
179
|
+
* and if so, which distinctive terms does the referenced (old) claim carry?
|
|
180
|
+
* `correction: true` means "resolution-shaped": corrects/retracts/supersedes/
|
|
181
|
+
* contradicts an earlier claim. `references` feeds an FTS query against the
|
|
182
|
+
* corpus; the shape call is the ONLY LLM work the scout adds per memory
|
|
183
|
+
* (non-corrections stop there), and it rides the same `complete()` surface +
|
|
184
|
+
* stage budget as every other call (#405 — there is no separate classify-tier
|
|
185
|
+
* ceiling to configure).
|
|
186
|
+
*/
|
|
187
|
+
export interface ScoutShape {
|
|
188
|
+
correction: boolean;
|
|
189
|
+
/** Distinctive terms of the referenced old claim ("" when not resolution-shaped). */
|
|
190
|
+
references: string;
|
|
191
|
+
/** Informational only — never gates behavior (no uncalibrated parameters). */
|
|
192
|
+
confidence: number;
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* Build the constrained correction-shape prompt (classify-tier cost profile:
|
|
196
|
+
* 1500-char truncation, supersession/verdict precedent). The wording asks for
|
|
197
|
+
* the OLD claim's distinctive terms — the field-failure mechanism is that a
|
|
198
|
+
* correction CONTAINS the words of what it corrects, even when the surrounding
|
|
199
|
+
* topics (and therefore the embedding cosine) are unrelated. Guard-C extends
|
|
200
|
+
* the question to contradictions: two records that disagree on the same
|
|
201
|
+
* quantity share even MORE wording than a cross-topic correction does.
|
|
202
|
+
*/
|
|
203
|
+
export declare function buildScoutShapePrompt(content: string): string;
|
|
204
|
+
/**
|
|
205
|
+
* Parse the scout shape reply. Null on unparseable JSON, a missing/non-boolean
|
|
206
|
+
* `correction`, or a missing/out-of-range `confidence` — the caller counts
|
|
207
|
+
* skipped_infra and moves on (parseSupersessionReply discipline: never
|
|
208
|
+
* mis-detect on ambiguity). `references` is lenient (missing/non-string → "")
|
|
209
|
+
* because an empty string simply yields no FTS hits — a harmless miss, not a
|
|
210
|
+
* mis-judgment.
|
|
153
211
|
*/
|
|
154
|
-
export
|
|
212
|
+
export declare function parseScoutShape(reply: string): ScoutShape | null;
|
|
155
213
|
/** Build the constrained pair-verdict prompt (1500-char truncation, supersession precedent). */
|
|
156
214
|
export declare function buildCorrectionVerdictPrompt(oldContent: string, newContent: string): string;
|
|
157
215
|
export interface CorrectionVerdict {
|
|
@@ -290,32 +348,55 @@ export declare function buildResolutionBands(floor: number, ceiling: number): Re
|
|
|
290
348
|
*/
|
|
291
349
|
export declare function bandForCosine(bands: ResolutionBand[], cosine: number): ResolutionBand | null;
|
|
292
350
|
/**
|
|
293
|
-
* Nightly reconsolidation stage (#384, #392 — THE unified resolution stage
|
|
351
|
+
* Nightly reconsolidation stage (#384, #392 — THE unified resolution stage;
|
|
352
|
+
* #439 apply-on-confirm).
|
|
294
353
|
*
|
|
295
|
-
* Phase
|
|
296
|
-
*
|
|
354
|
+
* Phase order (#393 guard-C): the deterministic merge zone (pairs >= the
|
|
355
|
+
* ceiling) runs LAST — after the scan (which now includes every judged-merge
|
|
356
|
+
* application and rewrite, #439). Judgment outranks the deterministic sweep:
|
|
357
|
+
* verdicts, marks, and binds land first and the zone merges only what no
|
|
358
|
+
* verdict claimed. With the zone first, a >=0.92 genuine-conflict pair was
|
|
359
|
+
* blended before the judge ever saw it (the planted-eval harm); running it
|
|
360
|
+
* last means a `conflicts` bind set by this run's scan guards the SAME run's
|
|
361
|
+
* zone. Zone internals (lock, backup, deadline, persistBand, fail-soft) are
|
|
362
|
+
* unchanged.
|
|
297
363
|
*
|
|
298
364
|
* Scan: every memory with rowid > reconsolidationCursor (no shape filter;
|
|
299
365
|
* absorbed candidates are skipped — invisible memories are not re-judged).
|
|
300
|
-
* Each candidate's pairs: incoming explicit marks (verified once, AC7) then
|
|
301
|
-
*
|
|
302
|
-
*
|
|
303
|
-
*
|
|
304
|
-
*
|
|
305
|
-
*
|
|
366
|
+
* Each candidate's pairs: incoming explicit marks (verified once, AC7), then
|
|
367
|
+
* ONE scout shape call (#393 B — flags correction shape; non-corrections stop
|
|
368
|
+
* there), then up-to-5 older KNN neighbors in [floor, ceiling) (verdict call
|
|
369
|
+
* per unlinked pair, AC2 — pairs at/above the ceiling are counted, never
|
|
370
|
+
* judged) plus the scout's FTS hits for correction-shaped memories (same
|
|
371
|
+
* verdict loop, NO similarity gate; guard-C: a scout hit whose KNN twin sits
|
|
372
|
+
* at/above the ceiling is re-tagged scout so the pair IS judged instead of
|
|
373
|
+
* being left for the zone to blend). Confirmed `corrects` pairs above the
|
|
374
|
+
* confidence gate on fact-shaped targets group by target; a `conflicts`
|
|
375
|
+
* verdict writes the conflicts link and nothing else (both live); everything
|
|
376
|
+
* else is mark-only.
|
|
306
377
|
*
|
|
307
|
-
*
|
|
308
|
-
*
|
|
309
|
-
*
|
|
378
|
+
* #439 BOUNDARY apply: at the END of each candidate iteration everything it
|
|
379
|
+
* confirmed applies IMMEDIATELY — merges first (each judged-merge pair via
|
|
380
|
+
* mergeMemoryIds in its own transaction, under the boundary's short lock
|
|
381
|
+
* window; ONE lazy pre-merge backup per run), then the iteration's rewrite
|
|
382
|
+
* groups (one rewrite LLM call + one applyRewriteGroup transaction each;
|
|
383
|
+
* dispositions resolved ACROSS the boundary's groups — the multi-target keep
|
|
384
|
+
* rule: a trigger absorbed only if every group's contract says absorb). A
|
|
385
|
+
* busy capture lock pushes the boundary's merges onto a same-run retry list
|
|
386
|
+
* (retried at the next boundary and once in a final drain after the scan);
|
|
387
|
+
* a deadline, a backup failure, or a rewrite-call refusal/infra error defers
|
|
388
|
+
* the remaining work and holds the cursor.
|
|
310
389
|
*
|
|
311
|
-
* Cursor discipline
|
|
312
|
-
*
|
|
313
|
-
*
|
|
314
|
-
*
|
|
315
|
-
*
|
|
316
|
-
*
|
|
317
|
-
*
|
|
318
|
-
*
|
|
390
|
+
* Cursor discipline: the cursor advances past a candidate only when its
|
|
391
|
+
* iteration's confirmed work has LANDED (or was refused-with-verdict-rendered:
|
|
392
|
+
* metadata mismatch, conflict-linked, mark-only fallback). A deferral holds
|
|
393
|
+
* the cursor BELOW the current candidate — bounded to ONE candidate's pairs,
|
|
394
|
+
* re-detected and re-judged next run (dup-over-loss — a confirmed resolution
|
|
395
|
+
* must never be silently dropped by the cursor passing it). The separate
|
|
396
|
+
* scan high-water (state.reconsolidationScannedRowid) records the max
|
|
397
|
+
* candidate rowid ENTERED and is never held back, so the report can split
|
|
398
|
+
* verdict calls into pairs_reevaluated (at/below the prior high-water) vs
|
|
399
|
+
* pairs_new — the convergence measurement.
|
|
319
400
|
*
|
|
320
401
|
* Dry-run: the zone's discovery + the free idempotency check only — zero LLM
|
|
321
402
|
* calls, zero writes, no cursor or band-stats persistence. Gate discovery is
|