@gamaze/hicortex 0.20.6 → 0.20.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -41
- package/dist/calibration.d.ts +174 -0
- package/dist/calibration.js +231 -0
- package/dist/capture.d.ts +15 -3
- package/dist/capture.js +10 -1
- package/dist/classify-domains.d.ts +6 -0
- package/dist/classify-domains.js +7 -1
- package/dist/cli.js +2 -3
- package/dist/config-read.d.ts +1 -1
- package/dist/config-read.js +96 -9
- package/dist/consolidate.d.ts +79 -68
- package/dist/consolidate.js +218 -174
- package/dist/dashboard.d.ts +4 -3
- package/dist/dedup.d.ts +34 -26
- package/dist/dedup.js +91 -57
- package/dist/distiller.js +1 -1
- package/dist/domain-classify.d.ts +7 -6
- package/dist/domain-classify.js +12 -10
- package/dist/eval/decay-eval.d.ts +3 -3
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/planted-eval.d.ts +26 -0
- package/dist/eval/planted-eval.js +97 -0
- package/dist/eval/planted-fixtures.d.ts +107 -0
- package/dist/eval/planted-fixtures.js +283 -0
- package/dist/eval/planted-harness.d.ts +176 -0
- package/dist/eval/planted-harness.js +649 -0
- package/dist/index.js +4 -3
- package/dist/init.d.ts +9 -3
- package/dist/init.js +52 -9
- package/dist/llm.d.ts +43 -58
- package/dist/llm.js +87 -101
- package/dist/mcp-server.js +29 -29
- package/dist/nightly.js +105 -103
- package/dist/nofit.d.ts +4 -11
- package/dist/nofit.js +6 -23
- package/dist/recall-index.d.ts +30 -28
- package/dist/recall-index.js +21 -18
- package/dist/recall-registry.d.ts +2 -1
- package/dist/recall-registry.js +35 -1
- package/dist/reconsolidation.d.ts +124 -72
- package/dist/reconsolidation.js +390 -169
- package/dist/relink.js +3 -4
- package/dist/retrieval.d.ts +68 -35
- package/dist/retrieval.js +292 -104
- package/dist/run-deadline.d.ts +62 -0
- package/dist/run-deadline.js +73 -0
- package/dist/schema-prototypes.d.ts +3 -3
- package/dist/schema-prototypes.js +3 -3
- package/dist/state.d.ts +2 -3
- package/dist/storage.d.ts +16 -16
- package/dist/storage.js +62 -24
- package/dist/telemetry.d.ts +8 -7
- package/dist/token-budget.js +3 -4
- package/dist/type-classify.js +4 -4
- package/dist/types.d.ts +95 -155
- package/domains.example.json +4 -5
- package/hermes-plugin/hicortex/README.md +2 -2
- package/openclaw.plugin.json +1 -1
- package/package.json +2 -1
- package/pi-extension/hicortex/README.md +1 -1
- package/server.json +3 -3
package/dist/recall-index.d.ts
CHANGED
|
@@ -51,9 +51,10 @@ import type { MemorySearchResult } from "./types.js";
|
|
|
51
51
|
import * as storage from "./storage.js";
|
|
52
52
|
import { SessionRecallRegistry } from "./recall-registry.js";
|
|
53
53
|
export interface RecallIndexOptions {
|
|
54
|
-
/** Minimum measured cosine for vector-only candidates (
|
|
55
|
-
*
|
|
56
|
-
*
|
|
54
|
+
/** Minimum measured cosine for vector-only candidates (release-managed
|
|
55
|
+
* since #408 — calibration.ts RECALL_MIN_SIMILARITY; this field is the
|
|
56
|
+
* eval/test seam). FTS-matched candidates pass regardless — a BM25
|
|
57
|
+
* text match is direct evidence of relevance. 0.62 (raised from 0.55
|
|
57
58
|
* on 2026-08-03 per a 0.01-step floor sweep on the rewritten corpus): steady
|
|
58
59
|
* ~3:1 noise:signal removal with no knee; 0.62 = +2.2pts precision, 10/98
|
|
59
60
|
* prompts silent, sits below the 0.63 local pessimum. The floor is a noise
|
|
@@ -61,41 +62,42 @@ export interface RecallIndexOptions {
|
|
|
61
62
|
* comes with ~1.5 wrongly-silenced (real signal); a non-cosine gate is the
|
|
62
63
|
* real silence fix (eval #3 §4). */
|
|
63
64
|
minSimilarity?: number;
|
|
64
|
-
/** Max index lines per response (
|
|
65
|
-
* (lowered from 6 on 2026-08-03). Per-slot
|
|
66
|
-
* slot 6 gives NO prompt its first relevant
|
|
67
|
-
* robust, prompt-set-independent finding, and
|
|
68
|
-
* is monotone (precision@4 33.7% > @6 30.6% >
|
|
69
|
-
* lower-noise — but the 4-vs-5 distinction rests on 5
|
|
70
|
-
* overfitting-fragile (K and the floor were tuned on
|
|
71
|
-
* with coverage at modest cost.
|
|
65
|
+
/** Max index lines per response (release-managed — calibration.ts
|
|
66
|
+
* RECALL_MAX_ITEMS; seam only). 5 (lowered from 6 on 2026-08-03). Per-slot
|
|
67
|
+
* decomposition at floor 0.62: slot 6 gives NO prompt its first relevant
|
|
68
|
+
* memory — "6 is wrong" is the robust, prompt-set-independent finding, and
|
|
69
|
+
* 5 captures it. The K-sweep is monotone (precision@4 33.7% > @6 30.6% >
|
|
70
|
+
* @8 28.3%), so 4 is lower-noise — but the 4-vs-5 distinction rests on 5
|
|
71
|
+
* of 98 prompts and is overfitting-fragile (K and the floor were tuned on
|
|
72
|
+
* the same set); 5 hedges with coverage at modest cost. */
|
|
72
73
|
maxItems?: number;
|
|
73
74
|
/** Prompts shorter than this are skipped (continuations, "yes", "do it"). */
|
|
74
75
|
minPromptLength?: number;
|
|
75
|
-
/** Max chars of the memory's first line shown in an index entry
|
|
76
|
-
*
|
|
77
|
-
*
|
|
78
|
-
* identical (0.6pts apart, N=40,
|
|
79
|
-
* per block. */
|
|
76
|
+
/** Max chars of the memory's first line shown in an index entry
|
|
77
|
+
* (release-managed — calibration.ts RECALL_TITLE_CHARS; seam only).
|
|
78
|
+
* 100 (reverted from 150 on 2026-08-03): the full-corpus relevance eval
|
|
79
|
+
* (#3, §5) found 100 vs 150 statistically identical (0.6pts apart, N=40,
|
|
80
|
+
* full CI overlap); 100 saves ~13% tokens per block. */
|
|
80
81
|
titleChars?: number;
|
|
81
82
|
/** Slots of `maxItems` guaranteed to the pure-prompt (unblended) search's
|
|
82
|
-
* top passing hit(s) — the #324 novelty floor.
|
|
83
|
-
*
|
|
84
|
-
* takeover
|
|
85
|
-
* Clamped to [0, maxItems]. */
|
|
83
|
+
* top passing hit(s) — the #324 novelty floor. Release-managed
|
|
84
|
+
* (calibration.ts NOVELTY_FLOOR_SLOTS; seam only); 2 mirrors
|
|
85
|
+
* coldExposureSlots sizing: small, a floor not a takeover. 0 disables the
|
|
86
|
+
* pure-prompt search entirely (the kill-switch). Clamped to [0, maxItems]. */
|
|
86
87
|
noveltyFloorSlots?: number;
|
|
87
88
|
}
|
|
88
|
-
/** Default #324 novelty-floor slots (
|
|
89
|
-
* coldExposureSlots sizing — enough to
|
|
90
|
-
* plus a runner-up, never a takeover of
|
|
91
|
-
* slots when a pure-prompt hit differs
|
|
92
|
-
* switch); continuing-intent sessions pay
|
|
93
|
-
* log's knob line (mcp-server
|
|
89
|
+
/** Default #324 novelty-floor slots (release-managed — calibration.ts
|
|
90
|
+
* NOVELTY_FLOOR_SLOTS). 2 mirrors coldExposureSlots sizing — enough to
|
|
91
|
+
* guarantee the pure-prompt top hit plus a runner-up, never a takeover of
|
|
92
|
+
* the index. The floor only SPENDS slots when a pure-prompt hit differs
|
|
93
|
+
* from the blended picks (topic switch); continuing-intent sessions pay
|
|
94
|
+
* nothing. Exported for the boot log's knob line (mcp-server prints the
|
|
95
|
+
* calibration constant here, once). */
|
|
94
96
|
export declare const DEFAULT_NOVELTY_FLOOR_SLOTS = 2;
|
|
95
97
|
/** Resolve the EFFECTIVE novelty floor (raw ?? default, clamped to
|
|
96
98
|
* [0, maxItems]) — one definition shared by the handler and the boot knob
|
|
97
99
|
* line so the logged value is what handleRecallIndex actually uses.
|
|
98
|
-
* maxItems may be the handler's already-resolved number OR raw
|
|
100
|
+
* maxItems may be the handler's already-resolved number OR raw/undefined
|
|
99
101
|
* (boot-log site) — raw is resolved with the handler's exact constants. */
|
|
100
102
|
export declare function resolveNoveltyFloorSlots(rawSlots: unknown, rawMaxItems: unknown): number;
|
|
101
103
|
export interface RecallIndexResult {
|
|
@@ -118,7 +120,7 @@ export declare function memoryTitle(content: string, maxLen?: number): string;
|
|
|
118
120
|
* Render one production index line. Exported (2026-08-02, relevance eval #v2)
|
|
119
121
|
* so the eval can measure the REAL rendered surface instead of reimplementing
|
|
120
122
|
* it — `maxLen` threads through to `memoryTitle` unchanged (default
|
|
121
|
-
* DEFAULT_TITLE_CHARS = 100,
|
|
123
|
+
* DEFAULT_TITLE_CHARS = 100, release-managed since #408) so the eval's snippet-length
|
|
122
124
|
* sweep (spec §4.2) can call this SAME function at 100/150/title1sent without
|
|
123
125
|
* duplicating the date/scope/agent/type meta-line logic.
|
|
124
126
|
*/
|
package/dist/recall-index.js
CHANGED
|
@@ -94,28 +94,31 @@ exports.formatMemoryGetText = formatMemoryGetText;
|
|
|
94
94
|
const storage = __importStar(require("./storage.js"));
|
|
95
95
|
const type_labels_js_1 = require("./type-labels.js");
|
|
96
96
|
const retrieval_js_1 = require("./retrieval.js");
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
*
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
97
|
+
const CALIBRATION = __importStar(require("./calibration.js"));
|
|
98
|
+
/** Relevance-gate floor for vector-only candidates (release-managed —
|
|
99
|
+
* calibration.ts RECALL_MIN_SIMILARITY). 0.62 (was 0.55; raised 2026-08-03
|
|
100
|
+
* on the fine-grain floor sweep — see the minSimilarity doc above). */
|
|
101
|
+
const DEFAULT_MIN_SIMILARITY = CALIBRATION.RECALL_MIN_SIMILARITY;
|
|
102
|
+
/** Max index lines per pushed recall block (release-managed — calibration.ts
|
|
103
|
+
* RECALL_MAX_ITEMS). 5 (was 6; lowered 2026-08-03 — slot 6 is pure padding
|
|
104
|
+
* at floor 0.62). */
|
|
105
|
+
const DEFAULT_MAX_ITEMS = CALIBRATION.RECALL_MAX_ITEMS;
|
|
106
|
+
const DEFAULT_MIN_PROMPT_LENGTH = CALIBRATION.RECALL_MIN_PROMPT_CHARS;
|
|
105
107
|
/** Default index-line title length. 100 (reverted from 150 on 2026-08-03:
|
|
106
108
|
* eval #3 §5 showed 100 vs 150 statistically identical; 100 saves ~13% tokens). */
|
|
107
|
-
const DEFAULT_TITLE_CHARS =
|
|
108
|
-
/** Default #324 novelty-floor slots (
|
|
109
|
-
* coldExposureSlots sizing — enough to
|
|
110
|
-
* plus a runner-up, never a takeover of
|
|
111
|
-
* slots when a pure-prompt hit differs
|
|
112
|
-
* switch); continuing-intent sessions pay
|
|
113
|
-
* log's knob line (mcp-server
|
|
114
|
-
|
|
109
|
+
const DEFAULT_TITLE_CHARS = CALIBRATION.RECALL_TITLE_CHARS;
|
|
110
|
+
/** Default #324 novelty-floor slots (release-managed — calibration.ts
|
|
111
|
+
* NOVELTY_FLOOR_SLOTS). 2 mirrors coldExposureSlots sizing — enough to
|
|
112
|
+
* guarantee the pure-prompt top hit plus a runner-up, never a takeover of
|
|
113
|
+
* the index. The floor only SPENDS slots when a pure-prompt hit differs
|
|
114
|
+
* from the blended picks (topic switch); continuing-intent sessions pay
|
|
115
|
+
* nothing. Exported for the boot log's knob line (mcp-server prints the
|
|
116
|
+
* calibration constant here, once). */
|
|
117
|
+
exports.DEFAULT_NOVELTY_FLOOR_SLOTS = CALIBRATION.NOVELTY_FLOOR_SLOTS;
|
|
115
118
|
/** Resolve the EFFECTIVE novelty floor (raw ?? default, clamped to
|
|
116
119
|
* [0, maxItems]) — one definition shared by the handler and the boot knob
|
|
117
120
|
* line so the logged value is what handleRecallIndex actually uses.
|
|
118
|
-
* maxItems may be the handler's already-resolved number OR raw
|
|
121
|
+
* maxItems may be the handler's already-resolved number OR raw/undefined
|
|
119
122
|
* (boot-log site) — raw is resolved with the handler's exact constants. */
|
|
120
123
|
function resolveNoveltyFloorSlots(rawSlots, rawMaxItems) {
|
|
121
124
|
const maxItems = typeof rawMaxItems === "number"
|
|
@@ -164,7 +167,7 @@ function formatDate(iso) {
|
|
|
164
167
|
* Render one production index line. Exported (2026-08-02, relevance eval #v2)
|
|
165
168
|
* so the eval can measure the REAL rendered surface instead of reimplementing
|
|
166
169
|
* it — `maxLen` threads through to `memoryTitle` unchanged (default
|
|
167
|
-
* DEFAULT_TITLE_CHARS = 100,
|
|
170
|
+
* DEFAULT_TITLE_CHARS = 100, release-managed since #408) so the eval's snippet-length
|
|
168
171
|
* sweep (spec §4.2) can call this SAME function at 100/150/title1sent without
|
|
169
172
|
* duplicating the date/scope/agent/type meta-line logic.
|
|
170
173
|
*/
|
|
@@ -36,7 +36,8 @@
|
|
|
36
36
|
* LRU beyond maxSessions so long-running servers don't accumulate state.
|
|
37
37
|
*/
|
|
38
38
|
export interface RecallRegistryOptions {
|
|
39
|
-
/** Turns a shown id stays suppressed.
|
|
39
|
+
/** Turns a shown id stays suppressed. Release-managed since #408 —
|
|
40
|
+
* calibration.ts RECALL_RESHOW_TURNS (30); this field is the eval/test seam. */
|
|
40
41
|
reshowTurns?: number;
|
|
41
42
|
/** Max tracked sessions before LRU eviction. */
|
|
42
43
|
maxSessions?: number;
|
package/dist/recall-registry.js
CHANGED
|
@@ -36,10 +36,44 @@
|
|
|
36
36
|
* early re-shows (~15 tokens each) — harmless by design. Sessions are pruned
|
|
37
37
|
* LRU beyond maxSessions so long-running servers don't accumulate state.
|
|
38
38
|
*/
|
|
39
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
40
|
+
if (k2 === undefined) k2 = k;
|
|
41
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
42
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
43
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
44
|
+
}
|
|
45
|
+
Object.defineProperty(o, k2, desc);
|
|
46
|
+
}) : (function(o, m, k, k2) {
|
|
47
|
+
if (k2 === undefined) k2 = k;
|
|
48
|
+
o[k2] = m[k];
|
|
49
|
+
}));
|
|
50
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
51
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
52
|
+
}) : function(o, v) {
|
|
53
|
+
o["default"] = v;
|
|
54
|
+
});
|
|
55
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
56
|
+
var ownKeys = function(o) {
|
|
57
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
58
|
+
var ar = [];
|
|
59
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
60
|
+
return ar;
|
|
61
|
+
};
|
|
62
|
+
return ownKeys(o);
|
|
63
|
+
};
|
|
64
|
+
return function (mod) {
|
|
65
|
+
if (mod && mod.__esModule) return mod;
|
|
66
|
+
var result = {};
|
|
67
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
68
|
+
__setModuleDefault(result, mod);
|
|
69
|
+
return result;
|
|
70
|
+
};
|
|
71
|
+
})();
|
|
39
72
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
40
73
|
exports.SessionRecallRegistry = exports.DEFAULT_RESHOW_TURNS = void 0;
|
|
41
74
|
const schema_prototypes_js_1 = require("./schema-prototypes.js");
|
|
42
|
-
|
|
75
|
+
const CALIBRATION = __importStar(require("./calibration.js"));
|
|
76
|
+
exports.DEFAULT_RESHOW_TURNS = CALIBRATION.RECALL_RESHOW_TURNS;
|
|
43
77
|
const DEFAULT_MAX_SESSIONS = 500;
|
|
44
78
|
class SessionRecallRegistry {
|
|
45
79
|
reshowTurns;
|
|
@@ -15,11 +15,39 @@
|
|
|
15
15
|
* #392 — one zone system, ONE verdict per pair: below `correctionMinSimilarity`
|
|
16
16
|
* (floor, 0.75) pairs are not candidates; in [floor, `dedupAutoMergeThreshold`)
|
|
17
17
|
* (ceiling, 0.92) each unlinked pair gets ONE verdict call whose action is
|
|
18
|
-
* `merge` | `corrects` | `supersedes` | `none`; at/above the
|
|
19
|
-
* deterministic merge zone (dedup.ts runDeterministicMergeZone —
|
|
20
|
-
* budget-free) owns the pair. The merge disposition reuses the
|
|
21
|
-
* execution (canonical pick, link re-point, dedup_log, metadata
|
|
22
|
-
* merge verdict below `correctionRewriteMinConfidence` keeps both
|
|
18
|
+
* `merge` | `corrects` | `supersedes` | `conflicts` | `none`; at/above the
|
|
19
|
+
* ceiling the deterministic merge zone (dedup.ts runDeterministicMergeZone —
|
|
20
|
+
* LLM-free, budget-free) owns the pair. The merge disposition reuses the
|
|
21
|
+
* dedup core's execution (canonical pick, link re-point, dedup_log, metadata
|
|
22
|
+
* rails); a merge verdict below `correctionRewriteMinConfidence` keeps both
|
|
23
|
+
* memories.
|
|
24
|
+
*
|
|
25
|
+
* #393 increment B — the SCOUT, a second detection source with the SAME
|
|
26
|
+
* judge: the similarity floor is structurally blind to corrections riding
|
|
27
|
+
* inside topically unrelated memories (the field failure — cosine ~0.5-0.6 to
|
|
28
|
+
* their target, zero `corrects` verdicts in the whole corpus baseline), so
|
|
29
|
+
* per NEW memory ONE classify-tier shape call asks whether it corrects/
|
|
30
|
+
* retracts/supersedes/CONTRADICTS something previously recorded (guard-C
|
|
31
|
+
* extended the question); correction-shaped
|
|
32
|
+
* memories FTS the corpus with the referenced claim's distinctive terms (the
|
|
33
|
+
* correction CONTAINS the words of what it corrects) and the hits become
|
|
34
|
+
* candidate pairs in the SAME verdict loop — no similarity gate for this
|
|
35
|
+
* source: cosine is a ranker/link strength, never a blocker. Per-source
|
|
36
|
+
* counters (scout_scanned / scout_correction_shaped / scout_candidates_found)
|
|
37
|
+
* ride the stage report; cosine band stats stay similarity-source-only.
|
|
38
|
+
*
|
|
39
|
+
* #393 guard-C — the conflicts flag + the zone-runs-last order: judgment
|
|
40
|
+
* OUTRANKS the deterministic sweep. A `conflicts` verdict writes a symmetric
|
|
41
|
+
* `conflicts` link (the pair genuinely disagrees — cannot both be true) and
|
|
42
|
+
* NOTHING else: no status change, no rewrite, no merge queue; both records
|
|
43
|
+
* stay live so the consumer sees both truths. Both merge paths (the zone's
|
|
44
|
+
* planDedup and the judged mergeMemoryIds) refuse to blend a conflicts-linked
|
|
45
|
+
* pair, counted as conflict_skipped. The zone therefore runs AFTER the
|
|
46
|
+
* rewrite phase — with the zone first, a >=0.92 conflict pair was blended
|
|
47
|
+
* before the judge ever saw it (the planted-eval harm: canonical=older, the
|
|
48
|
+
* newer truth erased); running it last means verdicts/marks/binds land first
|
|
49
|
+
* and the zone merges only what no verdict claimed — a conflicts bind set by
|
|
50
|
+
* this run's scan guards the SAME run's zone.
|
|
23
51
|
*
|
|
24
52
|
* Status vocabulary (code-defined, extensible — deliberately NOT config):
|
|
25
53
|
* NULL/'active' default | 'superseded' + 'retracted' demote in ranking |
|
|
@@ -41,40 +69,25 @@ import type { LlmClient, LlmUsage } from "./llm.js";
|
|
|
41
69
|
import type { ConsolidationReport } from "./types.js";
|
|
42
70
|
import type { EmbedFn } from "./retrieval.js";
|
|
43
71
|
import { acquireCaptureLock } from "./capture.js";
|
|
72
|
+
import type { RunDeadline } from "./run-deadline.js";
|
|
44
73
|
/** Stage label used for every budget.use()/recordUsage() call (#384). */
|
|
45
74
|
export declare const RECONSOLIDATION_STAGE_LABEL = "reconsolidation";
|
|
46
75
|
/**
|
|
47
|
-
* Default minimum COSINE similarity for a correction candidate pair
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
*
|
|
51
|
-
*
|
|
76
|
+
* Default minimum COSINE similarity for a correction candidate pair —
|
|
77
|
+
* RELEASE-MANAGED since #408 (calibration.ts CORRECTION_MIN_SIMILARITY;
|
|
78
|
+
* provenance there). Lower than the supersession stage's 0.80 on purpose: a
|
|
79
|
+
* retraction often rides inside an otherwise unrelated memory (the field
|
|
80
|
+
* failure that opened this issue), so the neighborhood gate must be a touch
|
|
81
|
+
* wider while the LLM verdict + confidence gate carry the precision load.
|
|
52
82
|
*/
|
|
53
83
|
export declare const DEFAULT_CORRECTION_MIN_SIMILARITY = 0.75;
|
|
54
84
|
/**
|
|
55
|
-
* Default minimum verdict confidence for the REWRITE fork
|
|
85
|
+
* Default minimum verdict confidence for the REWRITE fork — release-managed
|
|
86
|
+
* (calibration.ts CORRECTION_REWRITE_MIN_CONFIDENCE). Below this a
|
|
56
87
|
* `corrects` verdict degrades to mark-only — a weak mark is recoverable, a
|
|
57
88
|
* weak rewrite is corruption.
|
|
58
89
|
*/
|
|
59
90
|
export declare const DEFAULT_CORRECTION_REWRITE_MIN_CONFIDENCE = 0.8;
|
|
60
|
-
/**
|
|
61
|
-
* Default wall-clock bound for the stage, in minutes (#401). Checked at the
|
|
62
|
-
* top of the candidate scan loop (and before each rewrite contract call); on
|
|
63
|
-
* expiry the scan breaks cleanly at the last fully-considered candidate and
|
|
64
|
-
* the next run resumes from the persisted cursor. 120 sits safely under any
|
|
65
|
-
* sane process-level nightly timeout. 0 disables the bound. Invalid →
|
|
66
|
-
* default.
|
|
67
|
-
*/
|
|
68
|
-
export declare const DEFAULT_RECONSOLIDATION_MAX_MINUTES = 120;
|
|
69
|
-
/**
|
|
70
|
-
* Default per-run classify-call ceiling for the stage (#401) — the
|
|
71
|
-
* supersessionMaxCalls pattern with a NON-ZERO default ON PURPOSE: that
|
|
72
|
-
* knob's 0=unlimited default is what let the first full-corpus pass grow
|
|
73
|
-
* unbounded. Counts EVERY classify-tier call the stage makes (mark
|
|
74
|
-
* verifications, pair verdicts, rewrite contracts). 0 disables the cap.
|
|
75
|
-
* Invalid → default.
|
|
76
|
-
*/
|
|
77
|
-
export declare const DEFAULT_RECONSOLIDATION_MAX_CALLS = 600;
|
|
78
91
|
/** Head of the old content quoted in the provenance footer. */
|
|
79
92
|
export declare const FOOTER_HEAD_MAX_CHARS = 160;
|
|
80
93
|
/** The code-defined status vocabulary (see module doc). Not user-configurable. */
|
|
@@ -87,46 +100,34 @@ export declare const DEMOTED_STATUSES: readonly [MemoryStatus, MemoryStatus];
|
|
|
87
100
|
/** structural subset of consolidate.BudgetTracker (avoids an import cycle). */
|
|
88
101
|
export interface StageBudget {
|
|
89
102
|
readonly exhausted: boolean;
|
|
90
|
-
use(stage: string
|
|
103
|
+
use(stage: string): boolean;
|
|
91
104
|
recordUsage(stage: string, usage: LlmUsage | undefined): void;
|
|
92
105
|
}
|
|
93
106
|
export interface ReconsolidationOptions {
|
|
94
|
-
/**
|
|
107
|
+
/** Correction-pair cosine floor. Release-managed default (calibration.ts
|
|
108
|
+
* CORRECTION_MIN_SIMILARITY, 0.75); this field is the eval/test seam.
|
|
109
|
+
* Invalid → default. */
|
|
95
110
|
minSimilarity?: number;
|
|
96
|
-
/**
|
|
111
|
+
/** Rewrite-fork confidence floor. Release-managed default (calibration.ts
|
|
112
|
+
* CORRECTION_REWRITE_MIN_CONFIDENCE, 0.80); seam only. Invalid → default. */
|
|
97
113
|
rewriteMinConfidence?: number;
|
|
98
114
|
/**
|
|
99
|
-
*
|
|
100
|
-
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
* Invalid → default.
|
|
115
|
+
* The deterministic/LLM boundary of the unified resolution pass (#392):
|
|
116
|
+
* pairs at/above it merge via the LLM-free zone, pairs in [floor, ceiling)
|
|
117
|
+
* get the verdict. Release-managed default (calibration.ts
|
|
118
|
+
* DEDUP_AUTO_MERGE_THRESHOLD, 0.92); seam only. Invalid → default.
|
|
104
119
|
*/
|
|
105
120
|
autoMergeThreshold?: number;
|
|
106
121
|
/**
|
|
107
|
-
*
|
|
108
|
-
*
|
|
109
|
-
*
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
*
|
|
114
|
-
* deadline for the stage (#401), measured from stage start. Checked at the
|
|
115
|
-
* top of the candidate scan loop and before each rewrite contract call; on
|
|
116
|
-
* expiry the scan breaks cleanly — the cursor already points at the last
|
|
117
|
-
* fully-considered candidate, so the run ends consistent and the next
|
|
118
|
-
* nightly resumes from it. Invalid → default.
|
|
119
|
-
*/
|
|
120
|
-
maxMinutes?: number;
|
|
121
|
-
/**
|
|
122
|
-
* reconsolidationMaxCalls (config; default 600; 0 disables) — per-run
|
|
123
|
-
* ceiling on classify-tier calls for the stage (#401), the
|
|
124
|
-
* supersessionMaxCalls pattern with a NON-ZERO default (the 0=unlimited
|
|
125
|
-
* default there is what removed the last per-stage bound). Exhaustion
|
|
126
|
-
* mid-neighbor-loop or mid-rewrite-phase stops/defers cleanly at the
|
|
127
|
-
* current candidate boundary. Invalid → default.
|
|
122
|
+
* The run-wide pipeline deadline (#405 — successor of the stage-local
|
|
123
|
+
* reconsolidationMaxMinutes clock, #401): created at nightly start, shared
|
|
124
|
+
* with capture and every other stage, threaded here by runConsolidation.
|
|
125
|
+
* Checked at the top of the candidate scan loop and before each rewrite
|
|
126
|
+
* contract call; on expiry the scan breaks cleanly — the cursor already
|
|
127
|
+
* points at the last fully-considered candidate, so the run ends
|
|
128
|
+
* consistent and the next nightly resumes from it.
|
|
128
129
|
*/
|
|
129
|
-
|
|
130
|
+
deadline?: RunDeadline;
|
|
130
131
|
/**
|
|
131
132
|
* Capture-lock acquirer override (tests) — the deterministic zone and the
|
|
132
133
|
* judged-merge phase each hold a short lock window. Defaults to the real
|
|
@@ -146,12 +147,50 @@ export declare function isFactShapedTarget(mem: {
|
|
|
146
147
|
content: string;
|
|
147
148
|
}): boolean;
|
|
148
149
|
/**
|
|
149
|
-
* The unified resolution verdict (#392): ONE call per unlinked
|
|
150
|
-
* how the newer memory relates to the older — merge (same
|
|
151
|
-
* fact/verdict, differing in wording/qualifiers), corrects,
|
|
152
|
-
*
|
|
150
|
+
* The unified resolution verdict (#392, #393 guard-C): ONE call per unlinked
|
|
151
|
+
* pair decides how the newer memory relates to the older — merge (same
|
|
152
|
+
* underlying fact/verdict, differing in wording/qualifiers), corrects,
|
|
153
|
+
* supersedes, conflicts (genuine disagreement — cannot both be true; flag,
|
|
154
|
+
* keep both, never blend), or none (related but distinct).
|
|
155
|
+
*/
|
|
156
|
+
export type ResolutionAction = "merge" | "corrects" | "supersedes" | "conflicts" | "none";
|
|
157
|
+
/**
|
|
158
|
+
* The scout's correction-shape answer (#393 B, guard-C): does this NEW memory
|
|
159
|
+
* correct, retract, supersede, or CONTRADICT something previously recorded —
|
|
160
|
+
* and if so, which distinctive terms does the referenced (old) claim carry?
|
|
161
|
+
* `correction: true` means "resolution-shaped": corrects/retracts/supersedes/
|
|
162
|
+
* contradicts an earlier claim. `references` feeds an FTS query against the
|
|
163
|
+
* corpus; the shape call is the ONLY LLM work the scout adds per memory
|
|
164
|
+
* (non-corrections stop there), and it rides the same `complete()` surface +
|
|
165
|
+
* stage budget as every other call (#405 — there is no separate classify-tier
|
|
166
|
+
* ceiling to configure).
|
|
167
|
+
*/
|
|
168
|
+
export interface ScoutShape {
|
|
169
|
+
correction: boolean;
|
|
170
|
+
/** Distinctive terms of the referenced old claim ("" when not resolution-shaped). */
|
|
171
|
+
references: string;
|
|
172
|
+
/** Informational only — never gates behavior (no uncalibrated parameters). */
|
|
173
|
+
confidence: number;
|
|
174
|
+
}
|
|
175
|
+
/**
|
|
176
|
+
* Build the constrained correction-shape prompt (classify-tier cost profile:
|
|
177
|
+
* 1500-char truncation, supersession/verdict precedent). The wording asks for
|
|
178
|
+
* the OLD claim's distinctive terms — the field-failure mechanism is that a
|
|
179
|
+
* correction CONTAINS the words of what it corrects, even when the surrounding
|
|
180
|
+
* topics (and therefore the embedding cosine) are unrelated. Guard-C extends
|
|
181
|
+
* the question to contradictions: two records that disagree on the same
|
|
182
|
+
* quantity share even MORE wording than a cross-topic correction does.
|
|
183
|
+
*/
|
|
184
|
+
export declare function buildScoutShapePrompt(content: string): string;
|
|
185
|
+
/**
|
|
186
|
+
* Parse the scout shape reply. Null on unparseable JSON, a missing/non-boolean
|
|
187
|
+
* `correction`, or a missing/out-of-range `confidence` — the caller counts
|
|
188
|
+
* skipped_infra and moves on (parseSupersessionReply discipline: never
|
|
189
|
+
* mis-detect on ambiguity). `references` is lenient (missing/non-string → "")
|
|
190
|
+
* because an empty string simply yields no FTS hits — a harmless miss, not a
|
|
191
|
+
* mis-judgment.
|
|
153
192
|
*/
|
|
154
|
-
export
|
|
193
|
+
export declare function parseScoutShape(reply: string): ScoutShape | null;
|
|
155
194
|
/** Build the constrained pair-verdict prompt (1500-char truncation, supersession precedent). */
|
|
156
195
|
export declare function buildCorrectionVerdictPrompt(oldContent: string, newContent: string): string;
|
|
157
196
|
export interface CorrectionVerdict {
|
|
@@ -292,21 +331,34 @@ export declare function bandForCosine(bands: ResolutionBand[], cosine: number):
|
|
|
292
331
|
/**
|
|
293
332
|
* Nightly reconsolidation stage (#384, #392 — THE unified resolution stage).
|
|
294
333
|
*
|
|
295
|
-
* Phase
|
|
296
|
-
*
|
|
334
|
+
* Phase order (#393 guard-C): the deterministic merge zone (pairs >= the
|
|
335
|
+
* ceiling) runs LAST — after the scan, the judged-merge phase, and the
|
|
336
|
+
* rewrite phase. Judgment outranks the deterministic sweep: verdicts, marks,
|
|
337
|
+
* and binds land first and the zone merges only what no verdict claimed. With
|
|
338
|
+
* the zone first, a >=0.92 genuine-conflict pair was blended before the judge
|
|
339
|
+
* ever saw it (the planted-eval harm); running it last means a `conflicts`
|
|
340
|
+
* bind set by this run's scan guards the SAME run's zone. Zone internals
|
|
341
|
+
* (lock, backup, deadline, persistBand, fail-soft) are unchanged.
|
|
297
342
|
*
|
|
298
343
|
* Scan: every memory with rowid > reconsolidationCursor (no shape filter;
|
|
299
344
|
* absorbed candidates are skipped — invisible memories are not re-judged).
|
|
300
|
-
* Each candidate's pairs: incoming explicit marks (verified once, AC7) then
|
|
301
|
-
*
|
|
302
|
-
*
|
|
345
|
+
* Each candidate's pairs: incoming explicit marks (verified once, AC7), then
|
|
346
|
+
* ONE scout shape call (#393 B — flags correction shape; non-corrections stop
|
|
347
|
+
* there), then up-to-5 older KNN neighbors in [floor, ceiling) (verdict call
|
|
348
|
+
* per unlinked pair, AC2 — pairs at/above the ceiling are counted, never
|
|
349
|
+
* judged) plus the scout's FTS hits for correction-shaped memories (same
|
|
350
|
+
* verdict loop, NO similarity gate; guard-C: a scout hit whose KNN twin sits
|
|
351
|
+
* at/above the ceiling is re-tagged scout so the pair IS judged instead of
|
|
352
|
+
* being left for the zone to blend). Confirmed
|
|
303
353
|
* `corrects` pairs above the confidence gate on fact-shaped targets group by
|
|
304
354
|
* target into ONE rewrite call each (AC3); confirmed `merge` pairs queue for
|
|
305
|
-
* the merge phase;
|
|
355
|
+
* the merge phase; a `conflicts` verdict writes the conflicts link and
|
|
356
|
+
* nothing else (both live); everything else is mark-only.
|
|
306
357
|
*
|
|
307
358
|
* Merge phase (#392): queued pairs merge through the dedup core under one
|
|
308
|
-
* lock/backup window
|
|
309
|
-
*
|
|
359
|
+
* lock/backup window. A pair that cannot apply keeps both memories and holds
|
|
360
|
+
* the cursor; a conflicts-linked or metadata-mismatched refusal keeps both
|
|
361
|
+
* and advances (the verdict was rendered).
|
|
310
362
|
*
|
|
311
363
|
* Cursor discipline mirrors stageSupersession: the cursor advances past a
|
|
312
364
|
* candidate once its neighbor set has been considered, regardless of infra
|