@gamaze/hicortex 0.22.0 → 0.22.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +209 -76
- package/dist/calibration.d.ts +92 -12
- package/dist/calibration.js +102 -15
- package/dist/classify-domains.js +6 -1
- package/dist/consolidate.js +87 -19
- package/dist/dashboard.d.ts +11 -0
- package/dist/dashboard.js +9 -0
- package/dist/db.js +74 -0
- package/dist/eval/decay-eval.d.ts +4 -2
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/eval-clock.d.ts +32 -0
- package/dist/eval/eval-clock.js +47 -0
- package/dist/eval/graph-eval.d.ts +15 -2
- package/dist/eval/graph-eval.js +51 -5
- package/dist/eval/planted-eval.d.ts +4 -0
- package/dist/eval/planted-eval.js +27 -2
- package/dist/eval/planted-harness.d.ts +7 -0
- package/dist/eval/planted-harness.js +2 -0
- package/dist/eval/ranking-battery.d.ts +49 -2
- package/dist/eval/ranking-battery.js +110 -2
- package/dist/eval/ranking-eval.d.ts +26 -6
- package/dist/eval/ranking-eval.js +197 -34
- package/dist/eval/ranking-fixtures.d.ts +41 -1
- package/dist/eval/ranking-fixtures.js +261 -2
- package/dist/eval/recall-sweep.d.ts +7 -2
- package/dist/eval/recall-sweep.js +42 -13
- package/dist/eval/relevance-eval.d.ts +115 -1
- package/dist/eval/relevance-eval.js +318 -32
- package/dist/eval/run-eval.d.ts +7 -4
- package/dist/eval/run-eval.js +36 -9
- package/dist/mcp-server.js +37 -9
- package/dist/nightly.js +14 -0
- package/dist/recall-index.d.ts +46 -3
- package/dist/recall-index.js +83 -26
- package/dist/recall-precision.d.ts +212 -0
- package/dist/recall-precision.js +381 -0
- package/dist/retrieval.d.ts +34 -15
- package/dist/retrieval.js +132 -59
- package/dist/types.d.ts +7 -0
- package/package.json +1 -1
- package/server.json +3 -3
|
@@ -0,0 +1,381 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Memory Precision (#476) — the event store behind the console's Memory
|
|
4
|
+
* Precision card, and the two-level context model made visible:
|
|
5
|
+
*
|
|
6
|
+
* - Level 1 (pushed index): what every prompt receives unasked. One
|
|
7
|
+
* `recall_pushes` row per NON-skipped /recall-index call + one
|
|
8
|
+
* `recall_events` kind='shown' row per pushed line, carrying the RAW
|
|
9
|
+
* cosines (similarity to the prompt, redundancy vs standing context) —
|
|
10
|
+
* stored threshold-free, verdicts computed at render so recalibration
|
|
11
|
+
* never rewrites history.
|
|
12
|
+
* - Level 2 (recall depth): what the agent fetches when needed. One
|
|
13
|
+
* `recall_events` kind='fetch' row per handleMemoryGet — the ONE funnel
|
|
14
|
+
* the REST GET /memory and MCP hicortex_get paths share.
|
|
15
|
+
*
|
|
16
|
+
* Layering (the recall-index.ts / capture-health.ts convention): all write +
|
|
17
|
+
* aggregation logic lives here as pure functions over the db handle so tests
|
|
18
|
+
* exercise them without HTTP; mcp-server.ts only wires the record calls into
|
|
19
|
+
* the /recall-index and fetch funnels, and dashboard.ts spreads the window
|
|
20
|
+
* aggregation into /dashboard/data.
|
|
21
|
+
*
|
|
22
|
+
* FAIL-SOFT LAW: recording is telemetry. Any error here is caught by the
|
|
23
|
+
* CALLER (handleRecallIndex / handleMemoryGet) and swallowed — the recall
|
|
24
|
+
* response (status 200 + block) is never affected, and the exposure signal
|
|
25
|
+
* (touchMemoriesShown) never depends on it (the handler falls back to the
|
|
26
|
+
* plain exposure write when recording is absent or fails).
|
|
27
|
+
*
|
|
28
|
+
* The perf law (bounded hot-path cost): the recorder computes cosines with
|
|
29
|
+
* the request's ALREADY-MEMOIZED prompt embedding (createRecallRetrieveFn's
|
|
30
|
+
* embedPrompt — zero extra embeds) against STORED memory vectors, and writes
|
|
31
|
+
* one transaction of ≤ maxItems+1 INSERTs — microseconds inside an endpoint
|
|
32
|
+
* that already writes and spends ~100-300 ms embedding and searching.
|
|
33
|
+
*/
|
|
34
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
35
|
+
if (k2 === undefined) k2 = k;
|
|
36
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
37
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
38
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
39
|
+
}
|
|
40
|
+
Object.defineProperty(o, k2, desc);
|
|
41
|
+
}) : (function(o, m, k, k2) {
|
|
42
|
+
if (k2 === undefined) k2 = k;
|
|
43
|
+
o[k2] = m[k];
|
|
44
|
+
}));
|
|
45
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
46
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
47
|
+
}) : function(o, v) {
|
|
48
|
+
o["default"] = v;
|
|
49
|
+
});
|
|
50
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
51
|
+
var ownKeys = function(o) {
|
|
52
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
53
|
+
var ar = [];
|
|
54
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
55
|
+
return ar;
|
|
56
|
+
};
|
|
57
|
+
return ownKeys(o);
|
|
58
|
+
};
|
|
59
|
+
return function (mod) {
|
|
60
|
+
if (mod && mod.__esModule) return mod;
|
|
61
|
+
var result = {};
|
|
62
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
63
|
+
__setModuleDefault(result, mod);
|
|
64
|
+
return result;
|
|
65
|
+
};
|
|
66
|
+
})();
|
|
67
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
68
|
+
exports.recordRecallPush = recordRecallPush;
|
|
69
|
+
exports.recordRecallFetch = recordRecallFetch;
|
|
70
|
+
exports.pruneRecallPrecision = pruneRecallPrecision;
|
|
71
|
+
exports.identityServedToAllClients = identityServedToAllClients;
|
|
72
|
+
exports.createStandingContextBasis = createStandingContextBasis;
|
|
73
|
+
exports.readMemoryPrecision = readMemoryPrecision;
|
|
74
|
+
const storage = __importStar(require("./storage.js"));
|
|
75
|
+
const retrieval_js_1 = require("./retrieval.js");
|
|
76
|
+
const embedder_js_1 = require("./embedder.js");
|
|
77
|
+
const identity_store_js_1 = require("./identity-store.js");
|
|
78
|
+
const recall_index_js_1 = require("./recall-index.js");
|
|
79
|
+
const calibration_js_1 = require("./calibration.js");
|
|
80
|
+
/**
|
|
81
|
+
* Record one /recall-index push: the push row + (when lines were shown) the
|
|
82
|
+
* per-line kind='shown' event rows, in ONE transaction WITH the existing
|
|
83
|
+
* exposure write (storage.touchMemoriesShown — the spec's "same transaction";
|
|
84
|
+
* better-sqlite3 nests the inner transaction as a savepoint, so a recording
|
|
85
|
+
* failure rolls back the whole unit and the CALLER's fallback re-runs the
|
|
86
|
+
* exposure touch alone — shown_count never depends on telemetry).
|
|
87
|
+
*
|
|
88
|
+
* similarity = cosine(memory embedding, PURE prompt embedding), computed
|
|
89
|
+
* UNIFORMLY via cosineBetweenVectors against the stored vector — FTS-sourced
|
|
90
|
+
* picks (which bypass the cosine floor and carry similarity:null in
|
|
91
|
+
* MemorySearchResult) get a measured value here too. redundancy = MAX cosine
|
|
92
|
+
* against the standing-context basis vectors (NULL when the basis is empty —
|
|
93
|
+
* embed failure degrades to "unmeasured", never a guessed 0). Both stored
|
|
94
|
+
* raw; no thresholds touch this row.
|
|
95
|
+
*/
|
|
96
|
+
function recordRecallPush(db, entry, basis = []) {
|
|
97
|
+
const ts = entry.ts ?? new Date().toISOString();
|
|
98
|
+
// UTC YYYY-MM-DD of ts — day buckets are UTC everywhere (dashboard
|
|
99
|
+
// conventions; SQLite date('now') is UTC too).
|
|
100
|
+
const day = ts.slice(0, 10);
|
|
101
|
+
const excerpt = entry.prompt.slice(0, calibration_js_1.MEMORY_PRECISION_PROMPT_EXCERPT_CHARS);
|
|
102
|
+
const insertEvent = db.prepare(`INSERT INTO recall_events (ts, day, push_id, memory_id, similarity, redundancy, kind)
|
|
103
|
+
VALUES (?, ?, ?, ?, ?, ?, 'shown')`);
|
|
104
|
+
const tx = db.transaction(() => {
|
|
105
|
+
if (entry.ids.length > 0) {
|
|
106
|
+
// The exposure write rides INSIDE the recording transaction (spec: one
|
|
107
|
+
// atomic unit). Nested db.transaction → savepoint.
|
|
108
|
+
storage.touchMemoriesShown(db, entry.ids, ts);
|
|
109
|
+
}
|
|
110
|
+
const info = db
|
|
111
|
+
.prepare("INSERT INTO recall_pushes (ts, day, session_id, prompt_excerpt) VALUES (?, ?, ?, ?)")
|
|
112
|
+
.run(ts, day, entry.sessionId, excerpt);
|
|
113
|
+
const pushId = Number(info.lastInsertRowid);
|
|
114
|
+
if (entry.ids.length === 0)
|
|
115
|
+
return; // silent turn — push row only
|
|
116
|
+
const promptEmb = entry.promptEmbedding;
|
|
117
|
+
if (!promptEmb) {
|
|
118
|
+
// Contract violation (ids without an embedding): record the lines with
|
|
119
|
+
// NULL measures rather than guessing — honesty over fabrication.
|
|
120
|
+
for (const id of entry.ids)
|
|
121
|
+
insertEvent.run(ts, day, pushId, id, null, null);
|
|
122
|
+
return;
|
|
123
|
+
}
|
|
124
|
+
for (const id of entry.ids) {
|
|
125
|
+
const memEmb = storage.getStoredEmbedding(db, id);
|
|
126
|
+
const similarity = memEmb ? (0, retrieval_js_1.cosineBetweenVectors)(memEmb, promptEmb) : null;
|
|
127
|
+
let redundancy = null;
|
|
128
|
+
if (memEmb && basis.length > 0) {
|
|
129
|
+
for (const b of basis) {
|
|
130
|
+
const c = (0, retrieval_js_1.cosineBetweenVectors)(memEmb, b);
|
|
131
|
+
if (redundancy === null || c > redundancy)
|
|
132
|
+
redundancy = c;
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
insertEvent.run(ts, day, pushId, id, similarity, redundancy);
|
|
136
|
+
}
|
|
137
|
+
});
|
|
138
|
+
tx();
|
|
139
|
+
}
|
|
140
|
+
/**
|
|
141
|
+
* Record one explicit fetch (handleMemoryGet — the funnel both REST GET
|
|
142
|
+
* /memory and MCP hicortex_get route through, so the row is written exactly
|
|
143
|
+
* once per fetch). similarity/redundancy are NULL on fetch rows: Level 2 is
|
|
144
|
+
* the use signal, not a relevance measure. push_id NULL — a fetch stands
|
|
145
|
+
* alone. Fail-soft at the CALLER (handleMemoryGet wraps this).
|
|
146
|
+
*/
|
|
147
|
+
function recordRecallFetch(db, memoryId, ts) {
|
|
148
|
+
const nowIso = ts ?? new Date().toISOString();
|
|
149
|
+
db.prepare(`INSERT INTO recall_events (ts, day, push_id, memory_id, similarity, redundancy, kind)
|
|
150
|
+
VALUES (?, ?, NULL, ?, NULL, NULL, 'fetch')`).run(nowIso, nowIso.slice(0, 10), memoryId);
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* Nightly retention prune (zero-LLM): drop both event tables' rows outside
|
|
154
|
+
* the rolling window whose length IS the retention constant (the
|
|
155
|
+
* CAPTURE_HEALTH_WINDOW_DAYS single-constant law — the card can never claim a
|
|
156
|
+
* window the store no longer has rows for). Inclusive cutoff: the last N
|
|
157
|
+
* calendar days with today counted stay (day >= today−(N−1)), the complement
|
|
158
|
+
* is deleted — retained rows == window rows.
|
|
159
|
+
*/
|
|
160
|
+
function pruneRecallPrecision(db) {
|
|
161
|
+
db.exec(`DELETE FROM recall_pushes WHERE day < date('now', '-${calibration_js_1.MEMORY_PRECISION_WINDOW_DAYS - 1} days')`);
|
|
162
|
+
db.exec(`DELETE FROM recall_events WHERE day < date('now', '-${calibration_js_1.MEMORY_PRECISION_WINDOW_DAYS - 1} days')`);
|
|
163
|
+
}
|
|
164
|
+
// ---------------------------------------------------------------------------
|
|
165
|
+
// Standing-context basis — the redundancy reference (what every session
|
|
166
|
+
// ALREADY knows before the index pushes anything)
|
|
167
|
+
// ---------------------------------------------------------------------------
|
|
168
|
+
/** Cache TTL for the embedded basis (module constant, not a tuning value: it
|
|
169
|
+
* bounds how stale the redundancy reference may be — 24h mirrors the daily
|
|
170
|
+
* lesson/identity cadence the /learnings + /identity surfaces serve). */
|
|
171
|
+
const BASIS_TTL_MS = 24 * 60 * 60 * 1000;
|
|
172
|
+
/**
|
|
173
|
+
* True when the identity scope is served to EVERY known client — the only
|
|
174
|
+
* case where identity sections belong in the corpus-level standing-context
|
|
175
|
+
* basis. /recall-index carries no client type, so a SCOPED identity config
|
|
176
|
+
* (any subset) excludes identity rather than guessing which sections the
|
|
177
|
+
* calling session sees. `clients` is the boot-resolved
|
|
178
|
+
* resolveIdentityClientsConfig output; the absent-config default ["cc"] is
|
|
179
|
+
* NOT all → identity excluded.
|
|
180
|
+
*/
|
|
181
|
+
function identityServedToAllClients(clients) {
|
|
182
|
+
return identity_store_js_1.KNOWN_IDENTITY_CLIENTS.every((c) => clients.includes(c));
|
|
183
|
+
}
|
|
184
|
+
/**
|
|
185
|
+
* The standing-context basis: the top lessons in /learnings order
|
|
186
|
+
* (storage.getLessons(db, 30) — exactly what the /learnings handler serves
|
|
187
|
+
* and every client's session-start hook injects) PLUS identity sections ONLY
|
|
188
|
+
* when identityServedToAllClients (see above). Embedded with the LOCAL
|
|
189
|
+
* embedder — no LLM anywhere in this feature. Cached for 24h (single-flight:
|
|
190
|
+
* concurrent callers share the embedding pass); an embed failure degrades to
|
|
191
|
+
* an EMPTY cached basis (redundancy NULL — "unmeasured", never guessed) and
|
|
192
|
+
* is retried after the TTL.
|
|
193
|
+
*/
|
|
194
|
+
function createStandingContextBasis(config) {
|
|
195
|
+
const embedFn = config.embedFn ?? embedder_js_1.embedBatch;
|
|
196
|
+
const now = config.now ?? Date.now;
|
|
197
|
+
const ttlMs = config.ttlMs ?? BASIS_TTL_MS;
|
|
198
|
+
let cache = null;
|
|
199
|
+
let inFlight = null;
|
|
200
|
+
const build = async (db) => {
|
|
201
|
+
const texts = [];
|
|
202
|
+
// Lessons: the /learnings set + order (last 30 days, created_at DESC).
|
|
203
|
+
// Read errors degrade to an empty lesson half — the basis is a reference,
|
|
204
|
+
// not a ledger.
|
|
205
|
+
try {
|
|
206
|
+
for (const lesson of storage.getLessons(db, 30))
|
|
207
|
+
texts.push(lesson.content);
|
|
208
|
+
}
|
|
209
|
+
catch (err) {
|
|
210
|
+
console.warn(`[hicortex] standing-context basis: lesson read failed (degraded): ` +
|
|
211
|
+
`${err instanceof Error ? err.message : String(err)}`);
|
|
212
|
+
}
|
|
213
|
+
// Identity: only when the scope is served to EVERY client — never guessed.
|
|
214
|
+
if (identityServedToAllClients(config.clients())) {
|
|
215
|
+
try {
|
|
216
|
+
const { sections } = (0, identity_store_js_1.readSections)(config.identityDir());
|
|
217
|
+
for (const content of Object.values(sections))
|
|
218
|
+
texts.push(content);
|
|
219
|
+
}
|
|
220
|
+
catch (err) {
|
|
221
|
+
console.warn(`[hicortex] standing-context basis: identity read failed (degraded): ` +
|
|
222
|
+
`${err instanceof Error ? err.message : String(err)}`);
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
return texts.length > 0 ? await embedFn(texts) : [];
|
|
226
|
+
};
|
|
227
|
+
const provider = async (db) => {
|
|
228
|
+
if (cache && now() < cache.expiresAt)
|
|
229
|
+
return cache.vectors;
|
|
230
|
+
if (!inFlight) {
|
|
231
|
+
inFlight = build(db)
|
|
232
|
+
.catch((err) => {
|
|
233
|
+
// Embed failure → EMPTY basis cached for the TTL: redundancy reads
|
|
234
|
+
// NULL (unmeasured) on every push in the period, never a guessed 0.
|
|
235
|
+
console.warn(`[hicortex] standing-context basis: embed failed (redundancy NULL this TTL): ` +
|
|
236
|
+
`${err instanceof Error ? err.message : String(err)}`);
|
|
237
|
+
return [];
|
|
238
|
+
})
|
|
239
|
+
.then((vectors) => {
|
|
240
|
+
cache = { vectors, expiresAt: now() + ttlMs };
|
|
241
|
+
return vectors;
|
|
242
|
+
})
|
|
243
|
+
.finally(() => {
|
|
244
|
+
inFlight = null;
|
|
245
|
+
});
|
|
246
|
+
}
|
|
247
|
+
return inFlight;
|
|
248
|
+
};
|
|
249
|
+
const basis = provider;
|
|
250
|
+
basis.resetForTests = () => {
|
|
251
|
+
cache = null;
|
|
252
|
+
};
|
|
253
|
+
return basis;
|
|
254
|
+
}
|
|
255
|
+
const DAY_MS = 86_400_000;
|
|
256
|
+
const round4 = (v) => Math.round(v * 10000) / 10000;
|
|
257
|
+
/** Range string ("7d"|"30d"|…) → days; null when unparseable/"all" (the
|
|
258
|
+
* caller — handleDashboardData — has already validated against
|
|
259
|
+
* VALID_RANGES; anything unexpected degrades to the retention clamp). */
|
|
260
|
+
function rangeToDays(range) {
|
|
261
|
+
if (range === "all")
|
|
262
|
+
return null;
|
|
263
|
+
const n = parseInt(range, 10);
|
|
264
|
+
return Number.isFinite(n) && n > 0 ? n : null;
|
|
265
|
+
}
|
|
266
|
+
/**
|
|
267
|
+
* The window aggregation behind /dashboard/data's memory_precision block
|
|
268
|
+
* (the readCaptureHealthWindow pattern). Effective window = min(range,
|
|
269
|
+
* MEMORY_PRECISION_WINDOW_DAYS) for every range incl. 180d/all (the #452
|
|
270
|
+
* selector's wide options clamp to what the retention honestly holds).
|
|
271
|
+
*
|
|
272
|
+
* Level 1 groups the stored event rows over the window (day >= today−(N−1),
|
|
273
|
+
* the inclusive capture-health convention); Level 2 is the corpus-wide
|
|
274
|
+
* adoption delta — live sums minus the NEWEST non-null `adoption` in
|
|
275
|
+
* dashboard_snapshots at/before now−N days (cumulative counters telescope:
|
|
276
|
+
* a missing nightly widens the effective window rather than corrupting it;
|
|
277
|
+
* two snapshots on one day are harmless — the newest wins; no pre-window
|
|
278
|
+
* baseline → null, undefined rather than zero). Divergence joins the
|
|
279
|
+
* window's per-memory shown/fetch counts against LIVE (non-absorbed)
|
|
280
|
+
* memories and renders the top-3 through the shared index-line path.
|
|
281
|
+
*/
|
|
282
|
+
function readMemoryPrecision(db, range, liveAdoption, opts) {
|
|
283
|
+
const now = opts?.now ?? new Date();
|
|
284
|
+
const redundantAbove = opts?.redundantAbove ?? calibration_js_1.MEMORY_PRECISION_REDUNDANT_ABOVE;
|
|
285
|
+
const windowDays = Math.min(rangeToDays(range) ?? calibration_js_1.MEMORY_PRECISION_WINDOW_DAYS, calibration_js_1.MEMORY_PRECISION_WINDOW_DAYS);
|
|
286
|
+
// Inclusive window start (UTC YYYY-MM-DD): today counts as day 1.
|
|
287
|
+
const cutoffDay = new Date(now.getTime() - (windowDays - 1) * DAY_MS).toISOString().slice(0, 10);
|
|
288
|
+
// --- Level 1: group the window's shown events (raw values, no thresholds
|
|
289
|
+
// in SQL beyond the redundant-share counter — the verdict stays at render). ---
|
|
290
|
+
const agg = db.prepare(`SELECT COUNT(*) AS lines,
|
|
291
|
+
SUM(CASE WHEN similarity IS NOT NULL THEN similarity ELSE 0 END) AS sim_sum,
|
|
292
|
+
SUM(CASE WHEN similarity IS NOT NULL THEN 1 ELSE 0 END) AS sim_n,
|
|
293
|
+
SUM(CASE WHEN redundancy IS NOT NULL AND redundancy >= ? THEN 1 ELSE 0 END) AS red_n,
|
|
294
|
+
SUM(CASE WHEN redundancy IS NOT NULL THEN 1 ELSE 0 END) AS red_measured
|
|
295
|
+
FROM recall_events
|
|
296
|
+
WHERE day >= ? AND kind = 'shown'`).get(redundantAbove, cutoffDay);
|
|
297
|
+
const pushes = db.prepare("SELECT COUNT(*) AS c FROM recall_pushes WHERE day >= ?").get(cutoffDay).c;
|
|
298
|
+
// --- Level 2: corpus adoption delta over the window (snapshot series). ---
|
|
299
|
+
const baselineCutoff = new Date(now.getTime() - windowDays * DAY_MS).toISOString();
|
|
300
|
+
const snapRows = db
|
|
301
|
+
.prepare("SELECT run_at, metrics FROM dashboard_snapshots WHERE run_at <= ? ORDER BY run_at DESC")
|
|
302
|
+
.all(baselineCutoff);
|
|
303
|
+
let baseline = null;
|
|
304
|
+
for (const r of snapRows) {
|
|
305
|
+
// Backfilled rows OMIT adoption (point-in-time, not reconstructable) —
|
|
306
|
+
// walk to the newest row that actually carries it.
|
|
307
|
+
try {
|
|
308
|
+
const m = JSON.parse(r.metrics);
|
|
309
|
+
if (m.adoption &&
|
|
310
|
+
typeof m.adoption.shown_sum === "number" &&
|
|
311
|
+
typeof m.adoption.used_sum === "number") {
|
|
312
|
+
baseline = { shown_sum: m.adoption.shown_sum, used_sum: m.adoption.used_sum };
|
|
313
|
+
break;
|
|
314
|
+
}
|
|
315
|
+
}
|
|
316
|
+
catch {
|
|
317
|
+
// A malformed metrics blob is skipped, not fatal — the walk continues
|
|
318
|
+
// to the next-older candidate.
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
const shown = baseline ? liveAdoption.shown_sum - baseline.shown_sum : null;
|
|
322
|
+
const used = baseline ? liveAdoption.used_sum - baseline.used_sum : null;
|
|
323
|
+
// Divide guard + honesty: no baseline → null (undefined, never 0); a window
|
|
324
|
+
// with zero (or negative — a restored DB) shown delta has no ratio.
|
|
325
|
+
const usesPerShowing = shown !== null && shown > 0 && used !== null ? round4(used / shown) : null;
|
|
326
|
+
// --- Divergence: per-memory window counters, gated to live rows. ---
|
|
327
|
+
const perMemory = db
|
|
328
|
+
.prepare(`SELECT memory_id,
|
|
329
|
+
SUM(CASE WHEN kind = 'shown' THEN 1 ELSE 0 END) AS shown,
|
|
330
|
+
SUM(CASE WHEN kind = 'fetch' THEN 1 ELSE 0 END) AS fetches
|
|
331
|
+
FROM recall_events
|
|
332
|
+
WHERE day >= ?
|
|
333
|
+
GROUP BY memory_id`)
|
|
334
|
+
.all(cutoffDay);
|
|
335
|
+
const divergingIds = perMemory
|
|
336
|
+
.filter((r) => r.shown >= calibration_js_1.MEMORY_PRECISION_DIVERGENCE_MIN_SHOWN && r.fetches === 0)
|
|
337
|
+
.map((r) => ({ id: r.memory_id, shown: r.shown }));
|
|
338
|
+
const memStmt = db.prepare(`SELECT id, content, created_at, domain, project, source_agent, memory_type
|
|
339
|
+
FROM memories WHERE id = ? AND COALESCE(status, '') != 'absorbed'`);
|
|
340
|
+
const diverging = [];
|
|
341
|
+
for (const d of divergingIds) {
|
|
342
|
+
const row = memStmt.get(d.id);
|
|
343
|
+
if (!row)
|
|
344
|
+
continue; // absorbed/deleted since — not a LIVE divergence
|
|
345
|
+
diverging.push({
|
|
346
|
+
id: row.id,
|
|
347
|
+
shown: d.shown,
|
|
348
|
+
// The SHARED index-line path (what agents see in the pushed index) —
|
|
349
|
+
// formatIndexLine at the default title cap, same as the digest.
|
|
350
|
+
line: (0, recall_index_js_1.formatIndexLine)({
|
|
351
|
+
id: row.id,
|
|
352
|
+
content: row.content,
|
|
353
|
+
score: 0,
|
|
354
|
+
effective_strength: 0,
|
|
355
|
+
access_count: 0,
|
|
356
|
+
memory_type: row.memory_type,
|
|
357
|
+
project: row.project,
|
|
358
|
+
domain: row.domain,
|
|
359
|
+
source_agent: row.source_agent,
|
|
360
|
+
created_at: row.created_at,
|
|
361
|
+
connections: 0,
|
|
362
|
+
}, calibration_js_1.RECALL_TITLE_CHARS),
|
|
363
|
+
});
|
|
364
|
+
}
|
|
365
|
+
diverging.sort((a, b) => b.shown - a.shown);
|
|
366
|
+
return {
|
|
367
|
+
window_days: windowDays,
|
|
368
|
+
level1: {
|
|
369
|
+
pushes,
|
|
370
|
+
lines: agg.lines,
|
|
371
|
+
mean_similarity: agg.sim_n > 0 ? round4((agg.sim_sum ?? 0) / agg.sim_n) : null,
|
|
372
|
+
redundant_share: agg.red_measured > 0 ? round4(agg.red_n / agg.red_measured) : null,
|
|
373
|
+
thresholds: {
|
|
374
|
+
redundant_above: redundantAbove,
|
|
375
|
+
divergence_min_shown: calibration_js_1.MEMORY_PRECISION_DIVERGENCE_MIN_SHOWN,
|
|
376
|
+
},
|
|
377
|
+
},
|
|
378
|
+
level2: { shown, used, uses_per_showing: usesPerShowing },
|
|
379
|
+
divergence: { count: diverging.length, top: diverging.slice(0, 3) },
|
|
380
|
+
};
|
|
381
|
+
}
|
package/dist/retrieval.d.ts
CHANGED
|
@@ -55,14 +55,26 @@ export interface ScoringWeights {
|
|
|
55
55
|
similarity: number;
|
|
56
56
|
strength: number;
|
|
57
57
|
connections: number;
|
|
58
|
+
/**
|
|
59
|
+
* #449 log-saturation degree K for the connections term: the credit is
|
|
60
|
+
* min(1, log1p(k)/log1p(K)) — full at k = K, exactly +0 at k = 0. A count,
|
|
61
|
+
* not a weight: validated ≥ 1 (the rrfK non-weight seam precedent); no
|
|
62
|
+
* upper bound (the eval sweep widens K past the planted fixture degrees).
|
|
63
|
+
*/
|
|
64
|
+
connectionsSaturation: number;
|
|
58
65
|
recency: number;
|
|
59
|
-
|
|
60
|
-
|
|
66
|
+
/** #430 merged time curve: head amplitude at age 0 — the merged freshness
|
|
67
|
+
* job. A deliberate overshoot (max total 1.15 pre-clamp), NOT a fifth
|
|
68
|
+
* blend weight; the four blend weights still sum to 1.0. */
|
|
69
|
+
recencyHead: number;
|
|
70
|
+
/** #430 merged time curve: head window / join age in days. */
|
|
71
|
+
recencyHeadDays: number;
|
|
61
72
|
supersededDemotion: number;
|
|
62
|
-
/** #
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
73
|
+
/** #430 merged scope affinity: ONE term — max() of the project-match
|
|
74
|
+
* indicator (1 on exact match) and the max overlapping domain-tag weight,
|
|
75
|
+
* × this weight (born from #203's two additive boosts at their shared
|
|
76
|
+
* value). 0 (via the seam) disables the whole term. */
|
|
77
|
+
scopeAffinity: number;
|
|
66
78
|
/** #205 RRF k parameter (1/(k+rank+1)). Larger ⇒ shallower rank curve. */
|
|
67
79
|
rrfK: number;
|
|
68
80
|
/** #205 composite-score share of the final blend (RRF gets the remainder). */
|
|
@@ -232,12 +244,13 @@ export declare function effectiveStrength(baseStrength: number, lastAccessed: st
|
|
|
232
244
|
* Return a composite relevance score in [0, 1] for a candidate memory.
|
|
233
245
|
* Exported for exact-value tests of the similarity component (#145).
|
|
234
246
|
*
|
|
235
|
-
*
|
|
236
|
-
* graded, zero-boost-neutral
|
|
237
|
-
* and
|
|
238
|
-
* 0 when the scope is absent
|
|
239
|
-
* (
|
|
240
|
-
* soft-exclusion). See
|
|
247
|
+
* Scope affinity (options.scope + options.tagWeights; #203, merged #430):
|
|
248
|
+
* ONE additive, graded, zero-boost-neutral term — max() of the project-match
|
|
249
|
+
* indicator (exact match ⇒ 1) and the max overlapping memory_tags.weight,
|
|
250
|
+
* × the scopeAffinity weight. It is 0 when the scope is absent
|
|
251
|
+
* (byte-identical to pre-#203) and NEVER negative (a foreign memory adds 0,
|
|
252
|
+
* never a penalty — penalties re-introduce soft-exclusion). See
|
|
253
|
+
* `AffinityScope`.
|
|
241
254
|
*/
|
|
242
255
|
export interface AffinityScope {
|
|
243
256
|
/** Exact-match project from the client (CC/OC cwd-derived; /search project). */
|
|
@@ -246,12 +259,13 @@ export interface AffinityScope {
|
|
|
246
259
|
* vocabulary as memory_tags (config `domains`). */
|
|
247
260
|
missionDomains?: string[];
|
|
248
261
|
}
|
|
249
|
-
export declare function computeScore(memory: Memory, distance: number, connectionCount: number,
|
|
262
|
+
export declare function computeScore(memory: Memory, distance: number, connectionCount: number, now: Date, options?: {
|
|
250
263
|
superseded?: boolean;
|
|
251
|
-
/** #203: when present,
|
|
264
|
+
/** #203/#430: when present, the scope-affinity boost is applied. */
|
|
252
265
|
scope?: AffinityScope;
|
|
253
266
|
/** Candidate's graded domain tags (memory_tags rows). Loaded batched for
|
|
254
|
-
* the whole candidate set in retrieve(); used for
|
|
267
|
+
* the whole candidate set in retrieve(); used for the scope-affinity
|
|
268
|
+
* term's domain signal. */
|
|
255
269
|
tagWeights?: Array<{
|
|
256
270
|
tag: string;
|
|
257
271
|
weight: number | null;
|
|
@@ -308,6 +322,11 @@ export declare function retrieve(db: Database.Database, embedFn: EmbedFn, query:
|
|
|
308
322
|
ftsCandidates?: (fetchLimit: number, sourceAgent?: string) => Array<Memory & {
|
|
309
323
|
rank: number;
|
|
310
324
|
}>;
|
|
325
|
+
/** #458 eval clock pin — the instant the decay/recency terms score
|
|
326
|
+
* against. Eval-only: no production caller passes it, and unset means
|
|
327
|
+
* the live clock (byte-identical to pre-#458). Mirrors the
|
|
328
|
+
* injectable-clock idiom of run-deadline.ts; see eval/eval-clock.ts. */
|
|
329
|
+
now?: Date;
|
|
311
330
|
}): Promise<MemorySearchResult[]>;
|
|
312
331
|
/**
|
|
313
332
|
* Get recent context, optionally filtered by project.
|