@b4run/memory 0.8.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +49 -0
  3. package/dist/browse-budget.d.ts +40 -0
  4. package/dist/browse-budget.d.ts.map +1 -0
  5. package/dist/browse-budget.js +57 -0
  6. package/dist/browse-cursor.d.ts +22 -0
  7. package/dist/browse-cursor.d.ts.map +1 -0
  8. package/dist/browse-cursor.js +146 -0
  9. package/dist/browse-filter.d.ts +13 -0
  10. package/dist/browse-filter.d.ts.map +1 -0
  11. package/dist/browse-filter.js +16 -0
  12. package/dist/browse-order.d.ts +28 -0
  13. package/dist/browse-order.d.ts.map +1 -0
  14. package/dist/browse-order.js +37 -0
  15. package/dist/browse-range.d.ts +16 -0
  16. package/dist/browse-range.d.ts.map +1 -0
  17. package/dist/browse-range.js +52 -0
  18. package/dist/browse-validate.d.ts +31 -0
  19. package/dist/browse-validate.d.ts.map +1 -0
  20. package/dist/browse-validate.js +263 -0
  21. package/dist/browse.d.ts +15 -0
  22. package/dist/browse.d.ts.map +1 -0
  23. package/dist/browse.js +5 -0
  24. package/dist/distill.d.ts +81 -0
  25. package/dist/distill.d.ts.map +1 -0
  26. package/dist/distill.js +380 -0
  27. package/dist/hybrid.d.ts +26 -0
  28. package/dist/hybrid.d.ts.map +1 -0
  29. package/dist/hybrid.js +86 -0
  30. package/dist/index.d.ts +17 -0
  31. package/dist/index.d.ts.map +1 -0
  32. package/dist/index.js +13 -0
  33. package/dist/namespace.d.ts +18 -0
  34. package/dist/namespace.d.ts.map +1 -0
  35. package/dist/namespace.js +68 -0
  36. package/dist/reconcile.d.ts +51 -0
  37. package/dist/reconcile.d.ts.map +1 -0
  38. package/dist/reconcile.js +110 -0
  39. package/dist/score.d.ts +54 -0
  40. package/dist/score.d.ts.map +1 -0
  41. package/dist/score.js +66 -0
  42. package/dist/sqlite-browse-sql.d.ts +30 -0
  43. package/dist/sqlite-browse-sql.d.ts.map +1 -0
  44. package/dist/sqlite-browse-sql.js +178 -0
  45. package/dist/sqlite-store.d.ts +10 -0
  46. package/dist/sqlite-store.d.ts.map +1 -0
  47. package/dist/sqlite-store.js +521 -0
  48. package/dist/tokenize.d.ts +3 -0
  49. package/dist/tokenize.d.ts.map +1 -0
  50. package/dist/tokenize.js +14 -0
  51. package/dist/tsconfig.tsbuildinfo +1 -0
  52. package/dist/types.d.ts +201 -0
  53. package/dist/types.d.ts.map +1 -0
  54. package/dist/types.js +1 -0
  55. package/dist/vector.d.ts +22 -0
  56. package/dist/vector.d.ts.map +1 -0
  57. package/dist/vector.js +42 -0
  58. package/package.json +67 -0
@@ -0,0 +1,521 @@
1
+ import { mkdirSync } from "node:fs";
2
+ import { dirname } from "node:path";
3
+ import { DatabaseSync } from "node:sqlite";
4
+ import { browseCursorKey, browseQueryFingerprint, decodeBrowseCursor, encodeBrowseCursor, } from "./browse-cursor.js";
5
+ import { normalizeSetFilter } from "./browse-filter.js";
6
+ import { resolveBrowseOrder } from "./browse-order.js";
7
+ import { namespacePrefixUpperBound } from "./browse-range.js";
8
+ import { BROWSE_DEFAULT_LIMIT, validateBrowseQuery } from "./browse-validate.js";
9
+ import { fuseHybrid, rankKeywordCandidates } from "./hybrid.js";
10
+ import { DEFAULT_CANDIDATE_POOL } from "./score.js";
11
+ import { appendSqliteBrowseFilter, sqliteKeysetWhere } from "./sqlite-browse-sql.js";
12
+ import { tokenize } from "./tokenize.js";
13
+ import { cosineSimilarity, DEFAULT_VECTOR_K } from "./vector.js";
14
+ // ---------------------------------------------------------------------------
15
+ // Inline DB helpers (mirrors packages/sqlite-storage/src/internal — not
16
+ // re-exported from that package's public API, so we replicate the tiny shim).
17
+ // ---------------------------------------------------------------------------
18
+ function openDb(path) {
19
+ const isMemory = path === ":memory:";
20
+ if (!isMemory) {
21
+ mkdirSync(dirname(path), { recursive: true });
22
+ }
23
+ const db = new DatabaseSync(path);
24
+ if (!isMemory) {
25
+ db.exec("PRAGMA journal_mode = WAL");
26
+ }
27
+ db.exec("PRAGMA foreign_keys = ON");
28
+ db.exec("PRAGMA synchronous = NORMAL");
29
+ return db;
30
+ }
31
+ // Roll back best-effort — a failing ROLLBACK must not mask the original error.
32
+ function rollbackQuietly(db) {
33
+ try {
34
+ db.exec("ROLLBACK");
35
+ }
36
+ catch {
37
+ // Swallow: propagate the root-cause error from the caller's catch instead.
38
+ }
39
+ }
40
+ function runMigrations(db, migrations) {
41
+ db.exec("CREATE TABLE IF NOT EXISTS schema_version (version INTEGER PRIMARY KEY)");
42
+ const row = db.prepare("SELECT max(version) AS v FROM schema_version").get();
43
+ const current = row?.v ?? 0;
44
+ const sorted = [...migrations].sort((a, b) => a.version - b.version);
45
+ for (const m of sorted) {
46
+ if (m.version <= current)
47
+ continue;
48
+ db.exec("BEGIN");
49
+ try {
50
+ db.exec(m.up);
51
+ db.prepare("INSERT INTO schema_version(version) VALUES (?)").run(m.version);
52
+ db.exec("COMMIT");
53
+ }
54
+ catch (err) {
55
+ rollbackQuietly(db);
56
+ throw err;
57
+ }
58
+ }
59
+ }
60
+ // ---------------------------------------------------------------------------
61
+ // Schema
62
+ // ---------------------------------------------------------------------------
63
+ const MIGRATIONS = [
64
+ {
65
+ version: 1,
66
+ up: `
67
+ CREATE TABLE memories (
68
+ id TEXT PRIMARY KEY, kind TEXT NOT NULL, namespace TEXT NOT NULL,
69
+ content TEXT NOT NULL, data TEXT NOT NULL, source TEXT NOT NULL,
70
+ confidence REAL NOT NULL, tags TEXT NOT NULL, status TEXT NOT NULL,
71
+ supersedes TEXT, created_at TEXT NOT NULL, updated_at TEXT NOT NULL,
72
+ effective_at TEXT, expires_at TEXT
73
+ );
74
+ CREATE INDEX idx_mem_ns_status_updated ON memories(namespace, status, updated_at DESC);
75
+ CREATE TABLE memory_tokens (
76
+ memory_id TEXT NOT NULL REFERENCES memories(id) ON DELETE CASCADE, token TEXT NOT NULL
77
+ );
78
+ CREATE INDEX idx_memtok_token ON memory_tokens(token);
79
+ CREATE INDEX idx_memtok_mem ON memory_tokens(memory_id);
80
+ `,
81
+ },
82
+ {
83
+ version: 2,
84
+ up: `
85
+ ALTER TABLE memories ADD COLUMN embedding BLOB;
86
+ ALTER TABLE memories ADD COLUMN embedding_model TEXT;
87
+ `,
88
+ },
89
+ {
90
+ // Equality filter for kind-scoped windows; COALESCE(effective_at, created_at)
91
+ // ordering is intentionally unindexed at this scale.
92
+ version: 3,
93
+ up: `
94
+ CREATE INDEX IF NOT EXISTS idx_mem_ns_kind_effective ON memories (namespace, kind, effective_at DESC);
95
+ `,
96
+ },
97
+ {
98
+ // The global browse order — every poll tick's hot path, and the index the keyset
99
+ // guard seeks on. Directions are declared: a plain-ASC composite scanned backward
100
+ // reverses the id tie-break, which would make windows non-deterministic.
101
+ version: 4,
102
+ up: `
103
+ CREATE INDEX IF NOT EXISTS idx_mem_updated_id ON memories (updated_at DESC, id ASC);
104
+ `,
105
+ },
106
+ ];
107
+ // ---------------------------------------------------------------------------
108
+ // Row ↔ record conversion
109
+ // ---------------------------------------------------------------------------
110
+ function rowToRecord(row) {
111
+ return {
112
+ id: row.id,
113
+ kind: row.kind,
114
+ namespace: row.namespace,
115
+ content: row.content,
116
+ data: JSON.parse(row.data),
117
+ source: JSON.parse(row.source),
118
+ confidence: row.confidence,
119
+ tags: JSON.parse(row.tags),
120
+ status: row.status,
121
+ ...(row.supersedes ? { supersedes: JSON.parse(row.supersedes) } : {}),
122
+ createdAt: row.created_at,
123
+ updatedAt: row.updated_at,
124
+ ...(row.effective_at ? { effectiveAt: row.effective_at } : {}),
125
+ ...(row.expires_at ? { expiresAt: row.expires_at } : {}),
126
+ };
127
+ }
128
+ // Codepoint compare — matches SQLite BINARY collation, no ICU dependence.
129
+ function cmp(a, b) {
130
+ return a < b ? -1 : a > b ? 1 : 0;
131
+ }
132
+ function tokensFor(rec) {
133
+ const values = Object.values(rec.data).filter((v) => typeof v === "string");
134
+ return tokenize([rec.content, rec.tags.join(" "), values.join(" ")].join(" "));
135
+ }
136
+ // ---------------------------------------------------------------------------
137
+ // Factory
138
+ // ---------------------------------------------------------------------------
139
+ export function sqliteMemoryStore(opts) {
140
+ const db = openDb(opts.path);
141
+ runMigrations(db, MIGRATIONS);
142
+ function reindex(rec) {
143
+ db.prepare("DELETE FROM memory_tokens WHERE memory_id = ?").run(rec.id);
144
+ const ins = db.prepare("INSERT INTO memory_tokens(memory_id, token) VALUES (?, ?)");
145
+ for (const t of tokensFor(rec))
146
+ ins.run(rec.id, t);
147
+ }
148
+ function getById(id) {
149
+ const row = db.prepare("SELECT * FROM memories WHERE id = ?").get(id);
150
+ return row ? rowToRecord(row) : null;
151
+ }
152
+ function putRecord(rec, embed) {
153
+ const blob = embed?.embedding && embed.embeddingModel
154
+ ? Buffer.from(embed.embedding.buffer.slice(embed.embedding.byteOffset, embed.embedding.byteOffset + embed.embedding.byteLength))
155
+ : null;
156
+ const model = embed?.embedding && embed.embeddingModel ? embed.embeddingModel : null;
157
+ db.prepare(`INSERT OR REPLACE INTO memories
158
+ (id,kind,namespace,content,data,source,confidence,tags,status,supersedes,created_at,updated_at,effective_at,expires_at,embedding,embedding_model)
159
+ VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)`).run(rec.id, rec.kind, rec.namespace, rec.content, JSON.stringify(rec.data), JSON.stringify(rec.source), rec.confidence, JSON.stringify(rec.tags), rec.status, rec.supersedes ? JSON.stringify(rec.supersedes) : null, rec.createdAt, rec.updatedAt, rec.effectiveAt ?? null, rec.expiresAt ?? null, blob, model);
160
+ reindex(rec);
161
+ }
162
+ // Re-read a row's persisted embedding so update() does not drop it (putRecord
163
+ // rewrites the full row; without this the embedding columns would go null).
164
+ function getEmbeddingRow(id) {
165
+ const row = db
166
+ .prepare("SELECT embedding, embedding_model FROM memories WHERE id = ?")
167
+ .get(id);
168
+ if (row?.embedding && row.embedding_model) {
169
+ return {
170
+ embedding: new Float32Array(row.embedding.buffer.slice(row.embedding.byteOffset, row.embedding.byteOffset + row.embedding.byteLength)),
171
+ embeddingModel: row.embedding_model,
172
+ };
173
+ }
174
+ return {};
175
+ }
176
+ // Keyword-ranked list: candidate pool → live corpus stats → scoreMemory →
177
+ // stable sort (score DESC, updated_at DESC, id ASC). Returns the FULL sorted
178
+ // list (caller pages + tag-filters). `weightsOverride` lets the hybrid path
179
+ // rank by relevance ONLY; when absent the default recall weights apply — this
180
+ // is the exact behavior the shipped smarter-recall path exercises.
181
+ function rankKeyword(q, baseSql, baseParams, terms, weightsOverride) {
182
+ const rawPool = opts.recall?.candidatePool;
183
+ const pool = typeof rawPool === "number" && Number.isFinite(rawPool) && rawPool > 0
184
+ ? Math.floor(rawPool)
185
+ : DEFAULT_CANDIDATE_POOL;
186
+ const placeholders = terms.map(() => "?").join(",");
187
+ // 1) Candidate pool: rows matching ≥1 query token, newest first (pool
188
+ // truncation by recency is deterministic).
189
+ const candidateRows = db
190
+ .prepare(`SELECT m.* FROM memories m WHERE ${baseSql}
191
+ AND m.id IN (SELECT memory_id FROM memory_tokens WHERE token IN (${placeholders}))
192
+ ORDER BY m.updated_at DESC, m.id ASC LIMIT ?`)
193
+ .all(...baseParams, ...terms, pool);
194
+ const candidates = candidateRows.map(rowToRecord);
195
+ if (candidates.length === 0)
196
+ return [];
197
+ // 2) Corpus stats, computed live (nothing cached → nothing to go stale).
198
+ const corpusSize = db.prepare(`SELECT COUNT(*) AS n FROM memories m WHERE ${baseSql}`).get(...baseParams).n;
199
+ const dfRows = db
200
+ .prepare(`SELECT t.token AS token, COUNT(DISTINCT t.memory_id) AS df
201
+ FROM memory_tokens t JOIN memories m ON m.id = t.memory_id
202
+ WHERE ${baseSql} AND t.token IN (${placeholders}) GROUP BY t.token`)
203
+ .all(...baseParams, ...terms);
204
+ const dfByToken = new Map(dfRows.map((r) => [r.token, r.df]));
205
+ // 3) Score + sort via the shared pure ranking core. Candidate token sets are
206
+ // recomputed via the same `tokenize` reindex() uses, so they are
207
+ // guaranteed consistent with the table.
208
+ const options = weightsOverride || opts.recall
209
+ ? { ...opts.recall, ...(weightsOverride ? { weights: weightsOverride } : {}) }
210
+ : undefined;
211
+ return rankKeywordCandidates(candidates, dfByToken, corpusSize, terms, q.now, options, tokenize);
212
+ }
213
+ // Page (limit) then tag post-filter — today's ranked-path semantics, unchanged.
214
+ function pageAndTagFilter(records, limit, q) {
215
+ let out = records.slice(0, limit);
216
+ if (q.tags && q.tags.length > 0) {
217
+ const want = new Set(q.tags);
218
+ out = out.filter((r) => r.tags.some((t) => want.has(t)));
219
+ }
220
+ return out;
221
+ }
222
+ return {
223
+ async put(rec, opts) {
224
+ putRecord(rec, opts);
225
+ },
226
+ async get(id) {
227
+ return getById(id);
228
+ },
229
+ async search(q) {
230
+ const status = q.status ?? "active";
231
+ const limit = q.limit ?? 8;
232
+ const terms = q.query ? tokenize(q.query) : [];
233
+ // Shared base filter (namespace + status [+ kind] [+ time window]
234
+ // [+ expiry]) — the "corpus". Every search path (query-less, ranked,
235
+ // hybrid) AND the ranked path's df/N corpus stats interpolate this same
236
+ // clause, so window/expiry filtering and IDF stats stay coherent.
237
+ let baseSql = `m.namespace = ? AND m.status = ?`;
238
+ const baseParams = [q.namespace, status];
239
+ if (q.kind) {
240
+ baseSql += ` AND m.kind = ?`;
241
+ baseParams.push(q.kind);
242
+ }
243
+ if (q.since) {
244
+ baseSql += ` AND COALESCE(m.effective_at, m.created_at) >= ?`;
245
+ baseParams.push(q.since);
246
+ }
247
+ if (q.until) {
248
+ baseSql += ` AND COALESCE(m.effective_at, m.created_at) < ?`;
249
+ baseParams.push(q.until);
250
+ }
251
+ if (q.now) {
252
+ baseSql += ` AND (m.expires_at IS NULL OR m.expires_at > ?)`;
253
+ baseParams.push(q.now);
254
+ }
255
+ if (terms.length === 0) {
256
+ // Query-less path: unwindowed keeps EXACTLY the pre-ranking recency
257
+ // behavior (index fragment, listCandidates-adjacent consumers depend
258
+ // on pure recency order). A since/until window switches to event-time
259
+ // order — windowed queries are about "what happened then".
260
+ const order = q.since || q.until
261
+ ? "COALESCE(m.effective_at, m.created_at) DESC, m.id ASC"
262
+ : "m.updated_at DESC, m.id ASC";
263
+ const rows = db
264
+ .prepare(`SELECT m.* FROM memories m WHERE ${baseSql} ORDER BY ${order} LIMIT ?`)
265
+ .all(...baseParams, limit);
266
+ let records = rows.map(rowToRecord);
267
+ if (q.tags && q.tags.length > 0) {
268
+ const want = new Set(q.tags);
269
+ records = records.filter((r) => r.tags.some((t) => want.has(t)));
270
+ }
271
+ return records;
272
+ }
273
+ // Hybrid path — active only when the caller supplies a query embedding.
274
+ // Keyword ∪ vector-nearest, RRF-fused, then a bounded recency/confidence
275
+ // multiplier. See docs/superpowers/specs/2026-07-06-vector-recall-design.md.
276
+ if (q.queryEmbedding && q.embedderId) {
277
+ const v = q.vector ?? opts.vector ?? {};
278
+ const vectorK = typeof v.vectorK === "number" && Number.isFinite(v.vectorK) && v.vectorK > 0
279
+ ? Math.floor(v.vectorK)
280
+ : DEFAULT_VECTOR_K;
281
+ // Keyword-ranked records: reuse the ranked pool, ordered by relevance ONLY.
282
+ const kwRecords = rankKeyword(q, baseSql, baseParams, terms, {
283
+ relevance: 1,
284
+ recency: 0,
285
+ confidence: 0,
286
+ });
287
+ // Vector-ranked records: brute-force cosine over rows with a matching
288
+ // embedder tag and a non-null embedding, cosine-sorted then sliced to K.
289
+ const vecRows = db
290
+ .prepare(`SELECT m.id AS id, m.embedding AS embedding FROM memories m
291
+ WHERE ${baseSql} AND m.embedding_model = ? AND m.embedding IS NOT NULL`)
292
+ .all(...baseParams, q.embedderId);
293
+ const queryEmbedding = q.queryEmbedding;
294
+ const vectorRanked = vecRows
295
+ .map((r) => {
296
+ const emb = new Float32Array(r.embedding.buffer.slice(r.embedding.byteOffset, r.embedding.byteOffset + r.embedding.byteLength));
297
+ return { id: r.id, sim: cosineSimilarity(queryEmbedding, emb) };
298
+ })
299
+ .sort((a, b) => b.sim - a.sim || cmp(a.id, b.id))
300
+ .slice(0, vectorK)
301
+ .map((r) => getById(r.id))
302
+ .filter((r) => r !== null);
303
+ // One half-life knob: fall back to the shared recall tuning so a
304
+ // configured recall.recencyHalfLifeMs governs hybrid recency too.
305
+ const recencyHalfLifeMs = v.recencyHalfLifeMs ?? opts.recall?.recencyHalfLifeMs;
306
+ return pageAndTagFilter(fuseHybrid({
307
+ keywordRanked: kwRecords,
308
+ vectorRanked,
309
+ now: q.now,
310
+ options: {
311
+ ...v,
312
+ ...(recencyHalfLifeMs !== undefined ? { recencyHalfLifeMs } : {}),
313
+ },
314
+ }), limit, q);
315
+ }
316
+ // Ranked path — see docs/superpowers/specs/2026-07-05-smarter-recall-design.md.
317
+ return pageAndTagFilter(rankKeyword(q, baseSql, baseParams, terms), limit, q);
318
+ },
319
+ async update(id, patch) {
320
+ const current = getById(id);
321
+ if (!current)
322
+ throw new Error(`memory not found: ${id}`);
323
+ putRecord({ ...current, ...patch, id }, getEmbeddingRow(id));
324
+ },
325
+ async supersede(id, bySupersedingId) {
326
+ if (!getById(id))
327
+ throw new Error(`memory not found: ${id}`);
328
+ db.prepare("UPDATE memories SET status = 'superseded' WHERE id = ?").run(id);
329
+ const superseding = getById(bySupersedingId);
330
+ if (superseding) {
331
+ const links = new Set([...(superseding.supersedes ?? []), id]);
332
+ db.prepare("UPDATE memories SET supersedes = ? WHERE id = ?").run(JSON.stringify([...links]), bySupersedingId);
333
+ }
334
+ },
335
+ async delete(id) {
336
+ db.prepare("DELETE FROM memories WHERE id = ?").run(id);
337
+ },
338
+ async listCandidates(namespacePrefix) {
339
+ const rows = db
340
+ .prepare("SELECT * FROM memories WHERE status = 'candidate' AND namespace LIKE ? ORDER BY created_at DESC")
341
+ .all(`${namespacePrefix}%`);
342
+ return rows.map(rowToRecord);
343
+ },
344
+ async browse(q = {}) {
345
+ // Defence in depth: whatever the caller checked, a store that accepts nonsense
346
+ // returns an empty page that looks like an answer.
347
+ validateBrowseQuery(q);
348
+ const where = [];
349
+ const params = [];
350
+ if (q.namespace) {
351
+ where.push("namespace = ?");
352
+ params.push(q.namespace);
353
+ }
354
+ if (q.namespacePrefix) {
355
+ // Byte-exact, case-sensitive prefix as a half-open range — deliberately NOT
356
+ // LIKE, so %/_/\ stay literal. Sargable, but the ORDER BY picks the plan, so
357
+ // it is a trade: at 100k a selective prefix seeks (6.5 -> 0.13 ms) while a
358
+ // BROAD one gives up idx_mem_updated_id's ordered early exit for a temp
359
+ // B-tree sort (0.05 -> 41 ms). The COUNT(*) below seeks either way.
360
+ const upper = namespacePrefixUpperBound(q.namespacePrefix);
361
+ where.push(upper === undefined ? "namespace >= ?" : "namespace >= ? AND namespace < ?");
362
+ params.push(q.namespacePrefix);
363
+ if (upper !== undefined)
364
+ params.push(upper);
365
+ }
366
+ // A set becomes IN (?,?,…); an EMPTY set becomes a clause that matches
367
+ // nothing, which is the contract — not an absent filter.
368
+ const statuses = normalizeSetFilter(q.status);
369
+ if (statuses) {
370
+ where.push(statuses.length > 0 ? `status IN (${statuses.map(() => "?").join(",")})` : "0");
371
+ params.push(...statuses);
372
+ }
373
+ const kinds = normalizeSetFilter(q.kind);
374
+ if (kinds) {
375
+ where.push(kinds.length > 0 ? `kind IN (${kinds.map(() => "?").join(",")})` : "0");
376
+ params.push(...kinds);
377
+ }
378
+ if (q.sourceType) {
379
+ where.push("json_extract(source, '$.type') = ?");
380
+ params.push(q.sourceType);
381
+ }
382
+ if (q.since) {
383
+ where.push("COALESCE(effective_at, created_at) >= ?");
384
+ params.push(q.since);
385
+ }
386
+ if (q.until) {
387
+ where.push("COALESCE(effective_at, created_at) < ?");
388
+ params.push(q.until);
389
+ }
390
+ if (q.now) {
391
+ where.push("(expires_at IS NULL OR expires_at > ?)");
392
+ params.push(q.now);
393
+ }
394
+ for (const filter of q.filters ?? [])
395
+ appendSqliteBrowseFilter(filter, where, params);
396
+ // The COUNT must see the FILTERS ONLY: `total` is the size of the whole matching
397
+ // set, not of what is left after the cursor.
398
+ const filterParamCount = params.length;
399
+ const countClause = where.length > 0 ? `WHERE ${where.join(" AND ")}` : "";
400
+ const order = resolveBrowseOrder(q.orderBy);
401
+ // Every order terminates with `id ASC` so the total order is deterministic and
402
+ // a keyset window can never skip or repeat a row.
403
+ const orderSql = [
404
+ ...order.map((entry) => `${entry.column} ${entry.dir === "desc" ? "DESC" : "ASC"}`),
405
+ "id ASC",
406
+ ].join(", ");
407
+ const fingerprint = browseQueryFingerprint(q);
408
+ const rowWhere = [...where];
409
+ if (q.cursor) {
410
+ const payload = decodeBrowseCursor(q.cursor, fingerprint, order);
411
+ rowWhere.push(sqliteKeysetWhere(order, payload, params));
412
+ }
413
+ const rowsClause = rowWhere.length > 0 ? `WHERE ${rowWhere.join(" AND ")}` : "";
414
+ // A no-op since validateBrowseQuery rejects non-integers, limit < 1 and offset < 0.
415
+ // Kept as a floor because the two backends fail asymmetrically if one ever slips
416
+ // through: sqlite reads a negative LIMIT as unlimited, Postgres throws.
417
+ const limit = Math.max(0, Math.trunc(q.limit ?? BROWSE_DEFAULT_LIMIT));
418
+ const offset = Math.max(0, Math.trunc(q.offset ?? 0));
419
+ // One read snapshot across both statements, so `records` and `total` can never
420
+ // describe different versions of the table. The writer this defends against is
421
+ // always on ANOTHER connection — a second process against the same file — because
422
+ // these two statements are synchronous: no caller on this thread can run
423
+ // between them.
424
+ // COUNT(*) OVER () would collapse the pair to one statement and was measured on
425
+ // sqlite and rejected: the window aggregate materializes the entire filtered set
426
+ // (439 ms vs 5.3 ms at 1M rows) and destroys the lazy top-k path for the rows.
427
+ let rows;
428
+ let total;
429
+ // BEGIN sits outside the try: if it fails because a transaction is already open
430
+ // on this long-lived handle, the catch below would send its ROLLBACK into the
431
+ // CALLER's transaction and silently discard their uncommitted work.
432
+ db.exec("BEGIN DEFERRED");
433
+ try {
434
+ // Explicit columns: everything rowToRecord reads, EXCLUDING the embedding
435
+ // BLOB (~6KB/row) that a listing UI would otherwise fetch and discard.
436
+ rows = db
437
+ .prepare(`SELECT id, kind, namespace, content, data, source, confidence, tags, status,
438
+ supersedes, created_at, updated_at, effective_at, expires_at
439
+ FROM memories ${rowsClause} ORDER BY ${orderSql} LIMIT ? OFFSET ?`)
440
+ .all(...params, limit, offset);
441
+ total = db
442
+ .prepare(`SELECT COUNT(*) AS n FROM memories ${countClause}`)
443
+ .get(...params.slice(0, filterParamCount)).n;
444
+ db.exec("COMMIT");
445
+ }
446
+ catch (err) {
447
+ rollbackQuietly(db);
448
+ throw err;
449
+ }
450
+ const records = rows.map(rowToRecord);
451
+ const last = records.at(-1);
452
+ // Issued whenever the window FILLED, rather than over-fetching `limit + 1` to
453
+ // learn whether a further row exists: a walk over an exact multiple of `limit`
454
+ // therefore ends in one empty window.
455
+ const continuation = last && records.length === limit
456
+ ? encodeBrowseCursor(fingerprint, { key: browseCursorKey(last, order), id: last.id })
457
+ : null;
458
+ return { records, total, continuation };
459
+ },
460
+ async stats(opts = {}) {
461
+ // Byte-exact prefix match — substr(), not LIKE, so %/_/\ stay literal. This is
462
+ // NOT browse's half-open range: it cannot seek, and no task owns moving it.
463
+ const clause = opts.namespacePrefix ? "WHERE substr(namespace, 1, length(?)) = ?" : "";
464
+ const params = opts.namespacePrefix
465
+ ? [opts.namespacePrefix, opts.namespacePrefix]
466
+ : [];
467
+ const group = (expr) => Object.fromEntries(db
468
+ .prepare(`SELECT ${expr} AS k, COUNT(*) AS n FROM memories ${clause} GROUP BY k`)
469
+ .all(...params).map((r) => [r.k, r.n]));
470
+ const total = db.prepare(`SELECT COUNT(*) AS n FROM memories ${clause}`).get(...params).n;
471
+ return {
472
+ total,
473
+ byStatus: group("status"),
474
+ byKind: group("kind"),
475
+ byNamespace: group("namespace"),
476
+ bySourceType: group("json_extract(source, '$.type')"),
477
+ };
478
+ },
479
+ async prune(opts) {
480
+ // Byte-exact prefix match — substr(), not LIKE, so %/_/\ stay literal. This is
481
+ // NOT browse's half-open range: it cannot seek, and no task owns moving it.
482
+ const prefixParams = opts.namespacePrefix
483
+ ? [opts.namespacePrefix, opts.namespacePrefix]
484
+ : [];
485
+ const expiredSql = opts.namespacePrefix
486
+ ? `DELETE FROM memories WHERE expires_at IS NOT NULL AND expires_at <= ?
487
+ AND substr(namespace, 1, length(?)) = ?`
488
+ : "DELETE FROM memories WHERE expires_at IS NOT NULL AND expires_at <= ?";
489
+ const expired = db.prepare(expiredSql).run(opts.now, ...prefixParams);
490
+ let deletedOverCap = 0;
491
+ if (opts.cap !== undefined) {
492
+ const cap = Math.max(0, Math.trunc(opts.cap));
493
+ // Rank episodic rows per namespace by event time (newest first, id ASC
494
+ // tiebreak — the established cross-backend ordering) and delete beyond cap.
495
+ const overSql = opts.namespacePrefix
496
+ ? `DELETE FROM memories WHERE id IN (
497
+ SELECT id FROM (
498
+ SELECT id, ROW_NUMBER() OVER (
499
+ PARTITION BY namespace
500
+ ORDER BY COALESCE(effective_at, created_at) DESC, id ASC
501
+ ) AS rn
502
+ FROM memories
503
+ WHERE kind = 'episodic' AND substr(namespace, 1, length(?)) = ?
504
+ ) WHERE rn > ?
505
+ )`
506
+ : `DELETE FROM memories WHERE id IN (
507
+ SELECT id FROM (
508
+ SELECT id, ROW_NUMBER() OVER (
509
+ PARTITION BY namespace
510
+ ORDER BY COALESCE(effective_at, created_at) DESC, id ASC
511
+ ) AS rn
512
+ FROM memories WHERE kind = 'episodic'
513
+ ) WHERE rn > ?
514
+ )`;
515
+ const over = db.prepare(overSql).run(...prefixParams, cap);
516
+ deletedOverCap = Number(over.changes);
517
+ }
518
+ return { deletedExpired: Number(expired.changes), deletedOverCap };
519
+ },
520
+ };
521
+ }
@@ -0,0 +1,3 @@
1
+ /** Lowercase, split on non-alphanumerics, drop 1-char tokens, dedupe (insertion order). */
2
+ export declare function tokenize(text: string): string[];
3
+ //# sourceMappingURL=tokenize.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tokenize.d.ts","sourceRoot":"","sources":["../src/tokenize.ts"],"names":[],"mappings":"AAAA,2FAA2F;AAC3F,wBAAgB,QAAQ,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,EAAE,CAU/C"}
@@ -0,0 +1,14 @@
1
+ /** Lowercase, split on non-alphanumerics, drop 1-char tokens, dedupe (insertion order). */
2
+ export function tokenize(text) {
3
+ const seen = new Set();
4
+ const out = [];
5
+ for (const raw of text.toLowerCase().split(/[^a-z0-9]+/)) {
6
+ if (raw.length < 2)
7
+ continue;
8
+ if (seen.has(raw))
9
+ continue;
10
+ seen.add(raw);
11
+ out.push(raw);
12
+ }
13
+ return out;
14
+ }