@hanhnd/agent-kit 1.0.38 → 1.0.40
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +286 -140
- package/dist/entrypoints/server.js +3 -3
- package/dist/mcp/memory.d.ts +4 -9
- package/dist/mcp/memory.js +43 -19
- package/dist/server.js +445 -212
- package/dist/services/digest/digest-processor.test.js +1 -3
- package/dist/services/memory/index.d.ts +3 -1
- package/dist/services/memory/index.js +1 -0
- package/dist/services/memory/indexer.d.ts +5 -6
- package/dist/services/memory/indexer.js +71 -101
- package/dist/services/memory/indexer.test.js +347 -572
- package/dist/services/memory/memory.test.js +58 -0
- package/dist/services/memory/store.d.ts +4 -11
- package/dist/services/memory/store.js +200 -66
- package/dist/services/memory/store.test.js +359 -108
- package/dist/services/memory/subsystem.d.ts +19 -0
- package/dist/services/memory/subsystem.js +67 -0
- package/dist/services/memory/types.d.ts +48 -0
- package/dist/utils/async.d.ts +1 -0
- package/dist/utils/async.js +17 -0
- package/dist/utils/async.test.d.ts +1 -0
- package/dist/utils/async.test.js +48 -0
- package/package.json +1 -1
|
@@ -6,6 +6,7 @@ import * as path from 'node:path';
|
|
|
6
6
|
import { after, before, describe, test } from 'node:test';
|
|
7
7
|
import { MemoryStore, SCHEMA_VERSION } from './store.js';
|
|
8
8
|
import { EmbeddingModelName } from './types.js';
|
|
9
|
+
import { DENSE_SCORE_FLOOR, FETCH_MULTIPLIER, RECENCY_WEIGHT, RRF_K } from './constants.js';
|
|
9
10
|
const TEST_CONFIG = {
|
|
10
11
|
enabled: true,
|
|
11
12
|
wikiDir: '',
|
|
@@ -103,6 +104,103 @@ describe('MemoryStore', () => {
|
|
|
103
104
|
fs.rmSync(libsqlDir, { recursive: true, force: true });
|
|
104
105
|
}
|
|
105
106
|
});
|
|
107
|
+
// BC22: Task 1.3 — prove hybrid SQL shape (CTEs + ROW_NUMBER OVER + UNION + PARTITION BY + vector_top_k + FTS)
|
|
108
|
+
test('BC22: libsql supports hybrid SQL primitives — CTEs, ROW_NUMBER OVER, UNION, PARTITION BY, vector_top_k, bm25', () => {
|
|
109
|
+
const compatDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-hybrid-compat-'));
|
|
110
|
+
const db = new Database(path.join(compatDir, 'compat.db'));
|
|
111
|
+
try {
|
|
112
|
+
db.pragma('journal_mode = WAL');
|
|
113
|
+
db.exec(`
|
|
114
|
+
CREATE TABLE items (
|
|
115
|
+
rowid INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
116
|
+
id TEXT NOT NULL UNIQUE,
|
|
117
|
+
src TEXT NOT NULL,
|
|
118
|
+
embedding F32_BLOB(3) NOT NULL
|
|
119
|
+
);
|
|
120
|
+
CREATE INDEX idx_items_embedding ON items (libsql_vector_idx(embedding));
|
|
121
|
+
CREATE VIRTUAL TABLE items_fts USING fts5(id, src, content='items', content_rowid='rowid');
|
|
122
|
+
CREATE TRIGGER items_ai AFTER INSERT ON items BEGIN
|
|
123
|
+
INSERT INTO items_fts(rowid, id, src) VALUES (new.rowid, new.id, new.src);
|
|
124
|
+
END;
|
|
125
|
+
`);
|
|
126
|
+
const ins = db.prepare(`INSERT INTO items (id, src, embedding) VALUES (?, ?, vector32(?))`);
|
|
127
|
+
ins.run('a', 'source-alpha.md', '[1,0,0]');
|
|
128
|
+
ins.run('b', 'source-alpha.md', '[0,1,0]');
|
|
129
|
+
ins.run('c', 'source-beta.md', '[0,0,1]');
|
|
130
|
+
// Full hybrid SQL shape: CTEs + ROW_NUMBER OVER + UNION + PARTITION BY + vector_top_k + bm25
|
|
131
|
+
const queryVec = Buffer.from(new Float32Array([1, 0, 0]).buffer);
|
|
132
|
+
const rows = db
|
|
133
|
+
.prepare(`
|
|
134
|
+
WITH
|
|
135
|
+
dense_data AS (
|
|
136
|
+
SELECT items.id, items.src,
|
|
137
|
+
vector_distance_cos(items.embedding, vector32(?)) AS dense_dist
|
|
138
|
+
FROM vector_top_k('idx_items_embedding', vector32(?), ?) AS vtk
|
|
139
|
+
JOIN items ON items.rowid = vtk.id
|
|
140
|
+
),
|
|
141
|
+
dense_ranked AS (
|
|
142
|
+
SELECT *, ROW_NUMBER() OVER (ORDER BY dense_dist, id) AS dense_rank
|
|
143
|
+
FROM dense_data
|
|
144
|
+
),
|
|
145
|
+
bm25_data AS (
|
|
146
|
+
SELECT items.id, items.src,
|
|
147
|
+
bm25(items_fts, 1.0, 1.0) AS bm25_neg
|
|
148
|
+
FROM items_fts
|
|
149
|
+
JOIN items ON items.rowid = items_fts.rowid
|
|
150
|
+
WHERE items_fts MATCH ?
|
|
151
|
+
LIMIT ?
|
|
152
|
+
),
|
|
153
|
+
bm25_ranked AS (
|
|
154
|
+
SELECT *, ROW_NUMBER() OVER (ORDER BY bm25_neg, id) AS bm25_rank
|
|
155
|
+
FROM bm25_data
|
|
156
|
+
),
|
|
157
|
+
all_candidates AS (
|
|
158
|
+
SELECT d.id, d.src, d.dense_rank, d.dense_dist, b.bm25_rank,
|
|
159
|
+
CASE WHEN b.id IS NOT NULL THEN 'both' ELSE 'dense' END AS retriever
|
|
160
|
+
FROM dense_ranked d LEFT JOIN bm25_ranked b ON b.id = d.id
|
|
161
|
+
UNION ALL
|
|
162
|
+
SELECT b.id, b.src, NULL, NULL, b.bm25_rank, 'bm25' AS retriever
|
|
163
|
+
FROM bm25_ranked b LEFT JOIN dense_ranked d ON d.id = b.id
|
|
164
|
+
WHERE d.id IS NULL
|
|
165
|
+
),
|
|
166
|
+
recency_ranked AS (
|
|
167
|
+
SELECT id, ROW_NUMBER() OVER (ORDER BY id) AS recency_rank
|
|
168
|
+
FROM all_candidates
|
|
169
|
+
),
|
|
170
|
+
scored AS (
|
|
171
|
+
SELECT c.id, c.src, c.dense_rank, c.bm25_rank, c.retriever,
|
|
172
|
+
COALESCE(1.0 / (60 + c.dense_rank), 0.0) +
|
|
173
|
+
COALESCE(1.0 / (60 + c.bm25_rank), 0.0) AS total_score
|
|
174
|
+
FROM all_candidates c
|
|
175
|
+
JOIN recency_ranked r ON r.id = c.id
|
|
176
|
+
WHERE c.retriever != 'dense' OR (1.0 - c.dense_dist) >= 0.1
|
|
177
|
+
),
|
|
178
|
+
best_per_source AS (
|
|
179
|
+
SELECT *, ROW_NUMBER() OVER (PARTITION BY src ORDER BY total_score DESC, id) AS src_rank
|
|
180
|
+
FROM scored
|
|
181
|
+
)
|
|
182
|
+
SELECT id, src, retriever, total_score
|
|
183
|
+
FROM best_per_source
|
|
184
|
+
WHERE src_rank = 1
|
|
185
|
+
ORDER BY total_score DESC, src, id
|
|
186
|
+
LIMIT 5
|
|
187
|
+
`)
|
|
188
|
+
.all(queryVec, queryVec, 3, 'source*', 3);
|
|
189
|
+
// Should return at most one row per src, 'a' (dense=0 dist → score=1) should be top
|
|
190
|
+
assert.ok(rows.length > 0, 'Hybrid query must return results');
|
|
191
|
+
assert.equal(rows[0].id, 'a', 'Nearest dense candidate must rank first');
|
|
192
|
+
// Verify distinct sources
|
|
193
|
+
const srcs = rows.map((r) => r.src);
|
|
194
|
+
const uniqueSrcs = new Set(srcs);
|
|
195
|
+
assert.equal(srcs.length, uniqueSrcs.size, 'Each src must appear at most once (PARTITION BY dedup)');
|
|
196
|
+
// 'a' matched both dense (top) and BM25 (src matches 'source*')
|
|
197
|
+
assert.equal(rows[0].retriever, 'both', "'a' must be retriever=both");
|
|
198
|
+
}
|
|
199
|
+
finally {
|
|
200
|
+
db.close();
|
|
201
|
+
fs.rmSync(compatDir, { recursive: true, force: true });
|
|
202
|
+
}
|
|
203
|
+
});
|
|
106
204
|
test('vecAvailable is true after schema creation', () => {
|
|
107
205
|
assert.equal(store.vecAvailable, true);
|
|
108
206
|
});
|
|
@@ -113,13 +211,9 @@ describe('MemoryStore', () => {
|
|
|
113
211
|
});
|
|
114
212
|
test('upsert stores chunks and hashesBySource returns their ids', () => {
|
|
115
213
|
const chunk = makeChunk({ id: 'upsert-test-0001', source: 'upsert.md', sourceType: 'entity', fileMtimeAt: 2000 });
|
|
116
|
-
|
|
117
|
-
store.upsert([chunk], [embedding]);
|
|
214
|
+
store.upsert([chunk], [makeEmbedding(0)]);
|
|
118
215
|
const hashes = store.hashesBySource('upsert.md');
|
|
119
216
|
assert.ok(hashes.has('upsert-test-0001'));
|
|
120
|
-
const [stored] = store.getChunksByIds(['upsert-test-0001']);
|
|
121
|
-
assert.equal(stored.sourceType, 'entity');
|
|
122
|
-
assert.equal(stored.fileMtimeAt, 2000);
|
|
123
217
|
});
|
|
124
218
|
test('deleteBySource removes all chunks for that source', () => {
|
|
125
219
|
const c1 = makeChunk({
|
|
@@ -137,72 +231,6 @@ describe('MemoryStore', () => {
|
|
|
137
231
|
const hashes = store.hashesBySource('deleteme.md');
|
|
138
232
|
assert.equal(hashes.size, 0);
|
|
139
233
|
});
|
|
140
|
-
test('searchBm25 returns matching result after upsert', () => {
|
|
141
|
-
const chunk = makeChunk({
|
|
142
|
-
id: 'bm25-search-0001',
|
|
143
|
-
source: 'bm25.md',
|
|
144
|
-
content: 'uniqueKeywordXYZ for BM25 testing',
|
|
145
|
-
});
|
|
146
|
-
store.upsert([chunk], [makeEmbedding(3)]);
|
|
147
|
-
const results = store.searchBm25('uniqueKeywordXYZ', 5);
|
|
148
|
-
assert.ok(results.length > 0, 'Expected at least one BM25 result');
|
|
149
|
-
assert.ok(results.some((r) => r.id === 'bm25-search-0001'), `Expected chunk id in results, got: ${results.map((r) => r.id).join(', ')}`);
|
|
150
|
-
});
|
|
151
|
-
test('searchBm25 ignores filler words and matches preference source/content terms', () => {
|
|
152
|
-
const preference = makeChunk({
|
|
153
|
-
id: 'preference-search-0001',
|
|
154
|
-
source: 'compiled/preferences.md',
|
|
155
|
-
heading: '',
|
|
156
|
-
headingLevel: 0,
|
|
157
|
-
content: 'I like fish',
|
|
158
|
-
});
|
|
159
|
-
const unrelated = makeChunk({
|
|
160
|
-
id: 'preference-search-0002',
|
|
161
|
-
source: 'compiled/entities/worktree.md',
|
|
162
|
-
heading: 'Open Questions',
|
|
163
|
-
content: 'How should git manage worktree lifecycle decisions?',
|
|
164
|
-
});
|
|
165
|
-
store.upsert([preference, unrelated], [makeEmbedding(4), makeEmbedding(5)]);
|
|
166
|
-
const results = store.searchBm25('personal likes and preferences of the user', 5);
|
|
167
|
-
assert.ok(results.length > 0, 'Expected preference query to return results');
|
|
168
|
-
assert.equal(results[0].id, 'preference-search-0001');
|
|
169
|
-
});
|
|
170
|
-
test('searchBm25 returns empty array for empty query', () => {
|
|
171
|
-
const results = store.searchBm25(' ', 5);
|
|
172
|
-
assert.deepEqual(results, []);
|
|
173
|
-
});
|
|
174
|
-
test('getChunksByIds returns correct metadata for stored chunk', () => {
|
|
175
|
-
const chunk = makeChunk({
|
|
176
|
-
id: 'get-by-ids-0001',
|
|
177
|
-
source: 'metadata.md',
|
|
178
|
-
sourceType: 'concept',
|
|
179
|
-
heading: 'Metadata Section',
|
|
180
|
-
headingLevel: 2,
|
|
181
|
-
content: 'Content for metadata test',
|
|
182
|
-
lineStart: 10,
|
|
183
|
-
lineEnd: 20,
|
|
184
|
-
fileMtimeAt: 3000,
|
|
185
|
-
});
|
|
186
|
-
store.upsert([chunk], [makeEmbedding(6)]);
|
|
187
|
-
const results = store.getChunksByIds(['get-by-ids-0001']);
|
|
188
|
-
assert.equal(results.length, 1);
|
|
189
|
-
const [r] = results;
|
|
190
|
-
assert.equal(r.id, 'get-by-ids-0001');
|
|
191
|
-
assert.equal(r.source, 'metadata.md');
|
|
192
|
-
assert.equal(r.sourceType, 'concept');
|
|
193
|
-
assert.equal(r.heading, 'Metadata Section');
|
|
194
|
-
assert.equal(r.headingLevel, 2);
|
|
195
|
-
assert.equal(r.lineStart, 10);
|
|
196
|
-
assert.equal(r.lineEnd, 20);
|
|
197
|
-
assert.equal(r.fileMtimeAt, 3000);
|
|
198
|
-
});
|
|
199
|
-
test('getChunksByIds returns only found rows for mixed ids', () => {
|
|
200
|
-
const chunk = makeChunk({ id: 'partial-found-0001', source: 'partial.md', content: 'partial' });
|
|
201
|
-
store.upsert([chunk], [makeEmbedding(7)]);
|
|
202
|
-
const results = store.getChunksByIds(['partial-found-0001', 'does-not-exist-999']);
|
|
203
|
-
assert.equal(results.length, 1);
|
|
204
|
-
assert.equal(results[0].id, 'partial-found-0001');
|
|
205
|
-
});
|
|
206
234
|
test('indexedSources includes source after upsert', () => {
|
|
207
235
|
const chunk = makeChunk({
|
|
208
236
|
id: 'indexed-src-0001',
|
|
@@ -222,17 +250,6 @@ describe('MemoryStore', () => {
|
|
|
222
250
|
assert.ok(!hashes.has('del-ids-0001'), 'Deleted chunk id must be gone');
|
|
223
251
|
assert.ok(hashes.has('del-ids-0002'), 'Non-deleted chunk id must remain');
|
|
224
252
|
});
|
|
225
|
-
test('searchBm25 still works regardless of vecAvailable', () => {
|
|
226
|
-
// This verifies FTS5 degraded mode is always functional
|
|
227
|
-
const chunk = makeChunk({
|
|
228
|
-
id: 'fts5-degraded-0001',
|
|
229
|
-
source: 'fts5.md',
|
|
230
|
-
content: 'degradedModeTest keyword',
|
|
231
|
-
});
|
|
232
|
-
store.upsert([chunk], [makeEmbedding(11)]);
|
|
233
|
-
const results = store.searchBm25('degradedModeTest', 5);
|
|
234
|
-
assert.ok(results.length > 0, 'FTS5 search must work regardless of vecAvailable');
|
|
235
|
-
});
|
|
236
253
|
test('getRecentSources returns distinct sources ordered by newest chunk mtime', () => {
|
|
237
254
|
const recentDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-recent-store-'));
|
|
238
255
|
const recentStore = new MemoryStore(path.join(recentDir, 'index.db'), { ...TEST_CONFIG, wikiDir: recentDir });
|
|
@@ -282,28 +299,6 @@ describe('MemoryStore', () => {
|
|
|
282
299
|
fs.rmSync(recentDir, { recursive: true, force: true });
|
|
283
300
|
}
|
|
284
301
|
});
|
|
285
|
-
test('searchDense returns nearest vector with bounded scores', () => {
|
|
286
|
-
const denseDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-dense-store-'));
|
|
287
|
-
const denseStore = new MemoryStore(path.join(denseDir, 'index.db'), { ...TEST_CONFIG, wikiDir: denseDir });
|
|
288
|
-
try {
|
|
289
|
-
denseStore.upsert([
|
|
290
|
-
makeChunk({ id: 'dense-alpha', source: 'dense.md', content: 'alpha vector' }),
|
|
291
|
-
makeChunk({ id: 'dense-beta', source: 'dense.md', content: 'beta vector' }),
|
|
292
|
-
makeChunk({ id: 'dense-gamma', source: 'dense.md', content: 'gamma vector' }),
|
|
293
|
-
], [makeEmbedding(21), makeEmbedding(22), makeEmbedding(23)]);
|
|
294
|
-
const results = denseStore.searchDense(makeEmbedding(22), 3);
|
|
295
|
-
assert.ok(results.length > 0, 'Expected dense search results');
|
|
296
|
-
assert.equal(results[0].id, 'dense-beta');
|
|
297
|
-
assert.ok(results.every((result) => result.score >= 0 && result.score <= 1));
|
|
298
|
-
}
|
|
299
|
-
finally {
|
|
300
|
-
denseStore.close();
|
|
301
|
-
fs.rmSync(denseDir, { recursive: true, force: true });
|
|
302
|
-
}
|
|
303
|
-
});
|
|
304
|
-
test('searchDense returns empty array for invalid query embedding', () => {
|
|
305
|
-
assert.deepEqual(store.searchDense(new Float32Array(3), 5), []);
|
|
306
|
-
});
|
|
307
302
|
test('upsert skips invalid embeddings while persisting valid chunks', () => {
|
|
308
303
|
const validChunk = makeChunk({ id: 'valid-embedding-0001', source: 'embedding-validation.md' });
|
|
309
304
|
const invalidChunk = makeChunk({ id: 'invalid-embedding-0001', source: 'embedding-validation.md' });
|
|
@@ -379,14 +374,263 @@ describe('MemoryStore', () => {
|
|
|
379
374
|
content: 'postMigrationKeyword is searchable after migration',
|
|
380
375
|
});
|
|
381
376
|
migratedStore.upsert([migratedChunk], [makeEmbedding(16)]);
|
|
382
|
-
const results = migratedStore.
|
|
383
|
-
|
|
377
|
+
const results = migratedStore.searchHybrid({
|
|
378
|
+
query: 'postMigrationKeyword',
|
|
379
|
+
topK: 5,
|
|
380
|
+
fetchLimit: 20,
|
|
381
|
+
denseScoreFloor: 0,
|
|
382
|
+
recencyWeight: RECENCY_WEIGHT,
|
|
383
|
+
rrfK: RRF_K,
|
|
384
|
+
});
|
|
385
|
+
assert.ok(results.some((r) => r.chunk.id === 'migrated-bm25-0001'));
|
|
384
386
|
}
|
|
385
387
|
finally {
|
|
386
388
|
migratedStore.close();
|
|
387
389
|
fs.rmSync(migrationDir, { recursive: true, force: true });
|
|
388
390
|
}
|
|
389
391
|
});
|
|
392
|
+
// Task 3.4: Store-level hybrid ranking tests (BC1–BC12)
|
|
393
|
+
describe('searchHybrid', () => {
|
|
394
|
+
let hybridDir;
|
|
395
|
+
let hybridStore;
|
|
396
|
+
function makeHybridOpts(overrides = {}) {
|
|
397
|
+
return {
|
|
398
|
+
query: 'test query',
|
|
399
|
+
topK: 10,
|
|
400
|
+
fetchLimit: 10 * FETCH_MULTIPLIER,
|
|
401
|
+
denseScoreFloor: DENSE_SCORE_FLOOR,
|
|
402
|
+
recencyWeight: RECENCY_WEIGHT,
|
|
403
|
+
rrfK: RRF_K,
|
|
404
|
+
...overrides,
|
|
405
|
+
};
|
|
406
|
+
}
|
|
407
|
+
before(() => {
|
|
408
|
+
hybridDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-hybrid-store-'));
|
|
409
|
+
hybridStore = new MemoryStore(path.join(hybridDir, 'index.db'), { ...TEST_CONFIG, wikiDir: hybridDir });
|
|
410
|
+
// Insert test chunks with controlled embeddings:
|
|
411
|
+
// Embedding dims 40-49 for orthogonal test vectors (dimension 384)
|
|
412
|
+
// Each chunk at a unique dimension → query at dim 40 = nearest to 'alpha'
|
|
413
|
+
hybridStore.upsert([
|
|
414
|
+
// alpha: dense-only candidate, content has no keywords
|
|
415
|
+
makeChunk({
|
|
416
|
+
id: 'h-alpha',
|
|
417
|
+
source: 'source-alpha.md',
|
|
418
|
+
content: 'semantic content with no matching keywords',
|
|
419
|
+
fileMtimeAt: 1000,
|
|
420
|
+
sourceType: 'wiki',
|
|
421
|
+
}),
|
|
422
|
+
// beta: BM25-only candidate (keywords match, embedding far)
|
|
423
|
+
makeChunk({
|
|
424
|
+
id: 'h-beta',
|
|
425
|
+
source: 'source-beta.md',
|
|
426
|
+
content: 'hybridKeyword search text retrieval',
|
|
427
|
+
fileMtimeAt: 2000,
|
|
428
|
+
sourceType: 'concept',
|
|
429
|
+
}),
|
|
430
|
+
// gamma: dual-channel candidate (keywords + near embedding)
|
|
431
|
+
makeChunk({
|
|
432
|
+
id: 'h-gamma',
|
|
433
|
+
source: 'source-gamma.md',
|
|
434
|
+
content: 'hybridKeyword semantic content',
|
|
435
|
+
fileMtimeAt: 3000,
|
|
436
|
+
sourceType: 'wiki',
|
|
437
|
+
}),
|
|
438
|
+
// delta: below-floor dense-only (far from alpha query, no keywords)
|
|
439
|
+
makeChunk({
|
|
440
|
+
id: 'h-delta',
|
|
441
|
+
source: 'source-delta.md',
|
|
442
|
+
content: 'unrelated content with no relevant terms',
|
|
443
|
+
fileMtimeAt: 500,
|
|
444
|
+
sourceType: 'entity',
|
|
445
|
+
}),
|
|
446
|
+
// epsilon: same source as alpha (multi-chunk source for dedup test)
|
|
447
|
+
makeChunk({
|
|
448
|
+
id: 'h-epsilon',
|
|
449
|
+
source: 'source-alpha.md',
|
|
450
|
+
content: 'second chunk of alpha source, different content',
|
|
451
|
+
fileMtimeAt: 1000,
|
|
452
|
+
sourceType: 'wiki',
|
|
453
|
+
}),
|
|
454
|
+
// zeta: concept source for filter tests, has keywords
|
|
455
|
+
makeChunk({
|
|
456
|
+
id: 'h-zeta',
|
|
457
|
+
source: 'source-zeta.md',
|
|
458
|
+
content: 'hybridKeyword concept data for filtering',
|
|
459
|
+
fileMtimeAt: 4000,
|
|
460
|
+
sourceType: 'concept',
|
|
461
|
+
}),
|
|
462
|
+
], [
|
|
463
|
+
makeEmbedding(40), // alpha — at dim 40 (will match query at dim 40)
|
|
464
|
+
makeEmbedding(41), // beta — orthogonal to dim-40 query
|
|
465
|
+
makeEmbedding(40), // gamma — same embedding as alpha (near query)
|
|
466
|
+
makeEmbedding(45), // delta — orthogonal to dim-40 query
|
|
467
|
+
makeEmbedding(40), // epsilon — same source as alpha
|
|
468
|
+
makeEmbedding(41), // zeta — orthogonal to dim-40 query
|
|
469
|
+
]);
|
|
470
|
+
});
|
|
471
|
+
after(() => {
|
|
472
|
+
hybridStore.close();
|
|
473
|
+
fs.rmSync(hybridDir, { recursive: true, force: true });
|
|
474
|
+
});
|
|
475
|
+
// BC2: stopword-only query returns [] without malformed FTS SQL
|
|
476
|
+
test('BC2: stopword-only query returns empty array without malformed FTS SQL', () => {
|
|
477
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({ query: 'the and or is' }));
|
|
478
|
+
assert.deepEqual(results, []);
|
|
479
|
+
});
|
|
480
|
+
// BC4: BM25-only (no embedding) — returns finite normalized scores
|
|
481
|
+
test('BC4: BM25-only search returns results with finite scores when no embedding provided', () => {
|
|
482
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({ query: 'hybridKeyword', embedding: undefined }));
|
|
483
|
+
assert.ok(results.length > 0, 'Expected BM25-only results');
|
|
484
|
+
for (const r of results) {
|
|
485
|
+
assert.ok(Number.isFinite(r.score), `Score must be finite, got ${r.score}`);
|
|
486
|
+
assert.ok(r.score >= 0, 'Score must be >= 0');
|
|
487
|
+
assert.equal(r.retriever, 'bm25', 'retriever must be bm25 when no embedding');
|
|
488
|
+
}
|
|
489
|
+
});
|
|
490
|
+
// BC1 + BC3: dense+BM25 — returns topK ranked rows with correct retriever attribution
|
|
491
|
+
test('BC1/BC3: dual-channel search returns ranked rows with retriever attribution', () => {
|
|
492
|
+
const queryEmbed = makeEmbedding(40);
|
|
493
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
494
|
+
query: 'hybridKeyword',
|
|
495
|
+
embedding: queryEmbed,
|
|
496
|
+
topK: 5,
|
|
497
|
+
}));
|
|
498
|
+
assert.ok(results.length > 0, 'Expected results from dual-channel search');
|
|
499
|
+
assert.ok(results.length <= 5, 'Must not exceed topK');
|
|
500
|
+
for (const r of results) {
|
|
501
|
+
assert.ok(r.chunk, 'Each row must have a chunk');
|
|
502
|
+
assert.ok(Number.isFinite(r.score), 'Score must be finite');
|
|
503
|
+
assert.ok(['dense', 'bm25', 'both'].includes(r.retriever), 'retriever must be valid');
|
|
504
|
+
}
|
|
505
|
+
// gamma matches both channels (embedding at 40 + hybridKeyword content)
|
|
506
|
+
const gammaResult = results.find((r) => r.chunk.id === 'h-gamma');
|
|
507
|
+
assert.ok(gammaResult, 'h-gamma must be in results');
|
|
508
|
+
assert.equal(gammaResult?.retriever, 'both', 'gamma must have retriever=both');
|
|
509
|
+
});
|
|
510
|
+
// BC5: dual-channel candidate ranks above single-channel candidate
|
|
511
|
+
test('BC5: dual-channel candidate ranks above single-channel with same RRF weights', () => {
|
|
512
|
+
const queryEmbed = makeEmbedding(40);
|
|
513
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
514
|
+
query: 'hybridKeyword',
|
|
515
|
+
embedding: queryEmbed,
|
|
516
|
+
topK: 10,
|
|
517
|
+
}));
|
|
518
|
+
const gammaIdx = results.findIndex((r) => r.chunk.id === 'h-gamma');
|
|
519
|
+
const alphaIdx = results.findIndex((r) => r.chunk.id === 'h-alpha');
|
|
520
|
+
assert.ok(gammaIdx !== -1, 'gamma must be in results');
|
|
521
|
+
assert.ok(alphaIdx !== -1, 'alpha must be in results (above-floor dense)');
|
|
522
|
+
// gamma has both channels (higher RRF sum) so it should rank >= alpha (dense-only)
|
|
523
|
+
assert.ok(gammaIdx <= alphaIdx, `dual-channel gamma (idx ${gammaIdx}) must rank at or above dense-only alpha (idx ${alphaIdx})`);
|
|
524
|
+
});
|
|
525
|
+
// BC6: dense-only candidate below floor is dropped
|
|
526
|
+
test('BC6: dense-only candidate with score below denseScoreFloor is dropped', () => {
|
|
527
|
+
// delta is at dim 45, query at dim 40 → orthogonal → distance=1 → score=0 < 0.2
|
|
528
|
+
const queryEmbed = makeEmbedding(40);
|
|
529
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
530
|
+
query: 'hybridKeyword',
|
|
531
|
+
embedding: queryEmbed,
|
|
532
|
+
topK: 10,
|
|
533
|
+
denseScoreFloor: 0.5, // Higher floor to guarantee delta drops
|
|
534
|
+
}));
|
|
535
|
+
const deltaResult = results.find((r) => r.chunk.id === 'h-delta');
|
|
536
|
+
assert.ok(!deltaResult, 'h-delta (orthogonal, dense-only) must be dropped below floor');
|
|
537
|
+
});
|
|
538
|
+
// BC7: source dedup — at most one result per source (best chunk wins)
|
|
539
|
+
test('BC7: source dedup returns at most one result per source', () => {
|
|
540
|
+
const queryEmbed = makeEmbedding(40);
|
|
541
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
542
|
+
query: 'hybridKeyword',
|
|
543
|
+
embedding: queryEmbed,
|
|
544
|
+
topK: 10,
|
|
545
|
+
}));
|
|
546
|
+
const sources = results.map((r) => r.chunk.source);
|
|
547
|
+
const uniqueSources = new Set(sources);
|
|
548
|
+
assert.equal(sources.length, uniqueSources.size, 'Each source must appear at most once');
|
|
549
|
+
// source-alpha.md has two chunks (alpha + epsilon) — only one should surface
|
|
550
|
+
const alphaCount = sources.filter((s) => s === 'source-alpha.md').length;
|
|
551
|
+
assert.ok(alphaCount <= 1, `source-alpha.md must appear at most once, got ${alphaCount}`);
|
|
552
|
+
});
|
|
553
|
+
// BC8: deterministic order for tied candidates
|
|
554
|
+
test('BC8: repeated searchHybrid calls produce identical ordering', () => {
|
|
555
|
+
const queryEmbed = makeEmbedding(40);
|
|
556
|
+
const opts = makeHybridOpts({ query: 'hybridKeyword', embedding: queryEmbed, topK: 10 });
|
|
557
|
+
const first = hybridStore.searchHybrid(opts);
|
|
558
|
+
const second = hybridStore.searchHybrid(opts);
|
|
559
|
+
assert.equal(first.length, second.length, 'Result count must be stable');
|
|
560
|
+
for (let i = 0; i < first.length; i++) {
|
|
561
|
+
assert.equal(first[i].chunk.id, second[i].chunk.id, `Position ${i} must be deterministic`);
|
|
562
|
+
}
|
|
563
|
+
});
|
|
564
|
+
// BC9: sourceType filter applies to both channels
|
|
565
|
+
test('BC9: sourceType filter restricts results to matching source type', () => {
|
|
566
|
+
const queryEmbed = makeEmbedding(40);
|
|
567
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
568
|
+
query: 'hybridKeyword',
|
|
569
|
+
embedding: queryEmbed,
|
|
570
|
+
sourceType: 'concept',
|
|
571
|
+
topK: 10,
|
|
572
|
+
}));
|
|
573
|
+
assert.ok(results.length > 0, 'Filtered results must include concept sources');
|
|
574
|
+
for (const r of results) {
|
|
575
|
+
assert.equal(r.chunk.sourceType, 'concept', `All results must be concept type, got ${r.chunk.sourceType}`);
|
|
576
|
+
}
|
|
577
|
+
});
|
|
578
|
+
// BC9b: sinceMtimeAt filter restricts to recent sources
|
|
579
|
+
test('BC9b: sinceMtimeAt filter restricts results to sources newer than threshold', () => {
|
|
580
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
581
|
+
query: 'hybridKeyword',
|
|
582
|
+
sinceMtimeAt: 3000,
|
|
583
|
+
topK: 10,
|
|
584
|
+
}));
|
|
585
|
+
for (const r of results) {
|
|
586
|
+
assert.ok(r.chunk.fileMtimeAt >= 3000, `All results must have mtime >= 3000, got ${r.chunk.fileMtimeAt}`);
|
|
587
|
+
}
|
|
588
|
+
});
|
|
589
|
+
// BC11: debug fields present when includeDebug=true
|
|
590
|
+
test('BC11: debug fields are present on rows when includeDebug is true', () => {
|
|
591
|
+
const queryEmbed = makeEmbedding(40);
|
|
592
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
593
|
+
query: 'hybridKeyword',
|
|
594
|
+
embedding: queryEmbed,
|
|
595
|
+
includeDebug: true,
|
|
596
|
+
topK: 10,
|
|
597
|
+
}));
|
|
598
|
+
assert.ok(results.length > 0, 'Expected results with debug');
|
|
599
|
+
for (const r of results) {
|
|
600
|
+
assert.ok(r.debug, 'debug must be present when includeDebug=true');
|
|
601
|
+
assert.ok(Number.isFinite(r.debug.recencyRank), 'recencyRank must be finite');
|
|
602
|
+
assert.ok(Number.isFinite(r.debug.recencyScore), 'recencyScore must be finite');
|
|
603
|
+
assert.ok(Number.isFinite(r.debug.totalScore), 'totalScore must be finite');
|
|
604
|
+
}
|
|
605
|
+
});
|
|
606
|
+
// BC11b: debug fields absent when includeDebug is false/undefined
|
|
607
|
+
test('BC11b: debug fields are absent when includeDebug is not set', () => {
|
|
608
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({ query: 'hybridKeyword' }));
|
|
609
|
+
for (const r of results) {
|
|
610
|
+
assert.equal(r.debug, undefined, 'debug must be absent when includeDebug is not set');
|
|
611
|
+
}
|
|
612
|
+
});
|
|
613
|
+
// Empty channel: both dense and BM25 empty → []
|
|
614
|
+
test('empty channels: both channels return empty — searchHybrid returns []', () => {
|
|
615
|
+
// Query with no matching BM25 terms and embedding pointing to unmapped dimension
|
|
616
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
617
|
+
query: 'the and or', // stopwords → empty FTS
|
|
618
|
+
embedding: undefined,
|
|
619
|
+
}));
|
|
620
|
+
assert.deepEqual(results, []);
|
|
621
|
+
});
|
|
622
|
+
// Score range: normalized scores must be in [0, 1]
|
|
623
|
+
test('normalized scores are bounded in [0, 1]', () => {
|
|
624
|
+
const results = hybridStore.searchHybrid(makeHybridOpts({
|
|
625
|
+
query: 'hybridKeyword',
|
|
626
|
+
embedding: makeEmbedding(40),
|
|
627
|
+
topK: 10,
|
|
628
|
+
}));
|
|
629
|
+
for (const r of results) {
|
|
630
|
+
assert.ok(r.score >= 0 && r.score <= 1, `Score out of range [0,1]: ${r.score}`);
|
|
631
|
+
}
|
|
632
|
+
});
|
|
633
|
+
});
|
|
390
634
|
test('recreates schema when existing vector dimension differs from config', () => {
|
|
391
635
|
const dimensionDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-dimension-migration-'));
|
|
392
636
|
const dimensionDbPath = path.join(dimensionDir, 'index.db');
|
|
@@ -422,8 +666,15 @@ describe('MemoryStore', () => {
|
|
|
422
666
|
content: 'large dimension vector can be indexed',
|
|
423
667
|
});
|
|
424
668
|
largeStore.upsert([chunk], [makeEmbedding(18, 768)]);
|
|
425
|
-
const results = largeStore.
|
|
426
|
-
|
|
669
|
+
const results = largeStore.searchHybrid({
|
|
670
|
+
query: 'large dimension vector',
|
|
671
|
+
topK: 5,
|
|
672
|
+
fetchLimit: 20,
|
|
673
|
+
denseScoreFloor: 0,
|
|
674
|
+
recencyWeight: RECENCY_WEIGHT,
|
|
675
|
+
rrfK: RRF_K,
|
|
676
|
+
});
|
|
677
|
+
assert.ok(results.some((r) => r.chunk.id === 'new-dimension-row'));
|
|
427
678
|
}
|
|
428
679
|
finally {
|
|
429
680
|
largeStore.close();
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { type ProjectSettings } from '../../core/config/index.js';
|
|
2
|
+
import { MemoryIndexer } from './indexer.js';
|
|
3
|
+
import { MemoryStore } from './store.js';
|
|
4
|
+
import type { MemoryConfig, MemoryLifecycleStatus } from './types.js';
|
|
5
|
+
export interface MemorySubsystemOverrides {
|
|
6
|
+
indexer?: MemoryIndexer;
|
|
7
|
+
store?: MemoryStore;
|
|
8
|
+
config?: MemoryConfig;
|
|
9
|
+
settings?: ProjectSettings;
|
|
10
|
+
}
|
|
11
|
+
export interface MemorySubsystem {
|
|
12
|
+
readonly config: MemoryConfig;
|
|
13
|
+
readonly status: MemoryLifecycleStatus;
|
|
14
|
+
readonly indexer: MemoryIndexer | null;
|
|
15
|
+
readonly store: MemoryStore | null;
|
|
16
|
+
startWarmup(): Promise<void>;
|
|
17
|
+
close(): void;
|
|
18
|
+
}
|
|
19
|
+
export declare function createMemorySubsystem(workspaceRoot: string, overrides?: MemorySubsystemOverrides): MemorySubsystem | null;
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import * as path from 'path';
|
|
2
|
+
import { resolveMemoryConfig } from '../../core/config/index.js';
|
|
3
|
+
import { Embedder } from './embedder.js';
|
|
4
|
+
import { MemoryIndexer } from './indexer.js';
|
|
5
|
+
import { MemoryStore } from './store.js';
|
|
6
|
+
class DefaultMemorySubsystem {
|
|
7
|
+
config;
|
|
8
|
+
indexer;
|
|
9
|
+
store;
|
|
10
|
+
lifecycleStatus;
|
|
11
|
+
constructor(config, indexer, store, initialState, initialError) {
|
|
12
|
+
this.config = config;
|
|
13
|
+
this.indexer = indexer;
|
|
14
|
+
this.store = store;
|
|
15
|
+
this.lifecycleStatus = makeStatus(initialState, initialError);
|
|
16
|
+
}
|
|
17
|
+
get status() {
|
|
18
|
+
return this.lifecycleStatus;
|
|
19
|
+
}
|
|
20
|
+
async startWarmup() {
|
|
21
|
+
if (!this.indexer) {
|
|
22
|
+
this.lifecycleStatus = makeStatus('failed', this.lifecycleStatus.error ?? 'Memory indexer is unavailable');
|
|
23
|
+
return;
|
|
24
|
+
}
|
|
25
|
+
this.lifecycleStatus = makeStatus('warming');
|
|
26
|
+
try {
|
|
27
|
+
await this.indexer.startupIndex();
|
|
28
|
+
this.lifecycleStatus = makeStatus('ready');
|
|
29
|
+
}
|
|
30
|
+
catch (err) {
|
|
31
|
+
this.lifecycleStatus = makeStatus(this.store ? 'degraded' : 'failed', formatError(err));
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
close() {
|
|
35
|
+
this.store?.close();
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
function makeStatus(state, error) {
|
|
39
|
+
return {
|
|
40
|
+
state,
|
|
41
|
+
ready: state === 'ready',
|
|
42
|
+
warming: state === 'initializing' || state === 'warming',
|
|
43
|
+
degraded: state === 'degraded' || state === 'failed',
|
|
44
|
+
error,
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
function formatError(err) {
|
|
48
|
+
return err instanceof Error ? err.message : String(err);
|
|
49
|
+
}
|
|
50
|
+
export function createMemorySubsystem(workspaceRoot, overrides = {}) {
|
|
51
|
+
const config = overrides.config ?? resolveMemoryConfig(overrides.settings ?? {}, workspaceRoot);
|
|
52
|
+
if (config.enabled !== true)
|
|
53
|
+
return null;
|
|
54
|
+
let store = overrides.store ?? null;
|
|
55
|
+
let indexer = overrides.indexer ?? null;
|
|
56
|
+
try {
|
|
57
|
+
store = store ?? new MemoryStore(path.join(config.wikiDir, 'index.db'), config);
|
|
58
|
+
if (!indexer) {
|
|
59
|
+
const embedder = new Embedder(config.embeddingModel);
|
|
60
|
+
indexer = new MemoryIndexer(store, embedder, config);
|
|
61
|
+
}
|
|
62
|
+
return new DefaultMemorySubsystem(config, indexer, store, 'initializing');
|
|
63
|
+
}
|
|
64
|
+
catch (err) {
|
|
65
|
+
return new DefaultMemorySubsystem(config, indexer, store, 'failed', formatError(err));
|
|
66
|
+
}
|
|
67
|
+
}
|
|
@@ -23,6 +23,34 @@ export interface SearchResult {
|
|
|
23
23
|
retriever: 'dense' | 'bm25' | 'both';
|
|
24
24
|
contentSource: 'file' | 'fallback';
|
|
25
25
|
}
|
|
26
|
+
export interface HybridSearchOptions {
|
|
27
|
+
query: string;
|
|
28
|
+
embedding?: Float32Array;
|
|
29
|
+
topK: number;
|
|
30
|
+
fetchLimit: number;
|
|
31
|
+
denseScoreFloor: number;
|
|
32
|
+
recencyWeight: number;
|
|
33
|
+
rrfK: number;
|
|
34
|
+
sourceType?: SourceType;
|
|
35
|
+
sinceMtimeAt?: number;
|
|
36
|
+
includeDebug?: boolean;
|
|
37
|
+
}
|
|
38
|
+
export interface HybridRankDebug {
|
|
39
|
+
denseRank?: number;
|
|
40
|
+
denseScore?: number;
|
|
41
|
+
bm25Rank?: number;
|
|
42
|
+
bm25Score?: number;
|
|
43
|
+
recencyRank: number;
|
|
44
|
+
recencyScore: number;
|
|
45
|
+
totalScore: number;
|
|
46
|
+
droppedReason?: string;
|
|
47
|
+
}
|
|
48
|
+
export interface HybridSearchRow {
|
|
49
|
+
chunk: MemoryChunk;
|
|
50
|
+
score: number;
|
|
51
|
+
retriever: 'dense' | 'bm25' | 'both';
|
|
52
|
+
debug?: HybridRankDebug;
|
|
53
|
+
}
|
|
26
54
|
export interface MemoryConfig {
|
|
27
55
|
enabled: boolean;
|
|
28
56
|
wikiDir: string;
|
|
@@ -38,6 +66,26 @@ export interface IndexStats {
|
|
|
38
66
|
deleted: number;
|
|
39
67
|
skipped: number;
|
|
40
68
|
}
|
|
69
|
+
export type MemoryLifecycleState = 'disabled' | 'initializing' | 'warming' | 'ready' | 'degraded' | 'failed';
|
|
70
|
+
export interface MemoryLifecycleStatus {
|
|
71
|
+
state: MemoryLifecycleState;
|
|
72
|
+
ready: boolean;
|
|
73
|
+
warming: boolean;
|
|
74
|
+
degraded: boolean;
|
|
75
|
+
error?: string;
|
|
76
|
+
}
|
|
77
|
+
export interface IndexDirectoryOptions {
|
|
78
|
+
relativeBase?: string;
|
|
79
|
+
excludeFiles?: string[];
|
|
80
|
+
fileConcurrency?: number;
|
|
81
|
+
}
|
|
82
|
+
export interface PreparedIndexMutation {
|
|
83
|
+
source: string;
|
|
84
|
+
chunksToUpsert: MemoryChunk[];
|
|
85
|
+
embeddings: Float32Array[];
|
|
86
|
+
idsToDelete: string[];
|
|
87
|
+
stats: IndexStats;
|
|
88
|
+
}
|
|
41
89
|
export interface RecentSource {
|
|
42
90
|
source: string;
|
|
43
91
|
fileMtimeAt: number;
|