@hanhnd/agent-kit 1.0.38 → 1.0.40

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,6 +6,7 @@ import * as path from 'node:path';
6
6
  import { after, before, describe, test } from 'node:test';
7
7
  import { MemoryStore, SCHEMA_VERSION } from './store.js';
8
8
  import { EmbeddingModelName } from './types.js';
9
+ import { DENSE_SCORE_FLOOR, FETCH_MULTIPLIER, RECENCY_WEIGHT, RRF_K } from './constants.js';
9
10
  const TEST_CONFIG = {
10
11
  enabled: true,
11
12
  wikiDir: '',
@@ -103,6 +104,103 @@ describe('MemoryStore', () => {
103
104
  fs.rmSync(libsqlDir, { recursive: true, force: true });
104
105
  }
105
106
  });
107
+ // BC22: Task 1.3 — prove hybrid SQL shape (CTEs + ROW_NUMBER OVER + UNION + PARTITION BY + vector_top_k + FTS)
108
+ test('BC22: libsql supports hybrid SQL primitives — CTEs, ROW_NUMBER OVER, UNION, PARTITION BY, vector_top_k, bm25', () => {
109
+ const compatDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-hybrid-compat-'));
110
+ const db = new Database(path.join(compatDir, 'compat.db'));
111
+ try {
112
+ db.pragma('journal_mode = WAL');
113
+ db.exec(`
114
+ CREATE TABLE items (
115
+ rowid INTEGER PRIMARY KEY AUTOINCREMENT,
116
+ id TEXT NOT NULL UNIQUE,
117
+ src TEXT NOT NULL,
118
+ embedding F32_BLOB(3) NOT NULL
119
+ );
120
+ CREATE INDEX idx_items_embedding ON items (libsql_vector_idx(embedding));
121
+ CREATE VIRTUAL TABLE items_fts USING fts5(id, src, content='items', content_rowid='rowid');
122
+ CREATE TRIGGER items_ai AFTER INSERT ON items BEGIN
123
+ INSERT INTO items_fts(rowid, id, src) VALUES (new.rowid, new.id, new.src);
124
+ END;
125
+ `);
126
+ const ins = db.prepare(`INSERT INTO items (id, src, embedding) VALUES (?, ?, vector32(?))`);
127
+ ins.run('a', 'source-alpha.md', '[1,0,0]');
128
+ ins.run('b', 'source-alpha.md', '[0,1,0]');
129
+ ins.run('c', 'source-beta.md', '[0,0,1]');
130
+ // Full hybrid SQL shape: CTEs + ROW_NUMBER OVER + UNION + PARTITION BY + vector_top_k + bm25
131
+ const queryVec = Buffer.from(new Float32Array([1, 0, 0]).buffer);
132
+ const rows = db
133
+ .prepare(`
134
+ WITH
135
+ dense_data AS (
136
+ SELECT items.id, items.src,
137
+ vector_distance_cos(items.embedding, vector32(?)) AS dense_dist
138
+ FROM vector_top_k('idx_items_embedding', vector32(?), ?) AS vtk
139
+ JOIN items ON items.rowid = vtk.id
140
+ ),
141
+ dense_ranked AS (
142
+ SELECT *, ROW_NUMBER() OVER (ORDER BY dense_dist, id) AS dense_rank
143
+ FROM dense_data
144
+ ),
145
+ bm25_data AS (
146
+ SELECT items.id, items.src,
147
+ bm25(items_fts, 1.0, 1.0) AS bm25_neg
148
+ FROM items_fts
149
+ JOIN items ON items.rowid = items_fts.rowid
150
+ WHERE items_fts MATCH ?
151
+ LIMIT ?
152
+ ),
153
+ bm25_ranked AS (
154
+ SELECT *, ROW_NUMBER() OVER (ORDER BY bm25_neg, id) AS bm25_rank
155
+ FROM bm25_data
156
+ ),
157
+ all_candidates AS (
158
+ SELECT d.id, d.src, d.dense_rank, d.dense_dist, b.bm25_rank,
159
+ CASE WHEN b.id IS NOT NULL THEN 'both' ELSE 'dense' END AS retriever
160
+ FROM dense_ranked d LEFT JOIN bm25_ranked b ON b.id = d.id
161
+ UNION ALL
162
+ SELECT b.id, b.src, NULL, NULL, b.bm25_rank, 'bm25' AS retriever
163
+ FROM bm25_ranked b LEFT JOIN dense_ranked d ON d.id = b.id
164
+ WHERE d.id IS NULL
165
+ ),
166
+ recency_ranked AS (
167
+ SELECT id, ROW_NUMBER() OVER (ORDER BY id) AS recency_rank
168
+ FROM all_candidates
169
+ ),
170
+ scored AS (
171
+ SELECT c.id, c.src, c.dense_rank, c.bm25_rank, c.retriever,
172
+ COALESCE(1.0 / (60 + c.dense_rank), 0.0) +
173
+ COALESCE(1.0 / (60 + c.bm25_rank), 0.0) AS total_score
174
+ FROM all_candidates c
175
+ JOIN recency_ranked r ON r.id = c.id
176
+ WHERE c.retriever != 'dense' OR (1.0 - c.dense_dist) >= 0.1
177
+ ),
178
+ best_per_source AS (
179
+ SELECT *, ROW_NUMBER() OVER (PARTITION BY src ORDER BY total_score DESC, id) AS src_rank
180
+ FROM scored
181
+ )
182
+ SELECT id, src, retriever, total_score
183
+ FROM best_per_source
184
+ WHERE src_rank = 1
185
+ ORDER BY total_score DESC, src, id
186
+ LIMIT 5
187
+ `)
188
+ .all(queryVec, queryVec, 3, 'source*', 3);
189
+ // Should return at most one row per src, 'a' (dense=0 dist → score=1) should be top
190
+ assert.ok(rows.length > 0, 'Hybrid query must return results');
191
+ assert.equal(rows[0].id, 'a', 'Nearest dense candidate must rank first');
192
+ // Verify distinct sources
193
+ const srcs = rows.map((r) => r.src);
194
+ const uniqueSrcs = new Set(srcs);
195
+ assert.equal(srcs.length, uniqueSrcs.size, 'Each src must appear at most once (PARTITION BY dedup)');
196
+ // 'a' matched both dense (top) and BM25 (src matches 'source*')
197
+ assert.equal(rows[0].retriever, 'both', "'a' must be retriever=both");
198
+ }
199
+ finally {
200
+ db.close();
201
+ fs.rmSync(compatDir, { recursive: true, force: true });
202
+ }
203
+ });
106
204
  test('vecAvailable is true after schema creation', () => {
107
205
  assert.equal(store.vecAvailable, true);
108
206
  });
@@ -113,13 +211,9 @@ describe('MemoryStore', () => {
113
211
  });
114
212
  test('upsert stores chunks and hashesBySource returns their ids', () => {
115
213
  const chunk = makeChunk({ id: 'upsert-test-0001', source: 'upsert.md', sourceType: 'entity', fileMtimeAt: 2000 });
116
- const embedding = makeEmbedding(0);
117
- store.upsert([chunk], [embedding]);
214
+ store.upsert([chunk], [makeEmbedding(0)]);
118
215
  const hashes = store.hashesBySource('upsert.md');
119
216
  assert.ok(hashes.has('upsert-test-0001'));
120
- const [stored] = store.getChunksByIds(['upsert-test-0001']);
121
- assert.equal(stored.sourceType, 'entity');
122
- assert.equal(stored.fileMtimeAt, 2000);
123
217
  });
124
218
  test('deleteBySource removes all chunks for that source', () => {
125
219
  const c1 = makeChunk({
@@ -137,72 +231,6 @@ describe('MemoryStore', () => {
137
231
  const hashes = store.hashesBySource('deleteme.md');
138
232
  assert.equal(hashes.size, 0);
139
233
  });
140
- test('searchBm25 returns matching result after upsert', () => {
141
- const chunk = makeChunk({
142
- id: 'bm25-search-0001',
143
- source: 'bm25.md',
144
- content: 'uniqueKeywordXYZ for BM25 testing',
145
- });
146
- store.upsert([chunk], [makeEmbedding(3)]);
147
- const results = store.searchBm25('uniqueKeywordXYZ', 5);
148
- assert.ok(results.length > 0, 'Expected at least one BM25 result');
149
- assert.ok(results.some((r) => r.id === 'bm25-search-0001'), `Expected chunk id in results, got: ${results.map((r) => r.id).join(', ')}`);
150
- });
151
- test('searchBm25 ignores filler words and matches preference source/content terms', () => {
152
- const preference = makeChunk({
153
- id: 'preference-search-0001',
154
- source: 'compiled/preferences.md',
155
- heading: '',
156
- headingLevel: 0,
157
- content: 'I like fish',
158
- });
159
- const unrelated = makeChunk({
160
- id: 'preference-search-0002',
161
- source: 'compiled/entities/worktree.md',
162
- heading: 'Open Questions',
163
- content: 'How should git manage worktree lifecycle decisions?',
164
- });
165
- store.upsert([preference, unrelated], [makeEmbedding(4), makeEmbedding(5)]);
166
- const results = store.searchBm25('personal likes and preferences of the user', 5);
167
- assert.ok(results.length > 0, 'Expected preference query to return results');
168
- assert.equal(results[0].id, 'preference-search-0001');
169
- });
170
- test('searchBm25 returns empty array for empty query', () => {
171
- const results = store.searchBm25(' ', 5);
172
- assert.deepEqual(results, []);
173
- });
174
- test('getChunksByIds returns correct metadata for stored chunk', () => {
175
- const chunk = makeChunk({
176
- id: 'get-by-ids-0001',
177
- source: 'metadata.md',
178
- sourceType: 'concept',
179
- heading: 'Metadata Section',
180
- headingLevel: 2,
181
- content: 'Content for metadata test',
182
- lineStart: 10,
183
- lineEnd: 20,
184
- fileMtimeAt: 3000,
185
- });
186
- store.upsert([chunk], [makeEmbedding(6)]);
187
- const results = store.getChunksByIds(['get-by-ids-0001']);
188
- assert.equal(results.length, 1);
189
- const [r] = results;
190
- assert.equal(r.id, 'get-by-ids-0001');
191
- assert.equal(r.source, 'metadata.md');
192
- assert.equal(r.sourceType, 'concept');
193
- assert.equal(r.heading, 'Metadata Section');
194
- assert.equal(r.headingLevel, 2);
195
- assert.equal(r.lineStart, 10);
196
- assert.equal(r.lineEnd, 20);
197
- assert.equal(r.fileMtimeAt, 3000);
198
- });
199
- test('getChunksByIds returns only found rows for mixed ids', () => {
200
- const chunk = makeChunk({ id: 'partial-found-0001', source: 'partial.md', content: 'partial' });
201
- store.upsert([chunk], [makeEmbedding(7)]);
202
- const results = store.getChunksByIds(['partial-found-0001', 'does-not-exist-999']);
203
- assert.equal(results.length, 1);
204
- assert.equal(results[0].id, 'partial-found-0001');
205
- });
206
234
  test('indexedSources includes source after upsert', () => {
207
235
  const chunk = makeChunk({
208
236
  id: 'indexed-src-0001',
@@ -222,17 +250,6 @@ describe('MemoryStore', () => {
222
250
  assert.ok(!hashes.has('del-ids-0001'), 'Deleted chunk id must be gone');
223
251
  assert.ok(hashes.has('del-ids-0002'), 'Non-deleted chunk id must remain');
224
252
  });
225
- test('searchBm25 still works regardless of vecAvailable', () => {
226
- // This verifies FTS5 degraded mode is always functional
227
- const chunk = makeChunk({
228
- id: 'fts5-degraded-0001',
229
- source: 'fts5.md',
230
- content: 'degradedModeTest keyword',
231
- });
232
- store.upsert([chunk], [makeEmbedding(11)]);
233
- const results = store.searchBm25('degradedModeTest', 5);
234
- assert.ok(results.length > 0, 'FTS5 search must work regardless of vecAvailable');
235
- });
236
253
  test('getRecentSources returns distinct sources ordered by newest chunk mtime', () => {
237
254
  const recentDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-recent-store-'));
238
255
  const recentStore = new MemoryStore(path.join(recentDir, 'index.db'), { ...TEST_CONFIG, wikiDir: recentDir });
@@ -282,28 +299,6 @@ describe('MemoryStore', () => {
282
299
  fs.rmSync(recentDir, { recursive: true, force: true });
283
300
  }
284
301
  });
285
- test('searchDense returns nearest vector with bounded scores', () => {
286
- const denseDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-dense-store-'));
287
- const denseStore = new MemoryStore(path.join(denseDir, 'index.db'), { ...TEST_CONFIG, wikiDir: denseDir });
288
- try {
289
- denseStore.upsert([
290
- makeChunk({ id: 'dense-alpha', source: 'dense.md', content: 'alpha vector' }),
291
- makeChunk({ id: 'dense-beta', source: 'dense.md', content: 'beta vector' }),
292
- makeChunk({ id: 'dense-gamma', source: 'dense.md', content: 'gamma vector' }),
293
- ], [makeEmbedding(21), makeEmbedding(22), makeEmbedding(23)]);
294
- const results = denseStore.searchDense(makeEmbedding(22), 3);
295
- assert.ok(results.length > 0, 'Expected dense search results');
296
- assert.equal(results[0].id, 'dense-beta');
297
- assert.ok(results.every((result) => result.score >= 0 && result.score <= 1));
298
- }
299
- finally {
300
- denseStore.close();
301
- fs.rmSync(denseDir, { recursive: true, force: true });
302
- }
303
- });
304
- test('searchDense returns empty array for invalid query embedding', () => {
305
- assert.deepEqual(store.searchDense(new Float32Array(3), 5), []);
306
- });
307
302
  test('upsert skips invalid embeddings while persisting valid chunks', () => {
308
303
  const validChunk = makeChunk({ id: 'valid-embedding-0001', source: 'embedding-validation.md' });
309
304
  const invalidChunk = makeChunk({ id: 'invalid-embedding-0001', source: 'embedding-validation.md' });
@@ -379,14 +374,263 @@ describe('MemoryStore', () => {
379
374
  content: 'postMigrationKeyword is searchable after migration',
380
375
  });
381
376
  migratedStore.upsert([migratedChunk], [makeEmbedding(16)]);
382
- const results = migratedStore.searchBm25('postMigrationKeyword', 5);
383
- assert.ok(results.some((result) => result.id === 'migrated-bm25-0001'));
377
+ const results = migratedStore.searchHybrid({
378
+ query: 'postMigrationKeyword',
379
+ topK: 5,
380
+ fetchLimit: 20,
381
+ denseScoreFloor: 0,
382
+ recencyWeight: RECENCY_WEIGHT,
383
+ rrfK: RRF_K,
384
+ });
385
+ assert.ok(results.some((r) => r.chunk.id === 'migrated-bm25-0001'));
384
386
  }
385
387
  finally {
386
388
  migratedStore.close();
387
389
  fs.rmSync(migrationDir, { recursive: true, force: true });
388
390
  }
389
391
  });
392
+ // Task 3.4: Store-level hybrid ranking tests (BC1–BC12)
393
+ describe('searchHybrid', () => {
394
+ let hybridDir;
395
+ let hybridStore;
396
+ function makeHybridOpts(overrides = {}) {
397
+ return {
398
+ query: 'test query',
399
+ topK: 10,
400
+ fetchLimit: 10 * FETCH_MULTIPLIER,
401
+ denseScoreFloor: DENSE_SCORE_FLOOR,
402
+ recencyWeight: RECENCY_WEIGHT,
403
+ rrfK: RRF_K,
404
+ ...overrides,
405
+ };
406
+ }
407
+ before(() => {
408
+ hybridDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-hybrid-store-'));
409
+ hybridStore = new MemoryStore(path.join(hybridDir, 'index.db'), { ...TEST_CONFIG, wikiDir: hybridDir });
410
+ // Insert test chunks with controlled embeddings:
411
+ // Embedding dims 40-49 for orthogonal test vectors (dimension 384)
412
+ // Each chunk at a unique dimension → query at dim 40 = nearest to 'alpha'
413
+ hybridStore.upsert([
414
+ // alpha: dense-only candidate, content has no keywords
415
+ makeChunk({
416
+ id: 'h-alpha',
417
+ source: 'source-alpha.md',
418
+ content: 'semantic content with no matching keywords',
419
+ fileMtimeAt: 1000,
420
+ sourceType: 'wiki',
421
+ }),
422
+ // beta: BM25-only candidate (keywords match, embedding far)
423
+ makeChunk({
424
+ id: 'h-beta',
425
+ source: 'source-beta.md',
426
+ content: 'hybridKeyword search text retrieval',
427
+ fileMtimeAt: 2000,
428
+ sourceType: 'concept',
429
+ }),
430
+ // gamma: dual-channel candidate (keywords + near embedding)
431
+ makeChunk({
432
+ id: 'h-gamma',
433
+ source: 'source-gamma.md',
434
+ content: 'hybridKeyword semantic content',
435
+ fileMtimeAt: 3000,
436
+ sourceType: 'wiki',
437
+ }),
438
+ // delta: below-floor dense-only (far from alpha query, no keywords)
439
+ makeChunk({
440
+ id: 'h-delta',
441
+ source: 'source-delta.md',
442
+ content: 'unrelated content with no relevant terms',
443
+ fileMtimeAt: 500,
444
+ sourceType: 'entity',
445
+ }),
446
+ // epsilon: same source as alpha (multi-chunk source for dedup test)
447
+ makeChunk({
448
+ id: 'h-epsilon',
449
+ source: 'source-alpha.md',
450
+ content: 'second chunk of alpha source, different content',
451
+ fileMtimeAt: 1000,
452
+ sourceType: 'wiki',
453
+ }),
454
+ // zeta: concept source for filter tests, has keywords
455
+ makeChunk({
456
+ id: 'h-zeta',
457
+ source: 'source-zeta.md',
458
+ content: 'hybridKeyword concept data for filtering',
459
+ fileMtimeAt: 4000,
460
+ sourceType: 'concept',
461
+ }),
462
+ ], [
463
+ makeEmbedding(40), // alpha — at dim 40 (will match query at dim 40)
464
+ makeEmbedding(41), // beta — orthogonal to dim-40 query
465
+ makeEmbedding(40), // gamma — same embedding as alpha (near query)
466
+ makeEmbedding(45), // delta — orthogonal to dim-40 query
467
+ makeEmbedding(40), // epsilon — same source as alpha
468
+ makeEmbedding(41), // zeta — orthogonal to dim-40 query
469
+ ]);
470
+ });
471
+ after(() => {
472
+ hybridStore.close();
473
+ fs.rmSync(hybridDir, { recursive: true, force: true });
474
+ });
475
+ // BC2: stopword-only query returns [] without malformed FTS SQL
476
+ test('BC2: stopword-only query returns empty array without malformed FTS SQL', () => {
477
+ const results = hybridStore.searchHybrid(makeHybridOpts({ query: 'the and or is' }));
478
+ assert.deepEqual(results, []);
479
+ });
480
+ // BC4: BM25-only (no embedding) — returns finite normalized scores
481
+ test('BC4: BM25-only search returns results with finite scores when no embedding provided', () => {
482
+ const results = hybridStore.searchHybrid(makeHybridOpts({ query: 'hybridKeyword', embedding: undefined }));
483
+ assert.ok(results.length > 0, 'Expected BM25-only results');
484
+ for (const r of results) {
485
+ assert.ok(Number.isFinite(r.score), `Score must be finite, got ${r.score}`);
486
+ assert.ok(r.score >= 0, 'Score must be >= 0');
487
+ assert.equal(r.retriever, 'bm25', 'retriever must be bm25 when no embedding');
488
+ }
489
+ });
490
+ // BC1 + BC3: dense+BM25 — returns topK ranked rows with correct retriever attribution
491
+ test('BC1/BC3: dual-channel search returns ranked rows with retriever attribution', () => {
492
+ const queryEmbed = makeEmbedding(40);
493
+ const results = hybridStore.searchHybrid(makeHybridOpts({
494
+ query: 'hybridKeyword',
495
+ embedding: queryEmbed,
496
+ topK: 5,
497
+ }));
498
+ assert.ok(results.length > 0, 'Expected results from dual-channel search');
499
+ assert.ok(results.length <= 5, 'Must not exceed topK');
500
+ for (const r of results) {
501
+ assert.ok(r.chunk, 'Each row must have a chunk');
502
+ assert.ok(Number.isFinite(r.score), 'Score must be finite');
503
+ assert.ok(['dense', 'bm25', 'both'].includes(r.retriever), 'retriever must be valid');
504
+ }
505
+ // gamma matches both channels (embedding at 40 + hybridKeyword content)
506
+ const gammaResult = results.find((r) => r.chunk.id === 'h-gamma');
507
+ assert.ok(gammaResult, 'h-gamma must be in results');
508
+ assert.equal(gammaResult?.retriever, 'both', 'gamma must have retriever=both');
509
+ });
510
+ // BC5: dual-channel candidate ranks above single-channel candidate
511
+ test('BC5: dual-channel candidate ranks above single-channel with same RRF weights', () => {
512
+ const queryEmbed = makeEmbedding(40);
513
+ const results = hybridStore.searchHybrid(makeHybridOpts({
514
+ query: 'hybridKeyword',
515
+ embedding: queryEmbed,
516
+ topK: 10,
517
+ }));
518
+ const gammaIdx = results.findIndex((r) => r.chunk.id === 'h-gamma');
519
+ const alphaIdx = results.findIndex((r) => r.chunk.id === 'h-alpha');
520
+ assert.ok(gammaIdx !== -1, 'gamma must be in results');
521
+ assert.ok(alphaIdx !== -1, 'alpha must be in results (above-floor dense)');
522
+ // gamma has both channels (higher RRF sum) so it should rank >= alpha (dense-only)
523
+ assert.ok(gammaIdx <= alphaIdx, `dual-channel gamma (idx ${gammaIdx}) must rank at or above dense-only alpha (idx ${alphaIdx})`);
524
+ });
525
+ // BC6: dense-only candidate below floor is dropped
526
+ test('BC6: dense-only candidate with score below denseScoreFloor is dropped', () => {
527
+ // delta is at dim 45, query at dim 40 → orthogonal → distance=1 → score=0 < 0.2
528
+ const queryEmbed = makeEmbedding(40);
529
+ const results = hybridStore.searchHybrid(makeHybridOpts({
530
+ query: 'hybridKeyword',
531
+ embedding: queryEmbed,
532
+ topK: 10,
533
+ denseScoreFloor: 0.5, // Higher floor to guarantee delta drops
534
+ }));
535
+ const deltaResult = results.find((r) => r.chunk.id === 'h-delta');
536
+ assert.ok(!deltaResult, 'h-delta (orthogonal, dense-only) must be dropped below floor');
537
+ });
538
+ // BC7: source dedup — at most one result per source (best chunk wins)
539
+ test('BC7: source dedup returns at most one result per source', () => {
540
+ const queryEmbed = makeEmbedding(40);
541
+ const results = hybridStore.searchHybrid(makeHybridOpts({
542
+ query: 'hybridKeyword',
543
+ embedding: queryEmbed,
544
+ topK: 10,
545
+ }));
546
+ const sources = results.map((r) => r.chunk.source);
547
+ const uniqueSources = new Set(sources);
548
+ assert.equal(sources.length, uniqueSources.size, 'Each source must appear at most once');
549
+ // source-alpha.md has two chunks (alpha + epsilon) — only one should surface
550
+ const alphaCount = sources.filter((s) => s === 'source-alpha.md').length;
551
+ assert.ok(alphaCount <= 1, `source-alpha.md must appear at most once, got ${alphaCount}`);
552
+ });
553
+ // BC8: deterministic order for tied candidates
554
+ test('BC8: repeated searchHybrid calls produce identical ordering', () => {
555
+ const queryEmbed = makeEmbedding(40);
556
+ const opts = makeHybridOpts({ query: 'hybridKeyword', embedding: queryEmbed, topK: 10 });
557
+ const first = hybridStore.searchHybrid(opts);
558
+ const second = hybridStore.searchHybrid(opts);
559
+ assert.equal(first.length, second.length, 'Result count must be stable');
560
+ for (let i = 0; i < first.length; i++) {
561
+ assert.equal(first[i].chunk.id, second[i].chunk.id, `Position ${i} must be deterministic`);
562
+ }
563
+ });
564
+ // BC9: sourceType filter applies to both channels
565
+ test('BC9: sourceType filter restricts results to matching source type', () => {
566
+ const queryEmbed = makeEmbedding(40);
567
+ const results = hybridStore.searchHybrid(makeHybridOpts({
568
+ query: 'hybridKeyword',
569
+ embedding: queryEmbed,
570
+ sourceType: 'concept',
571
+ topK: 10,
572
+ }));
573
+ assert.ok(results.length > 0, 'Filtered results must include concept sources');
574
+ for (const r of results) {
575
+ assert.equal(r.chunk.sourceType, 'concept', `All results must be concept type, got ${r.chunk.sourceType}`);
576
+ }
577
+ });
578
+ // BC9b: sinceMtimeAt filter restricts to recent sources
579
+ test('BC9b: sinceMtimeAt filter restricts results to sources newer than threshold', () => {
580
+ const results = hybridStore.searchHybrid(makeHybridOpts({
581
+ query: 'hybridKeyword',
582
+ sinceMtimeAt: 3000,
583
+ topK: 10,
584
+ }));
585
+ for (const r of results) {
586
+ assert.ok(r.chunk.fileMtimeAt >= 3000, `All results must have mtime >= 3000, got ${r.chunk.fileMtimeAt}`);
587
+ }
588
+ });
589
+ // BC11: debug fields present when includeDebug=true
590
+ test('BC11: debug fields are present on rows when includeDebug is true', () => {
591
+ const queryEmbed = makeEmbedding(40);
592
+ const results = hybridStore.searchHybrid(makeHybridOpts({
593
+ query: 'hybridKeyword',
594
+ embedding: queryEmbed,
595
+ includeDebug: true,
596
+ topK: 10,
597
+ }));
598
+ assert.ok(results.length > 0, 'Expected results with debug');
599
+ for (const r of results) {
600
+ assert.ok(r.debug, 'debug must be present when includeDebug=true');
601
+ assert.ok(Number.isFinite(r.debug.recencyRank), 'recencyRank must be finite');
602
+ assert.ok(Number.isFinite(r.debug.recencyScore), 'recencyScore must be finite');
603
+ assert.ok(Number.isFinite(r.debug.totalScore), 'totalScore must be finite');
604
+ }
605
+ });
606
+ // BC11b: debug fields absent when includeDebug is false/undefined
607
+ test('BC11b: debug fields are absent when includeDebug is not set', () => {
608
+ const results = hybridStore.searchHybrid(makeHybridOpts({ query: 'hybridKeyword' }));
609
+ for (const r of results) {
610
+ assert.equal(r.debug, undefined, 'debug must be absent when includeDebug is not set');
611
+ }
612
+ });
613
+ // Empty channel: both dense and BM25 empty → []
614
+ test('empty channels: both channels return empty — searchHybrid returns []', () => {
615
+ // Query with no matching BM25 terms and embedding pointing to unmapped dimension
616
+ const results = hybridStore.searchHybrid(makeHybridOpts({
617
+ query: 'the and or', // stopwords → empty FTS
618
+ embedding: undefined,
619
+ }));
620
+ assert.deepEqual(results, []);
621
+ });
622
+ // Score range: normalized scores must be in [0, 1]
623
+ test('normalized scores are bounded in [0, 1]', () => {
624
+ const results = hybridStore.searchHybrid(makeHybridOpts({
625
+ query: 'hybridKeyword',
626
+ embedding: makeEmbedding(40),
627
+ topK: 10,
628
+ }));
629
+ for (const r of results) {
630
+ assert.ok(r.score >= 0 && r.score <= 1, `Score out of range [0,1]: ${r.score}`);
631
+ }
632
+ });
633
+ });
390
634
  test('recreates schema when existing vector dimension differs from config', () => {
391
635
  const dimensionDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-dimension-migration-'));
392
636
  const dimensionDbPath = path.join(dimensionDir, 'index.db');
@@ -422,8 +666,15 @@ describe('MemoryStore', () => {
422
666
  content: 'large dimension vector can be indexed',
423
667
  });
424
668
  largeStore.upsert([chunk], [makeEmbedding(18, 768)]);
425
- const results = largeStore.searchBm25('large dimension vector', 5);
426
- assert.ok(results.some((result) => result.id === 'new-dimension-row'));
669
+ const results = largeStore.searchHybrid({
670
+ query: 'large dimension vector',
671
+ topK: 5,
672
+ fetchLimit: 20,
673
+ denseScoreFloor: 0,
674
+ recencyWeight: RECENCY_WEIGHT,
675
+ rrfK: RRF_K,
676
+ });
677
+ assert.ok(results.some((r) => r.chunk.id === 'new-dimension-row'));
427
678
  }
428
679
  finally {
429
680
  largeStore.close();
@@ -0,0 +1,19 @@
1
+ import { type ProjectSettings } from '../../core/config/index.js';
2
+ import { MemoryIndexer } from './indexer.js';
3
+ import { MemoryStore } from './store.js';
4
+ import type { MemoryConfig, MemoryLifecycleStatus } from './types.js';
5
+ export interface MemorySubsystemOverrides {
6
+ indexer?: MemoryIndexer;
7
+ store?: MemoryStore;
8
+ config?: MemoryConfig;
9
+ settings?: ProjectSettings;
10
+ }
11
+ export interface MemorySubsystem {
12
+ readonly config: MemoryConfig;
13
+ readonly status: MemoryLifecycleStatus;
14
+ readonly indexer: MemoryIndexer | null;
15
+ readonly store: MemoryStore | null;
16
+ startWarmup(): Promise<void>;
17
+ close(): void;
18
+ }
19
+ export declare function createMemorySubsystem(workspaceRoot: string, overrides?: MemorySubsystemOverrides): MemorySubsystem | null;
@@ -0,0 +1,67 @@
1
+ import * as path from 'path';
2
+ import { resolveMemoryConfig } from '../../core/config/index.js';
3
+ import { Embedder } from './embedder.js';
4
+ import { MemoryIndexer } from './indexer.js';
5
+ import { MemoryStore } from './store.js';
6
+ class DefaultMemorySubsystem {
7
+ config;
8
+ indexer;
9
+ store;
10
+ lifecycleStatus;
11
+ constructor(config, indexer, store, initialState, initialError) {
12
+ this.config = config;
13
+ this.indexer = indexer;
14
+ this.store = store;
15
+ this.lifecycleStatus = makeStatus(initialState, initialError);
16
+ }
17
+ get status() {
18
+ return this.lifecycleStatus;
19
+ }
20
+ async startWarmup() {
21
+ if (!this.indexer) {
22
+ this.lifecycleStatus = makeStatus('failed', this.lifecycleStatus.error ?? 'Memory indexer is unavailable');
23
+ return;
24
+ }
25
+ this.lifecycleStatus = makeStatus('warming');
26
+ try {
27
+ await this.indexer.startupIndex();
28
+ this.lifecycleStatus = makeStatus('ready');
29
+ }
30
+ catch (err) {
31
+ this.lifecycleStatus = makeStatus(this.store ? 'degraded' : 'failed', formatError(err));
32
+ }
33
+ }
34
+ close() {
35
+ this.store?.close();
36
+ }
37
+ }
38
+ function makeStatus(state, error) {
39
+ return {
40
+ state,
41
+ ready: state === 'ready',
42
+ warming: state === 'initializing' || state === 'warming',
43
+ degraded: state === 'degraded' || state === 'failed',
44
+ error,
45
+ };
46
+ }
47
+ function formatError(err) {
48
+ return err instanceof Error ? err.message : String(err);
49
+ }
50
+ export function createMemorySubsystem(workspaceRoot, overrides = {}) {
51
+ const config = overrides.config ?? resolveMemoryConfig(overrides.settings ?? {}, workspaceRoot);
52
+ if (config.enabled !== true)
53
+ return null;
54
+ let store = overrides.store ?? null;
55
+ let indexer = overrides.indexer ?? null;
56
+ try {
57
+ store = store ?? new MemoryStore(path.join(config.wikiDir, 'index.db'), config);
58
+ if (!indexer) {
59
+ const embedder = new Embedder(config.embeddingModel);
60
+ indexer = new MemoryIndexer(store, embedder, config);
61
+ }
62
+ return new DefaultMemorySubsystem(config, indexer, store, 'initializing');
63
+ }
64
+ catch (err) {
65
+ return new DefaultMemorySubsystem(config, indexer, store, 'failed', formatError(err));
66
+ }
67
+ }
@@ -23,6 +23,34 @@ export interface SearchResult {
23
23
  retriever: 'dense' | 'bm25' | 'both';
24
24
  contentSource: 'file' | 'fallback';
25
25
  }
26
+ export interface HybridSearchOptions {
27
+ query: string;
28
+ embedding?: Float32Array;
29
+ topK: number;
30
+ fetchLimit: number;
31
+ denseScoreFloor: number;
32
+ recencyWeight: number;
33
+ rrfK: number;
34
+ sourceType?: SourceType;
35
+ sinceMtimeAt?: number;
36
+ includeDebug?: boolean;
37
+ }
38
+ export interface HybridRankDebug {
39
+ denseRank?: number;
40
+ denseScore?: number;
41
+ bm25Rank?: number;
42
+ bm25Score?: number;
43
+ recencyRank: number;
44
+ recencyScore: number;
45
+ totalScore: number;
46
+ droppedReason?: string;
47
+ }
48
+ export interface HybridSearchRow {
49
+ chunk: MemoryChunk;
50
+ score: number;
51
+ retriever: 'dense' | 'bm25' | 'both';
52
+ debug?: HybridRankDebug;
53
+ }
26
54
  export interface MemoryConfig {
27
55
  enabled: boolean;
28
56
  wikiDir: string;
@@ -38,6 +66,26 @@ export interface IndexStats {
38
66
  deleted: number;
39
67
  skipped: number;
40
68
  }
69
+ export type MemoryLifecycleState = 'disabled' | 'initializing' | 'warming' | 'ready' | 'degraded' | 'failed';
70
+ export interface MemoryLifecycleStatus {
71
+ state: MemoryLifecycleState;
72
+ ready: boolean;
73
+ warming: boolean;
74
+ degraded: boolean;
75
+ error?: string;
76
+ }
77
+ export interface IndexDirectoryOptions {
78
+ relativeBase?: string;
79
+ excludeFiles?: string[];
80
+ fileConcurrency?: number;
81
+ }
82
+ export interface PreparedIndexMutation {
83
+ source: string;
84
+ chunksToUpsert: MemoryChunk[];
85
+ embeddings: Float32Array[];
86
+ idsToDelete: string[];
87
+ stats: IndexStats;
88
+ }
41
89
  export interface RecentSource {
42
90
  source: string;
43
91
  fileMtimeAt: number;