@hanhnd/agent-kit 1.0.38 → 1.0.40

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,13 +1,12 @@
1
1
  import * as assert from 'node:assert/strict';
2
- import fsDefault from 'node:fs';
3
2
  import * as fs from 'node:fs';
4
- import { syncBuiltinESMExports } from 'node:module';
5
3
  import * as os from 'node:os';
6
4
  import * as path from 'node:path';
7
5
  import { after, before, describe, test } from 'node:test';
8
6
  import { deriveSourceType, MemoryIndexer } from './indexer.js';
9
7
  import { MemoryStore } from './store.js';
10
- import { EmbeddingModelName } from './types.js';
8
+ import { EmbeddingModelName, } from './types.js';
9
+ import { DENSE_SCORE_FLOOR, FETCH_MULTIPLIER, RECENCY_WEIGHT, RRF_K } from './constants.js';
11
10
  class StubEmbedder {
12
11
  async embed(texts) {
13
12
  return texts.map(() => new Float32Array(384).fill(0.05));
@@ -16,6 +15,36 @@ class StubEmbedder {
16
15
  return Promise.resolve();
17
16
  }
18
17
  }
18
+ class FailEmbedder {
19
+ async embed() {
20
+ throw new Error('embed failed');
21
+ }
22
+ initialize() {
23
+ return Promise.resolve();
24
+ }
25
+ }
26
+ class TrackingEmbedder {
27
+ failForText;
28
+ active = 0;
29
+ maxActive = 0;
30
+ constructor(failForText = undefined) {
31
+ this.failForText = failForText;
32
+ }
33
+ async embed(texts) {
34
+ const failForText = this.failForText;
35
+ if (failForText && texts.some((text) => text.includes(failForText))) {
36
+ throw new Error('targeted embed failure');
37
+ }
38
+ this.active++;
39
+ this.maxActive = Math.max(this.maxActive, this.active);
40
+ await new Promise((resolve) => setTimeout(resolve, 10));
41
+ this.active--;
42
+ return texts.map(() => new Float32Array(384).fill(0.05));
43
+ }
44
+ initialize() {
45
+ return Promise.resolve();
46
+ }
47
+ }
19
48
  function makeConfig(wikiDir) {
20
49
  return {
21
50
  enabled: true,
@@ -27,6 +56,23 @@ function makeConfig(wikiDir) {
27
56
  vectorDimension: 384,
28
57
  };
29
58
  }
59
+ function makeChunkFull(overrides = {}) {
60
+ return {
61
+ id: 'test-id',
62
+ source: 'compiled/entities/test.md',
63
+ sourceType: 'entity',
64
+ heading: 'Test',
65
+ headingLevel: 1,
66
+ content: 'test content',
67
+ lineStart: 1,
68
+ lineEnd: 2,
69
+ fileMtimeAt: 1000,
70
+ ...overrides,
71
+ };
72
+ }
73
+ function makeHybridRow(chunk, retriever = 'both') {
74
+ return { chunk, score: 0.8, retriever };
75
+ }
30
76
  describe('MemoryIndexer', () => {
31
77
  let tmpDir;
32
78
  let store;
@@ -52,9 +98,7 @@ describe('MemoryIndexer', () => {
52
98
  test('indexFile on unchanged file — indexed === 0, skipped > 0', async () => {
53
99
  const filePath = path.join(tmpDir, 'stable-file.md');
54
100
  fs.writeFileSync(filePath, '# Stable\nThis content does not change between runs.', 'utf8');
55
- // First run indexes it
56
101
  await indexer.indexFile(filePath);
57
- // Second run — same content
58
102
  const stats = await indexer.indexFile(filePath);
59
103
  assert.equal(stats.indexed, 0, `Expected indexed === 0, got ${stats.indexed}`);
60
104
  assert.ok(stats.skipped > 0, `Expected skipped > 0, got ${stats.skipped}`);
@@ -74,29 +118,121 @@ describe('MemoryIndexer', () => {
74
118
  const staleSource = path.relative(tmpDir, staleFile);
75
119
  const before = store.hashesBySource(staleSource);
76
120
  assert.ok(before.size > 0, 'Stale file must be indexed first');
77
- // Delete the file and re-index the directory
78
121
  fs.unlinkSync(staleFile);
79
122
  await indexer.indexDirectory(tmpDir);
80
123
  const afterDeletion = store.hashesBySource(staleSource);
81
124
  assert.equal(afterDeletion.size, 0, 'Stale source must be removed from store after directory scan');
82
125
  });
83
- test('search returns result with correct source for indexed content', async () => {
84
- const filePath = path.join(config.wikiDir, 'compiled', 'searchable.md');
85
- const fileContent = '# Searchable\nspecialUniqueTermForSearch is in this document.';
86
- fs.mkdirSync(path.dirname(filePath), { recursive: true });
87
- fs.writeFileSync(filePath, fileContent, 'utf8');
88
- await indexer.indexDirectory(path.join(config.wikiDir, 'compiled'), {
89
- relativeBase: config.wikiDir,
90
- });
91
- const results = await indexer.search('specialUniqueTermForSearch', 5);
92
- assert.ok(results.length > 0, 'Expected at least one search result');
93
- const expectedSource = path.relative(config.wikiDir, filePath);
94
- const match = results.find((r) => r.chunk.source === expectedSource);
95
- assert.ok(match, `Expected result with source=${expectedSource}, got: ${results.map((r) => r.chunk.source).join(', ')}`);
96
- assert.equal(match.chunk.content, fileContent);
97
- assert.equal(match.contentSource, 'file');
126
+ // BC13: delegation — vecAvailable=true embeds query and calls searchHybrid with correct options
127
+ test('BC13: search delegates to searchHybrid with embedding and constants when vecAvailable', async () => {
128
+ const chunk = makeChunkFull();
129
+ let capturedOptions;
130
+ const fakeStore = {
131
+ vecAvailable: true,
132
+ searchHybrid: (opts) => {
133
+ capturedOptions = opts;
134
+ return [makeHybridRow(chunk, 'both')];
135
+ },
136
+ };
137
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'bc13-'));
138
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
139
+ try {
140
+ const compiledEntities = path.join(testCfg.wikiDir, 'compiled', 'entities');
141
+ fs.mkdirSync(compiledEntities, { recursive: true });
142
+ fs.writeFileSync(path.join(compiledEntities, 'test.md'), 'test content', 'utf8');
143
+ const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
144
+ const results = await testIndexer.search('hello', 3);
145
+ assert.ok(capturedOptions, 'searchHybrid must be called');
146
+ assert.equal(capturedOptions.query, 'hello');
147
+ assert.equal(capturedOptions.topK, 3);
148
+ assert.equal(capturedOptions.fetchLimit, 3 * FETCH_MULTIPLIER);
149
+ assert.equal(capturedOptions.denseScoreFloor, DENSE_SCORE_FLOOR);
150
+ assert.equal(capturedOptions.recencyWeight, RECENCY_WEIGHT);
151
+ assert.equal(capturedOptions.rrfK, RRF_K);
152
+ assert.ok(capturedOptions.embedding instanceof Float32Array, 'embedding must be provided when vecAvailable');
153
+ assert.equal(results.length, 1);
154
+ }
155
+ finally {
156
+ fs.rmSync(testDir, { recursive: true, force: true });
157
+ }
158
+ });
159
+ // BC14: embedding failure → searchHybrid called without embedding
160
+ test('BC14: embedding failure causes search to call searchHybrid without embedding', async () => {
161
+ const chunk = makeChunkFull();
162
+ let capturedOptions;
163
+ const fakeStore = {
164
+ vecAvailable: true,
165
+ searchHybrid: (opts) => {
166
+ capturedOptions = opts;
167
+ return [makeHybridRow(chunk, 'bm25')];
168
+ },
169
+ };
170
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'bc14-'));
171
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
172
+ try {
173
+ const compiledEntities = path.join(testCfg.wikiDir, 'compiled', 'entities');
174
+ fs.mkdirSync(compiledEntities, { recursive: true });
175
+ fs.writeFileSync(path.join(compiledEntities, 'test.md'), 'test content', 'utf8');
176
+ const testIndexer = new MemoryIndexer(fakeStore, new FailEmbedder(), testCfg);
177
+ const results = await testIndexer.search('hello', 5);
178
+ assert.ok(capturedOptions, 'searchHybrid must be called even after embedding failure');
179
+ assert.equal(capturedOptions.embedding, undefined, 'embedding must be absent after failure');
180
+ assert.equal(results.length, 1);
181
+ }
182
+ finally {
183
+ fs.rmSync(testDir, { recursive: true, force: true });
184
+ }
185
+ });
186
+ // BC15: readable source file → full live content and contentSource: 'file'
187
+ test('BC15: readable source file is read asynchronously and contentSource is file', async () => {
188
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'bc15-'));
189
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
190
+ const sourceRelative = 'compiled/entities/test.md';
191
+ const sourceAbsolute = path.join(testCfg.wikiDir, sourceRelative);
192
+ const fileContent = '# Full\nLive file content from disk.';
193
+ try {
194
+ fs.mkdirSync(path.dirname(sourceAbsolute), { recursive: true });
195
+ fs.writeFileSync(sourceAbsolute, fileContent, 'utf8');
196
+ const chunk = makeChunkFull({ source: sourceRelative, content: 'stored chunk only' });
197
+ const fakeStore = {
198
+ vecAvailable: false,
199
+ searchHybrid: () => [makeHybridRow(chunk, 'bm25')],
200
+ };
201
+ const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
202
+ const results = await testIndexer.search('test', 5);
203
+ assert.equal(results.length, 1);
204
+ assert.equal(results[0].contentSource, 'file');
205
+ assert.equal(results[0].chunk.content, fileContent);
206
+ }
207
+ finally {
208
+ fs.rmSync(testDir, { recursive: true, force: true });
209
+ }
210
+ });
211
+ // BC16: missing source file → stored chunk content and contentSource: 'fallback'
212
+ test('BC16: missing source file leaves stored chunk content and sets contentSource fallback', async () => {
213
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'bc16-'));
214
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
215
+ try {
216
+ const chunk = makeChunkFull({
217
+ source: 'compiled/entities/missing.md',
218
+ content: 'stored chunk content',
219
+ });
220
+ const fakeStore = {
221
+ vecAvailable: false,
222
+ searchHybrid: () => [makeHybridRow(chunk, 'bm25')],
223
+ };
224
+ const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
225
+ const results = await testIndexer.search('test', 5);
226
+ assert.equal(results.length, 1);
227
+ assert.equal(results[0].contentSource, 'fallback');
228
+ assert.equal(results[0].chunk.content, 'stored chunk content');
229
+ }
230
+ finally {
231
+ fs.rmSync(testDir, { recursive: true, force: true });
232
+ }
98
233
  });
99
- test('indexDirectory walks nested markdown files and excludes configured basenames', async () => {
234
+ // BC17: indexDirectory deterministic traversal and stale deletion
235
+ test('BC17: indexDirectory walks nested markdown files and excludes configured basenames', async () => {
100
236
  const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'recursive-index-'));
101
237
  const testCfg = makeConfig(path.join(testDir, 'wiki'));
102
238
  const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
@@ -123,12 +259,114 @@ describe('MemoryIndexer', () => {
123
259
  fs.rmSync(testDir, { recursive: true, force: true });
124
260
  }
125
261
  });
126
- test('indexDirectory returns zero stats when root directory is missing', async () => {
262
+ // BC18: unreadable subtrees — log and continue
263
+ test('BC18: indexDirectory logs and continues when file or subdirectory is unreadable', async () => {
127
264
  const stats = await indexer.indexDirectory(path.join(config.wikiDir, 'missing'), {
128
265
  relativeBase: config.wikiDir,
129
266
  });
130
267
  assert.deepEqual(stats, { indexed: 0, deleted: 0, skipped: 0 });
131
268
  });
269
+ // BC19: save appends content asynchronously
270
+ test('BC19: save appends manual content to daily wiki raw file asynchronously', async () => {
271
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'save-daily-'));
272
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
273
+ const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
274
+ const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
275
+ const datePart = new Date().toISOString().slice(0, 10);
276
+ const savePath = path.join(testCfg.wikiDir, 'raw', `conv_save_${datePart}.md`);
277
+ try {
278
+ const firstStats = await testIndexer.save('first manual save');
279
+ const secondStats = await testIndexer.save('second manual save');
280
+ assert.deepEqual(firstStats, { indexed: 0, deleted: 0, skipped: 0 });
281
+ assert.deepEqual(secondStats, { indexed: 0, deleted: 0, skipped: 0 });
282
+ assert.equal(fs.existsSync(savePath), true);
283
+ const content = fs.readFileSync(savePath, 'utf8');
284
+ assert.match(content, /first manual save/);
285
+ assert.match(content, /second manual save/);
286
+ assert.equal(fs.readdirSync(path.dirname(savePath)).filter((name) => /^conv_save_.*\.md$/.test(name)).length, 1);
287
+ }
288
+ finally {
289
+ testStore.close();
290
+ fs.rmSync(testDir, { recursive: true, force: true });
291
+ }
292
+ });
293
+ test('search returns result with correct source for indexed content', async () => {
294
+ const filePath = path.join(config.wikiDir, 'compiled', 'searchable.md');
295
+ const fileContent = '# Searchable\nspecialUniqueTermForSearch is in this document.';
296
+ fs.mkdirSync(path.dirname(filePath), { recursive: true });
297
+ fs.writeFileSync(filePath, fileContent, 'utf8');
298
+ await indexer.indexDirectory(path.join(config.wikiDir, 'compiled'), {
299
+ relativeBase: config.wikiDir,
300
+ });
301
+ const results = await indexer.search('specialUniqueTermForSearch', 5);
302
+ assert.ok(results.length > 0, 'Expected at least one search result');
303
+ const expectedSource = path.relative(config.wikiDir, filePath);
304
+ const match = results.find((r) => r.chunk.source === expectedSource);
305
+ assert.ok(match, `Expected result with source=${expectedSource}, got: ${results.map((r) => r.chunk.source).join(', ')}`);
306
+ assert.equal(match.chunk.content, fileContent);
307
+ assert.equal(match.contentSource, 'file');
308
+ });
309
+ test('search on empty store returns []', async () => {
310
+ const emptyDir = fs.mkdtempSync(path.join(os.tmpdir(), 'empty-store-'));
311
+ const emptyCfg = makeConfig(emptyDir);
312
+ const emptyStore = new MemoryStore(path.join(emptyCfg.wikiDir, 'index.db'), emptyCfg);
313
+ const emptyIndexer = new MemoryIndexer(emptyStore, new StubEmbedder(), emptyCfg);
314
+ try {
315
+ const results = await emptyIndexer.search('anything', 5);
316
+ assert.deepEqual(results, []);
317
+ }
318
+ finally {
319
+ emptyStore.close();
320
+ fs.rmSync(emptyDir, { recursive: true, force: true });
321
+ }
322
+ });
323
+ test('search falls back to stored chunk content when source file is missing', async () => {
324
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'search-fallback-'));
325
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
326
+ const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
327
+ const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
328
+ const filePath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'missing.md');
329
+ const fileContent = '# Missing\nfallbackUniqueTerm stored chunk text';
330
+ try {
331
+ fs.mkdirSync(path.dirname(filePath), { recursive: true });
332
+ fs.writeFileSync(filePath, fileContent, 'utf8');
333
+ await testIndexer.indexDirectory(path.join(testCfg.wikiDir, 'compiled'), {
334
+ relativeBase: testCfg.wikiDir,
335
+ });
336
+ fs.unlinkSync(filePath);
337
+ const results = await testIndexer.search('fallbackUniqueTerm', 5);
338
+ assert.equal(results.length, 1);
339
+ assert.equal(results[0].contentSource, 'fallback');
340
+ assert.equal(results[0].chunk.content, fileContent);
341
+ }
342
+ finally {
343
+ testStore.close();
344
+ fs.rmSync(testDir, { recursive: true, force: true });
345
+ }
346
+ });
347
+ test('search returns distinct sources — one result per source', async () => {
348
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'search-dedup-'));
349
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
350
+ const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
351
+ const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
352
+ const filePath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'dedup.md');
353
+ // Multiple paragraphs create multiple chunks
354
+ const fileContent = '# Dedup\n\ndedupUniqueTermXYZ first chunk content\n\n---\n\ndedupUniqueTermXYZ second chunk content';
355
+ try {
356
+ fs.mkdirSync(path.dirname(filePath), { recursive: true });
357
+ fs.writeFileSync(filePath, fileContent, 'utf8');
358
+ await testIndexer.indexDirectory(path.join(testCfg.wikiDir, 'compiled'), {
359
+ relativeBase: testCfg.wikiDir,
360
+ });
361
+ const results = await testIndexer.search('dedupUniqueTermXYZ', 5);
362
+ const dedupSources = results.filter((r) => r.chunk.source === 'compiled/entities/dedup.md');
363
+ assert.equal(dedupSources.length, 1, 'At most one result per source after dedup');
364
+ }
365
+ finally {
366
+ testStore.close();
367
+ fs.rmSync(testDir, { recursive: true, force: true });
368
+ }
369
+ });
132
370
  test('indexDirectory removes stale pre-migration daily-file sources', async () => {
133
371
  const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'recursive-stale-'));
134
372
  const testCfg = makeConfig(path.join(testDir, 'wiki'));
@@ -162,583 +400,113 @@ describe('MemoryIndexer', () => {
162
400
  fs.rmSync(testDir, { recursive: true, force: true });
163
401
  }
164
402
  });
165
- test('search deduplicates sources and reads each matched source once', async () => {
166
- const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'search-dedup-'));
167
- const testCfg = {
168
- ...makeConfig(path.join(testDir, 'wiki')),
169
- chunkSize: 40,
170
- overlapLines: 0,
171
- };
403
+ test('indexDirectory prepares changed files with bounded parallelism', async () => {
404
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'parallel-index-'));
405
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
172
406
  const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
173
- const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
174
- const filePath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'dedup.md');
175
- const fileContent = [
176
- '# Dedup',
177
- 'dedupUniqueTerm first chunk text',
178
- 'dedupUniqueTerm second chunk text',
179
- 'dedupUniqueTerm third chunk text',
180
- ].join('\n');
181
- const originalReadFileSync = fsDefault.readFileSync;
182
- let readCount = 0;
407
+ const trackingEmbedder = new TrackingEmbedder();
408
+ const testIndexer = new MemoryIndexer(testStore, trackingEmbedder, testCfg);
409
+ const compiledDir = path.join(testCfg.wikiDir, 'compiled');
183
410
  try {
184
- fs.mkdirSync(path.dirname(filePath), { recursive: true });
185
- fs.writeFileSync(filePath, fileContent, 'utf8');
186
- await testIndexer.indexDirectory(path.join(testCfg.wikiDir, 'compiled'), {
411
+ fs.mkdirSync(compiledDir, { recursive: true });
412
+ fs.writeFileSync(path.join(compiledDir, 'a.md'), '# A\nparallel alpha content', 'utf8');
413
+ fs.writeFileSync(path.join(compiledDir, 'b.md'), '# B\nparallel beta content', 'utf8');
414
+ fs.writeFileSync(path.join(compiledDir, 'c.md'), '# C\nparallel gamma content', 'utf8');
415
+ const stats = await testIndexer.indexDirectory(compiledDir, {
187
416
  relativeBase: testCfg.wikiDir,
417
+ fileConcurrency: 2,
188
418
  });
189
- const expectedPath = path.join(testCfg.wikiDir, 'compiled/entities/dedup.md');
190
- fsDefault.readFileSync = ((targetPath, options) => {
191
- if (targetPath === expectedPath)
192
- readCount += 1;
193
- return originalReadFileSync(targetPath, options);
194
- });
195
- syncBuiltinESMExports();
196
- const results = await testIndexer.search('dedupUniqueTerm', 5);
197
- assert.equal(results.filter((r) => r.chunk.source === 'compiled/entities/dedup.md').length, 1);
198
- assert.equal(readCount, 1);
199
- assert.equal(results[0].chunk.content, fileContent);
200
- assert.equal(results[0].contentSource, 'file');
419
+ assert.ok(stats.indexed >= 3, `Expected at least 3 indexed chunks, got ${stats.indexed}`);
420
+ assert.equal(trackingEmbedder.maxActive, 2);
201
421
  }
202
422
  finally {
203
- fsDefault.readFileSync = originalReadFileSync;
204
- syncBuiltinESMExports();
205
423
  testStore.close();
206
424
  fs.rmSync(testDir, { recursive: true, force: true });
207
425
  }
208
426
  });
209
- test('search continues past duplicate sources until topK unique sources are returned', async () => {
210
- const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'search-unique-topk-'));
427
+ test('indexDirectory serializes store mutations after parallel preparation', async () => {
428
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'serialized-writes-'));
211
429
  const testCfg = makeConfig(path.join(testDir, 'wiki'));
212
- const firstPath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'first.md');
213
- const secondPath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'second.md');
214
- const firstContent = '# First\nfirst source full content';
215
- const secondContent = '# Second\nsecond source full content';
216
- const chunks = [
217
- {
218
- id: 'first-1',
219
- source: 'compiled/entities/first.md',
220
- heading: 'First',
221
- headingLevel: 1,
222
- content: 'first matching chunk one',
223
- lineStart: 1,
224
- lineEnd: 2,
225
- },
226
- {
227
- id: 'first-2',
228
- source: 'compiled/entities/first.md',
229
- heading: 'First',
230
- headingLevel: 1,
231
- content: 'first matching chunk two',
232
- lineStart: 3,
233
- lineEnd: 4,
234
- },
235
- {
236
- id: 'second-1',
237
- source: 'compiled/entities/second.md',
238
- heading: 'Second',
239
- headingLevel: 1,
240
- content: 'second matching chunk',
241
- lineStart: 1,
242
- lineEnd: 2,
243
- },
244
- ];
245
- const fakeStore = {
246
- vecAvailable: false,
247
- searchBm25: () => [
248
- { id: 'first-1', score: 1 },
249
- { id: 'first-2', score: 0.9 },
250
- { id: 'second-1', score: 0.8 },
251
- ],
252
- getChunksByIds: (ids) => chunks.filter((chunk) => ids.includes(chunk.id)),
253
- };
254
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
255
- try {
256
- fs.mkdirSync(path.dirname(firstPath), { recursive: true });
257
- fs.writeFileSync(firstPath, firstContent, 'utf8');
258
- fs.writeFileSync(secondPath, secondContent, 'utf8');
259
- const results = await testIndexer.search('duplicate source query', 2);
260
- assert.equal(results.length, 2);
261
- assert.deepEqual(results.map((result) => result.chunk.source), ['compiled/entities/first.md', 'compiled/entities/second.md']);
262
- assert.equal(results[0].chunk.content, firstContent);
263
- assert.equal(results[1].chunk.content, secondContent);
264
- }
265
- finally {
266
- fs.rmSync(testDir, { recursive: true, force: true });
267
- }
268
- });
269
- // Task 3.1: Rewrite — asserts union behavior (previously asserted the removed intersection filter).
270
- // BC1 + BC2: dense-only high-score candidate surfaces; dual-channel hit ranks first.
271
- test('search includes high-score dense-only results alongside BM25 matches', async () => {
272
- const testCfg = makeConfig('/tmp/search-union-behavior');
273
- const chunks = [
274
- {
275
- id: 'preference-1',
276
- source: 'compiled/preferences.md',
277
- heading: '',
278
- headingLevel: 0,
279
- content: 'I like fish',
280
- lineStart: 1,
281
- lineEnd: 1,
282
- },
283
- {
284
- id: 'dense-only-1',
285
- source: 'compiled/entities/worktree.md',
286
- heading: 'Worktree',
287
- headingLevel: 1,
288
- content: 'Unrelated worktree lifecycle content',
289
- lineStart: 1,
290
- lineEnd: 2,
291
- },
292
- ];
293
- const fakeStore = {
294
- vecAvailable: true,
295
- searchDense: () => [
296
- { id: 'dense-only-1', score: 0.99 }, // score >= DENSE_SCORE_FLOOR (0.2) → survives floor
297
- { id: 'preference-1', score: 0.98 },
298
- ],
299
- searchBm25: () => [{ id: 'preference-1', score: 1 }],
300
- getChunksByIds: (ids) => chunks.filter((chunk) => ids.includes(chunk.id)),
301
- };
302
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
303
- const results = await testIndexer.search('personal likes and preferences of the user', 5);
304
- const resultIds = results.map((r) => r.chunk.id);
305
- assert.ok(resultIds.includes('preference-1'), `Expected preference-1 in results, got: ${resultIds.join(', ')}`);
306
- assert.ok(resultIds.includes('dense-only-1'), `Expected dense-only-1 in results, got: ${resultIds.join(', ')}`);
307
- assert.equal(results[0].chunk.id, 'preference-1', 'dual-channel hit must rank first');
308
- assert.equal(results[0].retriever, 'both');
309
- const denseOnlyResult = results.find((r) => r.chunk.id === 'dense-only-1');
310
- assert.ok(denseOnlyResult, 'dense-only-1 must be present in results');
311
- assert.equal(denseOnlyResult.retriever, 'dense');
312
- });
313
- // Task 3.2 — Regression tests BC1–BC9
314
- // BC3: durable dual-channel memory ranks above a newer single-channel chunk despite lower recency score
315
- test('BC3: durable dual-channel chunk ranks above newer single-channel chunk', async () => {
316
- const testCfg = makeConfig('/tmp/bc3-durable-vs-recent');
317
- const chunks = [
318
- {
319
- id: 'durable-1',
320
- source: 'compiled/preferences.md',
321
- heading: '',
322
- headingLevel: 0,
323
- content: 'I prefer dark mode',
324
- lineStart: 1,
325
- lineEnd: 1,
326
- fileMtimeAt: 100, // old mtime
327
- },
328
- {
329
- id: 'recent-1',
330
- source: 'compiled/entities/meeting.md',
331
- heading: 'Meeting',
332
- headingLevel: 1,
333
- content: 'Meeting notes from today',
334
- lineStart: 1,
335
- lineEnd: 2,
336
- fileMtimeAt: 9_999_999_999_999, // very new mtime
337
- },
338
- ];
339
- const fakeStore = {
340
- vecAvailable: true,
341
- searchDense: () => [{ id: 'durable-1', score: 0.9 }],
342
- searchBm25: () => [
343
- { id: 'durable-1', score: 1 }, // rank 0 in bm25
344
- { id: 'recent-1', score: 0.8 }, // rank 1 in bm25 — newer mtime but single bm25 channel
345
- ],
346
- getChunksByIds: (ids) => chunks.filter((c) => ids.includes(c.id)),
347
- };
348
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
349
- const results = await testIndexer.search('preferences', 5);
350
- assert.ok(results.length >= 2, `Expected at least 2 results, got ${results.length}`);
351
- assert.equal(results[0].chunk.id, 'durable-1', 'dual-channel durable memory must rank above newer single-channel chunk');
352
- assert.equal(results[0].retriever, 'both');
353
- const recentResult = results.find((r) => r.chunk.id === 'recent-1');
354
- assert.ok(recentResult, 'recent-1 must still surface');
355
- });
356
- // BC4: search returns dense results when BM25 is empty, with recency channel applied
357
- test('BC4: search returns dense results and applies recency when BM25 has no matches', async () => {
358
- const testCfg = makeConfig('/tmp/bc4-dense-only-recency');
359
- const chunks = [
360
- {
361
- id: 'semantic-old',
362
- source: 'compiled/concepts/topic.md',
363
- heading: 'Topic',
364
- headingLevel: 1,
365
- content: 'Concept about topic',
366
- lineStart: 1,
367
- lineEnd: 2,
368
- fileMtimeAt: 1, // very old
369
- },
370
- {
371
- id: 'semantic-new',
372
- source: 'compiled/entities/recent-entity.md',
373
- heading: 'Recent',
374
- headingLevel: 1,
375
- content: 'Recent entity content',
376
- lineStart: 1,
377
- lineEnd: 2,
378
- fileMtimeAt: 9_999_999_999_999, // very new
379
- },
380
- ];
430
+ const compiledDir = path.join(testCfg.wikiDir, 'compiled');
431
+ const calls = [];
432
+ let activeMutations = 0;
433
+ let maxActiveMutations = 0;
381
434
  const fakeStore = {
382
435
  vecAvailable: true,
383
- searchDense: () => [
384
- { id: 'semantic-old', score: 0.9 }, // dense rank 0
385
- { id: 'semantic-new', score: 0.85 }, // dense rank 1
386
- ],
387
- searchBm25: () => [],
388
- getChunksByIds: (ids) => chunks.filter((c) => ids.includes(c.id)),
389
- };
390
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
391
- const results = await testIndexer.search('topic', 5);
392
- assert.ok(results.length >= 2, `Expected at least 2 dense results, got ${results.length}`);
393
- const ids = results.map((r) => r.chunk.id);
394
- assert.ok(ids.includes('semantic-old'), 'semantic-old must be present');
395
- assert.ok(ids.includes('semantic-new'), 'semantic-new must be present');
396
- for (const result of results) {
397
- assert.equal(result.retriever, 'dense', `Expected retriever 'dense', got '${result.retriever}'`);
398
- assert.ok(result.score > 0, 'Normalized score must be positive');
399
- }
400
- });
401
- // BC5: dense channel unavailable — bm25 results returned with finite normalized score, no throw
402
- test('BC5: bm25 results are returned with normalized score when dense channel is unavailable', async () => {
403
- const testCfg = makeConfig('/tmp/bc5-dense-unavailable');
404
- const chunks = [
405
- {
406
- id: 'keyword-1',
407
- source: 'compiled/concepts/keyword.md',
408
- heading: 'Keyword',
409
- headingLevel: 1,
410
- content: 'Keyword concept content',
411
- lineStart: 1,
412
- lineEnd: 2,
413
- },
414
- ];
415
- const fakeStore = {
416
- vecAvailable: false,
417
- searchBm25: () => [{ id: 'keyword-1', score: 1 }],
418
- getChunksByIds: (ids) => chunks.filter((c) => ids.includes(c.id)),
419
- };
420
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
421
- const results = await testIndexer.search('keyword', 5);
422
- assert.equal(results.length, 1, 'Must return bm25 result when dense is unavailable');
423
- assert.equal(results[0].chunk.id, 'keyword-1');
424
- assert.equal(results[0].retriever, 'bm25');
425
- assert.ok(Number.isFinite(results[0].score), 'Score must be finite');
426
- assert.ok(results[0].score >= 0 && results[0].score <= 1, `Score must be in [0,1], got ${results[0].score}`);
427
- });
428
- // BC6: both channels empty → search returns []
429
- test('BC6: search returns empty array when both dense and bm25 channels are empty', async () => {
430
- const testCfg = makeConfig('/tmp/bc6-both-empty');
431
- const fakeStore = {
432
- vecAvailable: false,
433
- searchBm25: () => [],
434
- getChunksByIds: () => [],
435
- };
436
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
437
- const results = await testIndexer.search('anything', 5);
438
- assert.deepEqual(results, []);
439
- });
440
- // BC7: per-source dedup yields up to topK distinct sources even when one source dominates the pool
441
- test('BC7: per-source dedup yields topK distinct sources when one source dominates the pool', async () => {
442
- const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'bc7-overfetch-'));
443
- const testCfg = makeConfig(path.join(testDir, 'wiki'));
444
- const allChunks = [
445
- {
446
- id: 'dom-1',
447
- source: 'compiled/entities/dominant.md',
448
- heading: '',
449
- headingLevel: 0,
450
- content: 'dom 1',
451
- lineStart: 1,
452
- lineEnd: 1,
453
- },
454
- {
455
- id: 'dom-2',
456
- source: 'compiled/entities/dominant.md',
457
- heading: '',
458
- headingLevel: 0,
459
- content: 'dom 2',
460
- lineStart: 2,
461
- lineEnd: 2,
462
- },
463
- {
464
- id: 'dom-3',
465
- source: 'compiled/entities/dominant.md',
466
- heading: '',
467
- headingLevel: 0,
468
- content: 'dom 3',
469
- lineStart: 3,
470
- lineEnd: 3,
436
+ hashesBySource: () => new Set(),
437
+ indexedSources: () => ['compiled/stale.md'],
438
+ deleteBySource: (source) => {
439
+ activeMutations++;
440
+ maxActiveMutations = Math.max(maxActiveMutations, activeMutations);
441
+ calls.push(`deleteBySource:${source}`);
442
+ activeMutations--;
471
443
  },
472
- {
473
- id: 'dom-4',
474
- source: 'compiled/entities/dominant.md',
475
- heading: '',
476
- headingLevel: 0,
477
- content: 'dom 4',
478
- lineStart: 4,
479
- lineEnd: 4,
444
+ upsert: (chunks) => {
445
+ activeMutations++;
446
+ maxActiveMutations = Math.max(maxActiveMutations, activeMutations);
447
+ calls.push(`upsert:${chunks[0]?.source}`);
448
+ activeMutations--;
480
449
  },
481
- {
482
- id: 'dom-5',
483
- source: 'compiled/entities/dominant.md',
484
- heading: '',
485
- headingLevel: 0,
486
- content: 'dom 5',
487
- lineStart: 5,
488
- lineEnd: 5,
450
+ deleteByIds: (ids) => {
451
+ activeMutations++;
452
+ maxActiveMutations = Math.max(maxActiveMutations, activeMutations);
453
+ calls.push(`deleteByIds:${ids.length}`);
454
+ activeMutations--;
489
455
  },
490
- {
491
- id: 'src-b-1',
492
- source: 'compiled/entities/source-b.md',
493
- heading: '',
494
- headingLevel: 0,
495
- content: 'source b',
496
- lineStart: 1,
497
- lineEnd: 1,
498
- },
499
- {
500
- id: 'src-c-1',
501
- source: 'compiled/entities/source-c.md',
502
- heading: '',
503
- headingLevel: 0,
504
- content: 'source c',
505
- lineStart: 1,
506
- lineEnd: 1,
507
- },
508
- ];
509
- const fakeStore = {
510
- vecAvailable: false,
511
- searchBm25: () => [
512
- { id: 'dom-1', score: 5 },
513
- { id: 'dom-2', score: 4.9 },
514
- { id: 'dom-3', score: 4.8 },
515
- { id: 'dom-4', score: 4.7 },
516
- { id: 'dom-5', score: 4.6 },
517
- { id: 'src-b-1', score: 3 },
518
- { id: 'src-c-1', score: 2 },
519
- ],
520
- getChunksByIds: (ids) => allChunks.filter((c) => ids.includes(c.id)),
521
456
  };
522
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
523
457
  try {
524
- const entitiesDir = path.join(testCfg.wikiDir, 'compiled', 'entities');
525
- fs.mkdirSync(entitiesDir, { recursive: true });
526
- fs.writeFileSync(path.join(entitiesDir, 'dominant.md'), 'dominant content', 'utf8');
527
- fs.writeFileSync(path.join(entitiesDir, 'source-b.md'), 'source b content', 'utf8');
528
- fs.writeFileSync(path.join(entitiesDir, 'source-c.md'), 'source c content', 'utf8');
529
- const results = await testIndexer.search('query', 3);
530
- assert.equal(results.length, 3, `Expected 3 results, got ${results.length}`);
531
- const sources = results.map((r) => r.chunk.source);
532
- const uniqueSources = new Set(sources);
533
- assert.equal(uniqueSources.size, 3, 'Each result must be from a distinct source');
534
- assert.ok(sources.includes('compiled/entities/dominant.md'), 'dominant source must be present');
535
- assert.ok(sources.includes('compiled/entities/source-b.md'), 'source-b must be present');
536
- assert.ok(sources.includes('compiled/entities/source-c.md'), 'source-c must be present');
458
+ fs.mkdirSync(compiledDir, { recursive: true });
459
+ fs.writeFileSync(path.join(compiledDir, 'a.md'), '# A\nserialized alpha content', 'utf8');
460
+ fs.writeFileSync(path.join(compiledDir, 'b.md'), '# B\nserialized beta content', 'utf8');
461
+ const trackingEmbedder = new TrackingEmbedder();
462
+ const testIndexer = new MemoryIndexer(fakeStore, trackingEmbedder, testCfg);
463
+ await testIndexer.indexDirectory(compiledDir, {
464
+ relativeBase: testCfg.wikiDir,
465
+ fileConcurrency: 2,
466
+ });
467
+ assert.equal(maxActiveMutations, 1);
468
+ assert.equal(trackingEmbedder.maxActive, 2);
469
+ assert.deepEqual(calls.slice(0, 1), ['deleteBySource:compiled/stale.md']);
470
+ assert.ok(calls.filter((call) => call.startsWith('upsert:')).length >= 2);
537
471
  }
538
472
  finally {
539
473
  fs.rmSync(testDir, { recursive: true, force: true });
540
474
  }
541
475
  });
542
- // BC8: dense-only candidate below DENSE_SCORE_FLOOR (0.2) is dropped; at/above floor is kept; bm25-only never dropped
543
- test('BC8: dense-only candidate below DENSE_SCORE_FLOOR is dropped; bm25-only and above-floor are kept', async () => {
544
- const testCfg = makeConfig('/tmp/bc8-floor');
545
- const chunks = [
546
- {
547
- id: 'below-floor',
548
- source: 'compiled/entities/below.md',
549
- heading: '',
550
- headingLevel: 0,
551
- content: 'below floor dense',
552
- lineStart: 1,
553
- lineEnd: 1,
554
- },
555
- {
556
- id: 'above-floor',
557
- source: 'compiled/entities/above.md',
558
- heading: '',
559
- headingLevel: 0,
560
- content: 'above floor dense',
561
- lineStart: 1,
562
- lineEnd: 1,
563
- },
564
- {
565
- id: 'bm25-overlap',
566
- source: 'compiled/concepts/overlap.md',
567
- heading: '',
568
- headingLevel: 0,
569
- content: 'overlap content',
570
- lineStart: 1,
571
- lineEnd: 1,
572
- },
573
- {
574
- id: 'bm25-only',
575
- source: 'compiled/concepts/keyword-only.md',
576
- heading: '',
577
- headingLevel: 0,
578
- content: 'keyword only',
579
- lineStart: 1,
580
- lineEnd: 1,
581
- },
582
- ];
476
+ test('indexDirectory isolates per-file embedding failure without deleting failed-file chunks', async () => {
477
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'failure-isolation-'));
478
+ const testCfg = makeConfig(path.join(testDir, 'wiki'));
479
+ const compiledDir = path.join(testCfg.wikiDir, 'compiled');
480
+ const upsertedSources = [];
481
+ const deletedIds = [];
583
482
  const fakeStore = {
584
483
  vecAvailable: true,
585
- // below-floor: dense-only, score 0.1 < DENSE_SCORE_FLOOR (0.2) → must be dropped
586
- // above-floor: dense-only, score 0.9 >= DENSE_SCORE_FLOOR (0.2) → must be kept
587
- // bm25-overlap: in both channels → never dropped regardless of dense score
588
- searchDense: () => [
589
- { id: 'below-floor', score: 0.1 },
590
- { id: 'above-floor', score: 0.9 },
591
- { id: 'bm25-overlap', score: 0.3 },
592
- ],
593
- searchBm25: () => [
594
- { id: 'bm25-overlap', score: 1 },
595
- { id: 'bm25-only', score: 0.8 },
596
- ],
597
- getChunksByIds: (ids) => chunks.filter((c) => ids.includes(c.id)),
598
- };
599
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
600
- const results = await testIndexer.search('query', 10);
601
- const resultIds = results.map((r) => r.chunk.id);
602
- assert.ok(!resultIds.includes('below-floor'), `below-floor (score 0.1 < 0.2) must be dropped; got: ${resultIds.join(', ')}`);
603
- assert.ok(resultIds.includes('above-floor'), `above-floor (score 0.9 >= 0.2) must be kept; got: ${resultIds.join(', ')}`);
604
- assert.ok(resultIds.includes('bm25-overlap'), `bm25-overlap (dual-channel) must always be kept`);
605
- assert.ok(resultIds.includes('bm25-only'), `bm25-only must never be dropped by the floor`);
606
- });
607
- // BC9: recency tiebreak is stable — repeated calls with tied/absent fileMtimeAt return identical order
608
- test('BC9: repeated search calls with equal-mtime candidates return deterministic order', async () => {
609
- const testCfg = makeConfig('/tmp/bc9-determinism');
610
- const chunks = [
611
- {
612
- id: 'c1',
613
- source: 'compiled/entities/c1.md',
614
- heading: '',
615
- headingLevel: 0,
616
- content: 'c1 content',
617
- lineStart: 1,
618
- lineEnd: 1,
619
- fileMtimeAt: 0,
484
+ hashesBySource: (source) => (source === 'compiled/bad.md' ? new Set(['bad-old-id']) : new Set()),
485
+ indexedSources: () => ['compiled/good.md', 'compiled/bad.md'],
486
+ deleteBySource: () => undefined,
487
+ upsert: (chunks) => {
488
+ upsertedSources.push(...chunks.map((chunk) => chunk.source));
620
489
  },
621
- {
622
- id: 'c2',
623
- source: 'compiled/entities/c2.md',
624
- heading: '',
625
- headingLevel: 0,
626
- content: 'c2 content',
627
- lineStart: 1,
628
- lineEnd: 1,
629
- fileMtimeAt: 0,
490
+ deleteByIds: (ids) => {
491
+ deletedIds.push(...ids);
630
492
  },
631
- {
632
- id: 'c3',
633
- source: 'compiled/entities/c3.md',
634
- heading: '',
635
- headingLevel: 0,
636
- content: 'c3 content',
637
- lineStart: 1,
638
- lineEnd: 1,
639
- fileMtimeAt: 0,
640
- },
641
- ];
642
- const fakeStore = {
643
- vecAvailable: true,
644
- searchDense: () => [
645
- { id: 'c1', score: 0.9 },
646
- { id: 'c2', score: 0.85 },
647
- { id: 'c3', score: 0.8 },
648
- ],
649
- searchBm25: () => [],
650
- getChunksByIds: (ids) => chunks.filter((c) => ids.includes(c.id)),
651
- };
652
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
653
- const firstCall = await testIndexer.search('query', 5);
654
- const secondCall = await testIndexer.search('query', 5);
655
- assert.equal(firstCall.length, secondCall.length, 'Result count must be identical across calls');
656
- for (let idx = 0; idx < firstCall.length; idx++) {
657
- assert.equal(firstCall[idx].chunk.id, secondCall[idx].chunk.id, `Position ${idx} must be deterministic: got '${firstCall[idx].chunk.id}' vs '${secondCall[idx].chunk.id}'`);
658
- }
659
- });
660
- test('search still returns dense-only results when BM25 has no matches', async () => {
661
- const testCfg = makeConfig('/tmp/search-dense-fallback');
662
- const chunks = [
663
- {
664
- id: 'semantic-1',
665
- source: 'compiled/preferences.md',
666
- heading: '',
667
- headingLevel: 0,
668
- content: 'I like fish',
669
- lineStart: 1,
670
- lineEnd: 1,
671
- },
672
- ];
673
- const fakeStore = {
674
- vecAvailable: true,
675
- searchDense: () => [{ id: 'semantic-1', score: 0.99 }],
676
- searchBm25: () => [],
677
- getChunksByIds: (ids) => chunks.filter((chunk) => ids.includes(chunk.id)),
678
493
  };
679
- const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
680
- const results = await testIndexer.search('favorite dish', 5);
681
- assert.deepEqual(results.map((result) => result.chunk.id), ['semantic-1']);
682
- assert.equal(results[0].retriever, 'dense');
683
- });
684
- test('search falls back to stored chunk content when source file is missing', async () => {
685
- const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'search-fallback-'));
686
- const testCfg = makeConfig(path.join(testDir, 'wiki'));
687
- const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
688
- const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
689
- const filePath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'missing.md');
690
- const fileContent = '# Missing\nfallbackUniqueTerm stored chunk text';
691
494
  try {
692
- fs.mkdirSync(path.dirname(filePath), { recursive: true });
693
- fs.writeFileSync(filePath, fileContent, 'utf8');
694
- await testIndexer.indexDirectory(path.join(testCfg.wikiDir, 'compiled'), {
495
+ fs.mkdirSync(compiledDir, { recursive: true });
496
+ fs.writeFileSync(path.join(compiledDir, 'good.md'), '# Good\nindexable good content', 'utf8');
497
+ fs.writeFileSync(path.join(compiledDir, 'bad.md'), '# Bad\nexplode this content', 'utf8');
498
+ const testIndexer = new MemoryIndexer(fakeStore, new TrackingEmbedder('explode'), testCfg);
499
+ const stats = await testIndexer.indexDirectory(compiledDir, {
695
500
  relativeBase: testCfg.wikiDir,
501
+ fileConcurrency: 2,
696
502
  });
697
- fs.unlinkSync(filePath);
698
- const results = await testIndexer.search('fallbackUniqueTerm', 5);
699
- assert.equal(results.length, 1);
700
- assert.equal(results[0].contentSource, 'fallback');
701
- assert.equal(results[0].chunk.content, fileContent);
503
+ assert.ok(stats.indexed > 0, `Expected good file to be indexed, got ${stats.indexed}`);
504
+ assert.equal(stats.skipped, 1);
505
+ assert.ok(upsertedSources.includes('compiled/good.md'));
506
+ assert.ok(!upsertedSources.includes('compiled/bad.md'));
507
+ assert.deepEqual(deletedIds, []);
702
508
  }
703
509
  finally {
704
- testStore.close();
705
- fs.rmSync(testDir, { recursive: true, force: true });
706
- }
707
- });
708
- test('search on empty store returns []', async () => {
709
- const emptyDir = fs.mkdtempSync(path.join(os.tmpdir(), 'empty-store-'));
710
- const emptyCfg = makeConfig(emptyDir);
711
- const emptyStore = new MemoryStore(path.join(emptyCfg.wikiDir, 'index.db'), emptyCfg);
712
- const emptyIndexer = new MemoryIndexer(emptyStore, new StubEmbedder(), emptyCfg);
713
- try {
714
- const results = await emptyIndexer.search('anything', 5);
715
- assert.deepEqual(results, []);
716
- }
717
- finally {
718
- emptyStore.close();
719
- fs.rmSync(emptyDir, { recursive: true, force: true });
720
- }
721
- });
722
- test('save appends manual content to daily wiki raw save file without indexing it', async () => {
723
- const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'save-daily-'));
724
- const testCfg = makeConfig(path.join(testDir, 'wiki'));
725
- const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
726
- const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
727
- const datePart = new Date().toISOString().slice(0, 10);
728
- const savePath = path.join(testCfg.wikiDir, 'raw', `conv_save_${datePart}.md`);
729
- try {
730
- const firstStats = await testIndexer.save('first manual save');
731
- const secondStats = await testIndexer.save('second manual save');
732
- assert.deepEqual(firstStats, { indexed: 0, deleted: 0, skipped: 0 });
733
- assert.deepEqual(secondStats, { indexed: 0, deleted: 0, skipped: 0 });
734
- assert.equal(fs.existsSync(savePath), true);
735
- const content = fs.readFileSync(savePath, 'utf8');
736
- assert.match(content, /first manual save/);
737
- assert.match(content, /second manual save/);
738
- assert.equal(fs.readdirSync(path.dirname(savePath)).filter((name) => /^conv_save_.*\.md$/.test(name)).length, 1);
739
- }
740
- finally {
741
- testStore.close();
742
510
  fs.rmSync(testDir, { recursive: true, force: true });
743
511
  }
744
512
  });
@@ -765,11 +533,18 @@ describe('MemoryIndexer', () => {
765
533
  assert.ok(stats.indexed > 0, `Expected indexed > 0, got ${stats.indexed}`);
766
534
  const ids = [...testStore.hashesBySource('compiled/provisional/conversation-digests/digest.md')];
767
535
  assert.ok(ids.length > 0);
768
- const chunks = testStore.getChunksByIds(ids);
769
- for (const chunk of chunks) {
770
- assert.equal(chunk.sourceType, 'digest');
771
- assert.equal(chunk.fileMtimeAt, expectedMtime);
772
- }
536
+ const results = testStore.searchHybrid({
537
+ query: 'mtime metadata unique',
538
+ topK: 5,
539
+ fetchLimit: 20,
540
+ denseScoreFloor: 0,
541
+ recencyWeight: RECENCY_WEIGHT,
542
+ rrfK: RRF_K,
543
+ });
544
+ const match = results.find((r) => r.chunk.source === 'compiled/provisional/conversation-digests/digest.md');
545
+ assert.ok(match, 'Expected searchHybrid to find the indexed chunk');
546
+ assert.equal(match.chunk.sourceType, 'digest');
547
+ assert.equal(match.chunk.fileMtimeAt, expectedMtime);
773
548
  }
774
549
  finally {
775
550
  testStore.close();