@hanhnd/agent-kit 1.0.28 → 1.0.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/kit-server.js +15737 -85
- package/dist/memory/__tests__/chunker.test.d.ts +1 -0
- package/dist/memory/__tests__/chunker.test.js +105 -0
- package/dist/memory/__tests__/indexer.test.d.ts +1 -0
- package/dist/memory/__tests__/indexer.test.js +392 -0
- package/dist/memory/__tests__/store.test.d.ts +1 -0
- package/dist/memory/__tests__/store.test.js +152 -0
- package/dist/memory/chunker.d.ts +4 -0
- package/dist/memory/chunker.js +147 -0
- package/dist/memory/embedder.d.ts +13 -0
- package/dist/memory/embedder.js +68 -0
- package/dist/memory/index.d.ts +5 -0
- package/dist/memory/index.js +4 -0
- package/dist/memory/indexer.d.ts +20 -0
- package/dist/memory/indexer.js +226 -0
- package/dist/memory/store.d.ts +29 -0
- package/dist/memory/store.js +273 -0
- package/dist/memory/types.d.ts +30 -0
- package/dist/memory/types.js +9 -0
- package/dist/tools/__tests__/memory.test.d.ts +1 -0
- package/dist/tools/__tests__/memory.test.js +185 -0
- package/dist/tools/config.d.ts +7 -0
- package/dist/tools/config.js +15 -0
- package/dist/tools/memory.d.ts +16 -0
- package/dist/tools/memory.js +75 -0
- package/dist/utils/utils.js +3 -1
- package/package.json +9 -3
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
import * as assert from 'node:assert/strict';
|
|
2
|
+
import { describe, test } from 'node:test';
|
|
3
|
+
import { chunkMarkdown, cleanContentForEmbedding, computeChunkId } from '../chunker.js';
|
|
4
|
+
const CFG = { chunkSize: 200, overlapLines: 2 };
|
|
5
|
+
const SRC = 'test.md';
|
|
6
|
+
describe('chunkMarkdown', () => {
|
|
7
|
+
test('returns [] for empty string', () => {
|
|
8
|
+
assert.deepEqual(chunkMarkdown('', SRC, CFG), []);
|
|
9
|
+
});
|
|
10
|
+
test('returns [] for whitespace-only string', () => {
|
|
11
|
+
assert.deepEqual(chunkMarkdown(' \n\n ', SRC, CFG), []);
|
|
12
|
+
});
|
|
13
|
+
test('single heading + short body produces one chunk with correct metadata', () => {
|
|
14
|
+
const text = '# My Heading\nThis is the body text.';
|
|
15
|
+
const chunks = chunkMarkdown(text, SRC, CFG);
|
|
16
|
+
assert.equal(chunks.length, 1);
|
|
17
|
+
const [c] = chunks;
|
|
18
|
+
assert.equal(c.heading, 'My Heading');
|
|
19
|
+
assert.equal(c.headingLevel, 1);
|
|
20
|
+
assert.equal(c.source, SRC);
|
|
21
|
+
assert.ok(c.lineStart >= 1);
|
|
22
|
+
assert.ok(c.lineEnd >= c.lineStart);
|
|
23
|
+
});
|
|
24
|
+
test('chunk id is deterministic — same content produces same id', () => {
|
|
25
|
+
const text = '# Section\nHello world.';
|
|
26
|
+
const [a] = chunkMarkdown(text, SRC, CFG);
|
|
27
|
+
const [b] = chunkMarkdown(text, 'other-source.md', CFG);
|
|
28
|
+
assert.equal(a.id, b.id, 'id must depend on content only, not source');
|
|
29
|
+
});
|
|
30
|
+
test('chunk id changes when content changes', () => {
|
|
31
|
+
const [a] = chunkMarkdown('# H\nVersion A', SRC, CFG);
|
|
32
|
+
const [b] = chunkMarkdown('# H\nVersion B', SRC, CFG);
|
|
33
|
+
assert.notEqual(a.id, b.id);
|
|
34
|
+
});
|
|
35
|
+
test('body exceeding chunkSize produces multiple chunks', () => {
|
|
36
|
+
const word = 'a'.repeat(50);
|
|
37
|
+
const body = Array.from({ length: 10 }, (_, i) => `Paragraph ${i}: ${word}`).join('\n\n');
|
|
38
|
+
const text = `# Big Section\n${body}`;
|
|
39
|
+
const chunks = chunkMarkdown(text, SRC, { chunkSize: 100, overlapLines: 0 });
|
|
40
|
+
assert.ok(chunks.length > 1, `Expected >1 chunks, got ${chunks.length}`);
|
|
41
|
+
for (const c of chunks) {
|
|
42
|
+
assert.equal(c.heading, 'Big Section');
|
|
43
|
+
}
|
|
44
|
+
});
|
|
45
|
+
test('multiple headings produce separate chunks with correct headingLevel', () => {
|
|
46
|
+
const text = [
|
|
47
|
+
'# Top Level',
|
|
48
|
+
'Top content.',
|
|
49
|
+
'## Sub Level',
|
|
50
|
+
'Sub content.',
|
|
51
|
+
'### Deep Level',
|
|
52
|
+
'Deep content.',
|
|
53
|
+
].join('\n');
|
|
54
|
+
const chunks = chunkMarkdown(text, SRC, CFG);
|
|
55
|
+
const levels = chunks.map((c) => c.headingLevel);
|
|
56
|
+
assert.ok(levels.includes(1), 'Expected headingLevel 1');
|
|
57
|
+
assert.ok(levels.includes(2), 'Expected headingLevel 2');
|
|
58
|
+
assert.ok(levels.includes(3), 'Expected headingLevel 3');
|
|
59
|
+
});
|
|
60
|
+
test('HTML comment is stripped from chunk content', () => {
|
|
61
|
+
const text = '# Section\nVisible text <!-- hidden comment --> more visible text.';
|
|
62
|
+
const [chunk] = chunkMarkdown(text, SRC, CFG);
|
|
63
|
+
assert.ok(!chunk.content.includes('hidden comment'), 'HTML comment must be stripped');
|
|
64
|
+
assert.ok(chunk.content.includes('Visible text'), 'Visible text must remain');
|
|
65
|
+
});
|
|
66
|
+
test('multiline HTML comment is stripped', () => {
|
|
67
|
+
const text = '# Section\nBefore.\n<!-- multi\nline\ncomment -->\nAfter.';
|
|
68
|
+
const [chunk] = chunkMarkdown(text, SRC, CFG);
|
|
69
|
+
assert.ok(!chunk.content.includes('multi'), 'Multiline comment must be stripped');
|
|
70
|
+
assert.ok(chunk.content.includes('Before.'), 'Text before comment must remain');
|
|
71
|
+
});
|
|
72
|
+
});
|
|
73
|
+
describe('computeChunkId', () => {
|
|
74
|
+
test('returns 16-character hex string', () => {
|
|
75
|
+
const id = computeChunkId('hello world');
|
|
76
|
+
assert.equal(id.length, 16);
|
|
77
|
+
assert.match(id, /^[0-9a-f]{16}$/);
|
|
78
|
+
});
|
|
79
|
+
test('same content always produces same id', () => {
|
|
80
|
+
assert.equal(computeChunkId('abc'), computeChunkId('abc'));
|
|
81
|
+
});
|
|
82
|
+
test('different content produces different id', () => {
|
|
83
|
+
assert.notEqual(computeChunkId('abc'), computeChunkId('xyz'));
|
|
84
|
+
});
|
|
85
|
+
});
|
|
86
|
+
describe('cleanContentForEmbedding', () => {
|
|
87
|
+
test('strips HTML comment', () => {
|
|
88
|
+
assert.equal(cleanContentForEmbedding('hello <!-- comment --> world'), 'hello world');
|
|
89
|
+
});
|
|
90
|
+
test('strips multiline HTML comment', () => {
|
|
91
|
+
const input = 'before <!-- line1\nline2 --> after';
|
|
92
|
+
const result = cleanContentForEmbedding(input);
|
|
93
|
+
assert.ok(!result.includes('line1'));
|
|
94
|
+
assert.ok(result.includes('before'));
|
|
95
|
+
assert.ok(result.includes('after'));
|
|
96
|
+
});
|
|
97
|
+
test('returns original string when no comments present', () => {
|
|
98
|
+
const text = 'plain text without comments';
|
|
99
|
+
assert.equal(cleanContentForEmbedding(text), text);
|
|
100
|
+
});
|
|
101
|
+
test('preserves fenced code block content', () => {
|
|
102
|
+
const text = '```js\nconst x = 1;\n```';
|
|
103
|
+
assert.equal(cleanContentForEmbedding(text), text);
|
|
104
|
+
});
|
|
105
|
+
});
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,392 @@
|
|
|
1
|
+
import * as assert from 'node:assert/strict';
|
|
2
|
+
import fsDefault from 'node:fs';
|
|
3
|
+
import * as fs from 'node:fs';
|
|
4
|
+
import { syncBuiltinESMExports } from 'node:module';
|
|
5
|
+
import * as os from 'node:os';
|
|
6
|
+
import * as path from 'node:path';
|
|
7
|
+
import { after, before, describe, test } from 'node:test';
|
|
8
|
+
import { MemoryIndexer } from '../indexer.js';
|
|
9
|
+
import { MemoryStore } from '../store.js';
|
|
10
|
+
// Stub embedder — returns deterministic non-zero vectors without loading any model
|
|
11
|
+
class StubEmbedder {
|
|
12
|
+
async embed(texts) {
|
|
13
|
+
return texts.map(() => new Float32Array(384).fill(0.05));
|
|
14
|
+
}
|
|
15
|
+
get dimension() {
|
|
16
|
+
return 384;
|
|
17
|
+
}
|
|
18
|
+
isReady() {
|
|
19
|
+
return true;
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
function makeConfig(wikiDir) {
|
|
23
|
+
return {
|
|
24
|
+
enabled: true,
|
|
25
|
+
wikiDir,
|
|
26
|
+
topK: 5,
|
|
27
|
+
chunkSize: 1500,
|
|
28
|
+
overlapLines: 2,
|
|
29
|
+
embeddingModel: 'Xenova/bge-small-en-v1.5',
|
|
30
|
+
vectorDimension: 384,
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
describe('MemoryIndexer', () => {
|
|
34
|
+
let tmpDir;
|
|
35
|
+
let store;
|
|
36
|
+
let indexer;
|
|
37
|
+
let config;
|
|
38
|
+
before(() => {
|
|
39
|
+
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-indexer-test-'));
|
|
40
|
+
config = makeConfig(tmpDir);
|
|
41
|
+
store = new MemoryStore(path.join(config.wikiDir, 'index.db'), config);
|
|
42
|
+
indexer = new MemoryIndexer(store, new StubEmbedder(), config);
|
|
43
|
+
});
|
|
44
|
+
after(() => {
|
|
45
|
+
store.close();
|
|
46
|
+
fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
47
|
+
});
|
|
48
|
+
test('indexFile on new file — indexed > 0, skipped === 0', async () => {
|
|
49
|
+
const filePath = path.join(tmpDir, 'new-file.md');
|
|
50
|
+
fs.writeFileSync(filePath, '# New File\nThis file has some content for indexing.', 'utf8');
|
|
51
|
+
const stats = await indexer.indexFile(filePath);
|
|
52
|
+
assert.ok(stats.indexed > 0, `Expected indexed > 0, got ${stats.indexed}`);
|
|
53
|
+
assert.equal(stats.skipped, 0);
|
|
54
|
+
});
|
|
55
|
+
test('indexFile on unchanged file — indexed === 0, skipped > 0', async () => {
|
|
56
|
+
const filePath = path.join(tmpDir, 'stable-file.md');
|
|
57
|
+
fs.writeFileSync(filePath, '# Stable\nThis content does not change between runs.', 'utf8');
|
|
58
|
+
// First run indexes it
|
|
59
|
+
await indexer.indexFile(filePath);
|
|
60
|
+
// Second run — same content
|
|
61
|
+
const stats = await indexer.indexFile(filePath);
|
|
62
|
+
assert.equal(stats.indexed, 0, `Expected indexed === 0, got ${stats.indexed}`);
|
|
63
|
+
assert.ok(stats.skipped > 0, `Expected skipped > 0, got ${stats.skipped}`);
|
|
64
|
+
});
|
|
65
|
+
test('indexFile after modification — only changed chunks re-indexed', async () => {
|
|
66
|
+
const filePath = path.join(tmpDir, 'modified-file.md');
|
|
67
|
+
fs.writeFileSync(filePath, '# Modified\nOriginal content.', 'utf8');
|
|
68
|
+
await indexer.indexFile(filePath);
|
|
69
|
+
fs.writeFileSync(filePath, '# Modified\nUpdated content that changed completely.', 'utf8');
|
|
70
|
+
const stats = await indexer.indexFile(filePath);
|
|
71
|
+
assert.ok(stats.indexed > 0, `Expected re-indexed chunks after modification`);
|
|
72
|
+
});
|
|
73
|
+
test('indexDirectory removes stale source when file is deleted', async () => {
|
|
74
|
+
const staleFile = path.join(tmpDir, 'stale-file.md');
|
|
75
|
+
fs.writeFileSync(staleFile, '# Stale\nThis file will be deleted.', 'utf8');
|
|
76
|
+
await indexer.indexFile(staleFile);
|
|
77
|
+
const staleSource = path.relative(tmpDir, staleFile);
|
|
78
|
+
const before = store.hashesBySource(staleSource);
|
|
79
|
+
assert.ok(before.size > 0, 'Stale file must be indexed first');
|
|
80
|
+
// Delete the file and re-index the directory
|
|
81
|
+
fs.unlinkSync(staleFile);
|
|
82
|
+
await indexer.indexDirectory(tmpDir);
|
|
83
|
+
const afterDeletion = store.hashesBySource(staleSource);
|
|
84
|
+
assert.equal(afterDeletion.size, 0, 'Stale source must be removed from store after directory scan');
|
|
85
|
+
});
|
|
86
|
+
test('search returns result with correct source for indexed content', async () => {
|
|
87
|
+
const filePath = path.join(config.wikiDir, 'compiled', 'searchable.md');
|
|
88
|
+
const fileContent = '# Searchable\nspecialUniqueTermForSearch is in this document.';
|
|
89
|
+
fs.mkdirSync(path.dirname(filePath), { recursive: true });
|
|
90
|
+
fs.writeFileSync(filePath, fileContent, 'utf8');
|
|
91
|
+
await indexer.indexDirectory(path.join(config.wikiDir, 'compiled'), {
|
|
92
|
+
relativeBase: config.wikiDir,
|
|
93
|
+
});
|
|
94
|
+
const results = await indexer.search('specialUniqueTermForSearch', 5);
|
|
95
|
+
assert.ok(results.length > 0, 'Expected at least one search result');
|
|
96
|
+
const expectedSource = path.relative(config.wikiDir, filePath);
|
|
97
|
+
const match = results.find((r) => r.chunk.source === expectedSource);
|
|
98
|
+
assert.ok(match, `Expected result with source=${expectedSource}, got: ${results.map((r) => r.chunk.source).join(', ')}`);
|
|
99
|
+
assert.equal(match.chunk.content, fileContent);
|
|
100
|
+
assert.equal(match.contentSource, 'file');
|
|
101
|
+
});
|
|
102
|
+
test('indexDirectory walks nested markdown files and excludes configured basenames', async () => {
|
|
103
|
+
const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'recursive-index-'));
|
|
104
|
+
const testCfg = makeConfig(path.join(testDir, 'wiki'));
|
|
105
|
+
const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
|
|
106
|
+
const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
|
|
107
|
+
const compiledDir = path.join(testCfg.wikiDir, 'compiled');
|
|
108
|
+
try {
|
|
109
|
+
fs.mkdirSync(path.join(compiledDir, 'entities'), { recursive: true });
|
|
110
|
+
fs.writeFileSync(path.join(compiledDir, 'entities', 'foo.md'), '# Foo\nrecursiveUniqueTerm', 'utf8');
|
|
111
|
+
fs.writeFileSync(path.join(compiledDir, 'entities', 'index.md'), '# Index\nskip me', 'utf8');
|
|
112
|
+
fs.writeFileSync(path.join(compiledDir, 'log.md'), '# Log\nskip me', 'utf8');
|
|
113
|
+
fs.writeFileSync(path.join(compiledDir, 'entities', 'notes.txt'), 'skip me', 'utf8');
|
|
114
|
+
const stats = await testIndexer.indexDirectory(compiledDir, {
|
|
115
|
+
relativeBase: testCfg.wikiDir,
|
|
116
|
+
excludeFiles: ['index.md', 'log.md'],
|
|
117
|
+
});
|
|
118
|
+
assert.ok(stats.indexed > 0, `Expected indexed > 0, got ${stats.indexed}`);
|
|
119
|
+
assert.ok(testStore.hashesBySource('compiled/entities/foo.md').size > 0);
|
|
120
|
+
assert.equal(testStore.hashesBySource('compiled/entities/index.md').size, 0);
|
|
121
|
+
assert.equal(testStore.hashesBySource('compiled/log.md').size, 0);
|
|
122
|
+
assert.equal(testStore.hashesBySource('compiled/entities/notes.txt').size, 0);
|
|
123
|
+
}
|
|
124
|
+
finally {
|
|
125
|
+
testStore.close();
|
|
126
|
+
fs.rmSync(testDir, { recursive: true, force: true });
|
|
127
|
+
}
|
|
128
|
+
});
|
|
129
|
+
test('indexDirectory returns zero stats when root directory is missing', async () => {
|
|
130
|
+
const stats = await indexer.indexDirectory(path.join(config.wikiDir, 'missing'), {
|
|
131
|
+
relativeBase: config.wikiDir,
|
|
132
|
+
});
|
|
133
|
+
assert.deepEqual(stats, { indexed: 0, deleted: 0, skipped: 0 });
|
|
134
|
+
});
|
|
135
|
+
test('indexDirectory removes stale pre-migration daily-file sources', async () => {
|
|
136
|
+
const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'recursive-stale-'));
|
|
137
|
+
const testCfg = makeConfig(path.join(testDir, 'wiki'));
|
|
138
|
+
const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
|
|
139
|
+
const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
|
|
140
|
+
const compiledDir = path.join(testCfg.wikiDir, 'compiled');
|
|
141
|
+
try {
|
|
142
|
+
fs.mkdirSync(compiledDir, { recursive: true });
|
|
143
|
+
testStore.upsert([{
|
|
144
|
+
id: 'stale-daily-file-0001',
|
|
145
|
+
source: '2026-05-18.md',
|
|
146
|
+
heading: 'Stale',
|
|
147
|
+
headingLevel: 1,
|
|
148
|
+
content: 'pre migration content',
|
|
149
|
+
lineStart: 1,
|
|
150
|
+
lineEnd: 2,
|
|
151
|
+
}], [new Float32Array(384)]);
|
|
152
|
+
assert.ok(testStore.hashesBySource('2026-05-18.md').size > 0);
|
|
153
|
+
const stats = await testIndexer.indexDirectory(compiledDir, {
|
|
154
|
+
relativeBase: testCfg.wikiDir,
|
|
155
|
+
});
|
|
156
|
+
assert.equal(stats.deleted, 1);
|
|
157
|
+
assert.equal(testStore.hashesBySource('2026-05-18.md').size, 0);
|
|
158
|
+
}
|
|
159
|
+
finally {
|
|
160
|
+
testStore.close();
|
|
161
|
+
fs.rmSync(testDir, { recursive: true, force: true });
|
|
162
|
+
}
|
|
163
|
+
});
|
|
164
|
+
test('search deduplicates sources and reads each matched source once', async () => {
|
|
165
|
+
const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'search-dedup-'));
|
|
166
|
+
const testCfg = {
|
|
167
|
+
...makeConfig(path.join(testDir, 'wiki')),
|
|
168
|
+
chunkSize: 40,
|
|
169
|
+
overlapLines: 0,
|
|
170
|
+
};
|
|
171
|
+
const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
|
|
172
|
+
const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
|
|
173
|
+
const filePath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'dedup.md');
|
|
174
|
+
const fileContent = [
|
|
175
|
+
'# Dedup',
|
|
176
|
+
'dedupUniqueTerm first chunk text',
|
|
177
|
+
'dedupUniqueTerm second chunk text',
|
|
178
|
+
'dedupUniqueTerm third chunk text',
|
|
179
|
+
].join('\n');
|
|
180
|
+
const originalReadFileSync = fsDefault.readFileSync;
|
|
181
|
+
let readCount = 0;
|
|
182
|
+
try {
|
|
183
|
+
fs.mkdirSync(path.dirname(filePath), { recursive: true });
|
|
184
|
+
fs.writeFileSync(filePath, fileContent, 'utf8');
|
|
185
|
+
await testIndexer.indexDirectory(path.join(testCfg.wikiDir, 'compiled'), {
|
|
186
|
+
relativeBase: testCfg.wikiDir,
|
|
187
|
+
});
|
|
188
|
+
const expectedPath = path.join(testCfg.wikiDir, 'compiled/entities/dedup.md');
|
|
189
|
+
fsDefault.readFileSync = ((targetPath, options) => {
|
|
190
|
+
if (targetPath === expectedPath)
|
|
191
|
+
readCount += 1;
|
|
192
|
+
return originalReadFileSync(targetPath, options);
|
|
193
|
+
});
|
|
194
|
+
syncBuiltinESMExports();
|
|
195
|
+
const results = await testIndexer.search('dedupUniqueTerm', 5);
|
|
196
|
+
assert.equal(results.filter((r) => r.chunk.source === 'compiled/entities/dedup.md').length, 1);
|
|
197
|
+
assert.equal(readCount, 1);
|
|
198
|
+
assert.equal(results[0].chunk.content, fileContent);
|
|
199
|
+
assert.equal(results[0].contentSource, 'file');
|
|
200
|
+
}
|
|
201
|
+
finally {
|
|
202
|
+
fsDefault.readFileSync = originalReadFileSync;
|
|
203
|
+
syncBuiltinESMExports();
|
|
204
|
+
testStore.close();
|
|
205
|
+
fs.rmSync(testDir, { recursive: true, force: true });
|
|
206
|
+
}
|
|
207
|
+
});
|
|
208
|
+
test('search continues past duplicate sources until topK unique sources are returned', async () => {
|
|
209
|
+
const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'search-unique-topk-'));
|
|
210
|
+
const testCfg = makeConfig(path.join(testDir, 'wiki'));
|
|
211
|
+
const firstPath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'first.md');
|
|
212
|
+
const secondPath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'second.md');
|
|
213
|
+
const firstContent = '# First\nfirst source full content';
|
|
214
|
+
const secondContent = '# Second\nsecond source full content';
|
|
215
|
+
const chunks = [
|
|
216
|
+
{
|
|
217
|
+
id: 'first-1',
|
|
218
|
+
source: 'compiled/entities/first.md',
|
|
219
|
+
heading: 'First',
|
|
220
|
+
headingLevel: 1,
|
|
221
|
+
content: 'first matching chunk one',
|
|
222
|
+
lineStart: 1,
|
|
223
|
+
lineEnd: 2,
|
|
224
|
+
},
|
|
225
|
+
{
|
|
226
|
+
id: 'first-2',
|
|
227
|
+
source: 'compiled/entities/first.md',
|
|
228
|
+
heading: 'First',
|
|
229
|
+
headingLevel: 1,
|
|
230
|
+
content: 'first matching chunk two',
|
|
231
|
+
lineStart: 3,
|
|
232
|
+
lineEnd: 4,
|
|
233
|
+
},
|
|
234
|
+
{
|
|
235
|
+
id: 'second-1',
|
|
236
|
+
source: 'compiled/entities/second.md',
|
|
237
|
+
heading: 'Second',
|
|
238
|
+
headingLevel: 1,
|
|
239
|
+
content: 'second matching chunk',
|
|
240
|
+
lineStart: 1,
|
|
241
|
+
lineEnd: 2,
|
|
242
|
+
},
|
|
243
|
+
];
|
|
244
|
+
const fakeStore = {
|
|
245
|
+
vecAvailable: false,
|
|
246
|
+
searchBm25: () => [
|
|
247
|
+
{ id: 'first-1', score: 1 },
|
|
248
|
+
{ id: 'first-2', score: 0.9 },
|
|
249
|
+
{ id: 'second-1', score: 0.8 },
|
|
250
|
+
],
|
|
251
|
+
getChunksByIds: (ids) => chunks.filter((chunk) => ids.includes(chunk.id)),
|
|
252
|
+
};
|
|
253
|
+
const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
|
|
254
|
+
try {
|
|
255
|
+
fs.mkdirSync(path.dirname(firstPath), { recursive: true });
|
|
256
|
+
fs.writeFileSync(firstPath, firstContent, 'utf8');
|
|
257
|
+
fs.writeFileSync(secondPath, secondContent, 'utf8');
|
|
258
|
+
const results = await testIndexer.search('duplicate source query', 2);
|
|
259
|
+
assert.equal(results.length, 2);
|
|
260
|
+
assert.deepEqual(results.map((result) => result.chunk.source), [
|
|
261
|
+
'compiled/entities/first.md',
|
|
262
|
+
'compiled/entities/second.md',
|
|
263
|
+
]);
|
|
264
|
+
assert.equal(results[0].chunk.content, firstContent);
|
|
265
|
+
assert.equal(results[1].chunk.content, secondContent);
|
|
266
|
+
}
|
|
267
|
+
finally {
|
|
268
|
+
fs.rmSync(testDir, { recursive: true, force: true });
|
|
269
|
+
}
|
|
270
|
+
});
|
|
271
|
+
test('search does not include dense-only results when BM25 has matches', async () => {
|
|
272
|
+
const testCfg = makeConfig('/tmp/search-dense-filter');
|
|
273
|
+
const chunks = [
|
|
274
|
+
{
|
|
275
|
+
id: 'preference-1',
|
|
276
|
+
source: 'compiled/preferences.md',
|
|
277
|
+
heading: '',
|
|
278
|
+
headingLevel: 0,
|
|
279
|
+
content: 'I like fish',
|
|
280
|
+
lineStart: 1,
|
|
281
|
+
lineEnd: 1,
|
|
282
|
+
},
|
|
283
|
+
{
|
|
284
|
+
id: 'dense-only-1',
|
|
285
|
+
source: 'compiled/entities/worktree.md',
|
|
286
|
+
heading: 'Worktree',
|
|
287
|
+
headingLevel: 1,
|
|
288
|
+
content: 'Unrelated worktree lifecycle content',
|
|
289
|
+
lineStart: 1,
|
|
290
|
+
lineEnd: 2,
|
|
291
|
+
},
|
|
292
|
+
];
|
|
293
|
+
const fakeStore = {
|
|
294
|
+
vecAvailable: true,
|
|
295
|
+
searchDense: () => [
|
|
296
|
+
{ id: 'dense-only-1', score: 0.99 },
|
|
297
|
+
{ id: 'preference-1', score: 0.98 },
|
|
298
|
+
],
|
|
299
|
+
searchBm25: () => [{ id: 'preference-1', score: 1 }],
|
|
300
|
+
getChunksByIds: (ids) => chunks.filter((chunk) => ids.includes(chunk.id)),
|
|
301
|
+
};
|
|
302
|
+
const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
|
|
303
|
+
const results = await testIndexer.search('personal likes and preferences of the user', 5);
|
|
304
|
+
assert.deepEqual(results.map((result) => result.chunk.id), ['preference-1']);
|
|
305
|
+
assert.equal(results[0].retriever, 'both');
|
|
306
|
+
});
|
|
307
|
+
test('search still returns dense-only results when BM25 has no matches', async () => {
|
|
308
|
+
const testCfg = makeConfig('/tmp/search-dense-fallback');
|
|
309
|
+
const chunks = [
|
|
310
|
+
{
|
|
311
|
+
id: 'semantic-1',
|
|
312
|
+
source: 'compiled/preferences.md',
|
|
313
|
+
heading: '',
|
|
314
|
+
headingLevel: 0,
|
|
315
|
+
content: 'I like fish',
|
|
316
|
+
lineStart: 1,
|
|
317
|
+
lineEnd: 1,
|
|
318
|
+
},
|
|
319
|
+
];
|
|
320
|
+
const fakeStore = {
|
|
321
|
+
vecAvailable: true,
|
|
322
|
+
searchDense: () => [{ id: 'semantic-1', score: 0.99 }],
|
|
323
|
+
searchBm25: () => [],
|
|
324
|
+
getChunksByIds: (ids) => chunks.filter((chunk) => ids.includes(chunk.id)),
|
|
325
|
+
};
|
|
326
|
+
const testIndexer = new MemoryIndexer(fakeStore, new StubEmbedder(), testCfg);
|
|
327
|
+
const results = await testIndexer.search('favorite dish', 5);
|
|
328
|
+
assert.deepEqual(results.map((result) => result.chunk.id), ['semantic-1']);
|
|
329
|
+
assert.equal(results[0].retriever, 'dense');
|
|
330
|
+
});
|
|
331
|
+
test('search falls back to stored chunk content when source file is missing', async () => {
|
|
332
|
+
const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'search-fallback-'));
|
|
333
|
+
const testCfg = makeConfig(path.join(testDir, 'wiki'));
|
|
334
|
+
const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
|
|
335
|
+
const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
|
|
336
|
+
const filePath = path.join(testCfg.wikiDir, 'compiled', 'entities', 'missing.md');
|
|
337
|
+
const fileContent = '# Missing\nfallbackUniqueTerm stored chunk text';
|
|
338
|
+
try {
|
|
339
|
+
fs.mkdirSync(path.dirname(filePath), { recursive: true });
|
|
340
|
+
fs.writeFileSync(filePath, fileContent, 'utf8');
|
|
341
|
+
await testIndexer.indexDirectory(path.join(testCfg.wikiDir, 'compiled'), {
|
|
342
|
+
relativeBase: testCfg.wikiDir,
|
|
343
|
+
});
|
|
344
|
+
fs.unlinkSync(filePath);
|
|
345
|
+
const results = await testIndexer.search('fallbackUniqueTerm', 5);
|
|
346
|
+
assert.equal(results.length, 1);
|
|
347
|
+
assert.equal(results[0].contentSource, 'fallback');
|
|
348
|
+
assert.equal(results[0].chunk.content, fileContent);
|
|
349
|
+
}
|
|
350
|
+
finally {
|
|
351
|
+
testStore.close();
|
|
352
|
+
fs.rmSync(testDir, { recursive: true, force: true });
|
|
353
|
+
}
|
|
354
|
+
});
|
|
355
|
+
test('search on empty store returns []', async () => {
|
|
356
|
+
const emptyDir = fs.mkdtempSync(path.join(os.tmpdir(), 'empty-store-'));
|
|
357
|
+
const emptyCfg = makeConfig(emptyDir);
|
|
358
|
+
const emptyStore = new MemoryStore(path.join(emptyCfg.wikiDir, 'index.db'), emptyCfg);
|
|
359
|
+
const emptyIndexer = new MemoryIndexer(emptyStore, new StubEmbedder(), emptyCfg);
|
|
360
|
+
try {
|
|
361
|
+
const results = await emptyIndexer.search('anything', 5);
|
|
362
|
+
assert.deepEqual(results, []);
|
|
363
|
+
}
|
|
364
|
+
finally {
|
|
365
|
+
emptyStore.close();
|
|
366
|
+
fs.rmSync(emptyDir, { recursive: true, force: true });
|
|
367
|
+
}
|
|
368
|
+
});
|
|
369
|
+
test('save appends manual content to daily wiki raw save file without indexing it', async () => {
|
|
370
|
+
const testDir = fs.mkdtempSync(path.join(os.tmpdir(), 'save-daily-'));
|
|
371
|
+
const testCfg = makeConfig(path.join(testDir, 'wiki'));
|
|
372
|
+
const testStore = new MemoryStore(path.join(testCfg.wikiDir, 'index.db'), testCfg);
|
|
373
|
+
const testIndexer = new MemoryIndexer(testStore, new StubEmbedder(), testCfg);
|
|
374
|
+
const datePart = new Date().toISOString().slice(0, 10);
|
|
375
|
+
const savePath = path.join(testCfg.wikiDir, 'raw', `conv_save_${datePart}.md`);
|
|
376
|
+
try {
|
|
377
|
+
const firstStats = await testIndexer.save('first manual save');
|
|
378
|
+
const secondStats = await testIndexer.save('second manual save');
|
|
379
|
+
assert.deepEqual(firstStats, { indexed: 0, deleted: 0, skipped: 0 });
|
|
380
|
+
assert.deepEqual(secondStats, { indexed: 0, deleted: 0, skipped: 0 });
|
|
381
|
+
assert.equal(fs.existsSync(savePath), true);
|
|
382
|
+
const content = fs.readFileSync(savePath, 'utf8');
|
|
383
|
+
assert.match(content, /first manual save/);
|
|
384
|
+
assert.match(content, /second manual save/);
|
|
385
|
+
assert.equal(fs.readdirSync(path.dirname(savePath)).filter((name) => /^conv_save_.*\.md$/.test(name)).length, 1);
|
|
386
|
+
}
|
|
387
|
+
finally {
|
|
388
|
+
testStore.close();
|
|
389
|
+
fs.rmSync(testDir, { recursive: true, force: true });
|
|
390
|
+
}
|
|
391
|
+
});
|
|
392
|
+
});
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
import * as assert from 'node:assert/strict';
|
|
2
|
+
import * as fs from 'node:fs';
|
|
3
|
+
import * as os from 'node:os';
|
|
4
|
+
import * as path from 'node:path';
|
|
5
|
+
import { after, before, describe, test } from 'node:test';
|
|
6
|
+
import { MemoryStore } from '../store.js';
|
|
7
|
+
const TEST_CONFIG = {
|
|
8
|
+
enabled: true,
|
|
9
|
+
wikiDir: '',
|
|
10
|
+
topK: 5,
|
|
11
|
+
chunkSize: 1500,
|
|
12
|
+
overlapLines: 2,
|
|
13
|
+
embeddingModel: 'Xenova/bge-small-en-v1.5',
|
|
14
|
+
vectorDimension: 384,
|
|
15
|
+
};
|
|
16
|
+
function makeChunk(overrides = {}) {
|
|
17
|
+
return {
|
|
18
|
+
id: 'test-id-0000001',
|
|
19
|
+
source: 'test.md',
|
|
20
|
+
heading: 'Test Heading',
|
|
21
|
+
headingLevel: 1,
|
|
22
|
+
content: 'This is test content for BM25 search.',
|
|
23
|
+
lineStart: 1,
|
|
24
|
+
lineEnd: 5,
|
|
25
|
+
...overrides,
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
describe('MemoryStore', () => {
|
|
29
|
+
let tmpDir;
|
|
30
|
+
let dbPath;
|
|
31
|
+
let store;
|
|
32
|
+
before(() => {
|
|
33
|
+
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'memory-store-test-'));
|
|
34
|
+
dbPath = path.join(tmpDir, 'index.db');
|
|
35
|
+
store = new MemoryStore(dbPath, TEST_CONFIG);
|
|
36
|
+
});
|
|
37
|
+
after(() => {
|
|
38
|
+
store.close();
|
|
39
|
+
fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
40
|
+
});
|
|
41
|
+
test('vecAvailable is a boolean', () => {
|
|
42
|
+
assert.equal(typeof store.vecAvailable, 'boolean');
|
|
43
|
+
});
|
|
44
|
+
test('hashesBySource returns empty set for unknown source', () => {
|
|
45
|
+
const hashes = store.hashesBySource('nonexistent.md');
|
|
46
|
+
assert.ok(hashes instanceof Set);
|
|
47
|
+
assert.equal(hashes.size, 0);
|
|
48
|
+
});
|
|
49
|
+
test('upsert stores chunks and hashesBySource returns their ids', () => {
|
|
50
|
+
const chunk = makeChunk({ id: 'upsert-test-0001', source: 'upsert.md' });
|
|
51
|
+
const embedding = new Float32Array(384).fill(0.1);
|
|
52
|
+
store.upsert([chunk], [embedding]);
|
|
53
|
+
const hashes = store.hashesBySource('upsert.md');
|
|
54
|
+
assert.ok(hashes.has('upsert-test-0001'));
|
|
55
|
+
});
|
|
56
|
+
test('deleteBySource removes all chunks for that source', () => {
|
|
57
|
+
const c1 = makeChunk({ id: 'del-src-0001', source: 'deleteme.md', content: 'delete source chunk 1' });
|
|
58
|
+
const c2 = makeChunk({ id: 'del-src-0002', source: 'deleteme.md', content: 'delete source chunk 2' });
|
|
59
|
+
store.upsert([c1, c2], [new Float32Array(384), new Float32Array(384)]);
|
|
60
|
+
store.deleteBySource('deleteme.md');
|
|
61
|
+
const hashes = store.hashesBySource('deleteme.md');
|
|
62
|
+
assert.equal(hashes.size, 0);
|
|
63
|
+
});
|
|
64
|
+
test('searchBm25 returns matching result after upsert', () => {
|
|
65
|
+
const chunk = makeChunk({
|
|
66
|
+
id: 'bm25-search-0001',
|
|
67
|
+
source: 'bm25.md',
|
|
68
|
+
content: 'uniqueKeywordXYZ for BM25 testing',
|
|
69
|
+
});
|
|
70
|
+
store.upsert([chunk], [new Float32Array(384)]);
|
|
71
|
+
const results = store.searchBm25('uniqueKeywordXYZ', 5);
|
|
72
|
+
assert.ok(results.length > 0, 'Expected at least one BM25 result');
|
|
73
|
+
assert.ok(results.some((r) => r.id === 'bm25-search-0001'), `Expected chunk id in results, got: ${results.map((r) => r.id).join(', ')}`);
|
|
74
|
+
});
|
|
75
|
+
test('searchBm25 ignores filler words and matches preference source/content terms', () => {
|
|
76
|
+
const preference = makeChunk({
|
|
77
|
+
id: 'preference-search-0001',
|
|
78
|
+
source: 'compiled/preferences.md',
|
|
79
|
+
heading: '',
|
|
80
|
+
headingLevel: 0,
|
|
81
|
+
content: 'I like fish',
|
|
82
|
+
});
|
|
83
|
+
const unrelated = makeChunk({
|
|
84
|
+
id: 'preference-search-0002',
|
|
85
|
+
source: 'compiled/entities/worktree.md',
|
|
86
|
+
heading: 'Open Questions',
|
|
87
|
+
content: 'How should git manage worktree lifecycle decisions?',
|
|
88
|
+
});
|
|
89
|
+
store.upsert([preference, unrelated], [new Float32Array(384), new Float32Array(384)]);
|
|
90
|
+
const results = store.searchBm25('personal likes and preferences of the user', 5);
|
|
91
|
+
assert.ok(results.length > 0, 'Expected preference query to return results');
|
|
92
|
+
assert.equal(results[0].id, 'preference-search-0001');
|
|
93
|
+
});
|
|
94
|
+
test('searchBm25 returns empty array for empty query', () => {
|
|
95
|
+
const results = store.searchBm25(' ', 5);
|
|
96
|
+
assert.deepEqual(results, []);
|
|
97
|
+
});
|
|
98
|
+
test('getChunksByIds returns correct metadata for stored chunk', () => {
|
|
99
|
+
const chunk = makeChunk({
|
|
100
|
+
id: 'get-by-ids-0001',
|
|
101
|
+
source: 'metadata.md',
|
|
102
|
+
heading: 'Metadata Section',
|
|
103
|
+
headingLevel: 2,
|
|
104
|
+
content: 'Content for metadata test',
|
|
105
|
+
lineStart: 10,
|
|
106
|
+
lineEnd: 20,
|
|
107
|
+
});
|
|
108
|
+
store.upsert([chunk], [new Float32Array(384)]);
|
|
109
|
+
const results = store.getChunksByIds(['get-by-ids-0001']);
|
|
110
|
+
assert.equal(results.length, 1);
|
|
111
|
+
const [r] = results;
|
|
112
|
+
assert.equal(r.id, 'get-by-ids-0001');
|
|
113
|
+
assert.equal(r.source, 'metadata.md');
|
|
114
|
+
assert.equal(r.heading, 'Metadata Section');
|
|
115
|
+
assert.equal(r.headingLevel, 2);
|
|
116
|
+
assert.equal(r.lineStart, 10);
|
|
117
|
+
assert.equal(r.lineEnd, 20);
|
|
118
|
+
});
|
|
119
|
+
test('getChunksByIds returns only found rows for mixed ids', () => {
|
|
120
|
+
const chunk = makeChunk({ id: 'partial-found-0001', source: 'partial.md', content: 'partial' });
|
|
121
|
+
store.upsert([chunk], [new Float32Array(384)]);
|
|
122
|
+
const results = store.getChunksByIds(['partial-found-0001', 'does-not-exist-999']);
|
|
123
|
+
assert.equal(results.length, 1);
|
|
124
|
+
assert.equal(results[0].id, 'partial-found-0001');
|
|
125
|
+
});
|
|
126
|
+
test('indexedSources includes source after upsert', () => {
|
|
127
|
+
const chunk = makeChunk({ id: 'indexed-src-0001', source: 'indexed-source.md', content: 'indexed' });
|
|
128
|
+
store.upsert([chunk], [new Float32Array(384)]);
|
|
129
|
+
const sources = store.indexedSources();
|
|
130
|
+
assert.ok(sources.includes('indexed-source.md'), `Expected 'indexed-source.md' in ${sources.join(', ')}`);
|
|
131
|
+
});
|
|
132
|
+
test('deleteByIds removes specific chunks', () => {
|
|
133
|
+
const c1 = makeChunk({ id: 'del-ids-0001', source: 'del-ids.md', content: 'delete by id 1' });
|
|
134
|
+
const c2 = makeChunk({ id: 'del-ids-0002', source: 'del-ids.md', content: 'delete by id 2' });
|
|
135
|
+
store.upsert([c1, c2], [new Float32Array(384), new Float32Array(384)]);
|
|
136
|
+
store.deleteByIds(['del-ids-0001']);
|
|
137
|
+
const hashes = store.hashesBySource('del-ids.md');
|
|
138
|
+
assert.ok(!hashes.has('del-ids-0001'), 'Deleted chunk id must be gone');
|
|
139
|
+
assert.ok(hashes.has('del-ids-0002'), 'Non-deleted chunk id must remain');
|
|
140
|
+
});
|
|
141
|
+
test('searchBm25 still works regardless of vecAvailable', () => {
|
|
142
|
+
// This verifies FTS5 degraded mode is always functional
|
|
143
|
+
const chunk = makeChunk({
|
|
144
|
+
id: 'fts5-degraded-0001',
|
|
145
|
+
source: 'fts5.md',
|
|
146
|
+
content: 'degradedModeTest keyword',
|
|
147
|
+
});
|
|
148
|
+
store.upsert([chunk], [new Float32Array(384)]);
|
|
149
|
+
const results = store.searchBm25('degradedModeTest', 5);
|
|
150
|
+
assert.ok(results.length > 0, 'FTS5 search must work regardless of vecAvailable');
|
|
151
|
+
});
|
|
152
|
+
});
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
import type { MemoryChunk, MemoryConfig } from './types.js';
|
|
2
|
+
export declare function computeChunkId(content: string): string;
|
|
3
|
+
export declare function cleanContentForEmbedding(text: string): string;
|
|
4
|
+
export declare function chunkMarkdown(text: string, source: string, config: Pick<MemoryConfig, 'chunkSize' | 'overlapLines'>): MemoryChunk[];
|