mnemonad-cli 0.1.1 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +173 -19
- package/bin/mnemonad.js +164 -11
- package/lib/bridgeCodec.js +12 -0
- package/lib/chainClient.js +28 -7
- package/lib/commands/buildIndex.js +171 -0
- package/lib/commands/burn.js +59 -0
- package/lib/commands/compact.js +7 -2
- package/lib/commands/diff.js +25 -6
- package/lib/commands/index.js +3 -0
- package/lib/commands/info.js +42 -2
- package/lib/commands/pull.js +13 -6
- package/lib/commands/push.js +61 -7
- package/lib/commands/search.js +91 -0
- package/lib/commands/shared.js +137 -11
- package/lib/commands/watch.js +66 -12
- package/lib/passkeyBridge.js +203 -0
- package/lib/search/Indexer.js +204 -0
- package/lib/search/Searcher.js +68 -0
- package/lib/search/VectorIndex.js +225 -0
- package/lib/search/browser/BrowserSearcher.js +131 -0
- package/lib/search/browser/WasmIndexReader.js +93 -0
- package/lib/search/browser.js +13 -0
- package/lib/search/chunking/ChunkingStrategy.js +23 -0
- package/lib/search/chunking/TextWindowChunkingStrategy.js +69 -0
- package/lib/search/embeddings/EmbeddingProvider.js +40 -0
- package/lib/search/embeddings/StaticEmbeddingProvider.js +180 -0
- package/lib/search/embeddings/modelFiles.browser.js +28 -0
- package/lib/search/embeddings/modelFiles.node.js +41 -0
- package/lib/search/embeddings/modelFiles.shared.js +44 -0
- package/lib/search/index.js +18 -0
- package/lib/search/providers.js +153 -0
- package/lib/search/schema.js +103 -0
- package/mnemonad.config.js +19 -5
- package/package.json +18 -3
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
import { existsSync, rmSync } from 'node:fs';
|
|
2
|
+
import VectorIndex from './VectorIndex.js';
|
|
3
|
+
import TextWindowChunkingStrategy from './chunking/TextWindowChunkingStrategy.js';
|
|
4
|
+
|
|
5
|
+
/** First N bytes checked for a null byte — the standard, cheap heuristic (same one git and
|
|
6
|
+
* most diff tools use) for "this is almost certainly binary, don't try to embed it as text". */
|
|
7
|
+
const BINARY_SNIFF_BYTES = 8000;
|
|
8
|
+
|
|
9
|
+
function looksLikeText(bytes) {
|
|
10
|
+
const n = Math.min(bytes.length, BINARY_SNIFF_BYTES);
|
|
11
|
+
for (let i = 0; i < n; i++) {
|
|
12
|
+
if (bytes[i] === 0) return false;
|
|
13
|
+
}
|
|
14
|
+
return true;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function toHex(bytes) {
|
|
18
|
+
return Array.from(bytes, (b) => b.toString(16).padStart(2, '0')).join('');
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Builds/updates a `VectorIndex` from a folder — anything shaped like
|
|
23
|
+
* `@fizzyflow/doublesync`'s `DoubleSyncFolder` (has an async `walk()` generator), which
|
|
24
|
+
* covers the CLI's real-disk `FSFolder` and any in-memory folder the same way. Incremental by
|
|
25
|
+
* construction: a file whose content hash already matches what's stored is skipped entirely,
|
|
26
|
+
* no re-embedding, no re-chunking, not even a read past the hash check.
|
|
27
|
+
*/
|
|
28
|
+
export default class Indexer {
|
|
29
|
+
/**
|
|
30
|
+
* @param {Object} [params]
|
|
31
|
+
* @param {?import('./embeddings/EmbeddingProvider.js').default} [params.embeddingProvider] -
|
|
32
|
+
* required by `index()`; `status()` never embeds, so it can go without one (see
|
|
33
|
+
* providers.js's createProvider for the usual way to get one).
|
|
34
|
+
* @param {import('./chunking/ChunkingStrategy.js').default} [params.chunkingStrategy]
|
|
35
|
+
*/
|
|
36
|
+
constructor({ embeddingProvider = null, chunkingStrategy } = {}) {
|
|
37
|
+
this.embeddingProvider = embeddingProvider;
|
|
38
|
+
this.chunkingStrategy = chunkingStrategy || new TextWindowChunkingStrategy();
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* What `index()` would do to `dbPath`, without doing it: no embedding, no model load, and
|
|
43
|
+
* the database opened read-only. Uses the same walk and the same rules as `index()` (see
|
|
44
|
+
* `_scan`), so it agrees exactly on which files count — a binary or empty file, which
|
|
45
|
+
* `index()` never records, isn't reported as missing from the index.
|
|
46
|
+
*
|
|
47
|
+
* @param {import('@fizzyflow/doublesync').DoubleSyncFolder} folder
|
|
48
|
+
* @param {string} dbPath
|
|
49
|
+
* @returns {Promise<?{changed: string[], added: string[], removed: string[]}>} null when
|
|
50
|
+
* there's no index at `dbPath` at all.
|
|
51
|
+
*/
|
|
52
|
+
async status(folder, dbPath) {
|
|
53
|
+
if (!existsSync(dbPath)) return null;
|
|
54
|
+
const index = new VectorIndex(dbPath, { readonly: true });
|
|
55
|
+
try {
|
|
56
|
+
const indexed = new Set(index.listFiles());
|
|
57
|
+
const changed = [];
|
|
58
|
+
const added = [];
|
|
59
|
+
const seen = new Set();
|
|
60
|
+
for await (const entry of this._scan(folder, index)) {
|
|
61
|
+
// An emptied file is left out of `seen`, so it comes out as removed below —
|
|
62
|
+
// exactly what index() would do to it.
|
|
63
|
+
if (!entry.unchanged && !entry.chunks) continue;
|
|
64
|
+
seen.add(entry.relPath);
|
|
65
|
+
if (entry.unchanged) continue;
|
|
66
|
+
(indexed.has(entry.relPath) ? changed : added).push(entry.relPath);
|
|
67
|
+
}
|
|
68
|
+
const removed = [...indexed].filter((p) => !seen.has(p));
|
|
69
|
+
return { changed, added, removed };
|
|
70
|
+
} finally {
|
|
71
|
+
index.close();
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* The walk both `index()` and `status()` run. Yields one entry per text file:
|
|
77
|
+
* `{relPath, unchanged: true}` when the index already has it at this content hash, or
|
|
78
|
+
* `{relPath, hash, chunks}` otherwise (`chunks` null for a file with no text to embed).
|
|
79
|
+
*/
|
|
80
|
+
async *_scan(folder, index) {
|
|
81
|
+
for await (const { path, file } of folder.walk()) {
|
|
82
|
+
const relPath = path.join('/');
|
|
83
|
+
const bytes = await file.getContent();
|
|
84
|
+
if (!looksLikeText(bytes)) continue;
|
|
85
|
+
|
|
86
|
+
const hash = toHex(await file.getFingerprint());
|
|
87
|
+
if (index.getFileHash(relPath) === hash) {
|
|
88
|
+
yield { relPath, unchanged: true };
|
|
89
|
+
continue;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const text = new TextDecoder('utf-8', { fatal: false }).decode(bytes);
|
|
93
|
+
const chunks = this.chunkingStrategy.chunk(text);
|
|
94
|
+
yield { relPath, hash, chunks: chunks.length > 0 ? chunks : null };
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* @param {import('@fizzyflow/doublesync').DoubleSyncFolder} folder
|
|
100
|
+
* @param {string} dbPath
|
|
101
|
+
* @param {Object} [opts]
|
|
102
|
+
* @param {boolean} [opts.rebuild] - start from an empty index instead of updating this one:
|
|
103
|
+
* every file is embedded again. The one way to switch an index to a different model — and,
|
|
104
|
+
* since it rewrites the whole file, the next push carries the full index again.
|
|
105
|
+
* @returns {Promise<{filesChanged: number, filesSkipped: number, filesRemoved: number, chunksAdded: number}>}
|
|
106
|
+
*/
|
|
107
|
+
async index(folder, dbPath, { rebuild = false } = {}) {
|
|
108
|
+
if (!this.embeddingProvider) throw new Error('Indexer.index: no embeddingProvider configured');
|
|
109
|
+
if (rebuild) {
|
|
110
|
+
rmSync(dbPath, { force: true });
|
|
111
|
+
rmSync(`${dbPath}-journal`, { force: true });
|
|
112
|
+
} else {
|
|
113
|
+
assertSameModel(VectorIndex.readMeta(dbPath), this.embeddingProvider, dbPath);
|
|
114
|
+
}
|
|
115
|
+
const index = new VectorIndex(dbPath);
|
|
116
|
+
try {
|
|
117
|
+
const seenPaths = new Set();
|
|
118
|
+
// One entry per changed/new file, chunked but not yet embedded — embedding happens
|
|
119
|
+
// in one batched call across every changed file below, rather than one call per
|
|
120
|
+
// file, since a single call amortizes the model's own fixed per-call overhead.
|
|
121
|
+
const pending = [];
|
|
122
|
+
let filesSkipped = 0;
|
|
123
|
+
|
|
124
|
+
for await (const entry of this._scan(folder, index)) {
|
|
125
|
+
// A file with no text to embed (emptied since last time) is deliberately left out
|
|
126
|
+
// of seenPaths, so the removal pass below drops its old chunks like a deleted file's.
|
|
127
|
+
if (entry.unchanged) {
|
|
128
|
+
seenPaths.add(entry.relPath);
|
|
129
|
+
filesSkipped++;
|
|
130
|
+
} else if (entry.chunks) {
|
|
131
|
+
seenPaths.add(entry.relPath);
|
|
132
|
+
pending.push(entry);
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
let chunksAdded = 0;
|
|
137
|
+
if (pending.length > 0) {
|
|
138
|
+
const allTexts = pending.flatMap((p) => p.chunks.map((c) => c.text));
|
|
139
|
+
const allVectors = await this.embeddingProvider.embed(allTexts);
|
|
140
|
+
index.initVectorColumn(this.embeddingProvider.dimension);
|
|
141
|
+
|
|
142
|
+
let cursor = 0;
|
|
143
|
+
for (const p of pending) {
|
|
144
|
+
const vectors = allVectors.slice(cursor, cursor + p.chunks.length);
|
|
145
|
+
cursor += p.chunks.length;
|
|
146
|
+
index.replaceFileChunks(p.relPath, p.hash, p.chunks.map((c, i) => ({
|
|
147
|
+
startOffset: c.startOffset,
|
|
148
|
+
endOffset: c.endOffset,
|
|
149
|
+
embedding: vectors[i],
|
|
150
|
+
})));
|
|
151
|
+
chunksAdded += p.chunks.length;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
index.setMeta('model', this.embeddingProvider.modelId);
|
|
155
|
+
if (this.embeddingProvider.embedder) index.setMeta('embedder', this.embeddingProvider.embedder);
|
|
156
|
+
index.setMeta('dimension', String(this.embeddingProvider.dimension));
|
|
157
|
+
if (this.embeddingProvider.dtype) index.setMeta('dtype', this.embeddingProvider.dtype);
|
|
158
|
+
index.setMeta('chunk_size', String(this.chunkingStrategy.size ?? ''));
|
|
159
|
+
index.setMeta('chunk_overlap', String(this.chunkingStrategy.overlap ?? ''));
|
|
160
|
+
} else {
|
|
161
|
+
// Nothing changed this run — if the index already has content from a previous
|
|
162
|
+
// run, the vector column still needs registering on this fresh connection (see
|
|
163
|
+
// VectorIndex.initVectorColumn's own doc) before removeFile()'s bookkeeping below
|
|
164
|
+
// touches the same tables. Not required for a genuinely empty/new index — there's
|
|
165
|
+
// no dimension to know yet, and nothing to query either.
|
|
166
|
+
const existingDim = index.getMeta('dimension');
|
|
167
|
+
if (existingDim) index.initVectorColumn(Number(existingDim));
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
let filesRemoved = 0;
|
|
171
|
+
for (const indexedPath of index.listFiles()) {
|
|
172
|
+
if (!seenPaths.has(indexedPath)) {
|
|
173
|
+
index.removeFile(indexedPath);
|
|
174
|
+
filesRemoved++;
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
return { filesChanged: pending.length, filesSkipped, filesRemoved, chunksAdded };
|
|
179
|
+
} finally {
|
|
180
|
+
index.close();
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Thrown when `index()` would add one model's vectors to an index built with another — two
|
|
187
|
+
* models' vectors aren't comparable (and usually aren't even the same length), so mixing them
|
|
188
|
+
* would quietly break every search. `rebuild` is the way through.
|
|
189
|
+
*/
|
|
190
|
+
export class ModelMismatchError extends Error {
|
|
191
|
+
constructor(dbPath, indexModel, providerModel) {
|
|
192
|
+
super(
|
|
193
|
+
`${dbPath} was built with ${indexModel}, not ${providerModel} — one index can't mix two models.`
|
|
194
|
+
);
|
|
195
|
+
this.name = 'ModelMismatchError';
|
|
196
|
+
this.indexModel = indexModel;
|
|
197
|
+
this.providerModel = providerModel;
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
function assertSameModel(meta, provider, dbPath) {
|
|
202
|
+
if (!meta?.model) return;
|
|
203
|
+
if (meta.model !== provider.modelId) throw new ModelMismatchError(dbPath, meta.model, provider.modelId);
|
|
204
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import VectorIndex from './VectorIndex.js';
|
|
2
|
+
import { LEGACY_DTYPE, EMBEDDER_TRANSFORMERS, embedderOfIndex } from './schema.js';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Queries an already-built `VectorIndex`. Deliberately the only piece that needs an embedding
|
|
6
|
+
* model loaded on the *searcher's* machine — and only to embed the query string itself, never
|
|
7
|
+
* the corpus, which is the entire point of pushing the index alongside the content: whoever
|
|
8
|
+
* pulls it gets every embedding for free, already paid for by whoever ran `Indexer` first.
|
|
9
|
+
*/
|
|
10
|
+
export default class Searcher {
|
|
11
|
+
/**
|
|
12
|
+
* @param {Object} params
|
|
13
|
+
* @param {(opts: {model: string, embedder: string, dtype: ?string}) => Promise<any>} [params.createProvider] -
|
|
14
|
+
* builds the provider the index's own metadata asks for (see providers.js) — the common
|
|
15
|
+
* case, opening someone else's index.
|
|
16
|
+
* @param {any} [params.embeddingProvider] - use this one instead, whatever the index recorded
|
|
17
|
+
* (tests; a custom provider). Its vectors must still match the index's dimension.
|
|
18
|
+
*/
|
|
19
|
+
constructor({ createProvider = null, embeddingProvider = null } = {}) {
|
|
20
|
+
if (!createProvider && !embeddingProvider) {
|
|
21
|
+
throw new Error('Searcher: pass createProvider or embeddingProvider');
|
|
22
|
+
}
|
|
23
|
+
this._createProvider = createProvider;
|
|
24
|
+
this._embeddingProvider = embeddingProvider;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* @param {string} dbPath
|
|
29
|
+
* @param {string} query
|
|
30
|
+
* @param {number} [k=5]
|
|
31
|
+
* @returns {Promise<{path: string, chunkIndex: number, startOffset: number, endOffset: number, distance: number}[]>}
|
|
32
|
+
*/
|
|
33
|
+
async search(dbPath, query, k = 5) {
|
|
34
|
+
// Read-only: the index is a file that gets pushed, and searching it must not change it.
|
|
35
|
+
const index = new VectorIndex(dbPath, { readonly: true });
|
|
36
|
+
try {
|
|
37
|
+
const dim = Number(index.getMeta('dimension'));
|
|
38
|
+
if (!dim) {
|
|
39
|
+
throw new Error(`${dbPath}: no index metadata found — run \`mnemonad index\` first`);
|
|
40
|
+
}
|
|
41
|
+
index.initVectorColumn(dim);
|
|
42
|
+
|
|
43
|
+
const provider = this._embeddingProvider || await this._createProvider(describeIndexModel(index));
|
|
44
|
+
const [queryVector] = await provider.embed([query]);
|
|
45
|
+
if (queryVector.length !== dim) {
|
|
46
|
+
throw new Error(
|
|
47
|
+
`Query embedding is ${queryVector.length}-dimensional but this index is ${dim}-dimensional — ` +
|
|
48
|
+
`the embedding model must match the one used to build it (recorded model: ${index.getMeta('model')})`
|
|
49
|
+
);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
return index.search(queryVector, k);
|
|
53
|
+
} finally {
|
|
54
|
+
index.close();
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** The model an index says it was built with, in the shape createProvider takes. */
|
|
60
|
+
export function describeIndexModel(index) {
|
|
61
|
+
const meta = { model: index.getMeta('model'), embedder: index.getMeta('embedder') };
|
|
62
|
+
const embedder = embedderOfIndex(meta);
|
|
63
|
+
return {
|
|
64
|
+
model: meta.model,
|
|
65
|
+
embedder,
|
|
66
|
+
dtype: embedder === EMBEDDER_TRANSFORMERS ? (index.getMeta('dtype') || LEGACY_DTYPE) : null,
|
|
67
|
+
};
|
|
68
|
+
}
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
import { existsSync } from 'node:fs';
|
|
2
|
+
import { createRequire } from 'node:module';
|
|
3
|
+
import Database from 'better-sqlite3';
|
|
4
|
+
import { VECTOR_INIT_SQL, SEARCH_SQL, vectorInitOptions, overfetchFor, mapSearchRow } from './schema.js';
|
|
5
|
+
|
|
6
|
+
// @sqliteai/sqlite-vector's own ESM build (dist/index.mjs) cannot find its platform binary
|
|
7
|
+
// package from a genuine ESM context — it tries a bare `require(packageName)` via a bundler-
|
|
8
|
+
// injected shim that only works when a real CJS `require` is already in scope, which never
|
|
9
|
+
// happens in native ESM (confirmed live: `import { getExtensionPath } from
|
|
10
|
+
// '@sqliteai/sqlite-vector'` throws `ExtensionNotFoundError` even when the platform package
|
|
11
|
+
// is correctly installed). `createRequire` is the standard Node workaround — it resolves the
|
|
12
|
+
// package's CJS build instead, where a real `require` exists and the same lookup works.
|
|
13
|
+
const require = createRequire(import.meta.url);
|
|
14
|
+
const { getExtensionPath } = require('@sqliteai/sqlite-vector');
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* A single, diff-friendly SQLite file holding embeddings for one synced folder — meant to be
|
|
18
|
+
* pushed alongside the folder's own content (a plain file, not dot-prefixed, so it rides
|
|
19
|
+
* through `push`/`pull`/`diff` unmodified) so a pull is immediately searchable with no
|
|
20
|
+
* re-embedding.
|
|
21
|
+
*
|
|
22
|
+
* The whole schema is designed around one constraint, verified empirically (not assumed):
|
|
23
|
+
* inserting rows into a plain table with a monotonic rowid touches only new/near-full pages,
|
|
24
|
+
* which content-defined chunking recognizes as mostly-unchanged on a re-push. A real
|
|
25
|
+
* measurement (200 existing rows, 5 new ones) showed ~2.9x overhead versus the new data's
|
|
26
|
+
* raw size — nowhere near a full-file rewrite, but *not* literally zero-touch either. What
|
|
27
|
+
* actually breaks this property: `VACUUM` (copies and rewrites every byte — never call it
|
|
28
|
+
* during routine reindexing) and `UPDATE`/`DELETE` on the (large, embedding-heavy)
|
|
29
|
+
* `search_index` table. `search_index_files` is deliberately the one place this class *does*
|
|
30
|
+
* mutate rows in place — it holds one row per file, so its churn is proportional to how many
|
|
31
|
+
* files actually changed, not to corpus size, and paying that small cost is what makes
|
|
32
|
+
* "which chunks are current" a correct query instead of a guess (see search_index_files' own
|
|
33
|
+
* doc below).
|
|
34
|
+
*/
|
|
35
|
+
export default class VectorIndex {
|
|
36
|
+
/**
|
|
37
|
+
* @param {string} dbPath - path to the SQLite file. Created if it doesn't exist yet
|
|
38
|
+
* (unless `readonly`).
|
|
39
|
+
* @param {Object} [options]
|
|
40
|
+
* @param {boolean} [options.readonly=false] - open an existing index without writing
|
|
41
|
+
* anything to it — not even the schema check below. For inspecting an index that's about
|
|
42
|
+
* to be pushed (see Indexer.status()), where touching the file would change what's pushed.
|
|
43
|
+
*/
|
|
44
|
+
constructor(dbPath, { readonly = false } = {}) {
|
|
45
|
+
this.path = dbPath;
|
|
46
|
+
this._db = readonly
|
|
47
|
+
? new Database(dbPath, { readonly: true, fileMustExist: true })
|
|
48
|
+
: new Database(dbPath);
|
|
49
|
+
// Deliberately not enabling WAL mode: WAL leaves a separate `-wal`/`-shm` file holding
|
|
50
|
+
// uncommitted pages, which would need an explicit checkpoint before every push (a real
|
|
51
|
+
// operational footgun for anyone forgetting it). Left at SQLite's own default rollback
|
|
52
|
+
// journal instead — this class always closes the database (see close()) before anything
|
|
53
|
+
// downstream would push it, so there's never a mid-transaction state to leave behind.
|
|
54
|
+
this._db.loadExtension(getExtensionPath());
|
|
55
|
+
if (!readonly) this._ensureSchema();
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
_ensureSchema() {
|
|
59
|
+
this._db.exec(`
|
|
60
|
+
CREATE TABLE IF NOT EXISTS search_index (
|
|
61
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
62
|
+
path TEXT NOT NULL,
|
|
63
|
+
chunk_index INTEGER NOT NULL,
|
|
64
|
+
content_hash TEXT NOT NULL,
|
|
65
|
+
start_offset INTEGER NOT NULL,
|
|
66
|
+
end_offset INTEGER NOT NULL,
|
|
67
|
+
embedding BLOB NOT NULL
|
|
68
|
+
)
|
|
69
|
+
`);
|
|
70
|
+
// One row per file, deliberately mutable (see class doc) — "content_hash" here is what
|
|
71
|
+
// decides which rows in the (append-only) search_index table above actually count as
|
|
72
|
+
// current: a chunk only counts if its own content_hash matches this table's row for the
|
|
73
|
+
// same path. That's what lets a stale reindex's chunks keep existing physically (pure
|
|
74
|
+
// insert, never deleted) while still being correctly excluded from search results —
|
|
75
|
+
// see search()'s own join.
|
|
76
|
+
this._db.exec(`
|
|
77
|
+
CREATE TABLE IF NOT EXISTS search_index_files (
|
|
78
|
+
path TEXT PRIMARY KEY,
|
|
79
|
+
content_hash TEXT NOT NULL,
|
|
80
|
+
chunk_count INTEGER NOT NULL
|
|
81
|
+
)
|
|
82
|
+
`);
|
|
83
|
+
this._db.exec(`
|
|
84
|
+
CREATE TABLE IF NOT EXISTS search_index_meta (
|
|
85
|
+
key TEXT PRIMARY KEY,
|
|
86
|
+
value TEXT NOT NULL
|
|
87
|
+
)
|
|
88
|
+
`);
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Registers `search_index`/`embedding` as a vector column with sqlite-vector — required on
|
|
93
|
+
* every fresh connection before `vector_full_scan` etc. work (confirmed live: omitting this
|
|
94
|
+
* fails with "vector_full_scan: unable to retrieve context", not a clearer error). Safe to
|
|
95
|
+
* call more than once; last call wins.
|
|
96
|
+
*
|
|
97
|
+
* @param {number} dim
|
|
98
|
+
* @param {'FLOAT32'|'INT8'} [type='FLOAT32']
|
|
99
|
+
*/
|
|
100
|
+
initVectorColumn(dim, type = 'FLOAT32') {
|
|
101
|
+
this._dim = dim;
|
|
102
|
+
this._vectorType = type;
|
|
103
|
+
this._db.prepare(VECTOR_INIT_SQL).run(vectorInitOptions(dim, type));
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/** @returns {?string} the file's currently-indexed content hash, or null if never indexed. */
|
|
107
|
+
getFileHash(path) {
|
|
108
|
+
const row = this._db.prepare('SELECT content_hash FROM search_index_files WHERE path = ?').get(path);
|
|
109
|
+
return row?.content_hash ?? null;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** @returns {string[]} every path currently recorded in the index (indexed, not necessarily still on disk). */
|
|
113
|
+
listFiles() {
|
|
114
|
+
return this._db.prepare('SELECT path FROM search_index_files').all().map((r) => r.path);
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Replaces one file's chunks. Never touches the file's *old* rows in `search_index` — it
|
|
119
|
+
* only ever inserts new ones (see class doc for why) — and upserts the single
|
|
120
|
+
* `search_index_files` row that makes the old ones stop counting as current.
|
|
121
|
+
*
|
|
122
|
+
* @param {string} path
|
|
123
|
+
* @param {string} contentHash - the file's new content hash (same hashing this package's
|
|
124
|
+
* caller already uses to detect the file changed in the first place).
|
|
125
|
+
* @param {{startOffset: number, endOffset: number, embedding: Float32Array}[]} chunks
|
|
126
|
+
*/
|
|
127
|
+
replaceFileChunks(path, contentHash, chunks) {
|
|
128
|
+
const insert = this._db.prepare(`
|
|
129
|
+
INSERT INTO search_index (path, chunk_index, content_hash, start_offset, end_offset, embedding)
|
|
130
|
+
VALUES (?, ?, ?, ?, ?, ?)
|
|
131
|
+
`);
|
|
132
|
+
const upsertFile = this._db.prepare(`
|
|
133
|
+
INSERT INTO search_index_files (path, content_hash, chunk_count) VALUES (?, ?, ?)
|
|
134
|
+
ON CONFLICT(path) DO UPDATE SET content_hash = excluded.content_hash, chunk_count = excluded.chunk_count
|
|
135
|
+
`);
|
|
136
|
+
const tx = this._db.transaction(() => {
|
|
137
|
+
chunks.forEach((chunk, index) => {
|
|
138
|
+
insert.run(path, index, contentHash, chunk.startOffset, chunk.endOffset, Buffer.from(chunk.embedding.buffer, chunk.embedding.byteOffset, chunk.embedding.byteLength));
|
|
139
|
+
});
|
|
140
|
+
upsertFile.run(path, contentHash, chunks.length);
|
|
141
|
+
});
|
|
142
|
+
tx();
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** A file no longer exists in the tree — stop counting its (still physically present) chunks as current. */
|
|
146
|
+
removeFile(path) {
|
|
147
|
+
this._db.prepare('DELETE FROM search_index_files WHERE path = ?').run(path);
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
setMeta(key, value) {
|
|
151
|
+
this._db.prepare(`
|
|
152
|
+
INSERT INTO search_index_meta (key, value) VALUES (?, ?)
|
|
153
|
+
ON CONFLICT(key) DO UPDATE SET value = excluded.value
|
|
154
|
+
`).run(key, String(value));
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* An existing index's metadata, read without writing anything — or null when there's no
|
|
159
|
+
* index at `dbPath` (or it has no metadata yet).
|
|
160
|
+
* @returns {?Object<string, string>}
|
|
161
|
+
*/
|
|
162
|
+
static readMeta(dbPath) {
|
|
163
|
+
if (!existsSync(dbPath)) return null;
|
|
164
|
+
const index = new VectorIndex(dbPath, { readonly: true });
|
|
165
|
+
try {
|
|
166
|
+
const rows = index._db.prepare('SELECT key, value FROM search_index_meta').all();
|
|
167
|
+
return rows.length ? Object.fromEntries(rows.map((r) => [r.key, r.value])) : null;
|
|
168
|
+
} catch {
|
|
169
|
+
// No meta table: not an index this package built.
|
|
170
|
+
return null;
|
|
171
|
+
} finally {
|
|
172
|
+
index.close();
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/** @returns {?string} */
|
|
177
|
+
getMeta(key) {
|
|
178
|
+
return this._db.prepare('SELECT value FROM search_index_meta WHERE key = ?').get(key)?.value ?? null;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Nearest-neighbor search, current chunks only. Over-fetches from `vector_full_scan` (exact,
|
|
183
|
+
* brute-force — appropriate for the corpus sizes a synced folder actually has; no ANN/
|
|
184
|
+
* partitioning index, which would break the diffability this whole design depends on) since
|
|
185
|
+
* some of its top candidates may turn out to be stale once joined against
|
|
186
|
+
* `search_index_files`, and stops short of the caller's requested `k` otherwise.
|
|
187
|
+
*
|
|
188
|
+
* @param {Float32Array} queryEmbedding
|
|
189
|
+
* @param {number} k
|
|
190
|
+
* @returns {{path: string, chunkIndex: number, startOffset: number, endOffset: number, distance: number}[]}
|
|
191
|
+
*/
|
|
192
|
+
search(queryEmbedding, k) {
|
|
193
|
+
const queryBuf = Buffer.from(queryEmbedding.buffer, queryEmbedding.byteOffset, queryEmbedding.byteLength);
|
|
194
|
+
// BigInt, not a plain number: better-sqlite3 binds every JS `number` parameter as SQLite
|
|
195
|
+
// REAL (confirmed live — even a whole number like 64), but vector_full_scan's own count
|
|
196
|
+
// argument strictly requires INTEGER and rejects REAL outright rather than coercing it.
|
|
197
|
+
// (The WASM build binds whole numbers as INTEGER on its own — see WasmIndexReader.)
|
|
198
|
+
return this._db
|
|
199
|
+
.prepare(SEARCH_SQL)
|
|
200
|
+
.all(queryBuf, BigInt(overfetchFor(k)), BigInt(k))
|
|
201
|
+
.map(mapSearchRow);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* The deliberate, rare, full-rewrite operation — drops every chunk that isn't current
|
|
206
|
+
* (superseded by a later reindex of the same file, or belonging to a file removed from the
|
|
207
|
+
* index entirely) and reclaims the space. Costs a CDC-diff equivalent to a full snapshot —
|
|
208
|
+
* same trade routine reindexing exists specifically to avoid — so this is opt-in, not run
|
|
209
|
+
* automatically. Mirrors the CLI's own `compact` command for stream history.
|
|
210
|
+
*/
|
|
211
|
+
compact() {
|
|
212
|
+
this._db.exec(`
|
|
213
|
+
DELETE FROM search_index
|
|
214
|
+
WHERE NOT EXISTS (
|
|
215
|
+
SELECT 1 FROM search_index_files f
|
|
216
|
+
WHERE f.path = search_index.path AND f.content_hash = search_index.content_hash
|
|
217
|
+
)
|
|
218
|
+
`);
|
|
219
|
+
this._db.exec('VACUUM');
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
close() {
|
|
223
|
+
this._db.close();
|
|
224
|
+
}
|
|
225
|
+
}
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import WasmIndexReader from './WasmIndexReader.js';
|
|
2
|
+
import { LEGACY_DTYPE, EMBEDDER_TRANSFORMERS, embedderOfIndex } from '../schema.js';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* The browser's `Searcher`: holds one open index (from bytes, not a path) and the embedding
|
|
6
|
+
* model that matches it, so repeated queries pay for neither again. `open()` again with new
|
|
7
|
+
* bytes when the index file changes; the model is only reloaded if the new index was built
|
|
8
|
+
* with a different one.
|
|
9
|
+
*
|
|
10
|
+
* Which provider to build for which index is the caller's job (`createProvider`), so this
|
|
11
|
+
* class — and the browser entry around it — never imports a model runtime itself: the
|
|
12
|
+
* explorer builds the static provider from its own files source, and loads transformers.js
|
|
13
|
+
* only for an index that needs it.
|
|
14
|
+
*/
|
|
15
|
+
export default class BrowserSearcher {
|
|
16
|
+
/**
|
|
17
|
+
* @param {Object} params
|
|
18
|
+
* @param {any} params.sqlite3 - an initialized `@sqliteai/sqlite-wasm` module.
|
|
19
|
+
* @param {(opts: {model: string, embedder: string, dtype: ?string, onProgress: ?Function}) => (any|Promise<any>)} [params.createProvider] -
|
|
20
|
+
* builds the provider an index's metadata asks for.
|
|
21
|
+
* @param {any} [params.embeddingProvider] - use this one for every index instead (tests).
|
|
22
|
+
* @param {(event: Object) => void} [params.onModelProgress] - passed to createProvider.
|
|
23
|
+
*/
|
|
24
|
+
constructor({ sqlite3, createProvider = null, embeddingProvider = null, onModelProgress = null } = {}) {
|
|
25
|
+
if (!sqlite3) throw new Error('BrowserSearcher: sqlite3 module is required');
|
|
26
|
+
if (!createProvider && !embeddingProvider) {
|
|
27
|
+
throw new Error('BrowserSearcher: pass createProvider or embeddingProvider');
|
|
28
|
+
}
|
|
29
|
+
this._sqlite3 = sqlite3;
|
|
30
|
+
this._createProvider = createProvider;
|
|
31
|
+
this._onModelProgress = onModelProgress;
|
|
32
|
+
this._fixedProvider = embeddingProvider;
|
|
33
|
+
this._provider = embeddingProvider;
|
|
34
|
+
this._providerKey = null;
|
|
35
|
+
this._reader = null;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** @param {Uint8Array} bytes - the whole search_index.db file. */
|
|
39
|
+
open(bytes) {
|
|
40
|
+
const reader = new WasmIndexReader(this._sqlite3, bytes);
|
|
41
|
+
const dim = Number(reader.getMeta('dimension'));
|
|
42
|
+
if (!dim) {
|
|
43
|
+
reader.close();
|
|
44
|
+
throw new Error('This search index has no metadata — it was never built, or is empty.');
|
|
45
|
+
}
|
|
46
|
+
reader.initVectorColumn(dim);
|
|
47
|
+
this.close();
|
|
48
|
+
this._reader = reader;
|
|
49
|
+
this._dim = dim;
|
|
50
|
+
this._model = reader.getMeta('model');
|
|
51
|
+
this._embedder = embedderOfIndex({ model: this._model, embedder: reader.getMeta('embedder') });
|
|
52
|
+
this._dtype = this._embedder === EMBEDDER_TRANSFORMERS ? (reader.getMeta('dtype') || LEGACY_DTYPE) : null;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
get isOpen() {
|
|
56
|
+
return this._reader !== null;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** @returns {{model: ?string, embedder: string, dtype: ?string, dimension: number}} */
|
|
60
|
+
get info() {
|
|
61
|
+
return { model: this._model, embedder: this._embedder, dtype: this._dtype, dimension: this._dim };
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/** @returns {Map<string, string>} see WasmIndexReader.getFileHashes */
|
|
65
|
+
getFileHashes() {
|
|
66
|
+
this._requireOpen();
|
|
67
|
+
return this._reader.getFileHashes();
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** @returns {boolean} whether the model the open index needs is already loaded. */
|
|
71
|
+
get isModelLoaded() {
|
|
72
|
+
if (!this._reader) return false;
|
|
73
|
+
if (this._fixedProvider) return true;
|
|
74
|
+
return this._providerKey === this._key() && !!this._provider?.isLoaded;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Loads the embedding model now (the slow, one-time part — a download on first use) so a
|
|
78
|
+
* caller can show that as its own step instead of as a slow first query. */
|
|
79
|
+
async loadModel() {
|
|
80
|
+
this._requireOpen();
|
|
81
|
+
const provider = await this._providerForIndex();
|
|
82
|
+
if (typeof provider.load === 'function') await provider.load();
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* @param {string} query
|
|
87
|
+
* @param {number} [k=5]
|
|
88
|
+
*/
|
|
89
|
+
async search(query, k = 5) {
|
|
90
|
+
this._requireOpen();
|
|
91
|
+
const provider = await this._providerForIndex();
|
|
92
|
+
const [queryVector] = await provider.embed([query]);
|
|
93
|
+
if (queryVector.length !== this._dim) {
|
|
94
|
+
throw new Error(
|
|
95
|
+
`Query embedding is ${queryVector.length}-dimensional but this index is ${this._dim}-dimensional — ` +
|
|
96
|
+
`the embedding model must match the one used to build it (recorded model: ${this._model})`
|
|
97
|
+
);
|
|
98
|
+
}
|
|
99
|
+
return this._reader.search(queryVector, k);
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
close() {
|
|
103
|
+
if (this._reader) {
|
|
104
|
+
this._reader.close();
|
|
105
|
+
this._reader = null;
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
_key() {
|
|
110
|
+
return `${this._embedder}|${this._model}|${this._dtype}`;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
async _providerForIndex() {
|
|
114
|
+
if (this._fixedProvider) return this._fixedProvider;
|
|
115
|
+
const key = this._key();
|
|
116
|
+
if (this._providerKey !== key) {
|
|
117
|
+
this._provider = await this._createProvider({
|
|
118
|
+
model: this._model,
|
|
119
|
+
embedder: this._embedder,
|
|
120
|
+
dtype: this._dtype,
|
|
121
|
+
onProgress: this._onModelProgress,
|
|
122
|
+
});
|
|
123
|
+
this._providerKey = key;
|
|
124
|
+
}
|
|
125
|
+
return this._provider;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
_requireOpen() {
|
|
129
|
+
if (!this._reader) throw new Error('BrowserSearcher: call open(bytes) first');
|
|
130
|
+
}
|
|
131
|
+
}
|