mnemonad-cli 0.1.1 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,204 @@
1
+ import { existsSync, rmSync } from 'node:fs';
2
+ import VectorIndex from './VectorIndex.js';
3
+ import TextWindowChunkingStrategy from './chunking/TextWindowChunkingStrategy.js';
4
+
5
+ /** First N bytes checked for a null byte — the standard, cheap heuristic (same one git and
6
+ * most diff tools use) for "this is almost certainly binary, don't try to embed it as text". */
7
+ const BINARY_SNIFF_BYTES = 8000;
8
+
9
+ function looksLikeText(bytes) {
10
+ const n = Math.min(bytes.length, BINARY_SNIFF_BYTES);
11
+ for (let i = 0; i < n; i++) {
12
+ if (bytes[i] === 0) return false;
13
+ }
14
+ return true;
15
+ }
16
+
17
+ function toHex(bytes) {
18
+ return Array.from(bytes, (b) => b.toString(16).padStart(2, '0')).join('');
19
+ }
20
+
21
+ /**
22
+ * Builds/updates a `VectorIndex` from a folder — anything shaped like
23
+ * `@fizzyflow/doublesync`'s `DoubleSyncFolder` (has an async `walk()` generator), which
24
+ * covers the CLI's real-disk `FSFolder` and any in-memory folder the same way. Incremental by
25
+ * construction: a file whose content hash already matches what's stored is skipped entirely,
26
+ * no re-embedding, no re-chunking, not even a read past the hash check.
27
+ */
28
+ export default class Indexer {
29
+ /**
30
+ * @param {Object} [params]
31
+ * @param {?import('./embeddings/EmbeddingProvider.js').default} [params.embeddingProvider] -
32
+ * required by `index()`; `status()` never embeds, so it can go without one (see
33
+ * providers.js's createProvider for the usual way to get one).
34
+ * @param {import('./chunking/ChunkingStrategy.js').default} [params.chunkingStrategy]
35
+ */
36
+ constructor({ embeddingProvider = null, chunkingStrategy } = {}) {
37
+ this.embeddingProvider = embeddingProvider;
38
+ this.chunkingStrategy = chunkingStrategy || new TextWindowChunkingStrategy();
39
+ }
40
+
41
+ /**
42
+ * What `index()` would do to `dbPath`, without doing it: no embedding, no model load, and
43
+ * the database opened read-only. Uses the same walk and the same rules as `index()` (see
44
+ * `_scan`), so it agrees exactly on which files count — a binary or empty file, which
45
+ * `index()` never records, isn't reported as missing from the index.
46
+ *
47
+ * @param {import('@fizzyflow/doublesync').DoubleSyncFolder} folder
48
+ * @param {string} dbPath
49
+ * @returns {Promise<?{changed: string[], added: string[], removed: string[]}>} null when
50
+ * there's no index at `dbPath` at all.
51
+ */
52
+ async status(folder, dbPath) {
53
+ if (!existsSync(dbPath)) return null;
54
+ const index = new VectorIndex(dbPath, { readonly: true });
55
+ try {
56
+ const indexed = new Set(index.listFiles());
57
+ const changed = [];
58
+ const added = [];
59
+ const seen = new Set();
60
+ for await (const entry of this._scan(folder, index)) {
61
+ // An emptied file is left out of `seen`, so it comes out as removed below —
62
+ // exactly what index() would do to it.
63
+ if (!entry.unchanged && !entry.chunks) continue;
64
+ seen.add(entry.relPath);
65
+ if (entry.unchanged) continue;
66
+ (indexed.has(entry.relPath) ? changed : added).push(entry.relPath);
67
+ }
68
+ const removed = [...indexed].filter((p) => !seen.has(p));
69
+ return { changed, added, removed };
70
+ } finally {
71
+ index.close();
72
+ }
73
+ }
74
+
75
+ /**
76
+ * The walk both `index()` and `status()` run. Yields one entry per text file:
77
+ * `{relPath, unchanged: true}` when the index already has it at this content hash, or
78
+ * `{relPath, hash, chunks}` otherwise (`chunks` null for a file with no text to embed).
79
+ */
80
+ async *_scan(folder, index) {
81
+ for await (const { path, file } of folder.walk()) {
82
+ const relPath = path.join('/');
83
+ const bytes = await file.getContent();
84
+ if (!looksLikeText(bytes)) continue;
85
+
86
+ const hash = toHex(await file.getFingerprint());
87
+ if (index.getFileHash(relPath) === hash) {
88
+ yield { relPath, unchanged: true };
89
+ continue;
90
+ }
91
+
92
+ const text = new TextDecoder('utf-8', { fatal: false }).decode(bytes);
93
+ const chunks = this.chunkingStrategy.chunk(text);
94
+ yield { relPath, hash, chunks: chunks.length > 0 ? chunks : null };
95
+ }
96
+ }
97
+
98
+ /**
99
+ * @param {import('@fizzyflow/doublesync').DoubleSyncFolder} folder
100
+ * @param {string} dbPath
101
+ * @param {Object} [opts]
102
+ * @param {boolean} [opts.rebuild] - start from an empty index instead of updating this one:
103
+ * every file is embedded again. The one way to switch an index to a different model — and,
104
+ * since it rewrites the whole file, the next push carries the full index again.
105
+ * @returns {Promise<{filesChanged: number, filesSkipped: number, filesRemoved: number, chunksAdded: number}>}
106
+ */
107
+ async index(folder, dbPath, { rebuild = false } = {}) {
108
+ if (!this.embeddingProvider) throw new Error('Indexer.index: no embeddingProvider configured');
109
+ if (rebuild) {
110
+ rmSync(dbPath, { force: true });
111
+ rmSync(`${dbPath}-journal`, { force: true });
112
+ } else {
113
+ assertSameModel(VectorIndex.readMeta(dbPath), this.embeddingProvider, dbPath);
114
+ }
115
+ const index = new VectorIndex(dbPath);
116
+ try {
117
+ const seenPaths = new Set();
118
+ // One entry per changed/new file, chunked but not yet embedded — embedding happens
119
+ // in one batched call across every changed file below, rather than one call per
120
+ // file, since a single call amortizes the model's own fixed per-call overhead.
121
+ const pending = [];
122
+ let filesSkipped = 0;
123
+
124
+ for await (const entry of this._scan(folder, index)) {
125
+ // A file with no text to embed (emptied since last time) is deliberately left out
126
+ // of seenPaths, so the removal pass below drops its old chunks like a deleted file's.
127
+ if (entry.unchanged) {
128
+ seenPaths.add(entry.relPath);
129
+ filesSkipped++;
130
+ } else if (entry.chunks) {
131
+ seenPaths.add(entry.relPath);
132
+ pending.push(entry);
133
+ }
134
+ }
135
+
136
+ let chunksAdded = 0;
137
+ if (pending.length > 0) {
138
+ const allTexts = pending.flatMap((p) => p.chunks.map((c) => c.text));
139
+ const allVectors = await this.embeddingProvider.embed(allTexts);
140
+ index.initVectorColumn(this.embeddingProvider.dimension);
141
+
142
+ let cursor = 0;
143
+ for (const p of pending) {
144
+ const vectors = allVectors.slice(cursor, cursor + p.chunks.length);
145
+ cursor += p.chunks.length;
146
+ index.replaceFileChunks(p.relPath, p.hash, p.chunks.map((c, i) => ({
147
+ startOffset: c.startOffset,
148
+ endOffset: c.endOffset,
149
+ embedding: vectors[i],
150
+ })));
151
+ chunksAdded += p.chunks.length;
152
+ }
153
+
154
+ index.setMeta('model', this.embeddingProvider.modelId);
155
+ if (this.embeddingProvider.embedder) index.setMeta('embedder', this.embeddingProvider.embedder);
156
+ index.setMeta('dimension', String(this.embeddingProvider.dimension));
157
+ if (this.embeddingProvider.dtype) index.setMeta('dtype', this.embeddingProvider.dtype);
158
+ index.setMeta('chunk_size', String(this.chunkingStrategy.size ?? ''));
159
+ index.setMeta('chunk_overlap', String(this.chunkingStrategy.overlap ?? ''));
160
+ } else {
161
+ // Nothing changed this run — if the index already has content from a previous
162
+ // run, the vector column still needs registering on this fresh connection (see
163
+ // VectorIndex.initVectorColumn's own doc) before removeFile()'s bookkeeping below
164
+ // touches the same tables. Not required for a genuinely empty/new index — there's
165
+ // no dimension to know yet, and nothing to query either.
166
+ const existingDim = index.getMeta('dimension');
167
+ if (existingDim) index.initVectorColumn(Number(existingDim));
168
+ }
169
+
170
+ let filesRemoved = 0;
171
+ for (const indexedPath of index.listFiles()) {
172
+ if (!seenPaths.has(indexedPath)) {
173
+ index.removeFile(indexedPath);
174
+ filesRemoved++;
175
+ }
176
+ }
177
+
178
+ return { filesChanged: pending.length, filesSkipped, filesRemoved, chunksAdded };
179
+ } finally {
180
+ index.close();
181
+ }
182
+ }
183
+ }
184
+
185
+ /**
186
+ * Thrown when `index()` would add one model's vectors to an index built with another — two
187
+ * models' vectors aren't comparable (and usually aren't even the same length), so mixing them
188
+ * would quietly break every search. `rebuild` is the way through.
189
+ */
190
+ export class ModelMismatchError extends Error {
191
+ constructor(dbPath, indexModel, providerModel) {
192
+ super(
193
+ `${dbPath} was built with ${indexModel}, not ${providerModel} — one index can't mix two models.`
194
+ );
195
+ this.name = 'ModelMismatchError';
196
+ this.indexModel = indexModel;
197
+ this.providerModel = providerModel;
198
+ }
199
+ }
200
+
201
+ function assertSameModel(meta, provider, dbPath) {
202
+ if (!meta?.model) return;
203
+ if (meta.model !== provider.modelId) throw new ModelMismatchError(dbPath, meta.model, provider.modelId);
204
+ }
@@ -0,0 +1,68 @@
1
+ import VectorIndex from './VectorIndex.js';
2
+ import { LEGACY_DTYPE, EMBEDDER_TRANSFORMERS, embedderOfIndex } from './schema.js';
3
+
4
+ /**
5
+ * Queries an already-built `VectorIndex`. Deliberately the only piece that needs an embedding
6
+ * model loaded on the *searcher's* machine — and only to embed the query string itself, never
7
+ * the corpus, which is the entire point of pushing the index alongside the content: whoever
8
+ * pulls it gets every embedding for free, already paid for by whoever ran `Indexer` first.
9
+ */
10
+ export default class Searcher {
11
+ /**
12
+ * @param {Object} params
13
+ * @param {(opts: {model: string, embedder: string, dtype: ?string}) => Promise<any>} [params.createProvider] -
14
+ * builds the provider the index's own metadata asks for (see providers.js) — the common
15
+ * case, opening someone else's index.
16
+ * @param {any} [params.embeddingProvider] - use this one instead, whatever the index recorded
17
+ * (tests; a custom provider). Its vectors must still match the index's dimension.
18
+ */
19
+ constructor({ createProvider = null, embeddingProvider = null } = {}) {
20
+ if (!createProvider && !embeddingProvider) {
21
+ throw new Error('Searcher: pass createProvider or embeddingProvider');
22
+ }
23
+ this._createProvider = createProvider;
24
+ this._embeddingProvider = embeddingProvider;
25
+ }
26
+
27
+ /**
28
+ * @param {string} dbPath
29
+ * @param {string} query
30
+ * @param {number} [k=5]
31
+ * @returns {Promise<{path: string, chunkIndex: number, startOffset: number, endOffset: number, distance: number}[]>}
32
+ */
33
+ async search(dbPath, query, k = 5) {
34
+ // Read-only: the index is a file that gets pushed, and searching it must not change it.
35
+ const index = new VectorIndex(dbPath, { readonly: true });
36
+ try {
37
+ const dim = Number(index.getMeta('dimension'));
38
+ if (!dim) {
39
+ throw new Error(`${dbPath}: no index metadata found — run \`mnemonad index\` first`);
40
+ }
41
+ index.initVectorColumn(dim);
42
+
43
+ const provider = this._embeddingProvider || await this._createProvider(describeIndexModel(index));
44
+ const [queryVector] = await provider.embed([query]);
45
+ if (queryVector.length !== dim) {
46
+ throw new Error(
47
+ `Query embedding is ${queryVector.length}-dimensional but this index is ${dim}-dimensional — ` +
48
+ `the embedding model must match the one used to build it (recorded model: ${index.getMeta('model')})`
49
+ );
50
+ }
51
+
52
+ return index.search(queryVector, k);
53
+ } finally {
54
+ index.close();
55
+ }
56
+ }
57
+ }
58
+
59
+ /** The model an index says it was built with, in the shape createProvider takes. */
60
+ export function describeIndexModel(index) {
61
+ const meta = { model: index.getMeta('model'), embedder: index.getMeta('embedder') };
62
+ const embedder = embedderOfIndex(meta);
63
+ return {
64
+ model: meta.model,
65
+ embedder,
66
+ dtype: embedder === EMBEDDER_TRANSFORMERS ? (index.getMeta('dtype') || LEGACY_DTYPE) : null,
67
+ };
68
+ }
@@ -0,0 +1,225 @@
1
+ import { existsSync } from 'node:fs';
2
+ import { createRequire } from 'node:module';
3
+ import Database from 'better-sqlite3';
4
+ import { VECTOR_INIT_SQL, SEARCH_SQL, vectorInitOptions, overfetchFor, mapSearchRow } from './schema.js';
5
+
6
+ // @sqliteai/sqlite-vector's own ESM build (dist/index.mjs) cannot find its platform binary
7
+ // package from a genuine ESM context — it tries a bare `require(packageName)` via a bundler-
8
+ // injected shim that only works when a real CJS `require` is already in scope, which never
9
+ // happens in native ESM (confirmed live: `import { getExtensionPath } from
10
+ // '@sqliteai/sqlite-vector'` throws `ExtensionNotFoundError` even when the platform package
11
+ // is correctly installed). `createRequire` is the standard Node workaround — it resolves the
12
+ // package's CJS build instead, where a real `require` exists and the same lookup works.
13
+ const require = createRequire(import.meta.url);
14
+ const { getExtensionPath } = require('@sqliteai/sqlite-vector');
15
+
16
+ /**
17
+ * A single, diff-friendly SQLite file holding embeddings for one synced folder — meant to be
18
+ * pushed alongside the folder's own content (a plain file, not dot-prefixed, so it rides
19
+ * through `push`/`pull`/`diff` unmodified) so a pull is immediately searchable with no
20
+ * re-embedding.
21
+ *
22
+ * The whole schema is designed around one constraint, verified empirically (not assumed):
23
+ * inserting rows into a plain table with a monotonic rowid touches only new/near-full pages,
24
+ * which content-defined chunking recognizes as mostly-unchanged on a re-push. A real
25
+ * measurement (200 existing rows, 5 new ones) showed ~2.9x overhead versus the new data's
26
+ * raw size — nowhere near a full-file rewrite, but *not* literally zero-touch either. What
27
+ * actually breaks this property: `VACUUM` (copies and rewrites every byte — never call it
28
+ * during routine reindexing) and `UPDATE`/`DELETE` on the (large, embedding-heavy)
29
+ * `search_index` table. `search_index_files` is deliberately the one place this class *does*
30
+ * mutate rows in place — it holds one row per file, so its churn is proportional to how many
31
+ * files actually changed, not to corpus size, and paying that small cost is what makes
32
+ * "which chunks are current" a correct query instead of a guess (see search_index_files' own
33
+ * doc below).
34
+ */
35
+ export default class VectorIndex {
36
+ /**
37
+ * @param {string} dbPath - path to the SQLite file. Created if it doesn't exist yet
38
+ * (unless `readonly`).
39
+ * @param {Object} [options]
40
+ * @param {boolean} [options.readonly=false] - open an existing index without writing
41
+ * anything to it — not even the schema check below. For inspecting an index that's about
42
+ * to be pushed (see Indexer.status()), where touching the file would change what's pushed.
43
+ */
44
+ constructor(dbPath, { readonly = false } = {}) {
45
+ this.path = dbPath;
46
+ this._db = readonly
47
+ ? new Database(dbPath, { readonly: true, fileMustExist: true })
48
+ : new Database(dbPath);
49
+ // Deliberately not enabling WAL mode: WAL leaves a separate `-wal`/`-shm` file holding
50
+ // uncommitted pages, which would need an explicit checkpoint before every push (a real
51
+ // operational footgun for anyone forgetting it). Left at SQLite's own default rollback
52
+ // journal instead — this class always closes the database (see close()) before anything
53
+ // downstream would push it, so there's never a mid-transaction state to leave behind.
54
+ this._db.loadExtension(getExtensionPath());
55
+ if (!readonly) this._ensureSchema();
56
+ }
57
+
58
+ _ensureSchema() {
59
+ this._db.exec(`
60
+ CREATE TABLE IF NOT EXISTS search_index (
61
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
62
+ path TEXT NOT NULL,
63
+ chunk_index INTEGER NOT NULL,
64
+ content_hash TEXT NOT NULL,
65
+ start_offset INTEGER NOT NULL,
66
+ end_offset INTEGER NOT NULL,
67
+ embedding BLOB NOT NULL
68
+ )
69
+ `);
70
+ // One row per file, deliberately mutable (see class doc) — "content_hash" here is what
71
+ // decides which rows in the (append-only) search_index table above actually count as
72
+ // current: a chunk only counts if its own content_hash matches this table's row for the
73
+ // same path. That's what lets a stale reindex's chunks keep existing physically (pure
74
+ // insert, never deleted) while still being correctly excluded from search results —
75
+ // see search()'s own join.
76
+ this._db.exec(`
77
+ CREATE TABLE IF NOT EXISTS search_index_files (
78
+ path TEXT PRIMARY KEY,
79
+ content_hash TEXT NOT NULL,
80
+ chunk_count INTEGER NOT NULL
81
+ )
82
+ `);
83
+ this._db.exec(`
84
+ CREATE TABLE IF NOT EXISTS search_index_meta (
85
+ key TEXT PRIMARY KEY,
86
+ value TEXT NOT NULL
87
+ )
88
+ `);
89
+ }
90
+
91
+ /**
92
+ * Registers `search_index`/`embedding` as a vector column with sqlite-vector — required on
93
+ * every fresh connection before `vector_full_scan` etc. work (confirmed live: omitting this
94
+ * fails with "vector_full_scan: unable to retrieve context", not a clearer error). Safe to
95
+ * call more than once; last call wins.
96
+ *
97
+ * @param {number} dim
98
+ * @param {'FLOAT32'|'INT8'} [type='FLOAT32']
99
+ */
100
+ initVectorColumn(dim, type = 'FLOAT32') {
101
+ this._dim = dim;
102
+ this._vectorType = type;
103
+ this._db.prepare(VECTOR_INIT_SQL).run(vectorInitOptions(dim, type));
104
+ }
105
+
106
+ /** @returns {?string} the file's currently-indexed content hash, or null if never indexed. */
107
+ getFileHash(path) {
108
+ const row = this._db.prepare('SELECT content_hash FROM search_index_files WHERE path = ?').get(path);
109
+ return row?.content_hash ?? null;
110
+ }
111
+
112
+ /** @returns {string[]} every path currently recorded in the index (indexed, not necessarily still on disk). */
113
+ listFiles() {
114
+ return this._db.prepare('SELECT path FROM search_index_files').all().map((r) => r.path);
115
+ }
116
+
117
+ /**
118
+ * Replaces one file's chunks. Never touches the file's *old* rows in `search_index` — it
119
+ * only ever inserts new ones (see class doc for why) — and upserts the single
120
+ * `search_index_files` row that makes the old ones stop counting as current.
121
+ *
122
+ * @param {string} path
123
+ * @param {string} contentHash - the file's new content hash (same hashing this package's
124
+ * caller already uses to detect the file changed in the first place).
125
+ * @param {{startOffset: number, endOffset: number, embedding: Float32Array}[]} chunks
126
+ */
127
+ replaceFileChunks(path, contentHash, chunks) {
128
+ const insert = this._db.prepare(`
129
+ INSERT INTO search_index (path, chunk_index, content_hash, start_offset, end_offset, embedding)
130
+ VALUES (?, ?, ?, ?, ?, ?)
131
+ `);
132
+ const upsertFile = this._db.prepare(`
133
+ INSERT INTO search_index_files (path, content_hash, chunk_count) VALUES (?, ?, ?)
134
+ ON CONFLICT(path) DO UPDATE SET content_hash = excluded.content_hash, chunk_count = excluded.chunk_count
135
+ `);
136
+ const tx = this._db.transaction(() => {
137
+ chunks.forEach((chunk, index) => {
138
+ insert.run(path, index, contentHash, chunk.startOffset, chunk.endOffset, Buffer.from(chunk.embedding.buffer, chunk.embedding.byteOffset, chunk.embedding.byteLength));
139
+ });
140
+ upsertFile.run(path, contentHash, chunks.length);
141
+ });
142
+ tx();
143
+ }
144
+
145
+ /** A file no longer exists in the tree — stop counting its (still physically present) chunks as current. */
146
+ removeFile(path) {
147
+ this._db.prepare('DELETE FROM search_index_files WHERE path = ?').run(path);
148
+ }
149
+
150
+ setMeta(key, value) {
151
+ this._db.prepare(`
152
+ INSERT INTO search_index_meta (key, value) VALUES (?, ?)
153
+ ON CONFLICT(key) DO UPDATE SET value = excluded.value
154
+ `).run(key, String(value));
155
+ }
156
+
157
+ /**
158
+ * An existing index's metadata, read without writing anything — or null when there's no
159
+ * index at `dbPath` (or it has no metadata yet).
160
+ * @returns {?Object<string, string>}
161
+ */
162
+ static readMeta(dbPath) {
163
+ if (!existsSync(dbPath)) return null;
164
+ const index = new VectorIndex(dbPath, { readonly: true });
165
+ try {
166
+ const rows = index._db.prepare('SELECT key, value FROM search_index_meta').all();
167
+ return rows.length ? Object.fromEntries(rows.map((r) => [r.key, r.value])) : null;
168
+ } catch {
169
+ // No meta table: not an index this package built.
170
+ return null;
171
+ } finally {
172
+ index.close();
173
+ }
174
+ }
175
+
176
+ /** @returns {?string} */
177
+ getMeta(key) {
178
+ return this._db.prepare('SELECT value FROM search_index_meta WHERE key = ?').get(key)?.value ?? null;
179
+ }
180
+
181
+ /**
182
+ * Nearest-neighbor search, current chunks only. Over-fetches from `vector_full_scan` (exact,
183
+ * brute-force — appropriate for the corpus sizes a synced folder actually has; no ANN/
184
+ * partitioning index, which would break the diffability this whole design depends on) since
185
+ * some of its top candidates may turn out to be stale once joined against
186
+ * `search_index_files`, and stops short of the caller's requested `k` otherwise.
187
+ *
188
+ * @param {Float32Array} queryEmbedding
189
+ * @param {number} k
190
+ * @returns {{path: string, chunkIndex: number, startOffset: number, endOffset: number, distance: number}[]}
191
+ */
192
+ search(queryEmbedding, k) {
193
+ const queryBuf = Buffer.from(queryEmbedding.buffer, queryEmbedding.byteOffset, queryEmbedding.byteLength);
194
+ // BigInt, not a plain number: better-sqlite3 binds every JS `number` parameter as SQLite
195
+ // REAL (confirmed live — even a whole number like 64), but vector_full_scan's own count
196
+ // argument strictly requires INTEGER and rejects REAL outright rather than coercing it.
197
+ // (The WASM build binds whole numbers as INTEGER on its own — see WasmIndexReader.)
198
+ return this._db
199
+ .prepare(SEARCH_SQL)
200
+ .all(queryBuf, BigInt(overfetchFor(k)), BigInt(k))
201
+ .map(mapSearchRow);
202
+ }
203
+
204
+ /**
205
+ * The deliberate, rare, full-rewrite operation — drops every chunk that isn't current
206
+ * (superseded by a later reindex of the same file, or belonging to a file removed from the
207
+ * index entirely) and reclaims the space. Costs a CDC-diff equivalent to a full snapshot —
208
+ * same trade routine reindexing exists specifically to avoid — so this is opt-in, not run
209
+ * automatically. Mirrors the CLI's own `compact` command for stream history.
210
+ */
211
+ compact() {
212
+ this._db.exec(`
213
+ DELETE FROM search_index
214
+ WHERE NOT EXISTS (
215
+ SELECT 1 FROM search_index_files f
216
+ WHERE f.path = search_index.path AND f.content_hash = search_index.content_hash
217
+ )
218
+ `);
219
+ this._db.exec('VACUUM');
220
+ }
221
+
222
+ close() {
223
+ this._db.close();
224
+ }
225
+ }
@@ -0,0 +1,131 @@
1
+ import WasmIndexReader from './WasmIndexReader.js';
2
+ import { LEGACY_DTYPE, EMBEDDER_TRANSFORMERS, embedderOfIndex } from '../schema.js';
3
+
4
+ /**
5
+ * The browser's `Searcher`: holds one open index (from bytes, not a path) and the embedding
6
+ * model that matches it, so repeated queries pay for neither again. `open()` again with new
7
+ * bytes when the index file changes; the model is only reloaded if the new index was built
8
+ * with a different one.
9
+ *
10
+ * Which provider to build for which index is the caller's job (`createProvider`), so this
11
+ * class — and the browser entry around it — never imports a model runtime itself: the
12
+ * explorer builds the static provider from its own files source, and loads transformers.js
13
+ * only for an index that needs it.
14
+ */
15
+ export default class BrowserSearcher {
16
+ /**
17
+ * @param {Object} params
18
+ * @param {any} params.sqlite3 - an initialized `@sqliteai/sqlite-wasm` module.
19
+ * @param {(opts: {model: string, embedder: string, dtype: ?string, onProgress: ?Function}) => (any|Promise<any>)} [params.createProvider] -
20
+ * builds the provider an index's metadata asks for.
21
+ * @param {any} [params.embeddingProvider] - use this one for every index instead (tests).
22
+ * @param {(event: Object) => void} [params.onModelProgress] - passed to createProvider.
23
+ */
24
+ constructor({ sqlite3, createProvider = null, embeddingProvider = null, onModelProgress = null } = {}) {
25
+ if (!sqlite3) throw new Error('BrowserSearcher: sqlite3 module is required');
26
+ if (!createProvider && !embeddingProvider) {
27
+ throw new Error('BrowserSearcher: pass createProvider or embeddingProvider');
28
+ }
29
+ this._sqlite3 = sqlite3;
30
+ this._createProvider = createProvider;
31
+ this._onModelProgress = onModelProgress;
32
+ this._fixedProvider = embeddingProvider;
33
+ this._provider = embeddingProvider;
34
+ this._providerKey = null;
35
+ this._reader = null;
36
+ }
37
+
38
+ /** @param {Uint8Array} bytes - the whole search_index.db file. */
39
+ open(bytes) {
40
+ const reader = new WasmIndexReader(this._sqlite3, bytes);
41
+ const dim = Number(reader.getMeta('dimension'));
42
+ if (!dim) {
43
+ reader.close();
44
+ throw new Error('This search index has no metadata — it was never built, or is empty.');
45
+ }
46
+ reader.initVectorColumn(dim);
47
+ this.close();
48
+ this._reader = reader;
49
+ this._dim = dim;
50
+ this._model = reader.getMeta('model');
51
+ this._embedder = embedderOfIndex({ model: this._model, embedder: reader.getMeta('embedder') });
52
+ this._dtype = this._embedder === EMBEDDER_TRANSFORMERS ? (reader.getMeta('dtype') || LEGACY_DTYPE) : null;
53
+ }
54
+
55
+ get isOpen() {
56
+ return this._reader !== null;
57
+ }
58
+
59
+ /** @returns {{model: ?string, embedder: string, dtype: ?string, dimension: number}} */
60
+ get info() {
61
+ return { model: this._model, embedder: this._embedder, dtype: this._dtype, dimension: this._dim };
62
+ }
63
+
64
+ /** @returns {Map<string, string>} see WasmIndexReader.getFileHashes */
65
+ getFileHashes() {
66
+ this._requireOpen();
67
+ return this._reader.getFileHashes();
68
+ }
69
+
70
+ /** @returns {boolean} whether the model the open index needs is already loaded. */
71
+ get isModelLoaded() {
72
+ if (!this._reader) return false;
73
+ if (this._fixedProvider) return true;
74
+ return this._providerKey === this._key() && !!this._provider?.isLoaded;
75
+ }
76
+
77
+ /** Loads the embedding model now (the slow, one-time part — a download on first use) so a
78
+ * caller can show that as its own step instead of as a slow first query. */
79
+ async loadModel() {
80
+ this._requireOpen();
81
+ const provider = await this._providerForIndex();
82
+ if (typeof provider.load === 'function') await provider.load();
83
+ }
84
+
85
+ /**
86
+ * @param {string} query
87
+ * @param {number} [k=5]
88
+ */
89
+ async search(query, k = 5) {
90
+ this._requireOpen();
91
+ const provider = await this._providerForIndex();
92
+ const [queryVector] = await provider.embed([query]);
93
+ if (queryVector.length !== this._dim) {
94
+ throw new Error(
95
+ `Query embedding is ${queryVector.length}-dimensional but this index is ${this._dim}-dimensional — ` +
96
+ `the embedding model must match the one used to build it (recorded model: ${this._model})`
97
+ );
98
+ }
99
+ return this._reader.search(queryVector, k);
100
+ }
101
+
102
+ close() {
103
+ if (this._reader) {
104
+ this._reader.close();
105
+ this._reader = null;
106
+ }
107
+ }
108
+
109
+ _key() {
110
+ return `${this._embedder}|${this._model}|${this._dtype}`;
111
+ }
112
+
113
+ async _providerForIndex() {
114
+ if (this._fixedProvider) return this._fixedProvider;
115
+ const key = this._key();
116
+ if (this._providerKey !== key) {
117
+ this._provider = await this._createProvider({
118
+ model: this._model,
119
+ embedder: this._embedder,
120
+ dtype: this._dtype,
121
+ onProgress: this._onModelProgress,
122
+ });
123
+ this._providerKey = key;
124
+ }
125
+ return this._provider;
126
+ }
127
+
128
+ _requireOpen() {
129
+ if (!this._reader) throw new Error('BrowserSearcher: call open(bytes) first');
130
+ }
131
+ }