mnemonad-cli 0.2.0 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,225 @@
1
+ import { existsSync } from 'node:fs';
2
+ import { createRequire } from 'node:module';
3
+ import Database from 'better-sqlite3';
4
+ import { VECTOR_INIT_SQL, SEARCH_SQL, vectorInitOptions, overfetchFor, mapSearchRow } from './schema.js';
5
+
6
+ // @sqliteai/sqlite-vector's own ESM build (dist/index.mjs) cannot find its platform binary
7
+ // package from a genuine ESM context — it tries a bare `require(packageName)` via a bundler-
8
+ // injected shim that only works when a real CJS `require` is already in scope, which never
9
+ // happens in native ESM (confirmed live: `import { getExtensionPath } from
10
+ // '@sqliteai/sqlite-vector'` throws `ExtensionNotFoundError` even when the platform package
11
+ // is correctly installed). `createRequire` is the standard Node workaround — it resolves the
12
+ // package's CJS build instead, where a real `require` exists and the same lookup works.
13
+ const require = createRequire(import.meta.url);
14
+ const { getExtensionPath } = require('@sqliteai/sqlite-vector');
15
+
16
+ /**
17
+ * A single, diff-friendly SQLite file holding embeddings for one synced folder — meant to be
18
+ * pushed alongside the folder's own content (a plain file, not dot-prefixed, so it rides
19
+ * through `push`/`pull`/`diff` unmodified) so a pull is immediately searchable with no
20
+ * re-embedding.
21
+ *
22
+ * The whole schema is designed around one constraint, verified empirically (not assumed):
23
+ * inserting rows into a plain table with a monotonic rowid touches only new/near-full pages,
24
+ * which content-defined chunking recognizes as mostly-unchanged on a re-push. A real
25
+ * measurement (200 existing rows, 5 new ones) showed ~2.9x overhead versus the new data's
26
+ * raw size — nowhere near a full-file rewrite, but *not* literally zero-touch either. What
27
+ * actually breaks this property: `VACUUM` (copies and rewrites every byte — never call it
28
+ * during routine reindexing) and `UPDATE`/`DELETE` on the (large, embedding-heavy)
29
+ * `search_index` table. `search_index_files` is deliberately the one place this class *does*
30
+ * mutate rows in place — it holds one row per file, so its churn is proportional to how many
31
+ * files actually changed, not to corpus size, and paying that small cost is what makes
32
+ * "which chunks are current" a correct query instead of a guess (see search_index_files' own
33
+ * doc below).
34
+ */
35
+ export default class VectorIndex {
36
+ /**
37
+ * @param {string} dbPath - path to the SQLite file. Created if it doesn't exist yet
38
+ * (unless `readonly`).
39
+ * @param {Object} [options]
40
+ * @param {boolean} [options.readonly=false] - open an existing index without writing
41
+ * anything to it — not even the schema check below. For inspecting an index that's about
42
+ * to be pushed (see Indexer.status()), where touching the file would change what's pushed.
43
+ */
44
+ constructor(dbPath, { readonly = false } = {}) {
45
+ this.path = dbPath;
46
+ this._db = readonly
47
+ ? new Database(dbPath, { readonly: true, fileMustExist: true })
48
+ : new Database(dbPath);
49
+ // Deliberately not enabling WAL mode: WAL leaves a separate `-wal`/`-shm` file holding
50
+ // uncommitted pages, which would need an explicit checkpoint before every push (a real
51
+ // operational footgun for anyone forgetting it). Left at SQLite's own default rollback
52
+ // journal instead — this class always closes the database (see close()) before anything
53
+ // downstream would push it, so there's never a mid-transaction state to leave behind.
54
+ this._db.loadExtension(getExtensionPath());
55
+ if (!readonly) this._ensureSchema();
56
+ }
57
+
58
+ _ensureSchema() {
59
+ this._db.exec(`
60
+ CREATE TABLE IF NOT EXISTS search_index (
61
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
62
+ path TEXT NOT NULL,
63
+ chunk_index INTEGER NOT NULL,
64
+ content_hash TEXT NOT NULL,
65
+ start_offset INTEGER NOT NULL,
66
+ end_offset INTEGER NOT NULL,
67
+ embedding BLOB NOT NULL
68
+ )
69
+ `);
70
+ // One row per file, deliberately mutable (see class doc) — "content_hash" here is what
71
+ // decides which rows in the (append-only) search_index table above actually count as
72
+ // current: a chunk only counts if its own content_hash matches this table's row for the
73
+ // same path. That's what lets a stale reindex's chunks keep existing physically (pure
74
+ // insert, never deleted) while still being correctly excluded from search results —
75
+ // see search()'s own join.
76
+ this._db.exec(`
77
+ CREATE TABLE IF NOT EXISTS search_index_files (
78
+ path TEXT PRIMARY KEY,
79
+ content_hash TEXT NOT NULL,
80
+ chunk_count INTEGER NOT NULL
81
+ )
82
+ `);
83
+ this._db.exec(`
84
+ CREATE TABLE IF NOT EXISTS search_index_meta (
85
+ key TEXT PRIMARY KEY,
86
+ value TEXT NOT NULL
87
+ )
88
+ `);
89
+ }
90
+
91
+ /**
92
+ * Registers `search_index`/`embedding` as a vector column with sqlite-vector — required on
93
+ * every fresh connection before `vector_full_scan` etc. work (confirmed live: omitting this
94
+ * fails with "vector_full_scan: unable to retrieve context", not a clearer error). Safe to
95
+ * call more than once; last call wins.
96
+ *
97
+ * @param {number} dim
98
+ * @param {'FLOAT32'|'INT8'} [type='FLOAT32']
99
+ */
100
+ initVectorColumn(dim, type = 'FLOAT32') {
101
+ this._dim = dim;
102
+ this._vectorType = type;
103
+ this._db.prepare(VECTOR_INIT_SQL).run(vectorInitOptions(dim, type));
104
+ }
105
+
106
+ /** @returns {?string} the file's currently-indexed content hash, or null if never indexed. */
107
+ getFileHash(path) {
108
+ const row = this._db.prepare('SELECT content_hash FROM search_index_files WHERE path = ?').get(path);
109
+ return row?.content_hash ?? null;
110
+ }
111
+
112
+ /** @returns {string[]} every path currently recorded in the index (indexed, not necessarily still on disk). */
113
+ listFiles() {
114
+ return this._db.prepare('SELECT path FROM search_index_files').all().map((r) => r.path);
115
+ }
116
+
117
+ /**
118
+ * Replaces one file's chunks. Never touches the file's *old* rows in `search_index` — it
119
+ * only ever inserts new ones (see class doc for why) — and upserts the single
120
+ * `search_index_files` row that makes the old ones stop counting as current.
121
+ *
122
+ * @param {string} path
123
+ * @param {string} contentHash - the file's new content hash (same hashing this package's
124
+ * caller already uses to detect the file changed in the first place).
125
+ * @param {{startOffset: number, endOffset: number, embedding: Float32Array}[]} chunks
126
+ */
127
+ replaceFileChunks(path, contentHash, chunks) {
128
+ const insert = this._db.prepare(`
129
+ INSERT INTO search_index (path, chunk_index, content_hash, start_offset, end_offset, embedding)
130
+ VALUES (?, ?, ?, ?, ?, ?)
131
+ `);
132
+ const upsertFile = this._db.prepare(`
133
+ INSERT INTO search_index_files (path, content_hash, chunk_count) VALUES (?, ?, ?)
134
+ ON CONFLICT(path) DO UPDATE SET content_hash = excluded.content_hash, chunk_count = excluded.chunk_count
135
+ `);
136
+ const tx = this._db.transaction(() => {
137
+ chunks.forEach((chunk, index) => {
138
+ insert.run(path, index, contentHash, chunk.startOffset, chunk.endOffset, Buffer.from(chunk.embedding.buffer, chunk.embedding.byteOffset, chunk.embedding.byteLength));
139
+ });
140
+ upsertFile.run(path, contentHash, chunks.length);
141
+ });
142
+ tx();
143
+ }
144
+
145
+ /** A file no longer exists in the tree — stop counting its (still physically present) chunks as current. */
146
+ removeFile(path) {
147
+ this._db.prepare('DELETE FROM search_index_files WHERE path = ?').run(path);
148
+ }
149
+
150
+ setMeta(key, value) {
151
+ this._db.prepare(`
152
+ INSERT INTO search_index_meta (key, value) VALUES (?, ?)
153
+ ON CONFLICT(key) DO UPDATE SET value = excluded.value
154
+ `).run(key, String(value));
155
+ }
156
+
157
+ /**
158
+ * An existing index's metadata, read without writing anything — or null when there's no
159
+ * index at `dbPath` (or it has no metadata yet).
160
+ * @returns {?Object<string, string>}
161
+ */
162
+ static readMeta(dbPath) {
163
+ if (!existsSync(dbPath)) return null;
164
+ const index = new VectorIndex(dbPath, { readonly: true });
165
+ try {
166
+ const rows = index._db.prepare('SELECT key, value FROM search_index_meta').all();
167
+ return rows.length ? Object.fromEntries(rows.map((r) => [r.key, r.value])) : null;
168
+ } catch {
169
+ // No meta table: not an index this package built.
170
+ return null;
171
+ } finally {
172
+ index.close();
173
+ }
174
+ }
175
+
176
+ /** @returns {?string} */
177
+ getMeta(key) {
178
+ return this._db.prepare('SELECT value FROM search_index_meta WHERE key = ?').get(key)?.value ?? null;
179
+ }
180
+
181
+ /**
182
+ * Nearest-neighbor search, current chunks only. Over-fetches from `vector_full_scan` (exact,
183
+ * brute-force — appropriate for the corpus sizes a synced folder actually has; no ANN/
184
+ * partitioning index, which would break the diffability this whole design depends on) since
185
+ * some of its top candidates may turn out to be stale once joined against
186
+ * `search_index_files`, and stops short of the caller's requested `k` otherwise.
187
+ *
188
+ * @param {Float32Array} queryEmbedding
189
+ * @param {number} k
190
+ * @returns {{path: string, chunkIndex: number, startOffset: number, endOffset: number, distance: number}[]}
191
+ */
192
+ search(queryEmbedding, k) {
193
+ const queryBuf = Buffer.from(queryEmbedding.buffer, queryEmbedding.byteOffset, queryEmbedding.byteLength);
194
+ // BigInt, not a plain number: better-sqlite3 binds every JS `number` parameter as SQLite
195
+ // REAL (confirmed live — even a whole number like 64), but vector_full_scan's own count
196
+ // argument strictly requires INTEGER and rejects REAL outright rather than coercing it.
197
+ // (The WASM build binds whole numbers as INTEGER on its own — see WasmIndexReader.)
198
+ return this._db
199
+ .prepare(SEARCH_SQL)
200
+ .all(queryBuf, BigInt(overfetchFor(k)), BigInt(k))
201
+ .map(mapSearchRow);
202
+ }
203
+
204
+ /**
205
+ * The deliberate, rare, full-rewrite operation — drops every chunk that isn't current
206
+ * (superseded by a later reindex of the same file, or belonging to a file removed from the
207
+ * index entirely) and reclaims the space. Costs a CDC-diff equivalent to a full snapshot —
208
+ * same trade routine reindexing exists specifically to avoid — so this is opt-in, not run
209
+ * automatically. Mirrors the CLI's own `compact` command for stream history.
210
+ */
211
+ compact() {
212
+ this._db.exec(`
213
+ DELETE FROM search_index
214
+ WHERE NOT EXISTS (
215
+ SELECT 1 FROM search_index_files f
216
+ WHERE f.path = search_index.path AND f.content_hash = search_index.content_hash
217
+ )
218
+ `);
219
+ this._db.exec('VACUUM');
220
+ }
221
+
222
+ close() {
223
+ this._db.close();
224
+ }
225
+ }
@@ -0,0 +1,131 @@
1
+ import WasmIndexReader from './WasmIndexReader.js';
2
+ import { LEGACY_DTYPE, EMBEDDER_TRANSFORMERS, embedderOfIndex } from '../schema.js';
3
+
4
+ /**
5
+ * The browser's `Searcher`: holds one open index (from bytes, not a path) and the embedding
6
+ * model that matches it, so repeated queries pay for neither again. `open()` again with new
7
+ * bytes when the index file changes; the model is only reloaded if the new index was built
8
+ * with a different one.
9
+ *
10
+ * Which provider to build for which index is the caller's job (`createProvider`), so this
11
+ * class — and the browser entry around it — never imports a model runtime itself: the
12
+ * explorer builds the static provider from its own files source, and loads transformers.js
13
+ * only for an index that needs it.
14
+ */
15
+ export default class BrowserSearcher {
16
+ /**
17
+ * @param {Object} params
18
+ * @param {any} params.sqlite3 - an initialized `@sqliteai/sqlite-wasm` module.
19
+ * @param {(opts: {model: string, embedder: string, dtype: ?string, onProgress: ?Function}) => (any|Promise<any>)} [params.createProvider] -
20
+ * builds the provider an index's metadata asks for.
21
+ * @param {any} [params.embeddingProvider] - use this one for every index instead (tests).
22
+ * @param {(event: Object) => void} [params.onModelProgress] - passed to createProvider.
23
+ */
24
+ constructor({ sqlite3, createProvider = null, embeddingProvider = null, onModelProgress = null } = {}) {
25
+ if (!sqlite3) throw new Error('BrowserSearcher: sqlite3 module is required');
26
+ if (!createProvider && !embeddingProvider) {
27
+ throw new Error('BrowserSearcher: pass createProvider or embeddingProvider');
28
+ }
29
+ this._sqlite3 = sqlite3;
30
+ this._createProvider = createProvider;
31
+ this._onModelProgress = onModelProgress;
32
+ this._fixedProvider = embeddingProvider;
33
+ this._provider = embeddingProvider;
34
+ this._providerKey = null;
35
+ this._reader = null;
36
+ }
37
+
38
+ /** @param {Uint8Array} bytes - the whole search_index.db file. */
39
+ open(bytes) {
40
+ const reader = new WasmIndexReader(this._sqlite3, bytes);
41
+ const dim = Number(reader.getMeta('dimension'));
42
+ if (!dim) {
43
+ reader.close();
44
+ throw new Error('This search index has no metadata — it was never built, or is empty.');
45
+ }
46
+ reader.initVectorColumn(dim);
47
+ this.close();
48
+ this._reader = reader;
49
+ this._dim = dim;
50
+ this._model = reader.getMeta('model');
51
+ this._embedder = embedderOfIndex({ model: this._model, embedder: reader.getMeta('embedder') });
52
+ this._dtype = this._embedder === EMBEDDER_TRANSFORMERS ? (reader.getMeta('dtype') || LEGACY_DTYPE) : null;
53
+ }
54
+
55
+ get isOpen() {
56
+ return this._reader !== null;
57
+ }
58
+
59
+ /** @returns {{model: ?string, embedder: string, dtype: ?string, dimension: number}} */
60
+ get info() {
61
+ return { model: this._model, embedder: this._embedder, dtype: this._dtype, dimension: this._dim };
62
+ }
63
+
64
+ /** @returns {Map<string, string>} see WasmIndexReader.getFileHashes */
65
+ getFileHashes() {
66
+ this._requireOpen();
67
+ return this._reader.getFileHashes();
68
+ }
69
+
70
+ /** @returns {boolean} whether the model the open index needs is already loaded. */
71
+ get isModelLoaded() {
72
+ if (!this._reader) return false;
73
+ if (this._fixedProvider) return true;
74
+ return this._providerKey === this._key() && !!this._provider?.isLoaded;
75
+ }
76
+
77
+ /** Loads the embedding model now (the slow, one-time part — a download on first use) so a
78
+ * caller can show that as its own step instead of as a slow first query. */
79
+ async loadModel() {
80
+ this._requireOpen();
81
+ const provider = await this._providerForIndex();
82
+ if (typeof provider.load === 'function') await provider.load();
83
+ }
84
+
85
+ /**
86
+ * @param {string} query
87
+ * @param {number} [k=5]
88
+ */
89
+ async search(query, k = 5) {
90
+ this._requireOpen();
91
+ const provider = await this._providerForIndex();
92
+ const [queryVector] = await provider.embed([query]);
93
+ if (queryVector.length !== this._dim) {
94
+ throw new Error(
95
+ `Query embedding is ${queryVector.length}-dimensional but this index is ${this._dim}-dimensional — ` +
96
+ `the embedding model must match the one used to build it (recorded model: ${this._model})`
97
+ );
98
+ }
99
+ return this._reader.search(queryVector, k);
100
+ }
101
+
102
+ close() {
103
+ if (this._reader) {
104
+ this._reader.close();
105
+ this._reader = null;
106
+ }
107
+ }
108
+
109
+ _key() {
110
+ return `${this._embedder}|${this._model}|${this._dtype}`;
111
+ }
112
+
113
+ async _providerForIndex() {
114
+ if (this._fixedProvider) return this._fixedProvider;
115
+ const key = this._key();
116
+ if (this._providerKey !== key) {
117
+ this._provider = await this._createProvider({
118
+ model: this._model,
119
+ embedder: this._embedder,
120
+ dtype: this._dtype,
121
+ onProgress: this._onModelProgress,
122
+ });
123
+ this._providerKey = key;
124
+ }
125
+ return this._provider;
126
+ }
127
+
128
+ _requireOpen() {
129
+ if (!this._reader) throw new Error('BrowserSearcher: call open(bytes) first');
130
+ }
131
+ }
@@ -0,0 +1,93 @@
1
+ import {
2
+ TABLE_FILES, TABLE_META, VECTOR_INIT_SQL, SEARCH_SQL,
3
+ vectorInitOptions, overfetchFor, mapSearchRow,
4
+ } from '../schema.js';
5
+
6
+ /**
7
+ * Read-only view of a `search_index.db` held in memory — the browser counterpart to
8
+ * `VectorIndex`, for a page that already has the file's bytes (the explorer does: after a
9
+ * restore, the whole folder, index included, lives in memory) and no filesystem to open it from.
10
+ *
11
+ * Runs on @sqliteai/sqlite-wasm, which ships the same sqlite-vector build the Node side loads
12
+ * as a native extension, so it answers with the same `vector_full_scan` query (shared through
13
+ * schema.js) and returns the same distances — not a second, JS-side scoring that could drift.
14
+ *
15
+ * The sqlite3 module is passed in rather than imported here: it's a ~3 MB WASM payload the
16
+ * caller decides when (and whether) to load, and it keeps this package free of a hard browser-
17
+ * only dependency: `(await import('@sqliteai/sqlite-wasm')).default()` is the usual way to get one.
18
+ */
19
+ export default class WasmIndexReader {
20
+ /**
21
+ * @param {any} sqlite3 - an initialized module from `@sqliteai/sqlite-wasm`'s default export.
22
+ * @param {Uint8Array} bytes - the whole database file.
23
+ */
24
+ constructor(sqlite3, bytes) {
25
+ this._sqlite3 = sqlite3;
26
+ this._db = new sqlite3.oo1.DB(':memory:');
27
+ // sqlite3_deserialize takes ownership of a WASM-heap copy (FREEONCLOSE), so the caller's
28
+ // buffer is never referenced again — safe to hand over bytes that later get replaced by a
29
+ // live poll. READONLY: nothing here should ever write, and a stray write would otherwise
30
+ // silently diverge this copy from the file actually in the stream.
31
+ const ptr = sqlite3.wasm.allocFromTypedArray(bytes);
32
+ const { capi } = sqlite3;
33
+ const rc = capi.sqlite3_deserialize(
34
+ this._db.pointer, 'main', ptr, bytes.byteLength, bytes.byteLength,
35
+ capi.SQLITE_DESERIALIZE_FREEONCLOSE | capi.SQLITE_DESERIALIZE_READONLY,
36
+ );
37
+ if (rc !== 0) {
38
+ this._db.close();
39
+ throw new Error(`WasmIndexReader: couldn't open the index (sqlite3_deserialize rc=${rc})`);
40
+ }
41
+ this._dim = null;
42
+ }
43
+
44
+ /** @returns {?string} */
45
+ getMeta(key) {
46
+ const value = this._db.selectValue(`SELECT value FROM ${TABLE_META} WHERE key = ?`, [key]);
47
+ return value ?? null;
48
+ }
49
+
50
+ /** @returns {Map<string, string>} every currently-indexed path → the content hash it was
51
+ * indexed at — what a caller compares against the live files to spot stale hits. */
52
+ getFileHashes() {
53
+ const rows = this._db.exec({
54
+ sql: `SELECT path, content_hash FROM ${TABLE_FILES}`,
55
+ rowMode: 'object',
56
+ returnValue: 'resultRows',
57
+ });
58
+ return new Map(rows.map((r) => [r.path, r.content_hash]));
59
+ }
60
+
61
+ /** Same role as `VectorIndex.initVectorColumn` — required once per connection before search. */
62
+ initVectorColumn(dim, type = 'FLOAT32') {
63
+ this._db.exec({ sql: VECTOR_INIT_SQL, bind: [vectorInitOptions(dim, type)] });
64
+ this._dim = dim;
65
+ }
66
+
67
+ /**
68
+ * @param {Float32Array} queryEmbedding
69
+ * @param {number} k
70
+ * @returns {{path: string, chunkIndex: number, startOffset: number, endOffset: number, distance: number}[]}
71
+ */
72
+ search(queryEmbedding, k) {
73
+ if (this._dim === null) {
74
+ const dim = Number(this.getMeta('dimension'));
75
+ if (!dim) throw new Error('WasmIndexReader: index has no dimension recorded — is it empty?');
76
+ this.initVectorColumn(dim);
77
+ }
78
+ const queryBlob = new Uint8Array(queryEmbedding.buffer, queryEmbedding.byteOffset, queryEmbedding.byteLength);
79
+ // Plain numbers are fine here, unlike better-sqlite3 (see VectorIndex.search): the WASM
80
+ // binding binds whole JS numbers as INTEGER, which vector_full_scan requires.
81
+ const rows = this._db.exec({
82
+ sql: SEARCH_SQL,
83
+ bind: [queryBlob, overfetchFor(k), k],
84
+ rowMode: 'object',
85
+ returnValue: 'resultRows',
86
+ });
87
+ return rows.map(mapSearchRow);
88
+ }
89
+
90
+ close() {
91
+ this._db.close();
92
+ }
93
+ }
@@ -0,0 +1,13 @@
1
+ // Browser-safe entry (the explorer imports this through a Vite alias): nothing here imports
2
+ // better-sqlite3, the native sqlite-vector extension or any `node:*` module. Searching only —
3
+ // building an index stays a Node job (`mnemonad index`), since that's where the folder's files
4
+ // live on disk and where embedding the whole corpus belongs; the browser only ever embeds a
5
+ // query string.
6
+ export { default as WasmIndexReader } from './browser/WasmIndexReader.js';
7
+ export { default as BrowserSearcher } from './browser/BrowserSearcher.js';
8
+ export { default as EmbeddingProvider } from './embeddings/EmbeddingProvider.js';
9
+ export { default as StaticEmbeddingProvider } from './embeddings/StaticEmbeddingProvider.js';
10
+ export { browserModelFiles } from './embeddings/modelFiles.browser.js';
11
+ export {
12
+ DEFAULT_DB_NAME, DEFAULT_MODEL, LEGACY_DTYPE, EMBEDDER_STATIC, EMBEDDER_TRANSFORMERS, embedderOfIndex,
13
+ } from './schema.js';
@@ -0,0 +1,23 @@
1
+ /**
2
+ * Base class for splitting one file's text content into embeddable chunks. A subclass fills
3
+ * in `chunk()`; everything else here is just the shared shape `Indexer` depends on.
4
+ *
5
+ * Deliberately separate from Mnemonad's own content-defined chunking (`FastCDC`, used for
6
+ * on-chain diffing): that one finds boundaries by content hash to maximize byte-level dedup
7
+ * between versions, which has nothing to do with what makes a good *semantic* unit to embed.
8
+ * The two chunkers operate on the same bytes for entirely unrelated reasons and should never
9
+ * be conflated.
10
+ */
11
+ export default class ChunkingStrategy {
12
+ /**
13
+ * @param {string} text - decoded file content (chunking only ever runs on text files —
14
+ * the caller is responsible for skipping binaries before this is reached).
15
+ * @returns {{startOffset: number, endOffset: number, text: string}[]} offsets are UTF-16
16
+ * code-unit offsets into `text` (i.e. plain JS string indices — `text.slice(startOffset,
17
+ * endOffset)` round-trips), not byte offsets — cheap to compute, and exact enough to
18
+ * locate a snippet back in the source file at display time.
19
+ */
20
+ chunk(_text) {
21
+ throw new Error(`${this.constructor.name}: chunk() not implemented`);
22
+ }
23
+ }
@@ -0,0 +1,69 @@
1
+ import ChunkingStrategy from './ChunkingStrategy.js';
2
+
3
+ /**
4
+ * Default chunker: fixed-size character windows with overlap, breaking on the nearest
5
+ * paragraph/sentence/word boundary before the target size rather than mid-word, when one is
6
+ * available within a reasonable lookback. Character-based rather than token-based
7
+ * deliberately — a token-accurate chunker needs a loaded tokenizer first, coupling chunking
8
+ * to whichever embedding model happens to be configured; character windows chunk perfectly
9
+ * well standalone, synchronously, with no model dependency, and 1000 characters is a
10
+ * conservative-enough stand-in for typical embedding models' token limits (roughly 4
11
+ * characters/token in English prose, so ~250 tokens — comfortably under the 256-512 token
12
+ * budget most small embedding models expect). A token-aware `ChunkingStrategy` reusing the
13
+ * configured `EmbeddingProvider`'s own tokenizer is a reasonable later swap-in — this stays
14
+ * the default because it works content-agnostically with zero setup.
15
+ */
16
+ export default class TextWindowChunkingStrategy extends ChunkingStrategy {
17
+ /**
18
+ * @param {Object} [params]
19
+ * @param {number} [params.size=1000] - target chunk size, in characters
20
+ * @param {number} [params.overlap=150] - characters of overlap between consecutive chunks
21
+ */
22
+ constructor({ size = 1000, overlap = 150 } = {}) {
23
+ super();
24
+ if (overlap >= size) {
25
+ throw new Error('TextWindowChunkingStrategy: overlap must be smaller than size');
26
+ }
27
+ this.size = size;
28
+ this.overlap = overlap;
29
+ }
30
+
31
+ chunk(text) {
32
+ if (!text) return [];
33
+ const chunks = [];
34
+ const step = this.size - this.overlap;
35
+ let start = 0;
36
+ while (start < text.length) {
37
+ let end = Math.min(start + this.size, text.length);
38
+ // Prefer breaking at a paragraph/sentence/word boundary over a hard mid-word cut —
39
+ // searched backward from `end`, but never past `start` (a very long unbroken run of
40
+ // text, e.g. a minified file, just gets a hard cut, which is fine: it wasn't going to
41
+ // chunk meaningfully either way).
42
+ if (end < text.length) {
43
+ const lookback = text.slice(start, end);
44
+ const boundary = _lastBoundary(lookback);
45
+ if (boundary > 0) end = start + boundary;
46
+ }
47
+ chunks.push({ startOffset: start, endOffset: end, text: text.slice(start, end) });
48
+ if (end >= text.length) break;
49
+ start = end - this.overlap;
50
+ }
51
+ return chunks;
52
+ }
53
+ }
54
+
55
+ /**
56
+ * Index just past the best break point found in `s` (paragraph > sentence > word), searched
57
+ * from the end backward, or -1 if none exists worth preferring over a hard cut.
58
+ * @param {string} s
59
+ * @returns {number}
60
+ */
61
+ function _lastBoundary(s) {
62
+ const paragraph = s.lastIndexOf('\n\n');
63
+ if (paragraph > s.length * 0.5) return paragraph + 2;
64
+ const sentence = Math.max(s.lastIndexOf('. '), s.lastIndexOf('.\n'));
65
+ if (sentence > s.length * 0.5) return sentence + 1;
66
+ const word = s.lastIndexOf(' ');
67
+ if (word > s.length * 0.5) return word + 1;
68
+ return -1;
69
+ }
@@ -0,0 +1,40 @@
1
+ /**
2
+ * Base class for turning text into vectors. A subclass fills in `embed()`/`dimension`;
3
+ * everything downstream (`VectorIndex`, `Indexer`, `Searcher`) only ever talks to this
4
+ * interface, never to a specific model or runtime — swapping the default local model for an
5
+ * API-based one later needs no change outside a new subclass.
6
+ */
7
+ export default class EmbeddingProvider {
8
+ /** @returns {number} vector length this provider produces — fixed for its lifetime. */
9
+ get dimension() {
10
+ throw new Error(`${this.constructor.name}: dimension getter not implemented`);
11
+ }
12
+
13
+ /** @returns {string} identifies this provider + model in `search_index_meta`, so a later
14
+ * `search` (possibly a different process entirely) knows what it's matching against. */
15
+ get modelId() {
16
+ throw new Error(`${this.constructor.name}: modelId getter not implemented`);
17
+ }
18
+
19
+ /** @returns {?string} which kind of embedder this is ('model2vec', 'transformers'), recorded
20
+ * in the index so a later search builds the same kind (see schema.js). */
21
+ get embedder() {
22
+ return null;
23
+ }
24
+
25
+ /** @returns {?string} weight precision (e.g. 'q8', 'fp32') when the model comes in more than
26
+ * one, recorded alongside `modelId` so a searcher loads the same weights. Null when the
27
+ * provider has no such choice. */
28
+ get dtype() {
29
+ return null;
30
+ }
31
+
32
+ /**
33
+ * @param {string[]} texts
34
+ * @returns {Promise<Float32Array[]>} one vector per input text, same order, each of length
35
+ * `this.dimension`.
36
+ */
37
+ async embed(_texts) {
38
+ throw new Error(`${this.constructor.name}: embed() not implemented`);
39
+ }
40
+ }